@furongjun1999/dsh-memory 0.4.11 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -16
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +142 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/hooks.js +36 -2
- package/lib/lib/roleplay_web.js +427 -427
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +368 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1327 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +285 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1005 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +300 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1536 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1097 -1097
- package/md_cg/crypto.py +437 -437
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +334 -334
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +580 -580
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +220 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +329 -329
- package/md_cg/hotcache.py +238 -214
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +199 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +622 -622
- package/md_cg/mcp_server.py +129 -32
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +176 -117
- package/md_cg/mdcos.py +79 -13
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +693 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +298 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +85 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +365 -365
- package/md_cg/scrub.py +852 -852
- package/md_cg/security.py +274 -274
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +151 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +562 -562
- package/md_cg/sources.py +815 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +48 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +1138 -1138
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branches.py +249 -249
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_en_pipeline.py +166 -166
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_identity_attribution.py +147 -147
- package/md_cg/test_index_durability.py +224 -224
- package/md_cg/test_interop.py +93 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +765 -765
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +298 -298
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +113 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +281 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_readcache_prodpath.py +155 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +340 -340
- package/md_cg/test_retr_s1b.py +209 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +384 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +175 -175
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +241 -241
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +397 -397
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +273 -273
- package/md_cg/tokens.py +677 -663
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +667 -667
- package/md_cg/vision_evidence.py +666 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +550 -542
- package/package.json +97 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +401 -401
- package/src/hooks.ts +38 -2
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +932 -932
- package/src/lib/token_store.ts +192 -192
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/crosscheck.py
CHANGED
|
@@ -1,1098 +1,1098 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
|
|
3
|
-
|
|
4
|
-
设计约束(与计划一致):
|
|
5
|
-
|
|
6
|
-
* **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
|
|
7
|
-
`python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
|
|
8
|
-
* **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
|
|
9
|
-
验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
|
|
10
|
-
* **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
|
|
11
|
-
理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
|
|
12
|
-
赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
|
|
13
|
-
* **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
|
|
14
|
-
拿不出证据的节点绝不写 `verification_basis`。
|
|
15
|
-
* **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
|
|
16
|
-
时撤销,否则计入 conflict 跳过。
|
|
17
|
-
|
|
18
|
-
`_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
|
|
19
|
-
"""
|
|
20
|
-
|
|
21
|
-
from __future__ import annotations
|
|
22
|
-
|
|
23
|
-
import argparse
|
|
24
|
-
import json
|
|
25
|
-
import os
|
|
26
|
-
import re
|
|
27
|
-
import sys
|
|
28
|
-
import time
|
|
29
|
-
from collections import OrderedDict
|
|
30
|
-
|
|
31
|
-
from . import crypto, nodefile
|
|
32
|
-
from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
|
|
33
|
-
SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
|
|
34
|
-
_ensure_comment, _entry_id, _readable_guard,
|
|
35
|
-
_remove_ccg_line, _sha, derive_fields)
|
|
36
|
-
from .consolidate import _has_ccg_line, _upsert_ccg_line
|
|
37
|
-
from .fsutil import append_jsonl, read_jsonl
|
|
38
|
-
from .mdcos import _ccg_field
|
|
39
|
-
|
|
40
|
-
# ---- 常量 ----------------------------------------------------------------
|
|
41
|
-
|
|
42
|
-
CROSSCHECK_LOG = "_crosscheck.jsonl"
|
|
43
|
-
CROSSCHECK_BATCH = "crosscheck"
|
|
44
|
-
|
|
45
|
-
REFLECT_UNIT = "reflect"
|
|
46
|
-
VERIFY_UNIT = "verify"
|
|
47
|
-
|
|
48
|
-
# 可写字段(本管线只动这一处,杜绝越界面)
|
|
49
|
-
WRITABLE_FIELDS = ("验证方式",)
|
|
50
|
-
FIELD_NORMALIZE = {
|
|
51
|
-
"验证方式": "验证方式",
|
|
52
|
-
"verification_basis": "验证方式",
|
|
53
|
-
"验证": "验证方式",
|
|
54
|
-
"verification": "验证方式",
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
# 赛道 → 来源执照策略
|
|
58
|
-
SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
|
|
59
|
-
|
|
60
|
-
# 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
|
|
61
|
-
SCIENCE_SUBJECT_HINTS = (
|
|
62
|
-
"数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
|
|
63
|
-
)
|
|
64
|
-
HUMANITIES_SUBJECT_HINTS = (
|
|
65
|
-
"语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
|
|
66
|
-
"文学", "哲学", "艺术", "音乐", "美术",
|
|
67
|
-
)
|
|
68
|
-
|
|
69
|
-
# 基底枚举 → 可读标签(条件化表述引用)
|
|
70
|
-
BASIS_LABEL = {
|
|
71
|
-
"compiler": "编译器/静态检查",
|
|
72
|
-
"test": "单元测试",
|
|
73
|
-
"measurement": "实测数据",
|
|
74
|
-
"formal_proof": "形式化证明",
|
|
75
|
-
"data": "数据统计",
|
|
76
|
-
"textbook": "人教版教材",
|
|
77
|
-
"public_kb": "公开知识库",
|
|
78
|
-
"other": "人工评审",
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
|
|
82
|
-
B_CLAIM = "B_valuation"
|
|
83
|
-
A_CLAIM = "A_fact"
|
|
84
|
-
|
|
85
|
-
# B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
|
|
86
|
-
VALUATION_MARKERS = (
|
|
87
|
-
"结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
|
|
88
|
-
"精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
|
|
89
|
-
"不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
|
|
90
|
-
"最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
|
|
91
|
-
"非凡",
|
|
92
|
-
)
|
|
93
|
-
VALUATION_PATTERNS = (
|
|
94
|
-
re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
|
|
95
|
-
re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
|
|
96
|
-
)
|
|
97
|
-
|
|
98
|
-
CONDITION_MARK = "〔来源限定〕"
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
# ---- 赛道与来源执照 ------------------------------------------------------
|
|
102
|
-
|
|
103
|
-
# 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
|
|
104
|
-
def classify_track(fm: dict, content: str = "") -> str:
|
|
105
|
-
"""判定节点赛道:`humanities` / `science` / `undetermined`。
|
|
106
|
-
|
|
107
|
-
只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
|
|
108
|
-
"""
|
|
109
|
-
fm = fm or {}
|
|
110
|
-
explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
|
|
111
|
-
if explicit in ("humanities", "文科", "arts"):
|
|
112
|
-
return "humanities"
|
|
113
|
-
if explicit in ("science", "理科", "stem"):
|
|
114
|
-
return "science"
|
|
115
|
-
st = fm.get("state_attributes")
|
|
116
|
-
name = _as_text(st.get("name")) if isinstance(st, dict) else ""
|
|
117
|
-
parts = [
|
|
118
|
-
name,
|
|
119
|
-
_as_text(fm.get("title")),
|
|
120
|
-
_as_text(fm.get("discipline")),
|
|
121
|
-
_as_text(fm.get("subject")),
|
|
122
|
-
" ".join(str(t) for t in (fm.get("tags") or [])),
|
|
123
|
-
_ccg_field(content, "功能名"),
|
|
124
|
-
_ccg_field(content, "执行"),
|
|
125
|
-
_ccg_field(content, "子功能"),
|
|
126
|
-
]
|
|
127
|
-
text = " ".join(p for p in parts if p)
|
|
128
|
-
has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
|
|
129
|
-
has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
|
|
130
|
-
if has_sci and not has_hum:
|
|
131
|
-
return "science"
|
|
132
|
-
if has_hum and not has_sci:
|
|
133
|
-
return "humanities"
|
|
134
|
-
return "undetermined"
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
# 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
|
|
138
|
-
def source_policy(track: str) -> str:
|
|
139
|
-
"""赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
|
|
140
|
-
return SOURCE_POLICY.get(track or "", "")
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
# 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
|
|
144
|
-
def allowed_basis(track: str) -> tuple:
|
|
145
|
-
if track == "science":
|
|
146
|
-
return tuple(nodefile.REPRODUCIBLE_BASIS)
|
|
147
|
-
if track == "humanities":
|
|
148
|
-
return tuple(nodefile.CONSISTENCY_BASIS)
|
|
149
|
-
return ()
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
# 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
|
|
153
|
-
def basis_licensed(track: str, basis) -> bool:
|
|
154
|
-
"""来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
|
|
155
|
-
return bool(basis) and str(basis) in allowed_basis(track)
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
# 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
|
|
159
|
-
def normalize_field(field) -> str:
|
|
160
|
-
return FIELD_NORMALIZE.get(str(field or "").strip(), "")
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
# 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
|
|
164
|
-
def _as_source(v) -> list:
|
|
165
|
-
if v is None:
|
|
166
|
-
return []
|
|
167
|
-
items = list(v) if isinstance(v, (list, tuple)) else [v]
|
|
168
|
-
return [str(x).strip() for x in items if str(x).strip()]
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
# ---- B 型识别与条件化改写 ------------------------------------------------
|
|
172
|
-
|
|
173
|
-
# 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
|
|
174
|
-
def claim_type(text) -> str:
|
|
175
|
-
"""`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
|
|
176
|
-
s = str(text or "")
|
|
177
|
-
if not s.strip():
|
|
178
|
-
return A_CLAIM
|
|
179
|
-
if any(m in s for m in VALUATION_MARKERS):
|
|
180
|
-
return B_CLAIM
|
|
181
|
-
if any(p.search(s) for p in VALUATION_PATTERNS):
|
|
182
|
-
return B_CLAIM
|
|
183
|
-
return A_CLAIM
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
# 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
|
|
187
|
-
def is_conditioned(text) -> bool:
|
|
188
|
-
return str(text or "").strip().startswith(CONDITION_MARK)
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
# 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
|
|
192
|
-
def conditioned_claim(text, label, source):
|
|
193
|
-
"""把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
|
|
194
|
-
|
|
195
|
-
形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
|
|
196
|
-
原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
|
|
197
|
-
已条件化的文本原样返回(幂等)。
|
|
198
|
-
"""
|
|
199
|
-
body = str(text or "").strip()
|
|
200
|
-
if not body or not label:
|
|
201
|
-
return None
|
|
202
|
-
if is_conditioned(body):
|
|
203
|
-
return body
|
|
204
|
-
src = ";".join(_as_source(source))
|
|
205
|
-
if not src:
|
|
206
|
-
return None
|
|
207
|
-
return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
# 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
|
|
211
|
-
def extract_claims(fm: dict, content: str) -> list:
|
|
212
|
-
"""提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
|
|
213
|
-
|
|
214
|
-
每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
|
|
215
|
-
占位标记不成为断言。
|
|
216
|
-
"""
|
|
217
|
-
out, seen = [], set()
|
|
218
|
-
|
|
219
|
-
# 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
|
|
220
|
-
def _push(text, where, field=""):
|
|
221
|
-
s = str(text or "").strip()
|
|
222
|
-
if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
|
|
223
|
-
return
|
|
224
|
-
seen.add(s)
|
|
225
|
-
out.append({"text": s, "type": claim_type(s), "where": where,
|
|
226
|
-
"field": field})
|
|
227
|
-
|
|
228
|
-
for f in CLAIM_FIELDS:
|
|
229
|
-
v = _ccg_field(content, f)
|
|
230
|
-
if v:
|
|
231
|
-
_push(v, "ccg", f)
|
|
232
|
-
c = _comment(fm)
|
|
233
|
-
for f in CLAIM_FIELDS:
|
|
234
|
-
v = c.get(f)
|
|
235
|
-
if isinstance(v, list):
|
|
236
|
-
for item in v:
|
|
237
|
-
_push(item, "comment", f)
|
|
238
|
-
elif v:
|
|
239
|
-
_push(v, "comment", f)
|
|
240
|
-
for line in (content or "").split("\n"):
|
|
241
|
-
raw = line.strip()
|
|
242
|
-
if not raw:
|
|
243
|
-
continue
|
|
244
|
-
body = raw.lstrip("#").strip()
|
|
245
|
-
if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
|
|
246
|
-
continue # 声明行已按 CCG 字段处理,不重复断言
|
|
247
|
-
for sent in re.split(r"[。!?]", body):
|
|
248
|
-
s = sent.strip()
|
|
249
|
-
if len(s) >= 8 and not s.startswith(CONDITION_MARK):
|
|
250
|
-
_push(s, "body", "")
|
|
251
|
-
return out
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
# 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
|
|
255
|
-
def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
|
|
256
|
-
"""节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
|
|
257
|
-
where, field = claim.get("where"), claim.get("field")
|
|
258
|
-
before = str(claim.get("text") or "")
|
|
259
|
-
if where == "ccg" and field:
|
|
260
|
-
if _ccg_field(content, field).strip() == before.strip():
|
|
261
|
-
return _upsert_ccg_line(content, field, new_text), True
|
|
262
|
-
return content, False
|
|
263
|
-
if where == "comment" and field:
|
|
264
|
-
c = _comment(fm)
|
|
265
|
-
v = c.get(field)
|
|
266
|
-
if isinstance(v, list):
|
|
267
|
-
if before in v:
|
|
268
|
-
c[field] = [new_text if x == before else x for x in v]
|
|
269
|
-
return content, True
|
|
270
|
-
return content, False
|
|
271
|
-
if str(v or "").strip() == before.strip():
|
|
272
|
-
c[field] = new_text
|
|
273
|
-
return content, True
|
|
274
|
-
return content, False
|
|
275
|
-
if where == "body":
|
|
276
|
-
if before and content.count(before) == 1:
|
|
277
|
-
return content.replace(before, new_text), True
|
|
278
|
-
return content, False
|
|
279
|
-
return content, False
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
# ---- 工单 ----------------------------------------------------------------
|
|
283
|
-
|
|
284
|
-
# 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
|
|
285
|
-
def _need(fm: dict, content: str) -> list:
|
|
286
|
-
need = []
|
|
287
|
-
if not nodefile.verification_basis_valid(fm):
|
|
288
|
-
need.append("verification_basis")
|
|
289
|
-
if not _has_ccg_line(content, "验证方式"):
|
|
290
|
-
need.append("验证方式")
|
|
291
|
-
return need
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
# 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
|
|
295
|
-
def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
|
|
296
|
-
track = classify_track(fm, content)
|
|
297
|
-
return {
|
|
298
|
-
"id": nid,
|
|
299
|
-
"layer": e.get("layer"),
|
|
300
|
-
"track": track,
|
|
301
|
-
"claims": extract_claims(fm, content),
|
|
302
|
-
"need": _need(fm, content),
|
|
303
|
-
"source_policy": source_policy(track),
|
|
304
|
-
}
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
# 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
|
|
308
|
-
def _is_placeholder_shell(fm: dict, content: str) -> bool:
|
|
309
|
-
"""空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
|
|
310
|
-
|
|
311
|
-
口径(宁漏判不误判):
|
|
312
|
-
1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
|
|
313
|
-
且值为占位标记 → 空壳;
|
|
314
|
-
2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
|
|
315
|
-
`kp_archaeo_*` 类节点因此不被误判为空壳。
|
|
316
|
-
"""
|
|
317
|
-
decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
|
|
318
|
-
if decl:
|
|
319
|
-
return nodefile.is_placeholder_text(decl)
|
|
320
|
-
return nodefile.is_placeholder_text(content)
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
# 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
|
|
324
|
-
def _scan(cg, layer=None, ids=None, prefix=None):
|
|
325
|
-
"""逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
|
|
326
|
-
|
|
327
|
-
`prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
|
|
328
|
-
避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
|
|
329
|
-
"""
|
|
330
|
-
want = set(ids) if ids else None
|
|
331
|
-
for nid, e in list((cg.index.get("nodes") or {}).items()):
|
|
332
|
-
if want is not None and nid not in want:
|
|
333
|
-
continue
|
|
334
|
-
if prefix and not str(nid).startswith(prefix):
|
|
335
|
-
continue
|
|
336
|
-
if layer and e.get("layer") != layer:
|
|
337
|
-
continue
|
|
338
|
-
if e.get("layer") in INTERNAL_LAYERS:
|
|
339
|
-
yield {"status": "skip", "reason": "internal", "id": nid}
|
|
340
|
-
continue
|
|
341
|
-
if not _readable_guard(cg, e):
|
|
342
|
-
yield {"status": "skip", "reason": "denied", "id": nid}
|
|
343
|
-
continue
|
|
344
|
-
fm, content = cg._read(e)
|
|
345
|
-
if fm is None:
|
|
346
|
-
yield {"status": "skip", "reason": "unreadable", "id": nid}
|
|
347
|
-
continue
|
|
348
|
-
if crypto.is_encrypted(content):
|
|
349
|
-
yield {"status": "skip", "reason": "locked", "id": nid}
|
|
350
|
-
continue
|
|
351
|
-
if e.get("layer") in SKIP_LAYERS or any(
|
|
352
|
-
t in SKIP_TAGS for t in (fm.get("tags") or [])):
|
|
353
|
-
yield {"status": "skip", "reason": "derived", "id": nid}
|
|
354
|
-
continue
|
|
355
|
-
need = _need(fm, content)
|
|
356
|
-
if not need: # 已齐备
|
|
357
|
-
yield {"status": "skip", "reason": "present", "id": nid}
|
|
358
|
-
continue
|
|
359
|
-
ph = []
|
|
360
|
-
der = derive_fields(fm, content, placeholder_out=ph)
|
|
361
|
-
if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
|
|
362
|
-
yield {"status": "skip", "reason": "placeholder", "id": nid,
|
|
363
|
-
"placeholder_fields": ph}
|
|
364
|
-
continue
|
|
365
|
-
yield {"status": "row", "id": nid,
|
|
366
|
-
"row": _worklist_row(nid, e, fm, content)}
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
_SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
|
|
370
|
-
"present": "skipped_present", "denied": "skipped_denied",
|
|
371
|
-
"unreadable": "skipped_unreadable", "internal": "skipped_internal"}
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
# 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
|
|
375
|
-
def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
|
|
376
|
-
"""生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
|
|
377
|
-
|
|
378
|
-
`prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
|
|
379
|
-
内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
|
|
380
|
-
"""
|
|
381
|
-
cg = _as_cg(x)
|
|
382
|
-
rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
|
|
383
|
-
"prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
|
|
384
|
-
"skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
|
|
385
|
-
"skipped_denied": 0, "skipped_placeholder": 0,
|
|
386
|
-
"skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
|
|
387
|
-
"targeted": 0, "items": []}
|
|
388
|
-
for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
|
|
389
|
-
rep["nodes_scanned"] += 1
|
|
390
|
-
if scan["status"] == "skip":
|
|
391
|
-
reason = scan["reason"]
|
|
392
|
-
if reason == "placeholder":
|
|
393
|
-
rep["skipped_placeholder"] += 1
|
|
394
|
-
rep["placeholder_ids"].append(scan["id"])
|
|
395
|
-
else:
|
|
396
|
-
key = _SKIP_KEY.get(reason)
|
|
397
|
-
if key:
|
|
398
|
-
rep[key] += 1
|
|
399
|
-
continue
|
|
400
|
-
row = scan["row"]
|
|
401
|
-
if row["track"] == "undetermined":
|
|
402
|
-
rep["undetermined"] += 1
|
|
403
|
-
rep["targeted"] += 1
|
|
404
|
-
if limit is None or len(rep["items"]) < limit:
|
|
405
|
-
rep["items"].append(row)
|
|
406
|
-
rep["planned_ids"] = [r["id"] for r in rep["items"]]
|
|
407
|
-
return rep
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
# ---- 子代理接口(提示词 + 解析) -----------------------------------------
|
|
411
|
-
|
|
412
|
-
_REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
|
|
413
|
-
赛道:{track};来源策略:{policy};待补字段:{need}
|
|
414
|
-
节点标题:{title}
|
|
415
|
-
已声明断言:
|
|
416
|
-
{claims}
|
|
417
|
-
正文:
|
|
418
|
-
{body}
|
|
419
|
-
|
|
420
|
-
只输出 JSON 数组,元素形如:
|
|
421
|
-
{{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
|
|
422
|
-
硬约束:
|
|
423
|
-
1. 文科(humanities)basis 只能是 textbook / public_kb;
|
|
424
|
-
2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
|
|
425
|
-
3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
|
|
426
|
-
4. 只能补 field=验证方式,禁止新增其它字段。"""
|
|
427
|
-
|
|
428
|
-
_VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
|
|
429
|
-
赛道:{track};来源策略:{policy}
|
|
430
|
-
候选(JSON):
|
|
431
|
-
{candidates}
|
|
432
|
-
|
|
433
|
-
只输出 JSON 数组,元素形如:
|
|
434
|
-
{{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
|
|
435
|
-
硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
# 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
|
|
439
|
-
def reflect_prompt(row: dict, fm: dict, content: str) -> str:
|
|
440
|
-
claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
|
|
441
|
-
return _REFLECT_TEMPLATE.format(
|
|
442
|
-
track=row.get("track"), policy=row.get("source_policy") or "(未定)",
|
|
443
|
-
need="、".join(row.get("need") or []),
|
|
444
|
-
title=_as_text(fm.get("title")) or row.get("id"),
|
|
445
|
-
claims=claims or "(无)",
|
|
446
|
-
body=(content or "")[:1200])
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
# 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
|
|
450
|
-
def verify_prompt(row: dict, rows: list) -> str:
|
|
451
|
-
cands = [{"field": r.get("field"), "value": r.get("value"),
|
|
452
|
-
"basis": r.get("basis"), "source": r.get("source")} for r in rows]
|
|
453
|
-
return _VERIFY_TEMPLATE.format(
|
|
454
|
-
track=row.get("track"), policy=row.get("source_policy") or "(未定)",
|
|
455
|
-
candidates=json.dumps(cands, ensure_ascii=False))
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
# 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
|
|
459
|
-
def _extract_json(raw, want_list=True):
|
|
460
|
-
"""从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
|
|
461
|
-
s = str(raw or "")
|
|
462
|
-
pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
|
|
463
|
-
for op, cl in pairs:
|
|
464
|
-
i, j = s.find(op), s.rfind(cl)
|
|
465
|
-
if i < 0 or j <= i:
|
|
466
|
-
continue
|
|
467
|
-
try:
|
|
468
|
-
obj = json.loads(s[i:j + 1])
|
|
469
|
-
except ValueError:
|
|
470
|
-
continue
|
|
471
|
-
if isinstance(obj, list):
|
|
472
|
-
return obj
|
|
473
|
-
if isinstance(obj, dict):
|
|
474
|
-
for key in ("rows", "items", "verdicts", "candidates", "data"):
|
|
475
|
-
if isinstance(obj.get(key), list):
|
|
476
|
-
return obj[key]
|
|
477
|
-
return [obj]
|
|
478
|
-
return []
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
# 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
|
|
482
|
-
def _norm_row(item) -> dict:
|
|
483
|
-
if not isinstance(item, dict):
|
|
484
|
-
return {}
|
|
485
|
-
return {
|
|
486
|
-
"field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
|
|
487
|
-
"value": _as_text(item.get("value")),
|
|
488
|
-
"basis": str(item.get("basis") or "").strip(),
|
|
489
|
-
"source": _as_source(item.get("source")),
|
|
490
|
-
"verdict": str(item.get("verdict") or "").strip().lower(),
|
|
491
|
-
"reason": str(item.get("reason") or "").strip(),
|
|
492
|
-
}
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
# 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
|
|
496
|
-
def parse_reflect_rows(raw) -> list:
|
|
497
|
-
out = []
|
|
498
|
-
for item in _extract_json(raw, want_list=True):
|
|
499
|
-
r = _norm_row(item)
|
|
500
|
-
if not r or r["field"] not in ("验证方式", "verification_basis"):
|
|
501
|
-
continue
|
|
502
|
-
r["field"] = "验证方式"
|
|
503
|
-
out.append(r)
|
|
504
|
-
return out
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
# 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
|
|
508
|
-
def parse_verify_rows(raw) -> list:
|
|
509
|
-
out = []
|
|
510
|
-
for item in _extract_json(raw, want_list=False):
|
|
511
|
-
r = _norm_row(item)
|
|
512
|
-
if not r or not r["field"]:
|
|
513
|
-
continue
|
|
514
|
-
out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
|
|
515
|
-
return out
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
# ---- 白箱闸门与双单元折叠 ------------------------------------------------
|
|
519
|
-
|
|
520
|
-
# 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
|
|
521
|
-
def gate_rows(rows: list, track: str) -> tuple:
|
|
522
|
-
"""零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
|
|
523
|
-
|
|
524
|
-
返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
|
|
525
|
-
"""
|
|
526
|
-
kept, gated = [], []
|
|
527
|
-
for r in rows:
|
|
528
|
-
f = normalize_field(r.get("field"))
|
|
529
|
-
row = dict(r, field=f)
|
|
530
|
-
if f not in WRITABLE_FIELDS:
|
|
531
|
-
gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
|
|
532
|
-
continue
|
|
533
|
-
if not basis_licensed(track, r.get("basis")):
|
|
534
|
-
gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
|
|
535
|
-
continue
|
|
536
|
-
if not r.get("source"):
|
|
537
|
-
gated.append(dict(row, reason="无来源,不写"))
|
|
538
|
-
continue
|
|
539
|
-
value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
|
|
540
|
-
if not value:
|
|
541
|
-
gated.append(dict(row, reason="无验证方式声明"))
|
|
542
|
-
continue
|
|
543
|
-
kept.append(dict(row, value=value, verdict="accept"))
|
|
544
|
-
return kept, gated
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
# 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
|
|
548
|
-
def fold_verdicts(rows: list) -> tuple:
|
|
549
|
-
"""把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
|
|
550
|
-
|
|
551
|
-
接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
|
|
552
|
-
任一 verify-drop 即否决(veto 优先)。
|
|
553
|
-
"""
|
|
554
|
-
groups = OrderedDict()
|
|
555
|
-
for r in rows:
|
|
556
|
-
key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
|
|
557
|
-
r.get("value"))
|
|
558
|
-
groups.setdefault(key, []).append(r)
|
|
559
|
-
accepted, deferred, dropped = [], [], []
|
|
560
|
-
for (nid, field, value), rs in groups.items():
|
|
561
|
-
refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
|
|
562
|
-
ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
|
|
563
|
-
base = {"id": nid, "field": field or "验证方式", "value": value,
|
|
564
|
-
"basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
|
|
565
|
-
"source": next((r.get("source") for r in refl if r.get("source")), []),
|
|
566
|
-
"track": next((r.get("track") for r in refl if r.get("track")), "")}
|
|
567
|
-
if not refl or not ver:
|
|
568
|
-
base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
|
|
569
|
-
deferred.append(base)
|
|
570
|
-
continue
|
|
571
|
-
if any(r.get("verdict") == "drop" for r in ver):
|
|
572
|
-
base["reason"] = next((r.get("reason") for r in ver
|
|
573
|
-
if r.get("verdict") == "drop"), "验证单元否决")
|
|
574
|
-
dropped.append(base)
|
|
575
|
-
continue
|
|
576
|
-
ra = any(r.get("verdict") == "accept" for r in refl)
|
|
577
|
-
va = any(r.get("verdict") == "accept" for r in ver)
|
|
578
|
-
if ra and va:
|
|
579
|
-
base["units"] = {
|
|
580
|
-
"reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
|
|
581
|
-
"verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
|
|
582
|
-
}
|
|
583
|
-
accepted.append(base)
|
|
584
|
-
else:
|
|
585
|
-
base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
|
|
586
|
-
deferred.append(base)
|
|
587
|
-
return accepted, deferred, dropped
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
# 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
|
|
591
|
-
def detect_self_verify(rows: list) -> bool:
|
|
592
|
-
"""同一执行者同时充当反思与验证 = 自证(禁止)。"""
|
|
593
|
-
r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
|
|
594
|
-
v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
|
|
595
|
-
r.discard("")
|
|
596
|
-
v.discard("")
|
|
597
|
-
return bool(r & v)
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
# ---- 落库写入 ------------------------------------------------------------
|
|
601
|
-
|
|
602
|
-
# 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
|
|
603
|
-
def _norm_verdicts(verdicts):
|
|
604
|
-
"""外部裁决(子代理落盘)→ `{id: [rows]}`。"""
|
|
605
|
-
if verdicts is None:
|
|
606
|
-
return None
|
|
607
|
-
out = OrderedDict()
|
|
608
|
-
if isinstance(verdicts, dict):
|
|
609
|
-
for nid, rs in verdicts.items():
|
|
610
|
-
out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
|
|
611
|
-
return out
|
|
612
|
-
for r in verdicts or []:
|
|
613
|
-
if not isinstance(r, dict):
|
|
614
|
-
continue
|
|
615
|
-
nid = str(r.get("id") or "")
|
|
616
|
-
if nid:
|
|
617
|
-
out.setdefault(nid, []).append(dict(r))
|
|
618
|
-
return out
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
# 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
|
|
622
|
-
def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
|
|
623
|
-
condition_claims=True):
|
|
624
|
-
"""把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
|
|
625
|
-
a = accepted[0]
|
|
626
|
-
basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
|
|
627
|
-
source = _as_source(a.get("source"))
|
|
628
|
-
content_before = content
|
|
629
|
-
value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
|
|
630
|
-
|
|
631
|
-
vb_before = {"had": "verification_basis" in fm,
|
|
632
|
-
"value": fm.get("verification_basis")}
|
|
633
|
-
prov_before = {"had": "verification_evidence" in fm,
|
|
634
|
-
"value": fm.get("verification_evidence")}
|
|
635
|
-
line_before = {"had": _has_ccg_line(content, "验证方式"),
|
|
636
|
-
"value": _ccg_field(content, "验证方式")}
|
|
637
|
-
comment0 = _comment(fm)
|
|
638
|
-
cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
|
|
639
|
-
|
|
640
|
-
# 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
|
|
641
|
-
content = _upsert_ccg_line(content, "验证方式", value)
|
|
642
|
-
_ensure_comment(fm)["验证方式"] = value
|
|
643
|
-
if nodefile.verification_basis_valid(fm):
|
|
644
|
-
basis = str(fm.get("verification_basis"))
|
|
645
|
-
else:
|
|
646
|
-
fm["verification_basis"] = basis
|
|
647
|
-
|
|
648
|
-
# 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
|
|
649
|
-
conditioned = []
|
|
650
|
-
if condition_claims:
|
|
651
|
-
label = BASIS_LABEL.get(basis, basis)
|
|
652
|
-
for c in row.get("claims") or []:
|
|
653
|
-
if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
|
|
654
|
-
continue
|
|
655
|
-
new = conditioned_claim(c.get("text"), label, source)
|
|
656
|
-
if not new or new == c.get("text"):
|
|
657
|
-
continue
|
|
658
|
-
nc, ok = _rewrite_claim(fm, content, c, new)
|
|
659
|
-
if not ok:
|
|
660
|
-
continue
|
|
661
|
-
content = nc
|
|
662
|
-
conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
|
|
663
|
-
"before": c.get("text"), "after": new})
|
|
664
|
-
|
|
665
|
-
wid = _sha(f"{nid}|{batch}|{time.time()}")
|
|
666
|
-
fm["verification_evidence"] = {
|
|
667
|
-
"at": round(time.time(), 3), "batch": batch, "basis": basis,
|
|
668
|
-
"source": source, "track": row.get("track"),
|
|
669
|
-
"policy": row.get("source_policy"), "write_id": wid,
|
|
670
|
-
"units": a.get("units") or {},
|
|
671
|
-
"conditioned": len(conditioned),
|
|
672
|
-
}
|
|
673
|
-
|
|
674
|
-
cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
|
|
675
|
-
durable=True)
|
|
676
|
-
return {
|
|
677
|
-
"action": "crosscheck", "ts": time.time(), "batch": batch,
|
|
678
|
-
"actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
|
|
679
|
-
"node": nid, "layer": e.get("layer"), "track": row.get("track"),
|
|
680
|
-
"policy": row.get("source_policy"), "basis": basis,
|
|
681
|
-
"verification_value": value, "source": source,
|
|
682
|
-
"units": a.get("units") or {},
|
|
683
|
-
"fm_before": {"verification_basis": vb_before,
|
|
684
|
-
"verification_evidence": prov_before,
|
|
685
|
-
"comment_verification": cv_before},
|
|
686
|
-
"verification_line_before": line_before,
|
|
687
|
-
"claims_conditioned": conditioned,
|
|
688
|
-
"content_hash_before": _sha(content_before),
|
|
689
|
-
"content_hash_after": _sha(content),
|
|
690
|
-
}
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
# ---- 主流程 --------------------------------------------------------------
|
|
694
|
-
|
|
695
|
-
# 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
|
|
696
|
-
def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
|
|
697
|
-
"""调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
|
|
698
|
-
prompt = reflect_prompt(scan_row, fm, content)
|
|
699
|
-
cands = parse_reflect_rows(reflect_fn(prompt))
|
|
700
|
-
rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
|
|
701
|
-
track=scan_row["track"]) for c in cands]
|
|
702
|
-
if not rrows:
|
|
703
|
-
return [], []
|
|
704
|
-
vrows = []
|
|
705
|
-
if verify_fn is not None:
|
|
706
|
-
vp = verify_prompt(scan_row, rrows)
|
|
707
|
-
vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
|
|
708
|
-
track=scan_row["track"],
|
|
709
|
-
value=c.get("value") or next(
|
|
710
|
-
(x.get("value") for x in rrows
|
|
711
|
-
if x.get("field") == c.get("field")), ""))
|
|
712
|
-
for c in parse_verify_rows(verify_fn(vp))]
|
|
713
|
-
return rrows, vrows
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
# 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
|
|
717
|
-
def _rows_from_verdicts(nid, track, pre):
|
|
718
|
-
"""从外部裁决中拆出 (reflect, verify) 两组行。"""
|
|
719
|
-
rrows, vrows = [], []
|
|
720
|
-
for r in pre:
|
|
721
|
-
unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
|
|
722
|
-
base = dict(r, id=nid, unit=unit, track=track)
|
|
723
|
-
(vrows if unit == VERIFY_UNIT else rrows).append(base)
|
|
724
|
-
return rrows, vrows
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
# 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
|
|
728
|
-
def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
|
|
729
|
-
verify_fn=None, verdicts=None, apply=False,
|
|
730
|
-
batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
|
|
731
|
-
allow_self_verify=False, reflect_actor=None, verify_actor=None,
|
|
732
|
-
condition_claims=True, verbose=True, prefix=None) -> dict:
|
|
733
|
-
"""批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
|
|
734
|
-
|
|
735
|
-
`reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
|
|
736
|
-
`verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
|
|
737
|
-
二选一。`apply=True` 才写盘。
|
|
738
|
-
"""
|
|
739
|
-
cg = _as_cg(x)
|
|
740
|
-
batch = batch or CROSSCHECK_BATCH
|
|
741
|
-
vmap = _norm_verdicts(verdicts)
|
|
742
|
-
r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
|
|
743
|
-
v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
|
|
744
|
-
|
|
745
|
-
rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
|
|
746
|
-
"batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
|
|
747
|
-
"targeted": 0,
|
|
748
|
-
"accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
|
|
749
|
-
"claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
|
|
750
|
-
"skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
|
|
751
|
-
"skipped_unreadable": 0, "skipped_placeholder": 0,
|
|
752
|
-
"placeholder_ids": [], "undetermined": 0,
|
|
753
|
-
"reasons": {}, "samples": [], "entry_ids": []}
|
|
754
|
-
|
|
755
|
-
# 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
|
|
756
|
-
def _bump(reason):
|
|
757
|
-
rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
|
|
758
|
-
|
|
759
|
-
# 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
|
|
760
|
-
def _sample(kind, nid, detail=""):
|
|
761
|
-
if verbose and len(rep["samples"]) < 20:
|
|
762
|
-
rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
|
|
763
|
-
|
|
764
|
-
seen_targets = 0
|
|
765
|
-
for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
|
|
766
|
-
rep["nodes_scanned"] += 1
|
|
767
|
-
if scan["status"] == "skip":
|
|
768
|
-
reason = scan["reason"]
|
|
769
|
-
if reason == "placeholder":
|
|
770
|
-
rep["skipped_placeholder"] += 1
|
|
771
|
-
rep["placeholder_ids"].append(scan["id"])
|
|
772
|
-
else:
|
|
773
|
-
key = _SKIP_KEY.get(reason)
|
|
774
|
-
if key:
|
|
775
|
-
rep[key] += 1
|
|
776
|
-
continue
|
|
777
|
-
row = scan["row"]
|
|
778
|
-
if row["track"] == "undetermined":
|
|
779
|
-
rep["undetermined"] += 1
|
|
780
|
-
if limit is not None and seen_targets >= limit:
|
|
781
|
-
continue
|
|
782
|
-
seen_targets += 1
|
|
783
|
-
rep["targeted"] += 1
|
|
784
|
-
nid = row["id"]
|
|
785
|
-
e = cg.index["nodes"].get(nid)
|
|
786
|
-
fm, content = cg._read(e) if e else (None, None)
|
|
787
|
-
if fm is None or crypto.is_encrypted(content):
|
|
788
|
-
rep["skipped_locked"] += 1
|
|
789
|
-
continue
|
|
790
|
-
|
|
791
|
-
# ---- 取两单元裁决 ----
|
|
792
|
-
if vmap is not None:
|
|
793
|
-
pre = vmap.get(nid)
|
|
794
|
-
if not pre:
|
|
795
|
-
rep["deferred"] += 1
|
|
796
|
-
_bump("no_verdict")
|
|
797
|
-
continue
|
|
798
|
-
rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
|
|
799
|
-
elif reflect_fn is not None:
|
|
800
|
-
try:
|
|
801
|
-
rrows, vrows = _rows_for(row, fm, content, reflect_fn,
|
|
802
|
-
verify_fn, r_actor, v_actor)
|
|
803
|
-
except Exception as exc: # noqa: BLE001
|
|
804
|
-
rep["deferred"] += 1
|
|
805
|
-
_bump(f"unit_error:{type(exc).__name__}")
|
|
806
|
-
continue
|
|
807
|
-
else:
|
|
808
|
-
rep["deferred"] += 1
|
|
809
|
-
_bump("no_reflect")
|
|
810
|
-
continue
|
|
811
|
-
|
|
812
|
-
# 自证:同一执行者既反思又验证 → 拒收
|
|
813
|
-
if not allow_self_verify and detect_self_verify(rrows + vrows):
|
|
814
|
-
rep["deferred"] += 1
|
|
815
|
-
_bump("self_verify_disallowed")
|
|
816
|
-
_sample("self_verify", nid)
|
|
817
|
-
continue
|
|
818
|
-
|
|
819
|
-
# ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
|
|
820
|
-
kept, gated = gate_rows(rrows, row["track"])
|
|
821
|
-
for g in gated:
|
|
822
|
-
_bump(f"gate:{g.get('reason')[:24]}")
|
|
823
|
-
if not kept:
|
|
824
|
-
rep["deferred"] += 1
|
|
825
|
-
_bump("no_candidate")
|
|
826
|
-
_sample("gated", nid, gated[0].get("reason") if gated else "")
|
|
827
|
-
continue
|
|
828
|
-
if require_verify and not vrows:
|
|
829
|
-
rep["deferred"] += 1
|
|
830
|
-
_bump("verify_unavailable")
|
|
831
|
-
continue
|
|
832
|
-
|
|
833
|
-
rows = kept + vrows
|
|
834
|
-
accepted, deferred, dropped = fold_verdicts(rows)
|
|
835
|
-
if dropped and not accepted:
|
|
836
|
-
rep["rejected"] += 1
|
|
837
|
-
_bump("verify_veto")
|
|
838
|
-
_sample("veto", nid, dropped[0].get("reason", ""))
|
|
839
|
-
continue
|
|
840
|
-
if not accepted:
|
|
841
|
-
rep["deferred"] += 1
|
|
842
|
-
_bump("verdict_deferred")
|
|
843
|
-
_sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
|
|
844
|
-
continue
|
|
845
|
-
|
|
846
|
-
rep["accepted"] += 1
|
|
847
|
-
rep["claims_conditioned"] += sum(
|
|
848
|
-
1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
|
|
849
|
-
if apply:
|
|
850
|
-
rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
|
|
851
|
-
actor, condition_claims=condition_claims)
|
|
852
|
-
append_jsonl(_log_path(cg), rec)
|
|
853
|
-
rep["written"] += 1
|
|
854
|
-
rep["entry_ids"].append(rec["entry_id"])
|
|
855
|
-
_sample("accepted", nid, accepted[0].get("basis", ""))
|
|
856
|
-
|
|
857
|
-
if rep["written"]:
|
|
858
|
-
cg.rebuild_index()
|
|
859
|
-
return rep
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
# ---- 留痕查询 / 回滚 -----------------------------------------------------
|
|
863
|
-
|
|
864
|
-
# 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
|
|
865
|
-
def _log_path(cg) -> str:
|
|
866
|
-
return os.path.join(cg.root, CROSSCHECK_LOG)
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
# 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
|
|
870
|
-
def _reattach(fm: dict, content: str, box: dict, key: str):
|
|
871
|
-
"""把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
|
|
872
|
-
if not isinstance(box, dict):
|
|
873
|
-
return False
|
|
874
|
-
had, value = box.get("had"), box.get("value")
|
|
875
|
-
if key == "comment_verification":
|
|
876
|
-
c = _ensure_comment(fm)
|
|
877
|
-
if had:
|
|
878
|
-
c["验证方式"] = value
|
|
879
|
-
else:
|
|
880
|
-
c.pop("验证方式", None)
|
|
881
|
-
return True
|
|
882
|
-
if had:
|
|
883
|
-
fm[key] = value
|
|
884
|
-
else:
|
|
885
|
-
fm.pop(key, None)
|
|
886
|
-
return True
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
# 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
|
|
890
|
-
def _rewind_claim(fm: dict, content: str, c: dict):
|
|
891
|
-
"""撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
|
|
892
|
-
where, field = c.get("where"), c.get("field")
|
|
893
|
-
before, after = str(c.get("before") or ""), str(c.get("after") or "")
|
|
894
|
-
if not after:
|
|
895
|
-
return content, False
|
|
896
|
-
if where == "ccg" and field:
|
|
897
|
-
if _ccg_field(content, field).strip() == after.strip():
|
|
898
|
-
return _upsert_ccg_line(content, field, before), True
|
|
899
|
-
return content, False
|
|
900
|
-
if where == "comment" and field:
|
|
901
|
-
cc = _comment(fm)
|
|
902
|
-
v = cc.get(field)
|
|
903
|
-
if isinstance(v, list):
|
|
904
|
-
if after in v:
|
|
905
|
-
cc[field] = [before if x == after else x for x in v]
|
|
906
|
-
return content, True
|
|
907
|
-
return content, False
|
|
908
|
-
if str(v or "").strip() == after.strip():
|
|
909
|
-
cc[field] = before
|
|
910
|
-
return content, True
|
|
911
|
-
return content, False
|
|
912
|
-
if where == "body":
|
|
913
|
-
if after in content:
|
|
914
|
-
return content.replace(after, before, 1), True
|
|
915
|
-
return content, False
|
|
916
|
-
return content, False
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
# 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
|
|
920
|
-
def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
|
|
921
|
-
"""按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
|
|
922
|
-
cg = _as_cg(x)
|
|
923
|
-
want = set(entry_ids) if entry_ids else None
|
|
924
|
-
done = set()
|
|
925
|
-
rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
|
|
926
|
-
"batch": batch, "actor": actor, "planned": 0, "reverted": 0,
|
|
927
|
-
"skipped_drift": 0, "conflict": 0, "entry_ids": []}
|
|
928
|
-
recs = [r for r in (read_jsonl(_log_path(cg)) or [])
|
|
929
|
-
if r.get("action") == "crosscheck"
|
|
930
|
-
and (batch is None or r.get("batch") == batch)
|
|
931
|
-
and (want is None or r.get("entry_id") in want)]
|
|
932
|
-
rep["planned"] = len(recs)
|
|
933
|
-
for rec in recs:
|
|
934
|
-
nid = rec.get("node")
|
|
935
|
-
if not nid or nid in done:
|
|
936
|
-
continue
|
|
937
|
-
e = cg.index["nodes"].get(nid)
|
|
938
|
-
if not e:
|
|
939
|
-
rep["skipped_drift"] += 1
|
|
940
|
-
continue
|
|
941
|
-
fm, content = cg._read(e)
|
|
942
|
-
if fm is None or crypto.is_encrypted(content):
|
|
943
|
-
rep["skipped_drift"] += 1
|
|
944
|
-
continue
|
|
945
|
-
# 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
|
|
946
|
-
wid = (fm.get("verification_evidence") or {}).get("write_id")
|
|
947
|
-
if wid and rec.get("write_id") and wid != rec.get("write_id"):
|
|
948
|
-
rep["conflict"] += 1
|
|
949
|
-
continue
|
|
950
|
-
# 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
|
|
951
|
-
reverted = []
|
|
952
|
-
for c in rec.get("claims_conditioned") or []:
|
|
953
|
-
content, ok = _rewind_claim(fm, content, c)
|
|
954
|
-
if ok:
|
|
955
|
-
reverted.append(c.get("field") or c.get("where"))
|
|
956
|
-
# 2) 撤销「验证方式」行
|
|
957
|
-
lb = rec.get("verification_line_before") or {}
|
|
958
|
-
cur_line = _ccg_field(content, "验证方式")
|
|
959
|
-
if cur_line.strip() and cur_line.strip() == str(
|
|
960
|
-
rec.get("verification_value") or "").strip():
|
|
961
|
-
content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
|
|
962
|
-
if lb.get("had") else _remove_ccg_line(content, "验证方式"))
|
|
963
|
-
reverted.append("验证方式")
|
|
964
|
-
# 3) 还原 frontmatter 现场
|
|
965
|
-
for key, box in (rec.get("fm_before") or {}).items():
|
|
966
|
-
_reattach(fm, content, box, key)
|
|
967
|
-
if not reverted:
|
|
968
|
-
rep["conflict"] += 1
|
|
969
|
-
continue
|
|
970
|
-
cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
|
|
971
|
-
durable=True)
|
|
972
|
-
append_jsonl(_log_path(cg), {
|
|
973
|
-
"action": "crosscheck_rollback", "ts": time.time(),
|
|
974
|
-
"batch": rec.get("batch"), "actor": actor,
|
|
975
|
-
"entry_id": rec.get("entry_id"), "node": nid,
|
|
976
|
-
"reverted": reverted, "content_hash_after": _sha(content)})
|
|
977
|
-
rep["reverted"] += 1
|
|
978
|
-
rep["entry_ids"].append(rec.get("entry_id"))
|
|
979
|
-
done.add(nid)
|
|
980
|
-
if rep["reverted"]:
|
|
981
|
-
cg.rebuild_index()
|
|
982
|
-
return rep
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
# 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
|
|
986
|
-
def history(x, limit=100, action=None, batch=None) -> dict:
|
|
987
|
-
cg = _as_cg(x)
|
|
988
|
-
recs = []
|
|
989
|
-
for rec in read_jsonl(_log_path(cg)) or []:
|
|
990
|
-
if action and rec.get("action") != action:
|
|
991
|
-
continue
|
|
992
|
-
if batch and rec.get("batch") != batch:
|
|
993
|
-
continue
|
|
994
|
-
recs.append(rec)
|
|
995
|
-
total = len(recs)
|
|
996
|
-
if limit is not None and limit >= 0:
|
|
997
|
-
recs = recs[-limit:]
|
|
998
|
-
return {"root": cg.root, "total": total, "returned": len(recs),
|
|
999
|
-
"records": recs}
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
# ---- 权限与 CLI ----------------------------------------------------------
|
|
1003
|
-
|
|
1004
|
-
# 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
|
|
1005
|
-
def can_write_knowledge(principal) -> bool:
|
|
1006
|
-
"""落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
|
|
1007
|
-
if principal is None:
|
|
1008
|
-
return False
|
|
1009
|
-
try:
|
|
1010
|
-
if principal.expired() or not principal.can_write:
|
|
1011
|
-
return False
|
|
1012
|
-
return bool(principal.allows_layer("knowledge"))
|
|
1013
|
-
except Exception: # noqa: BLE001
|
|
1014
|
-
return False
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
# 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
|
|
1018
|
-
def _load_verdicts(path: str):
|
|
1019
|
-
if not path:
|
|
1020
|
-
return None
|
|
1021
|
-
if not os.path.exists(path):
|
|
1022
|
-
raise SystemExit(f"裁决文件不存在:{path}")
|
|
1023
|
-
with open(path, "r", encoding="utf-8") as fh:
|
|
1024
|
-
text = fh.read().strip()
|
|
1025
|
-
if not text:
|
|
1026
|
-
return []
|
|
1027
|
-
try:
|
|
1028
|
-
return json.loads(text)
|
|
1029
|
-
except ValueError:
|
|
1030
|
-
rows = []
|
|
1031
|
-
for line in text.splitlines():
|
|
1032
|
-
line = line.strip()
|
|
1033
|
-
if not line or line.startswith("//"):
|
|
1034
|
-
continue
|
|
1035
|
-
rows.append(json.loads(line))
|
|
1036
|
-
return rows
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
# 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
|
|
1040
|
-
def _cli(argv=None) -> int:
|
|
1041
|
-
ap = argparse.ArgumentParser(
|
|
1042
|
-
prog="python -m md_cg.crosscheck",
|
|
1043
|
-
description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
|
|
1044
|
-
ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
|
|
1045
|
-
ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
|
|
1046
|
-
ap.add_argument("--token-file", default=None)
|
|
1047
|
-
ap.add_argument("--action", default="worklist",
|
|
1048
|
-
choices=("worklist", "crosscheck", "rollback", "history"))
|
|
1049
|
-
ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
|
|
1050
|
-
ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
|
|
1051
|
-
ap.add_argument("--batch", default=CROSSCHECK_BATCH)
|
|
1052
|
-
ap.add_argument("--limit", type=int, default=None)
|
|
1053
|
-
ap.add_argument("--layer", default=None)
|
|
1054
|
-
ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
|
|
1055
|
-
ap.add_argument("--ids", default="", help="逗号分隔节点 id")
|
|
1056
|
-
ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
|
|
1057
|
-
ap.add_argument("--allow-self-verify", action="store_true")
|
|
1058
|
-
ap.add_argument("--reflect-actor", default=None)
|
|
1059
|
-
ap.add_argument("--verify-actor", default=None)
|
|
1060
|
-
args = ap.parse_args(argv)
|
|
1061
|
-
|
|
1062
|
-
from . import tokens
|
|
1063
|
-
principal = None
|
|
1064
|
-
if args.token:
|
|
1065
|
-
try:
|
|
1066
|
-
principal = tokens.verify_token(args.token, path=args.token_file)
|
|
1067
|
-
except tokens.TokenError as exc:
|
|
1068
|
-
raise SystemExit(f"令牌校验失败:{exc}")
|
|
1069
|
-
actor = getattr(principal, "actor", None)
|
|
1070
|
-
ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
|
|
1071
|
-
|
|
1072
|
-
if args.action == "worklist":
|
|
1073
|
-
rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
|
|
1074
|
-
ids=ids, prefix=args.prefix)
|
|
1075
|
-
elif args.action == "history":
|
|
1076
|
-
rep = history(args.root, limit=args.limit if args.limit is not None else 100,
|
|
1077
|
-
batch=args.batch)
|
|
1078
|
-
elif args.action == "rollback":
|
|
1079
|
-
if not can_write_knowledge(principal):
|
|
1080
|
-
raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
|
|
1081
|
-
rep = rollback(args.root, batch=args.batch, actor=actor)
|
|
1082
|
-
else:
|
|
1083
|
-
if args.apply and not can_write_knowledge(principal):
|
|
1084
|
-
raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
|
|
1085
|
-
rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
|
|
1086
|
-
prefix=args.prefix,
|
|
1087
|
-
verdicts=_load_verdicts(args.verdicts), apply=args.apply,
|
|
1088
|
-
batch=args.batch,
|
|
1089
|
-
require_verify=not args.no_verify,
|
|
1090
|
-
allow_self_verify=args.allow_self_verify,
|
|
1091
|
-
reflect_actor=args.reflect_actor,
|
|
1092
|
-
verify_actor=args.verify_actor, actor=actor)
|
|
1093
|
-
print(json.dumps(rep, ensure_ascii=False, indent=2))
|
|
1094
|
-
return 0
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
if __name__ == "__main__": # pragma: no cover
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
|
|
3
|
+
|
|
4
|
+
设计约束(与计划一致):
|
|
5
|
+
|
|
6
|
+
* **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
|
|
7
|
+
`python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
|
|
8
|
+
* **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
|
|
9
|
+
验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
|
|
10
|
+
* **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
|
|
11
|
+
理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
|
|
12
|
+
赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
|
|
13
|
+
* **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
|
|
14
|
+
拿不出证据的节点绝不写 `verification_basis`。
|
|
15
|
+
* **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
|
|
16
|
+
时撤销,否则计入 conflict 跳过。
|
|
17
|
+
|
|
18
|
+
`_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
import re
|
|
27
|
+
import sys
|
|
28
|
+
import time
|
|
29
|
+
from collections import OrderedDict
|
|
30
|
+
|
|
31
|
+
from . import crypto, nodefile
|
|
32
|
+
from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
|
|
33
|
+
SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
|
|
34
|
+
_ensure_comment, _entry_id, _readable_guard,
|
|
35
|
+
_remove_ccg_line, _sha, derive_fields)
|
|
36
|
+
from .consolidate import _has_ccg_line, _upsert_ccg_line
|
|
37
|
+
from .fsutil import append_jsonl, read_jsonl
|
|
38
|
+
from .mdcos import _ccg_field
|
|
39
|
+
|
|
40
|
+
# ---- 常量 ----------------------------------------------------------------
|
|
41
|
+
|
|
42
|
+
CROSSCHECK_LOG = "_crosscheck.jsonl"
|
|
43
|
+
CROSSCHECK_BATCH = "crosscheck"
|
|
44
|
+
|
|
45
|
+
REFLECT_UNIT = "reflect"
|
|
46
|
+
VERIFY_UNIT = "verify"
|
|
47
|
+
|
|
48
|
+
# 可写字段(本管线只动这一处,杜绝越界面)
|
|
49
|
+
WRITABLE_FIELDS = ("验证方式",)
|
|
50
|
+
FIELD_NORMALIZE = {
|
|
51
|
+
"验证方式": "验证方式",
|
|
52
|
+
"verification_basis": "验证方式",
|
|
53
|
+
"验证": "验证方式",
|
|
54
|
+
"verification": "验证方式",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
# 赛道 → 来源执照策略
|
|
58
|
+
SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
|
|
59
|
+
|
|
60
|
+
# 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
|
|
61
|
+
SCIENCE_SUBJECT_HINTS = (
|
|
62
|
+
"数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
|
|
63
|
+
)
|
|
64
|
+
HUMANITIES_SUBJECT_HINTS = (
|
|
65
|
+
"语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
|
|
66
|
+
"文学", "哲学", "艺术", "音乐", "美术",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
# 基底枚举 → 可读标签(条件化表述引用)
|
|
70
|
+
BASIS_LABEL = {
|
|
71
|
+
"compiler": "编译器/静态检查",
|
|
72
|
+
"test": "单元测试",
|
|
73
|
+
"measurement": "实测数据",
|
|
74
|
+
"formal_proof": "形式化证明",
|
|
75
|
+
"data": "数据统计",
|
|
76
|
+
"textbook": "人教版教材",
|
|
77
|
+
"public_kb": "公开知识库",
|
|
78
|
+
"other": "人工评审",
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
|
|
82
|
+
B_CLAIM = "B_valuation"
|
|
83
|
+
A_CLAIM = "A_fact"
|
|
84
|
+
|
|
85
|
+
# B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
|
|
86
|
+
VALUATION_MARKERS = (
|
|
87
|
+
"结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
|
|
88
|
+
"精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
|
|
89
|
+
"不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
|
|
90
|
+
"最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
|
|
91
|
+
"非凡",
|
|
92
|
+
)
|
|
93
|
+
VALUATION_PATTERNS = (
|
|
94
|
+
re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
|
|
95
|
+
re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
CONDITION_MARK = "〔来源限定〕"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ---- 赛道与来源执照 ------------------------------------------------------
|
|
102
|
+
|
|
103
|
+
# 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
|
|
104
|
+
def classify_track(fm: dict, content: str = "") -> str:
|
|
105
|
+
"""判定节点赛道:`humanities` / `science` / `undetermined`。
|
|
106
|
+
|
|
107
|
+
只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
|
|
108
|
+
"""
|
|
109
|
+
fm = fm or {}
|
|
110
|
+
explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
|
|
111
|
+
if explicit in ("humanities", "文科", "arts"):
|
|
112
|
+
return "humanities"
|
|
113
|
+
if explicit in ("science", "理科", "stem"):
|
|
114
|
+
return "science"
|
|
115
|
+
st = fm.get("state_attributes")
|
|
116
|
+
name = _as_text(st.get("name")) if isinstance(st, dict) else ""
|
|
117
|
+
parts = [
|
|
118
|
+
name,
|
|
119
|
+
_as_text(fm.get("title")),
|
|
120
|
+
_as_text(fm.get("discipline")),
|
|
121
|
+
_as_text(fm.get("subject")),
|
|
122
|
+
" ".join(str(t) for t in (fm.get("tags") or [])),
|
|
123
|
+
_ccg_field(content, "功能名"),
|
|
124
|
+
_ccg_field(content, "执行"),
|
|
125
|
+
_ccg_field(content, "子功能"),
|
|
126
|
+
]
|
|
127
|
+
text = " ".join(p for p in parts if p)
|
|
128
|
+
has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
|
|
129
|
+
has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
|
|
130
|
+
if has_sci and not has_hum:
|
|
131
|
+
return "science"
|
|
132
|
+
if has_hum and not has_sci:
|
|
133
|
+
return "humanities"
|
|
134
|
+
return "undetermined"
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
|
|
138
|
+
def source_policy(track: str) -> str:
|
|
139
|
+
"""赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
|
|
140
|
+
return SOURCE_POLICY.get(track or "", "")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
|
|
144
|
+
def allowed_basis(track: str) -> tuple:
|
|
145
|
+
if track == "science":
|
|
146
|
+
return tuple(nodefile.REPRODUCIBLE_BASIS)
|
|
147
|
+
if track == "humanities":
|
|
148
|
+
return tuple(nodefile.CONSISTENCY_BASIS)
|
|
149
|
+
return ()
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
# 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
|
|
153
|
+
def basis_licensed(track: str, basis) -> bool:
|
|
154
|
+
"""来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
|
|
155
|
+
return bool(basis) and str(basis) in allowed_basis(track)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
|
|
159
|
+
def normalize_field(field) -> str:
|
|
160
|
+
return FIELD_NORMALIZE.get(str(field or "").strip(), "")
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
|
|
164
|
+
def _as_source(v) -> list:
|
|
165
|
+
if v is None:
|
|
166
|
+
return []
|
|
167
|
+
items = list(v) if isinstance(v, (list, tuple)) else [v]
|
|
168
|
+
return [str(x).strip() for x in items if str(x).strip()]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# ---- B 型识别与条件化改写 ------------------------------------------------
|
|
172
|
+
|
|
173
|
+
# 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
|
|
174
|
+
def claim_type(text) -> str:
|
|
175
|
+
"""`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
|
|
176
|
+
s = str(text or "")
|
|
177
|
+
if not s.strip():
|
|
178
|
+
return A_CLAIM
|
|
179
|
+
if any(m in s for m in VALUATION_MARKERS):
|
|
180
|
+
return B_CLAIM
|
|
181
|
+
if any(p.search(s) for p in VALUATION_PATTERNS):
|
|
182
|
+
return B_CLAIM
|
|
183
|
+
return A_CLAIM
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
|
|
187
|
+
def is_conditioned(text) -> bool:
|
|
188
|
+
return str(text or "").strip().startswith(CONDITION_MARK)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
|
|
192
|
+
def conditioned_claim(text, label, source):
|
|
193
|
+
"""把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
|
|
194
|
+
|
|
195
|
+
形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
|
|
196
|
+
原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
|
|
197
|
+
已条件化的文本原样返回(幂等)。
|
|
198
|
+
"""
|
|
199
|
+
body = str(text or "").strip()
|
|
200
|
+
if not body or not label:
|
|
201
|
+
return None
|
|
202
|
+
if is_conditioned(body):
|
|
203
|
+
return body
|
|
204
|
+
src = ";".join(_as_source(source))
|
|
205
|
+
if not src:
|
|
206
|
+
return None
|
|
207
|
+
return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
|
|
211
|
+
def extract_claims(fm: dict, content: str) -> list:
|
|
212
|
+
"""提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
|
|
213
|
+
|
|
214
|
+
每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
|
|
215
|
+
占位标记不成为断言。
|
|
216
|
+
"""
|
|
217
|
+
out, seen = [], set()
|
|
218
|
+
|
|
219
|
+
# 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
|
|
220
|
+
def _push(text, where, field=""):
|
|
221
|
+
s = str(text or "").strip()
|
|
222
|
+
if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
|
|
223
|
+
return
|
|
224
|
+
seen.add(s)
|
|
225
|
+
out.append({"text": s, "type": claim_type(s), "where": where,
|
|
226
|
+
"field": field})
|
|
227
|
+
|
|
228
|
+
for f in CLAIM_FIELDS:
|
|
229
|
+
v = _ccg_field(content, f)
|
|
230
|
+
if v:
|
|
231
|
+
_push(v, "ccg", f)
|
|
232
|
+
c = _comment(fm)
|
|
233
|
+
for f in CLAIM_FIELDS:
|
|
234
|
+
v = c.get(f)
|
|
235
|
+
if isinstance(v, list):
|
|
236
|
+
for item in v:
|
|
237
|
+
_push(item, "comment", f)
|
|
238
|
+
elif v:
|
|
239
|
+
_push(v, "comment", f)
|
|
240
|
+
for line in (content or "").split("\n"):
|
|
241
|
+
raw = line.strip()
|
|
242
|
+
if not raw:
|
|
243
|
+
continue
|
|
244
|
+
body = raw.lstrip("#").strip()
|
|
245
|
+
if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
|
|
246
|
+
continue # 声明行已按 CCG 字段处理,不重复断言
|
|
247
|
+
for sent in re.split(r"[。!?]", body):
|
|
248
|
+
s = sent.strip()
|
|
249
|
+
if len(s) >= 8 and not s.startswith(CONDITION_MARK):
|
|
250
|
+
_push(s, "body", "")
|
|
251
|
+
return out
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
# 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
|
|
255
|
+
def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
|
|
256
|
+
"""节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
|
|
257
|
+
where, field = claim.get("where"), claim.get("field")
|
|
258
|
+
before = str(claim.get("text") or "")
|
|
259
|
+
if where == "ccg" and field:
|
|
260
|
+
if _ccg_field(content, field).strip() == before.strip():
|
|
261
|
+
return _upsert_ccg_line(content, field, new_text), True
|
|
262
|
+
return content, False
|
|
263
|
+
if where == "comment" and field:
|
|
264
|
+
c = _comment(fm)
|
|
265
|
+
v = c.get(field)
|
|
266
|
+
if isinstance(v, list):
|
|
267
|
+
if before in v:
|
|
268
|
+
c[field] = [new_text if x == before else x for x in v]
|
|
269
|
+
return content, True
|
|
270
|
+
return content, False
|
|
271
|
+
if str(v or "").strip() == before.strip():
|
|
272
|
+
c[field] = new_text
|
|
273
|
+
return content, True
|
|
274
|
+
return content, False
|
|
275
|
+
if where == "body":
|
|
276
|
+
if before and content.count(before) == 1:
|
|
277
|
+
return content.replace(before, new_text), True
|
|
278
|
+
return content, False
|
|
279
|
+
return content, False
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
# ---- 工单 ----------------------------------------------------------------
|
|
283
|
+
|
|
284
|
+
# 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
|
|
285
|
+
def _need(fm: dict, content: str) -> list:
|
|
286
|
+
need = []
|
|
287
|
+
if not nodefile.verification_basis_valid(fm):
|
|
288
|
+
need.append("verification_basis")
|
|
289
|
+
if not _has_ccg_line(content, "验证方式"):
|
|
290
|
+
need.append("验证方式")
|
|
291
|
+
return need
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
# 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
|
|
295
|
+
def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
|
|
296
|
+
track = classify_track(fm, content)
|
|
297
|
+
return {
|
|
298
|
+
"id": nid,
|
|
299
|
+
"layer": e.get("layer"),
|
|
300
|
+
"track": track,
|
|
301
|
+
"claims": extract_claims(fm, content),
|
|
302
|
+
"need": _need(fm, content),
|
|
303
|
+
"source_policy": source_policy(track),
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
# 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
|
|
308
|
+
def _is_placeholder_shell(fm: dict, content: str) -> bool:
|
|
309
|
+
"""空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
|
|
310
|
+
|
|
311
|
+
口径(宁漏判不误判):
|
|
312
|
+
1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
|
|
313
|
+
且值为占位标记 → 空壳;
|
|
314
|
+
2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
|
|
315
|
+
`kp_archaeo_*` 类节点因此不被误判为空壳。
|
|
316
|
+
"""
|
|
317
|
+
decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
|
|
318
|
+
if decl:
|
|
319
|
+
return nodefile.is_placeholder_text(decl)
|
|
320
|
+
return nodefile.is_placeholder_text(content)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
# 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
|
|
324
|
+
def _scan(cg, layer=None, ids=None, prefix=None):
|
|
325
|
+
"""逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
|
|
326
|
+
|
|
327
|
+
`prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
|
|
328
|
+
避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
|
|
329
|
+
"""
|
|
330
|
+
want = set(ids) if ids else None
|
|
331
|
+
for nid, e in list((cg.index.get("nodes") or {}).items()):
|
|
332
|
+
if want is not None and nid not in want:
|
|
333
|
+
continue
|
|
334
|
+
if prefix and not str(nid).startswith(prefix):
|
|
335
|
+
continue
|
|
336
|
+
if layer and e.get("layer") != layer:
|
|
337
|
+
continue
|
|
338
|
+
if e.get("layer") in INTERNAL_LAYERS:
|
|
339
|
+
yield {"status": "skip", "reason": "internal", "id": nid}
|
|
340
|
+
continue
|
|
341
|
+
if not _readable_guard(cg, e):
|
|
342
|
+
yield {"status": "skip", "reason": "denied", "id": nid}
|
|
343
|
+
continue
|
|
344
|
+
fm, content = cg._read(e)
|
|
345
|
+
if fm is None:
|
|
346
|
+
yield {"status": "skip", "reason": "unreadable", "id": nid}
|
|
347
|
+
continue
|
|
348
|
+
if crypto.is_encrypted(content):
|
|
349
|
+
yield {"status": "skip", "reason": "locked", "id": nid}
|
|
350
|
+
continue
|
|
351
|
+
if e.get("layer") in SKIP_LAYERS or any(
|
|
352
|
+
t in SKIP_TAGS for t in (fm.get("tags") or [])):
|
|
353
|
+
yield {"status": "skip", "reason": "derived", "id": nid}
|
|
354
|
+
continue
|
|
355
|
+
need = _need(fm, content)
|
|
356
|
+
if not need: # 已齐备
|
|
357
|
+
yield {"status": "skip", "reason": "present", "id": nid}
|
|
358
|
+
continue
|
|
359
|
+
ph = []
|
|
360
|
+
der = derive_fields(fm, content, placeholder_out=ph)
|
|
361
|
+
if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
|
|
362
|
+
yield {"status": "skip", "reason": "placeholder", "id": nid,
|
|
363
|
+
"placeholder_fields": ph}
|
|
364
|
+
continue
|
|
365
|
+
yield {"status": "row", "id": nid,
|
|
366
|
+
"row": _worklist_row(nid, e, fm, content)}
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
_SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
|
|
370
|
+
"present": "skipped_present", "denied": "skipped_denied",
|
|
371
|
+
"unreadable": "skipped_unreadable", "internal": "skipped_internal"}
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
# 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
|
|
375
|
+
def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
|
|
376
|
+
"""生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
|
|
377
|
+
|
|
378
|
+
`prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
|
|
379
|
+
内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
|
|
380
|
+
"""
|
|
381
|
+
cg = _as_cg(x)
|
|
382
|
+
rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
|
|
383
|
+
"prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
|
|
384
|
+
"skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
|
|
385
|
+
"skipped_denied": 0, "skipped_placeholder": 0,
|
|
386
|
+
"skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
|
|
387
|
+
"targeted": 0, "items": []}
|
|
388
|
+
for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
|
|
389
|
+
rep["nodes_scanned"] += 1
|
|
390
|
+
if scan["status"] == "skip":
|
|
391
|
+
reason = scan["reason"]
|
|
392
|
+
if reason == "placeholder":
|
|
393
|
+
rep["skipped_placeholder"] += 1
|
|
394
|
+
rep["placeholder_ids"].append(scan["id"])
|
|
395
|
+
else:
|
|
396
|
+
key = _SKIP_KEY.get(reason)
|
|
397
|
+
if key:
|
|
398
|
+
rep[key] += 1
|
|
399
|
+
continue
|
|
400
|
+
row = scan["row"]
|
|
401
|
+
if row["track"] == "undetermined":
|
|
402
|
+
rep["undetermined"] += 1
|
|
403
|
+
rep["targeted"] += 1
|
|
404
|
+
if limit is None or len(rep["items"]) < limit:
|
|
405
|
+
rep["items"].append(row)
|
|
406
|
+
rep["planned_ids"] = [r["id"] for r in rep["items"]]
|
|
407
|
+
return rep
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
# ---- 子代理接口(提示词 + 解析) -----------------------------------------
|
|
411
|
+
|
|
412
|
+
_REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
|
|
413
|
+
赛道:{track};来源策略:{policy};待补字段:{need}
|
|
414
|
+
节点标题:{title}
|
|
415
|
+
已声明断言:
|
|
416
|
+
{claims}
|
|
417
|
+
正文:
|
|
418
|
+
{body}
|
|
419
|
+
|
|
420
|
+
只输出 JSON 数组,元素形如:
|
|
421
|
+
{{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
|
|
422
|
+
硬约束:
|
|
423
|
+
1. 文科(humanities)basis 只能是 textbook / public_kb;
|
|
424
|
+
2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
|
|
425
|
+
3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
|
|
426
|
+
4. 只能补 field=验证方式,禁止新增其它字段。"""
|
|
427
|
+
|
|
428
|
+
_VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
|
|
429
|
+
赛道:{track};来源策略:{policy}
|
|
430
|
+
候选(JSON):
|
|
431
|
+
{candidates}
|
|
432
|
+
|
|
433
|
+
只输出 JSON 数组,元素形如:
|
|
434
|
+
{{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
|
|
435
|
+
硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
# 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
|
|
439
|
+
def reflect_prompt(row: dict, fm: dict, content: str) -> str:
|
|
440
|
+
claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
|
|
441
|
+
return _REFLECT_TEMPLATE.format(
|
|
442
|
+
track=row.get("track"), policy=row.get("source_policy") or "(未定)",
|
|
443
|
+
need="、".join(row.get("need") or []),
|
|
444
|
+
title=_as_text(fm.get("title")) or row.get("id"),
|
|
445
|
+
claims=claims or "(无)",
|
|
446
|
+
body=(content or "")[:1200])
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
# 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
|
|
450
|
+
def verify_prompt(row: dict, rows: list) -> str:
|
|
451
|
+
cands = [{"field": r.get("field"), "value": r.get("value"),
|
|
452
|
+
"basis": r.get("basis"), "source": r.get("source")} for r in rows]
|
|
453
|
+
return _VERIFY_TEMPLATE.format(
|
|
454
|
+
track=row.get("track"), policy=row.get("source_policy") or "(未定)",
|
|
455
|
+
candidates=json.dumps(cands, ensure_ascii=False))
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
# 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
|
|
459
|
+
def _extract_json(raw, want_list=True):
|
|
460
|
+
"""从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
|
|
461
|
+
s = str(raw or "")
|
|
462
|
+
pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
|
|
463
|
+
for op, cl in pairs:
|
|
464
|
+
i, j = s.find(op), s.rfind(cl)
|
|
465
|
+
if i < 0 or j <= i:
|
|
466
|
+
continue
|
|
467
|
+
try:
|
|
468
|
+
obj = json.loads(s[i:j + 1])
|
|
469
|
+
except ValueError:
|
|
470
|
+
continue
|
|
471
|
+
if isinstance(obj, list):
|
|
472
|
+
return obj
|
|
473
|
+
if isinstance(obj, dict):
|
|
474
|
+
for key in ("rows", "items", "verdicts", "candidates", "data"):
|
|
475
|
+
if isinstance(obj.get(key), list):
|
|
476
|
+
return obj[key]
|
|
477
|
+
return [obj]
|
|
478
|
+
return []
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
# 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
|
|
482
|
+
def _norm_row(item) -> dict:
|
|
483
|
+
if not isinstance(item, dict):
|
|
484
|
+
return {}
|
|
485
|
+
return {
|
|
486
|
+
"field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
|
|
487
|
+
"value": _as_text(item.get("value")),
|
|
488
|
+
"basis": str(item.get("basis") or "").strip(),
|
|
489
|
+
"source": _as_source(item.get("source")),
|
|
490
|
+
"verdict": str(item.get("verdict") or "").strip().lower(),
|
|
491
|
+
"reason": str(item.get("reason") or "").strip(),
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
# 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
|
|
496
|
+
def parse_reflect_rows(raw) -> list:
|
|
497
|
+
out = []
|
|
498
|
+
for item in _extract_json(raw, want_list=True):
|
|
499
|
+
r = _norm_row(item)
|
|
500
|
+
if not r or r["field"] not in ("验证方式", "verification_basis"):
|
|
501
|
+
continue
|
|
502
|
+
r["field"] = "验证方式"
|
|
503
|
+
out.append(r)
|
|
504
|
+
return out
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
# 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
|
|
508
|
+
def parse_verify_rows(raw) -> list:
|
|
509
|
+
out = []
|
|
510
|
+
for item in _extract_json(raw, want_list=False):
|
|
511
|
+
r = _norm_row(item)
|
|
512
|
+
if not r or not r["field"]:
|
|
513
|
+
continue
|
|
514
|
+
out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
|
|
515
|
+
return out
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
# ---- 白箱闸门与双单元折叠 ------------------------------------------------
|
|
519
|
+
|
|
520
|
+
# 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
|
|
521
|
+
def gate_rows(rows: list, track: str) -> tuple:
|
|
522
|
+
"""零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
|
|
523
|
+
|
|
524
|
+
返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
|
|
525
|
+
"""
|
|
526
|
+
kept, gated = [], []
|
|
527
|
+
for r in rows:
|
|
528
|
+
f = normalize_field(r.get("field"))
|
|
529
|
+
row = dict(r, field=f)
|
|
530
|
+
if f not in WRITABLE_FIELDS:
|
|
531
|
+
gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
|
|
532
|
+
continue
|
|
533
|
+
if not basis_licensed(track, r.get("basis")):
|
|
534
|
+
gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
|
|
535
|
+
continue
|
|
536
|
+
if not r.get("source"):
|
|
537
|
+
gated.append(dict(row, reason="无来源,不写"))
|
|
538
|
+
continue
|
|
539
|
+
value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
|
|
540
|
+
if not value:
|
|
541
|
+
gated.append(dict(row, reason="无验证方式声明"))
|
|
542
|
+
continue
|
|
543
|
+
kept.append(dict(row, value=value, verdict="accept"))
|
|
544
|
+
return kept, gated
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
# 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
|
|
548
|
+
def fold_verdicts(rows: list) -> tuple:
|
|
549
|
+
"""把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
|
|
550
|
+
|
|
551
|
+
接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
|
|
552
|
+
任一 verify-drop 即否决(veto 优先)。
|
|
553
|
+
"""
|
|
554
|
+
groups = OrderedDict()
|
|
555
|
+
for r in rows:
|
|
556
|
+
key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
|
|
557
|
+
r.get("value"))
|
|
558
|
+
groups.setdefault(key, []).append(r)
|
|
559
|
+
accepted, deferred, dropped = [], [], []
|
|
560
|
+
for (nid, field, value), rs in groups.items():
|
|
561
|
+
refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
|
|
562
|
+
ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
|
|
563
|
+
base = {"id": nid, "field": field or "验证方式", "value": value,
|
|
564
|
+
"basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
|
|
565
|
+
"source": next((r.get("source") for r in refl if r.get("source")), []),
|
|
566
|
+
"track": next((r.get("track") for r in refl if r.get("track")), "")}
|
|
567
|
+
if not refl or not ver:
|
|
568
|
+
base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
|
|
569
|
+
deferred.append(base)
|
|
570
|
+
continue
|
|
571
|
+
if any(r.get("verdict") == "drop" for r in ver):
|
|
572
|
+
base["reason"] = next((r.get("reason") for r in ver
|
|
573
|
+
if r.get("verdict") == "drop"), "验证单元否决")
|
|
574
|
+
dropped.append(base)
|
|
575
|
+
continue
|
|
576
|
+
ra = any(r.get("verdict") == "accept" for r in refl)
|
|
577
|
+
va = any(r.get("verdict") == "accept" for r in ver)
|
|
578
|
+
if ra and va:
|
|
579
|
+
base["units"] = {
|
|
580
|
+
"reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
|
|
581
|
+
"verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
|
|
582
|
+
}
|
|
583
|
+
accepted.append(base)
|
|
584
|
+
else:
|
|
585
|
+
base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
|
|
586
|
+
deferred.append(base)
|
|
587
|
+
return accepted, deferred, dropped
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
# 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
|
|
591
|
+
def detect_self_verify(rows: list) -> bool:
|
|
592
|
+
"""同一执行者同时充当反思与验证 = 自证(禁止)。"""
|
|
593
|
+
r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
|
|
594
|
+
v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
|
|
595
|
+
r.discard("")
|
|
596
|
+
v.discard("")
|
|
597
|
+
return bool(r & v)
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
# ---- 落库写入 ------------------------------------------------------------
|
|
601
|
+
|
|
602
|
+
# 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
|
|
603
|
+
def _norm_verdicts(verdicts):
|
|
604
|
+
"""外部裁决(子代理落盘)→ `{id: [rows]}`。"""
|
|
605
|
+
if verdicts is None:
|
|
606
|
+
return None
|
|
607
|
+
out = OrderedDict()
|
|
608
|
+
if isinstance(verdicts, dict):
|
|
609
|
+
for nid, rs in verdicts.items():
|
|
610
|
+
out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
|
|
611
|
+
return out
|
|
612
|
+
for r in verdicts or []:
|
|
613
|
+
if not isinstance(r, dict):
|
|
614
|
+
continue
|
|
615
|
+
nid = str(r.get("id") or "")
|
|
616
|
+
if nid:
|
|
617
|
+
out.setdefault(nid, []).append(dict(r))
|
|
618
|
+
return out
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
# 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
|
|
622
|
+
def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
|
|
623
|
+
condition_claims=True):
|
|
624
|
+
"""把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
|
|
625
|
+
a = accepted[0]
|
|
626
|
+
basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
|
|
627
|
+
source = _as_source(a.get("source"))
|
|
628
|
+
content_before = content
|
|
629
|
+
value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
|
|
630
|
+
|
|
631
|
+
vb_before = {"had": "verification_basis" in fm,
|
|
632
|
+
"value": fm.get("verification_basis")}
|
|
633
|
+
prov_before = {"had": "verification_evidence" in fm,
|
|
634
|
+
"value": fm.get("verification_evidence")}
|
|
635
|
+
line_before = {"had": _has_ccg_line(content, "验证方式"),
|
|
636
|
+
"value": _ccg_field(content, "验证方式")}
|
|
637
|
+
comment0 = _comment(fm)
|
|
638
|
+
cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
|
|
639
|
+
|
|
640
|
+
# 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
|
|
641
|
+
content = _upsert_ccg_line(content, "验证方式", value)
|
|
642
|
+
_ensure_comment(fm)["验证方式"] = value
|
|
643
|
+
if nodefile.verification_basis_valid(fm):
|
|
644
|
+
basis = str(fm.get("verification_basis"))
|
|
645
|
+
else:
|
|
646
|
+
fm["verification_basis"] = basis
|
|
647
|
+
|
|
648
|
+
# 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
|
|
649
|
+
conditioned = []
|
|
650
|
+
if condition_claims:
|
|
651
|
+
label = BASIS_LABEL.get(basis, basis)
|
|
652
|
+
for c in row.get("claims") or []:
|
|
653
|
+
if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
|
|
654
|
+
continue
|
|
655
|
+
new = conditioned_claim(c.get("text"), label, source)
|
|
656
|
+
if not new or new == c.get("text"):
|
|
657
|
+
continue
|
|
658
|
+
nc, ok = _rewrite_claim(fm, content, c, new)
|
|
659
|
+
if not ok:
|
|
660
|
+
continue
|
|
661
|
+
content = nc
|
|
662
|
+
conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
|
|
663
|
+
"before": c.get("text"), "after": new})
|
|
664
|
+
|
|
665
|
+
wid = _sha(f"{nid}|{batch}|{time.time()}")
|
|
666
|
+
fm["verification_evidence"] = {
|
|
667
|
+
"at": round(time.time(), 3), "batch": batch, "basis": basis,
|
|
668
|
+
"source": source, "track": row.get("track"),
|
|
669
|
+
"policy": row.get("source_policy"), "write_id": wid,
|
|
670
|
+
"units": a.get("units") or {},
|
|
671
|
+
"conditioned": len(conditioned),
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
|
|
675
|
+
durable=True)
|
|
676
|
+
return {
|
|
677
|
+
"action": "crosscheck", "ts": time.time(), "batch": batch,
|
|
678
|
+
"actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
|
|
679
|
+
"node": nid, "layer": e.get("layer"), "track": row.get("track"),
|
|
680
|
+
"policy": row.get("source_policy"), "basis": basis,
|
|
681
|
+
"verification_value": value, "source": source,
|
|
682
|
+
"units": a.get("units") or {},
|
|
683
|
+
"fm_before": {"verification_basis": vb_before,
|
|
684
|
+
"verification_evidence": prov_before,
|
|
685
|
+
"comment_verification": cv_before},
|
|
686
|
+
"verification_line_before": line_before,
|
|
687
|
+
"claims_conditioned": conditioned,
|
|
688
|
+
"content_hash_before": _sha(content_before),
|
|
689
|
+
"content_hash_after": _sha(content),
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
# ---- 主流程 --------------------------------------------------------------
|
|
694
|
+
|
|
695
|
+
# 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
|
|
696
|
+
def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
|
|
697
|
+
"""调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
|
|
698
|
+
prompt = reflect_prompt(scan_row, fm, content)
|
|
699
|
+
cands = parse_reflect_rows(reflect_fn(prompt))
|
|
700
|
+
rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
|
|
701
|
+
track=scan_row["track"]) for c in cands]
|
|
702
|
+
if not rrows:
|
|
703
|
+
return [], []
|
|
704
|
+
vrows = []
|
|
705
|
+
if verify_fn is not None:
|
|
706
|
+
vp = verify_prompt(scan_row, rrows)
|
|
707
|
+
vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
|
|
708
|
+
track=scan_row["track"],
|
|
709
|
+
value=c.get("value") or next(
|
|
710
|
+
(x.get("value") for x in rrows
|
|
711
|
+
if x.get("field") == c.get("field")), ""))
|
|
712
|
+
for c in parse_verify_rows(verify_fn(vp))]
|
|
713
|
+
return rrows, vrows
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
# 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
|
|
717
|
+
def _rows_from_verdicts(nid, track, pre):
|
|
718
|
+
"""从外部裁决中拆出 (reflect, verify) 两组行。"""
|
|
719
|
+
rrows, vrows = [], []
|
|
720
|
+
for r in pre:
|
|
721
|
+
unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
|
|
722
|
+
base = dict(r, id=nid, unit=unit, track=track)
|
|
723
|
+
(vrows if unit == VERIFY_UNIT else rrows).append(base)
|
|
724
|
+
return rrows, vrows
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
# 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
|
|
728
|
+
def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
|
|
729
|
+
verify_fn=None, verdicts=None, apply=False,
|
|
730
|
+
batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
|
|
731
|
+
allow_self_verify=False, reflect_actor=None, verify_actor=None,
|
|
732
|
+
condition_claims=True, verbose=True, prefix=None) -> dict:
|
|
733
|
+
"""批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
|
|
734
|
+
|
|
735
|
+
`reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
|
|
736
|
+
`verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
|
|
737
|
+
二选一。`apply=True` 才写盘。
|
|
738
|
+
"""
|
|
739
|
+
cg = _as_cg(x)
|
|
740
|
+
batch = batch or CROSSCHECK_BATCH
|
|
741
|
+
vmap = _norm_verdicts(verdicts)
|
|
742
|
+
r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
|
|
743
|
+
v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
|
|
744
|
+
|
|
745
|
+
rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
|
|
746
|
+
"batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
|
|
747
|
+
"targeted": 0,
|
|
748
|
+
"accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
|
|
749
|
+
"claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
|
|
750
|
+
"skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
|
|
751
|
+
"skipped_unreadable": 0, "skipped_placeholder": 0,
|
|
752
|
+
"placeholder_ids": [], "undetermined": 0,
|
|
753
|
+
"reasons": {}, "samples": [], "entry_ids": []}
|
|
754
|
+
|
|
755
|
+
# 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
|
|
756
|
+
def _bump(reason):
|
|
757
|
+
rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
|
|
758
|
+
|
|
759
|
+
# 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
|
|
760
|
+
def _sample(kind, nid, detail=""):
|
|
761
|
+
if verbose and len(rep["samples"]) < 20:
|
|
762
|
+
rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
|
|
763
|
+
|
|
764
|
+
seen_targets = 0
|
|
765
|
+
for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
|
|
766
|
+
rep["nodes_scanned"] += 1
|
|
767
|
+
if scan["status"] == "skip":
|
|
768
|
+
reason = scan["reason"]
|
|
769
|
+
if reason == "placeholder":
|
|
770
|
+
rep["skipped_placeholder"] += 1
|
|
771
|
+
rep["placeholder_ids"].append(scan["id"])
|
|
772
|
+
else:
|
|
773
|
+
key = _SKIP_KEY.get(reason)
|
|
774
|
+
if key:
|
|
775
|
+
rep[key] += 1
|
|
776
|
+
continue
|
|
777
|
+
row = scan["row"]
|
|
778
|
+
if row["track"] == "undetermined":
|
|
779
|
+
rep["undetermined"] += 1
|
|
780
|
+
if limit is not None and seen_targets >= limit:
|
|
781
|
+
continue
|
|
782
|
+
seen_targets += 1
|
|
783
|
+
rep["targeted"] += 1
|
|
784
|
+
nid = row["id"]
|
|
785
|
+
e = cg.index["nodes"].get(nid)
|
|
786
|
+
fm, content = cg._read(e) if e else (None, None)
|
|
787
|
+
if fm is None or crypto.is_encrypted(content):
|
|
788
|
+
rep["skipped_locked"] += 1
|
|
789
|
+
continue
|
|
790
|
+
|
|
791
|
+
# ---- 取两单元裁决 ----
|
|
792
|
+
if vmap is not None:
|
|
793
|
+
pre = vmap.get(nid)
|
|
794
|
+
if not pre:
|
|
795
|
+
rep["deferred"] += 1
|
|
796
|
+
_bump("no_verdict")
|
|
797
|
+
continue
|
|
798
|
+
rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
|
|
799
|
+
elif reflect_fn is not None:
|
|
800
|
+
try:
|
|
801
|
+
rrows, vrows = _rows_for(row, fm, content, reflect_fn,
|
|
802
|
+
verify_fn, r_actor, v_actor)
|
|
803
|
+
except Exception as exc: # noqa: BLE001
|
|
804
|
+
rep["deferred"] += 1
|
|
805
|
+
_bump(f"unit_error:{type(exc).__name__}")
|
|
806
|
+
continue
|
|
807
|
+
else:
|
|
808
|
+
rep["deferred"] += 1
|
|
809
|
+
_bump("no_reflect")
|
|
810
|
+
continue
|
|
811
|
+
|
|
812
|
+
# 自证:同一执行者既反思又验证 → 拒收
|
|
813
|
+
if not allow_self_verify and detect_self_verify(rrows + vrows):
|
|
814
|
+
rep["deferred"] += 1
|
|
815
|
+
_bump("self_verify_disallowed")
|
|
816
|
+
_sample("self_verify", nid)
|
|
817
|
+
continue
|
|
818
|
+
|
|
819
|
+
# ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
|
|
820
|
+
kept, gated = gate_rows(rrows, row["track"])
|
|
821
|
+
for g in gated:
|
|
822
|
+
_bump(f"gate:{g.get('reason')[:24]}")
|
|
823
|
+
if not kept:
|
|
824
|
+
rep["deferred"] += 1
|
|
825
|
+
_bump("no_candidate")
|
|
826
|
+
_sample("gated", nid, gated[0].get("reason") if gated else "")
|
|
827
|
+
continue
|
|
828
|
+
if require_verify and not vrows:
|
|
829
|
+
rep["deferred"] += 1
|
|
830
|
+
_bump("verify_unavailable")
|
|
831
|
+
continue
|
|
832
|
+
|
|
833
|
+
rows = kept + vrows
|
|
834
|
+
accepted, deferred, dropped = fold_verdicts(rows)
|
|
835
|
+
if dropped and not accepted:
|
|
836
|
+
rep["rejected"] += 1
|
|
837
|
+
_bump("verify_veto")
|
|
838
|
+
_sample("veto", nid, dropped[0].get("reason", ""))
|
|
839
|
+
continue
|
|
840
|
+
if not accepted:
|
|
841
|
+
rep["deferred"] += 1
|
|
842
|
+
_bump("verdict_deferred")
|
|
843
|
+
_sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
|
|
844
|
+
continue
|
|
845
|
+
|
|
846
|
+
rep["accepted"] += 1
|
|
847
|
+
rep["claims_conditioned"] += sum(
|
|
848
|
+
1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
|
|
849
|
+
if apply:
|
|
850
|
+
rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
|
|
851
|
+
actor, condition_claims=condition_claims)
|
|
852
|
+
append_jsonl(_log_path(cg), rec)
|
|
853
|
+
rep["written"] += 1
|
|
854
|
+
rep["entry_ids"].append(rec["entry_id"])
|
|
855
|
+
_sample("accepted", nid, accepted[0].get("basis", ""))
|
|
856
|
+
|
|
857
|
+
if rep["written"]:
|
|
858
|
+
cg.rebuild_index()
|
|
859
|
+
return rep
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
# ---- 留痕查询 / 回滚 -----------------------------------------------------
|
|
863
|
+
|
|
864
|
+
# 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
|
|
865
|
+
def _log_path(cg) -> str:
|
|
866
|
+
return os.path.join(cg.root, CROSSCHECK_LOG)
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
# 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
|
|
870
|
+
def _reattach(fm: dict, content: str, box: dict, key: str):
|
|
871
|
+
"""把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
|
|
872
|
+
if not isinstance(box, dict):
|
|
873
|
+
return False
|
|
874
|
+
had, value = box.get("had"), box.get("value")
|
|
875
|
+
if key == "comment_verification":
|
|
876
|
+
c = _ensure_comment(fm)
|
|
877
|
+
if had:
|
|
878
|
+
c["验证方式"] = value
|
|
879
|
+
else:
|
|
880
|
+
c.pop("验证方式", None)
|
|
881
|
+
return True
|
|
882
|
+
if had:
|
|
883
|
+
fm[key] = value
|
|
884
|
+
else:
|
|
885
|
+
fm.pop(key, None)
|
|
886
|
+
return True
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
# 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
|
|
890
|
+
def _rewind_claim(fm: dict, content: str, c: dict):
|
|
891
|
+
"""撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
|
|
892
|
+
where, field = c.get("where"), c.get("field")
|
|
893
|
+
before, after = str(c.get("before") or ""), str(c.get("after") or "")
|
|
894
|
+
if not after:
|
|
895
|
+
return content, False
|
|
896
|
+
if where == "ccg" and field:
|
|
897
|
+
if _ccg_field(content, field).strip() == after.strip():
|
|
898
|
+
return _upsert_ccg_line(content, field, before), True
|
|
899
|
+
return content, False
|
|
900
|
+
if where == "comment" and field:
|
|
901
|
+
cc = _comment(fm)
|
|
902
|
+
v = cc.get(field)
|
|
903
|
+
if isinstance(v, list):
|
|
904
|
+
if after in v:
|
|
905
|
+
cc[field] = [before if x == after else x for x in v]
|
|
906
|
+
return content, True
|
|
907
|
+
return content, False
|
|
908
|
+
if str(v or "").strip() == after.strip():
|
|
909
|
+
cc[field] = before
|
|
910
|
+
return content, True
|
|
911
|
+
return content, False
|
|
912
|
+
if where == "body":
|
|
913
|
+
if after in content:
|
|
914
|
+
return content.replace(after, before, 1), True
|
|
915
|
+
return content, False
|
|
916
|
+
return content, False
|
|
917
|
+
|
|
918
|
+
|
|
919
|
+
# 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
|
|
920
|
+
def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
|
|
921
|
+
"""按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
|
|
922
|
+
cg = _as_cg(x)
|
|
923
|
+
want = set(entry_ids) if entry_ids else None
|
|
924
|
+
done = set()
|
|
925
|
+
rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
|
|
926
|
+
"batch": batch, "actor": actor, "planned": 0, "reverted": 0,
|
|
927
|
+
"skipped_drift": 0, "conflict": 0, "entry_ids": []}
|
|
928
|
+
recs = [r for r in (read_jsonl(_log_path(cg)) or [])
|
|
929
|
+
if r.get("action") == "crosscheck"
|
|
930
|
+
and (batch is None or r.get("batch") == batch)
|
|
931
|
+
and (want is None or r.get("entry_id") in want)]
|
|
932
|
+
rep["planned"] = len(recs)
|
|
933
|
+
for rec in recs:
|
|
934
|
+
nid = rec.get("node")
|
|
935
|
+
if not nid or nid in done:
|
|
936
|
+
continue
|
|
937
|
+
e = cg.index["nodes"].get(nid)
|
|
938
|
+
if not e:
|
|
939
|
+
rep["skipped_drift"] += 1
|
|
940
|
+
continue
|
|
941
|
+
fm, content = cg._read(e)
|
|
942
|
+
if fm is None or crypto.is_encrypted(content):
|
|
943
|
+
rep["skipped_drift"] += 1
|
|
944
|
+
continue
|
|
945
|
+
# 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
|
|
946
|
+
wid = (fm.get("verification_evidence") or {}).get("write_id")
|
|
947
|
+
if wid and rec.get("write_id") and wid != rec.get("write_id"):
|
|
948
|
+
rep["conflict"] += 1
|
|
949
|
+
continue
|
|
950
|
+
# 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
|
|
951
|
+
reverted = []
|
|
952
|
+
for c in rec.get("claims_conditioned") or []:
|
|
953
|
+
content, ok = _rewind_claim(fm, content, c)
|
|
954
|
+
if ok:
|
|
955
|
+
reverted.append(c.get("field") or c.get("where"))
|
|
956
|
+
# 2) 撤销「验证方式」行
|
|
957
|
+
lb = rec.get("verification_line_before") or {}
|
|
958
|
+
cur_line = _ccg_field(content, "验证方式")
|
|
959
|
+
if cur_line.strip() and cur_line.strip() == str(
|
|
960
|
+
rec.get("verification_value") or "").strip():
|
|
961
|
+
content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
|
|
962
|
+
if lb.get("had") else _remove_ccg_line(content, "验证方式"))
|
|
963
|
+
reverted.append("验证方式")
|
|
964
|
+
# 3) 还原 frontmatter 现场
|
|
965
|
+
for key, box in (rec.get("fm_before") or {}).items():
|
|
966
|
+
_reattach(fm, content, box, key)
|
|
967
|
+
if not reverted:
|
|
968
|
+
rep["conflict"] += 1
|
|
969
|
+
continue
|
|
970
|
+
cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
|
|
971
|
+
durable=True)
|
|
972
|
+
append_jsonl(_log_path(cg), {
|
|
973
|
+
"action": "crosscheck_rollback", "ts": time.time(),
|
|
974
|
+
"batch": rec.get("batch"), "actor": actor,
|
|
975
|
+
"entry_id": rec.get("entry_id"), "node": nid,
|
|
976
|
+
"reverted": reverted, "content_hash_after": _sha(content)})
|
|
977
|
+
rep["reverted"] += 1
|
|
978
|
+
rep["entry_ids"].append(rec.get("entry_id"))
|
|
979
|
+
done.add(nid)
|
|
980
|
+
if rep["reverted"]:
|
|
981
|
+
cg.rebuild_index()
|
|
982
|
+
return rep
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
# 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
|
|
986
|
+
def history(x, limit=100, action=None, batch=None) -> dict:
|
|
987
|
+
cg = _as_cg(x)
|
|
988
|
+
recs = []
|
|
989
|
+
for rec in read_jsonl(_log_path(cg)) or []:
|
|
990
|
+
if action and rec.get("action") != action:
|
|
991
|
+
continue
|
|
992
|
+
if batch and rec.get("batch") != batch:
|
|
993
|
+
continue
|
|
994
|
+
recs.append(rec)
|
|
995
|
+
total = len(recs)
|
|
996
|
+
if limit is not None and limit >= 0:
|
|
997
|
+
recs = recs[-limit:]
|
|
998
|
+
return {"root": cg.root, "total": total, "returned": len(recs),
|
|
999
|
+
"records": recs}
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
# ---- 权限与 CLI ----------------------------------------------------------
|
|
1003
|
+
|
|
1004
|
+
# 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
|
|
1005
|
+
def can_write_knowledge(principal) -> bool:
|
|
1006
|
+
"""落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
|
|
1007
|
+
if principal is None:
|
|
1008
|
+
return False
|
|
1009
|
+
try:
|
|
1010
|
+
if principal.expired() or not principal.can_write:
|
|
1011
|
+
return False
|
|
1012
|
+
return bool(principal.allows_layer("knowledge"))
|
|
1013
|
+
except Exception: # noqa: BLE001
|
|
1014
|
+
return False
|
|
1015
|
+
|
|
1016
|
+
|
|
1017
|
+
# 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
|
|
1018
|
+
def _load_verdicts(path: str):
|
|
1019
|
+
if not path:
|
|
1020
|
+
return None
|
|
1021
|
+
if not os.path.exists(path):
|
|
1022
|
+
raise SystemExit(f"裁决文件不存在:{path}")
|
|
1023
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
1024
|
+
text = fh.read().strip()
|
|
1025
|
+
if not text:
|
|
1026
|
+
return []
|
|
1027
|
+
try:
|
|
1028
|
+
return json.loads(text)
|
|
1029
|
+
except ValueError:
|
|
1030
|
+
rows = []
|
|
1031
|
+
for line in text.splitlines():
|
|
1032
|
+
line = line.strip()
|
|
1033
|
+
if not line or line.startswith("//"):
|
|
1034
|
+
continue
|
|
1035
|
+
rows.append(json.loads(line))
|
|
1036
|
+
return rows
|
|
1037
|
+
|
|
1038
|
+
|
|
1039
|
+
# 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
|
|
1040
|
+
def _cli(argv=None) -> int:
|
|
1041
|
+
ap = argparse.ArgumentParser(
|
|
1042
|
+
prog="python -m md_cg.crosscheck",
|
|
1043
|
+
description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
|
|
1044
|
+
ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
|
|
1045
|
+
ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
|
|
1046
|
+
ap.add_argument("--token-file", default=None)
|
|
1047
|
+
ap.add_argument("--action", default="worklist",
|
|
1048
|
+
choices=("worklist", "crosscheck", "rollback", "history"))
|
|
1049
|
+
ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
|
|
1050
|
+
ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
|
|
1051
|
+
ap.add_argument("--batch", default=CROSSCHECK_BATCH)
|
|
1052
|
+
ap.add_argument("--limit", type=int, default=None)
|
|
1053
|
+
ap.add_argument("--layer", default=None)
|
|
1054
|
+
ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
|
|
1055
|
+
ap.add_argument("--ids", default="", help="逗号分隔节点 id")
|
|
1056
|
+
ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
|
|
1057
|
+
ap.add_argument("--allow-self-verify", action="store_true")
|
|
1058
|
+
ap.add_argument("--reflect-actor", default=None)
|
|
1059
|
+
ap.add_argument("--verify-actor", default=None)
|
|
1060
|
+
args = ap.parse_args(argv)
|
|
1061
|
+
|
|
1062
|
+
from . import tokens
|
|
1063
|
+
principal = None
|
|
1064
|
+
if args.token:
|
|
1065
|
+
try:
|
|
1066
|
+
principal = tokens.verify_token(args.token, path=args.token_file)
|
|
1067
|
+
except tokens.TokenError as exc:
|
|
1068
|
+
raise SystemExit(f"令牌校验失败:{exc}")
|
|
1069
|
+
actor = getattr(principal, "actor", None)
|
|
1070
|
+
ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
|
|
1071
|
+
|
|
1072
|
+
if args.action == "worklist":
|
|
1073
|
+
rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
|
|
1074
|
+
ids=ids, prefix=args.prefix)
|
|
1075
|
+
elif args.action == "history":
|
|
1076
|
+
rep = history(args.root, limit=args.limit if args.limit is not None else 100,
|
|
1077
|
+
batch=args.batch)
|
|
1078
|
+
elif args.action == "rollback":
|
|
1079
|
+
if not can_write_knowledge(principal):
|
|
1080
|
+
raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
|
|
1081
|
+
rep = rollback(args.root, batch=args.batch, actor=actor)
|
|
1082
|
+
else:
|
|
1083
|
+
if args.apply and not can_write_knowledge(principal):
|
|
1084
|
+
raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
|
|
1085
|
+
rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
|
|
1086
|
+
prefix=args.prefix,
|
|
1087
|
+
verdicts=_load_verdicts(args.verdicts), apply=args.apply,
|
|
1088
|
+
batch=args.batch,
|
|
1089
|
+
require_verify=not args.no_verify,
|
|
1090
|
+
allow_self_verify=args.allow_self_verify,
|
|
1091
|
+
reflect_actor=args.reflect_actor,
|
|
1092
|
+
verify_actor=args.verify_actor, actor=actor)
|
|
1093
|
+
print(json.dumps(rep, ensure_ascii=False, indent=2))
|
|
1094
|
+
return 0
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
if __name__ == "__main__": # pragma: no cover
|
|
1098
1098
|
sys.exit(_cli())
|