@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,512 +1,512 @@
1
- """
2
- lexer.py · 词法分析器 v2.0(生产版)
3
- 中文分词 + Token 识别
4
- 支持道德经助记符、九章算术结构、中文标点
5
-
6
- v2.0 算法(两阶段):
7
- 阶段一(粗分):逐字符扫描,按"类型"分组
8
- - 中文串(连续 CJK)
9
- - 字母串(连续 ASCII 字母/数字/下划线)
10
- - 数字串(连续数字,含小数点)
11
- - 标点(单个)
12
- - 空白(跳过)
13
- - 字符串(引号包裹)
14
-
15
- 阶段二(精分):对每个中文串/字母串做关键字切分
16
- - 使用正向最大匹配
17
- - 关键字 → Token
18
- - 非关键字 → IDENTIFIER
19
- """
20
-
21
- from enum import Enum, auto
22
- from dataclasses import dataclass
23
- from typing import List, Tuple
24
-
25
-
26
- # =============================================================================
27
- # Token 类型定义
28
- # =============================================================================
29
-
30
- class TokenType(Enum):
31
- """Token 类型枚举"""
32
-
33
- # === 道德经助记符(指令集) ===
34
- DAO = auto() # 道
35
- DE = auto() # 德
36
- ZIRAN = auto() # 自然
37
- WUWEI = auto() # 无为
38
- GU = auto() # 谷
39
- PIN = auto() # 牝
40
- ROU = auto() # 柔
41
- PU = auto() # 朴
42
- ZHI = auto() # 止
43
- ZHIZU = auto() # 知足
44
-
45
- # === 九章算术结构关键字 ===
46
- WENYUE = auto() # 问曰
47
- DAYUE = auto() # 答曰
48
- SHUYUE = auto() # 术曰
49
-
50
- # === 条件/逻辑关键字 ===
51
- RUO = auto() # 若
52
- ZE = auto() # 则
53
- FOUZE = auto() # 否则
54
- YU = auto() # 于
55
- WEI = auto() # 为
56
- BUWEI = auto() # 不为
57
- QIE = auto() # 且
58
- HUO = auto() # 或
59
- FEI = auto() # 非
60
- DENGYU = auto() # 等于
61
- DAYU = auto() # 大于
62
- XIAOYU = auto() # 小于
63
-
64
- # === 循环关键字(当…执行:白箱循环语法)===
65
- DANG = auto() # 当
66
- ZHIXING = auto() # 执行
67
-
68
- # === 函数关键字(定义…返回:函数抽象)===
69
- DINGYI = auto() # 定义
70
- FANHUI = auto() # 返回
71
-
72
- # === 标识符与常量 ===
73
- IDENTIFIER = auto() # 标识符
74
- NUMBER = auto() # 数值常量
75
- STRING = auto() # 字符串常量
76
-
77
- # === 标点符号 ===
78
- COMMA = auto() # ,
79
- PERIOD = auto() # 。
80
- SEMICOLON = auto() # ;
81
- COLON = auto() # :
82
- LPAREN = auto() # (
83
- RPAREN = auto() # )
84
- ARROW = auto() # →
85
- QUESTION = auto() # ?
86
- EXCLAM = auto() # !
87
- EQUALS = auto() # = 或 =(赋值符号)
88
-
89
- # === 算术运算符(循环体/表达式需要)===
90
- OP_ADD = auto() # +
91
- OP_SUB = auto() # -
92
- OP_MUL = auto() # *
93
- OP_DIV = auto() # /
94
-
95
- # === 特殊 ===
96
- COMMENT = auto() # 注释
97
- NEWLINE = auto() # 换行
98
- EOF = auto() # 文件结束
99
- UNKNOWN = auto() # 未知
100
-
101
-
102
- # =============================================================================
103
- # Token 数据结构
104
- # =============================================================================
105
-
106
- @dataclass
107
- class Token:
108
- """Token 数据类"""
109
- type: TokenType
110
- value: str
111
- line: int = 1
112
- column: int = 1
113
-
114
- def __repr__(self) -> str:
115
- """Token 调试表示:类型·值·位置一行式。"""
116
- v = str(self.value)[:30]
117
- if len(str(self.value)) > 30:
118
- v += "..."
119
- return f"Token({self.type.name}, '{v}', L{self.line}:C{self.column})"
120
-
121
-
122
- # =============================================================================
123
- # 关键字映射表
124
- # =============================================================================
125
-
126
- # 所有关键字集合
127
- KEYWORDS = {
128
- # 道德经
129
- "道", "德", "自然", "无为", "谷", "牝", "柔", "朴", "止", "知足",
130
- # 九章算术
131
- "问曰", "答曰", "术曰",
132
- # 条件/逻辑
133
- "若", "则", "否则", "于", "为", "不为", "且", "或", "非",
134
- "等于", "大于", "小于",
135
- # 循环(当…执行:白箱循环语法)
136
- "当", "执行",
137
- # 函数(定义…返回:函数抽象)
138
- "定义", "返回",
139
- # 中文算术词(加/减/乘/除 → 运算符)
140
- "加", "减", "乘", "除",
141
- }
142
-
143
- # 关键字 → TokenType
144
- KEYWORD_MAP = {
145
- "道": TokenType.DAO,
146
- "德": TokenType.DE,
147
- "自然": TokenType.ZIRAN,
148
- "无为": TokenType.WUWEI,
149
- "谷": TokenType.GU,
150
- "牝": TokenType.PIN,
151
- "柔": TokenType.ROU,
152
- "朴": TokenType.PU,
153
- "止": TokenType.ZHI,
154
- "知足": TokenType.ZHIZU,
155
- "问曰": TokenType.WENYUE,
156
- "答曰": TokenType.DAYUE,
157
- "术曰": TokenType.SHUYUE,
158
- "若": TokenType.RUO,
159
- "则": TokenType.ZE,
160
- "否则": TokenType.FOUZE,
161
- "于": TokenType.YU,
162
- "为": TokenType.WEI,
163
- "不为": TokenType.BUWEI,
164
- "且": TokenType.QIE,
165
- "或": TokenType.HUO,
166
- "非": TokenType.FEI,
167
- "等于": TokenType.DENGYU,
168
- "大于": TokenType.DAYU,
169
- "小于": TokenType.XIAOYU,
170
- "当": TokenType.DANG,
171
- "执行": TokenType.ZHIXING,
172
- "定义": TokenType.DINGYI,
173
- "返回": TokenType.FANHUI,
174
- "加": TokenType.OP_ADD,
175
- "减": TokenType.OP_SUB,
176
- "乘": TokenType.OP_MUL,
177
- "除": TokenType.OP_DIV,
178
- }
179
-
180
- # 按长度降序排列
181
- SORTED_KW = sorted(KEYWORDS, key=len, reverse=True)
182
-
183
- # 中文标点映射
184
- PUNCTUATION_MAP = {
185
- ",": TokenType.COMMA, ",": TokenType.COMMA,
186
- "。": TokenType.PERIOD, ".": TokenType.PERIOD,
187
- ";": TokenType.SEMICOLON, ";": TokenType.SEMICOLON,
188
- ":": TokenType.COLON, ":": TokenType.COLON,
189
- "(": TokenType.LPAREN, "(": TokenType.LPAREN,
190
- ")": TokenType.RPAREN, ")": TokenType.RPAREN,
191
- "→": TokenType.ARROW,
192
- "?": TokenType.QUESTION, "?": TokenType.QUESTION,
193
- "!": TokenType.EXCLAM, "!": TokenType.EXCLAM,
194
- "=": TokenType.EQUALS, "=": TokenType.EQUALS,
195
- "+": TokenType.OP_ADD, "+": TokenType.OP_ADD,
196
- "-": TokenType.OP_SUB, "-": TokenType.OP_SUB,
197
- "*": TokenType.OP_MUL, "×": TokenType.OP_MUL,
198
- "/": TokenType.OP_DIV, "÷": TokenType.OP_DIV,
199
- }
200
-
201
-
202
- # =============================================================================
203
- # 工具函数
204
- # =============================================================================
205
-
206
- def _is_cjk(ch: str) -> bool:
207
- """是否为 CJK 统一汉字"""
208
- code = ord(ch)
209
- return 0x4E00 <= code <= 0x9FFF
210
-
211
- def _is_cjk_or_alpha(ch: str) -> bool:
212
- """是否为中文、字母、下划线"""
213
- return _is_cjk(ch) or ch.isalpha() or ch == "_"
214
-
215
-
216
- # =============================================================================
217
- # 词法分析器
218
- # =============================================================================
219
-
220
- class Lexer:
221
- """
222
- 词法分析器 v2.0
223
-
224
- 两阶段分词:
225
- 阶段一:粗分(按字符类型分组)
226
- 阶段二:精分(对中文/字母串做关键字切分)
227
- """
228
-
229
- def __init__(self, source: str):
230
- """Token 构造:类型·字面值·行号列号。"""
231
- self.source = source
232
- self.pos = 0
233
- self.line = 1
234
- self.column = 1
235
- self.tokens: List[Token] = []
236
- self.errors: List[str] = []
237
-
238
- def tokenize(self) -> Tuple[List[Token], List[str]]:
239
- """执行词法分析"""
240
- self.tokens = []
241
- self.errors = []
242
- self.pos = 0
243
- self.line = 1
244
- self.column = 1
245
-
246
- while self.pos < len(self.source):
247
- ch = self.source[self.pos]
248
-
249
- # 换行
250
- if ch == "\n":
251
- self.line += 1
252
- self.column = 1
253
- self.pos += 1
254
- continue
255
-
256
- # 空白
257
- if ch in (" ", "\t", "\r"):
258
- self.column += 1
259
- self.pos += 1
260
- continue
261
-
262
- # 注释 //
263
- if ch == "/" and self.pos + 1 < len(self.source) and self.source[self.pos + 1] == "/":
264
- self._skip_comment()
265
- continue
266
-
267
- # 字符串
268
- if ch == '"' or ch == "\u201c":
269
- self._read_string()
270
- continue
271
-
272
- # 数字(包括 . 开头的小数)
273
- if ch.isdigit() or ch == "." or (ch == "-" and self._peek_isdigit()):
274
- self._read_number()
275
- continue
276
-
277
- # 中文标点
278
- if ch in PUNCTUATION_MAP:
279
- self._emit(PUNCTUATION_MAP[ch], ch)
280
- self.pos += 1
281
- self.column += 1
282
- continue
283
-
284
- # 赋值符号 = (ASCII)或 =(全角)
285
- if ch == "=" or ch == "\uFF1D":
286
- self._emit(TokenType.EQUALS, ch)
287
- self.pos += 1
288
- self.column += 1
289
- continue
290
-
291
- # 中文/字母/下划线 → 粗分 + 精分
292
- if _is_cjk_or_alpha(ch):
293
- self._read_and_segment()
294
- continue
295
-
296
- # 未知字符
297
- self.errors.append(f"L{self.line}:C{self.column} 未知字符: '{ch}' (U+{ord(ch):04X})")
298
- self.pos += 1
299
- self.column += 1
300
-
301
- self._emit(TokenType.EOF, "")
302
- return self.tokens, self.errors
303
-
304
- # ---- 阶段一:粗分 ----
305
-
306
- def _read_and_segment(self):
307
- """
308
- 读取连续的中文/字母/数字/下划线,然后做关键字精分
309
-
310
- 这是核心方法:
311
- 1. 贪婪读取所有"词字符"(中文/字母/数字/下划线)
312
- 2. 对结果做正向最大匹配切分
313
- """
314
- start_col = self.column
315
- start_pos = self.pos
316
-
317
- # 贪婪读取
318
- while self.pos < len(self.source):
319
- ch = self.source[self.pos]
320
- if _is_cjk_or_alpha(ch) or ch.isdigit():
321
- self.pos += 1
322
- else:
323
- break
324
-
325
- text = self.source[start_pos:self.pos]
326
- self.column += (self.pos - start_pos)
327
-
328
- # 阶段二:精分
329
- self._segment(text, start_col)
330
-
331
- # ---- 阶段二:精分 ----
332
-
333
- def _segment(self, text: str, start_col: int):
334
- """
335
- 正向最大匹配(Forward Maximum Matching)
336
-
337
- 对 text 中的每个位置,找最长的匹配关键字。
338
- 如果找不到关键字,发出单个字符作为 IDENTIFIER。
339
-
340
- 关键改进:使用位置指针 i 遍历 text,
341
- 每次从 i 开始找最长关键字。
342
- 找到后 i 跳过该关键字长度。
343
- 找不到时 i 前进 1(发出单个字符)。
344
-
345
- 标识符规则:连续 CJK 串优先整体为标识符(如「阶乘」含关键词「乘」,
346
- 但整体不是关键词 → 保持为标识符,避免误切分)。
347
- """
348
- i = 0
349
- col = start_col
350
-
351
- while i < len(text):
352
- # 尝试从位置 i 找最长关键字
353
- matched_kw = None
354
- matched_len = 0
355
-
356
- for kw in SORTED_KW:
357
- if text.startswith(kw, i):
358
- if len(kw) > matched_len:
359
- matched_kw = kw
360
- matched_len = len(kw)
361
-
362
- if matched_kw is not None:
363
- # 发出关键字 token
364
- self._emit(KEYWORD_MAP[matched_kw], matched_kw, col)
365
- i += matched_len
366
- col += matched_len
367
- else:
368
- # 不是关键字 → 收集连续的非关键字字符作为标识符
369
- # 中文规则:连续 CJK 串优先整体为标识符(如「阶乘」含关键词「乘」,
370
- # 但整串「阶乘」非关键词 → 保持整体,避免误切分)
371
- ident_start = i
372
- ident_col = col
373
-
374
- if _is_cjk(text[i]):
375
- while i < len(text) and _is_cjk(text[i]):
376
- i += 1
377
- col += 1
378
- else:
379
- while i < len(text):
380
- ch = text[i]
381
- has_kw_ahead = any(text.startswith(kw, i) for kw in SORTED_KW)
382
- if has_kw_ahead:
383
- break
384
- i += 1
385
- col += 1
386
-
387
- ident = text[ident_start:i]
388
- if ident:
389
- self._emit(TokenType.IDENTIFIER, ident, ident_col)
390
-
391
- # ---- 数字读取 ----
392
-
393
- def _peek_isdigit(self) -> bool:
394
- """前瞻当前字符是否数字(多位数聚合判断)。"""
395
- return self.pos + 1 < len(self.source) and self.source[self.pos + 1].isdigit()
396
-
397
- def _read_number(self):
398
- """读取数值(支持整数、小数、负数)"""
399
- start_col = self.column
400
- result = []
401
-
402
- # 负号
403
- if self.source[self.pos] == "-":
404
- result.append("-")
405
- self.pos += 1
406
- self.column += 1
407
-
408
- # 整数部分
409
- while self.pos < len(self.source) and self.source[self.pos].isdigit():
410
- result.append(self.source[self.pos])
411
- self.pos += 1
412
- self.column += 1
413
-
414
- # 小数部分
415
- if self.pos < len(self.source) and self.source[self.pos] == ".":
416
- result.append(".")
417
- self.pos += 1
418
- self.column += 1
419
- while self.pos < len(self.source) and self.source[self.pos].isdigit():
420
- result.append(self.source[self.pos])
421
- self.pos += 1
422
- self.column += 1
423
-
424
- num_str = "".join(result)
425
- try:
426
- float(num_str)
427
- self._emit(TokenType.NUMBER, num_str)
428
- except ValueError:
429
- self.errors.append(f"L{self.line}:C{start_col} 非法数值: '{num_str}'")
430
- self._emit(TokenType.NUMBER, num_str)
431
-
432
- # ---- 字符串读取 ----
433
-
434
- def _read_string(self):
435
- """读取字符串"""
436
- quote_char = self.source[self.pos]
437
- end_quote = '"' if quote_char == '"' else "\u201d"
438
-
439
- self.pos += 1
440
- self.column += 1
441
- result = []
442
-
443
- while self.pos < len(self.source) and self.source[self.pos] != end_quote:
444
- ch = self.source[self.pos]
445
- if ch == "\n":
446
- self.errors.append(f"L{self.line}:C{self.column} 字符串未闭合")
447
- break
448
- result.append(ch)
449
- self.pos += 1
450
- self.column += 1
451
-
452
- if self.pos < len(self.source):
453
- self.pos += 1 # 跳过结束引号
454
- self.column += 1
455
-
456
- self._emit(TokenType.STRING, "".join(result))
457
-
458
- # ---- 注释跳过 ----
459
-
460
- def _skip_comment(self):
461
- """跳过注释至本行结束。"""
462
- while self.pos < len(self.source) and self.source[self.pos] != "\n":
463
- self.pos += 1
464
-
465
- # ---- Token 输出 ----
466
-
467
- def _emit(self, token_type: TokenType, value: str, col: int = None):
468
- """输出一个 Token"""
469
- c = col if col is not None else self.column
470
- self.tokens.append(Token(token_type, value, self.line, c))
471
-
472
-
473
- # =============================================================================
474
- # 便捷函数
475
- # =============================================================================
476
-
477
- def tokenize(source: str) -> Tuple[List[Token], List[str]]:
478
- """便捷函数"""
479
- lexer = Lexer(source)
480
- return lexer.tokenize()
481
-
482
-
483
- # =============================================================================
484
- # 测试
485
- # =============================================================================
486
-
487
- if __name__ == "__main__":
488
- test_code = """若条件空间为伴侣,则止情感权重于0.15。
489
- 道 新信任路径
490
- 问曰:如何验证信任?
491
- 答曰:信任值大于0.7。
492
- 术曰:1。德 累积信任值;2。自然 恢复默认。"""
493
-
494
- print("=" * 60)
495
- print("词法分析器 v2.0 测试")
496
- print("=" * 60)
497
- print(f"源代码:\n{test_code}\n")
498
-
499
- tokens, errors = tokenize(test_code)
500
-
501
- print("Token 序列:")
502
- for t in tokens:
503
- if t.type != TokenType.EOF:
504
- print(f" {t}")
505
-
506
- if errors:
507
- print(f"\n错误:")
508
- for e in errors:
509
- print(f" ❌ {e}")
510
-
511
- valid = len([t for t in tokens if t.type != TokenType.EOF])
512
- print(f"\n总计:{valid} 个 Token,{len(errors)} 个错误")
1
+ """
2
+ lexer.py · 词法分析器 v2.0(生产版)
3
+ 中文分词 + Token 识别
4
+ 支持道德经助记符、九章算术结构、中文标点
5
+
6
+ v2.0 算法(两阶段):
7
+ 阶段一(粗分):逐字符扫描,按"类型"分组
8
+ - 中文串(连续 CJK)
9
+ - 字母串(连续 ASCII 字母/数字/下划线)
10
+ - 数字串(连续数字,含小数点)
11
+ - 标点(单个)
12
+ - 空白(跳过)
13
+ - 字符串(引号包裹)
14
+
15
+ 阶段二(精分):对每个中文串/字母串做关键字切分
16
+ - 使用正向最大匹配
17
+ - 关键字 → Token
18
+ - 非关键字 → IDENTIFIER
19
+ """
20
+
21
+ from enum import Enum, auto
22
+ from dataclasses import dataclass
23
+ from typing import List, Tuple
24
+
25
+
26
+ # =============================================================================
27
+ # Token 类型定义
28
+ # =============================================================================
29
+
30
+ class TokenType(Enum):
31
+ """Token 类型枚举"""
32
+
33
+ # === 道德经助记符(指令集) ===
34
+ DAO = auto() # 道
35
+ DE = auto() # 德
36
+ ZIRAN = auto() # 自然
37
+ WUWEI = auto() # 无为
38
+ GU = auto() # 谷
39
+ PIN = auto() # 牝
40
+ ROU = auto() # 柔
41
+ PU = auto() # 朴
42
+ ZHI = auto() # 止
43
+ ZHIZU = auto() # 知足
44
+
45
+ # === 九章算术结构关键字 ===
46
+ WENYUE = auto() # 问曰
47
+ DAYUE = auto() # 答曰
48
+ SHUYUE = auto() # 术曰
49
+
50
+ # === 条件/逻辑关键字 ===
51
+ RUO = auto() # 若
52
+ ZE = auto() # 则
53
+ FOUZE = auto() # 否则
54
+ YU = auto() # 于
55
+ WEI = auto() # 为
56
+ BUWEI = auto() # 不为
57
+ QIE = auto() # 且
58
+ HUO = auto() # 或
59
+ FEI = auto() # 非
60
+ DENGYU = auto() # 等于
61
+ DAYU = auto() # 大于
62
+ XIAOYU = auto() # 小于
63
+
64
+ # === 循环关键字(当…执行:白箱循环语法)===
65
+ DANG = auto() # 当
66
+ ZHIXING = auto() # 执行
67
+
68
+ # === 函数关键字(定义…返回:函数抽象)===
69
+ DINGYI = auto() # 定义
70
+ FANHUI = auto() # 返回
71
+
72
+ # === 标识符与常量 ===
73
+ IDENTIFIER = auto() # 标识符
74
+ NUMBER = auto() # 数值常量
75
+ STRING = auto() # 字符串常量
76
+
77
+ # === 标点符号 ===
78
+ COMMA = auto() # ,
79
+ PERIOD = auto() # 。
80
+ SEMICOLON = auto() # ;
81
+ COLON = auto() # :
82
+ LPAREN = auto() # (
83
+ RPAREN = auto() # )
84
+ ARROW = auto() # →
85
+ QUESTION = auto() # ?
86
+ EXCLAM = auto() # !
87
+ EQUALS = auto() # = 或 =(赋值符号)
88
+
89
+ # === 算术运算符(循环体/表达式需要)===
90
+ OP_ADD = auto() # +
91
+ OP_SUB = auto() # -
92
+ OP_MUL = auto() # *
93
+ OP_DIV = auto() # /
94
+
95
+ # === 特殊 ===
96
+ COMMENT = auto() # 注释
97
+ NEWLINE = auto() # 换行
98
+ EOF = auto() # 文件结束
99
+ UNKNOWN = auto() # 未知
100
+
101
+
102
+ # =============================================================================
103
+ # Token 数据结构
104
+ # =============================================================================
105
+
106
+ @dataclass
107
+ class Token:
108
+ """Token 数据类"""
109
+ type: TokenType
110
+ value: str
111
+ line: int = 1
112
+ column: int = 1
113
+
114
+ def __repr__(self) -> str:
115
+ """Token 调试表示:类型·值·位置一行式。"""
116
+ v = str(self.value)[:30]
117
+ if len(str(self.value)) > 30:
118
+ v += "..."
119
+ return f"Token({self.type.name}, '{v}', L{self.line}:C{self.column})"
120
+
121
+
122
+ # =============================================================================
123
+ # 关键字映射表
124
+ # =============================================================================
125
+
126
+ # 所有关键字集合
127
+ KEYWORDS = {
128
+ # 道德经
129
+ "道", "德", "自然", "无为", "谷", "牝", "柔", "朴", "止", "知足",
130
+ # 九章算术
131
+ "问曰", "答曰", "术曰",
132
+ # 条件/逻辑
133
+ "若", "则", "否则", "于", "为", "不为", "且", "或", "非",
134
+ "等于", "大于", "小于",
135
+ # 循环(当…执行:白箱循环语法)
136
+ "当", "执行",
137
+ # 函数(定义…返回:函数抽象)
138
+ "定义", "返回",
139
+ # 中文算术词(加/减/乘/除 → 运算符)
140
+ "加", "减", "乘", "除",
141
+ }
142
+
143
+ # 关键字 → TokenType
144
+ KEYWORD_MAP = {
145
+ "道": TokenType.DAO,
146
+ "德": TokenType.DE,
147
+ "自然": TokenType.ZIRAN,
148
+ "无为": TokenType.WUWEI,
149
+ "谷": TokenType.GU,
150
+ "牝": TokenType.PIN,
151
+ "柔": TokenType.ROU,
152
+ "朴": TokenType.PU,
153
+ "止": TokenType.ZHI,
154
+ "知足": TokenType.ZHIZU,
155
+ "问曰": TokenType.WENYUE,
156
+ "答曰": TokenType.DAYUE,
157
+ "术曰": TokenType.SHUYUE,
158
+ "若": TokenType.RUO,
159
+ "则": TokenType.ZE,
160
+ "否则": TokenType.FOUZE,
161
+ "于": TokenType.YU,
162
+ "为": TokenType.WEI,
163
+ "不为": TokenType.BUWEI,
164
+ "且": TokenType.QIE,
165
+ "或": TokenType.HUO,
166
+ "非": TokenType.FEI,
167
+ "等于": TokenType.DENGYU,
168
+ "大于": TokenType.DAYU,
169
+ "小于": TokenType.XIAOYU,
170
+ "当": TokenType.DANG,
171
+ "执行": TokenType.ZHIXING,
172
+ "定义": TokenType.DINGYI,
173
+ "返回": TokenType.FANHUI,
174
+ "加": TokenType.OP_ADD,
175
+ "减": TokenType.OP_SUB,
176
+ "乘": TokenType.OP_MUL,
177
+ "除": TokenType.OP_DIV,
178
+ }
179
+
180
+ # 按长度降序排列
181
+ SORTED_KW = sorted(KEYWORDS, key=len, reverse=True)
182
+
183
+ # 中文标点映射
184
+ PUNCTUATION_MAP = {
185
+ ",": TokenType.COMMA, ",": TokenType.COMMA,
186
+ "。": TokenType.PERIOD, ".": TokenType.PERIOD,
187
+ ";": TokenType.SEMICOLON, ";": TokenType.SEMICOLON,
188
+ ":": TokenType.COLON, ":": TokenType.COLON,
189
+ "(": TokenType.LPAREN, "(": TokenType.LPAREN,
190
+ ")": TokenType.RPAREN, ")": TokenType.RPAREN,
191
+ "→": TokenType.ARROW,
192
+ "?": TokenType.QUESTION, "?": TokenType.QUESTION,
193
+ "!": TokenType.EXCLAM, "!": TokenType.EXCLAM,
194
+ "=": TokenType.EQUALS, "=": TokenType.EQUALS,
195
+ "+": TokenType.OP_ADD, "+": TokenType.OP_ADD,
196
+ "-": TokenType.OP_SUB, "-": TokenType.OP_SUB,
197
+ "*": TokenType.OP_MUL, "×": TokenType.OP_MUL,
198
+ "/": TokenType.OP_DIV, "÷": TokenType.OP_DIV,
199
+ }
200
+
201
+
202
+ # =============================================================================
203
+ # 工具函数
204
+ # =============================================================================
205
+
206
+ def _is_cjk(ch: str) -> bool:
207
+ """是否为 CJK 统一汉字"""
208
+ code = ord(ch)
209
+ return 0x4E00 <= code <= 0x9FFF
210
+
211
+ def _is_cjk_or_alpha(ch: str) -> bool:
212
+ """是否为中文、字母、下划线"""
213
+ return _is_cjk(ch) or ch.isalpha() or ch == "_"
214
+
215
+
216
+ # =============================================================================
217
+ # 词法分析器
218
+ # =============================================================================
219
+
220
+ class Lexer:
221
+ """
222
+ 词法分析器 v2.0
223
+
224
+ 两阶段分词:
225
+ 阶段一:粗分(按字符类型分组)
226
+ 阶段二:精分(对中文/字母串做关键字切分)
227
+ """
228
+
229
+ def __init__(self, source: str):
230
+ """Token 构造:类型·字面值·行号列号。"""
231
+ self.source = source
232
+ self.pos = 0
233
+ self.line = 1
234
+ self.column = 1
235
+ self.tokens: List[Token] = []
236
+ self.errors: List[str] = []
237
+
238
+ def tokenize(self) -> Tuple[List[Token], List[str]]:
239
+ """执行词法分析"""
240
+ self.tokens = []
241
+ self.errors = []
242
+ self.pos = 0
243
+ self.line = 1
244
+ self.column = 1
245
+
246
+ while self.pos < len(self.source):
247
+ ch = self.source[self.pos]
248
+
249
+ # 换行
250
+ if ch == "\n":
251
+ self.line += 1
252
+ self.column = 1
253
+ self.pos += 1
254
+ continue
255
+
256
+ # 空白
257
+ if ch in (" ", "\t", "\r"):
258
+ self.column += 1
259
+ self.pos += 1
260
+ continue
261
+
262
+ # 注释 //
263
+ if ch == "/" and self.pos + 1 < len(self.source) and self.source[self.pos + 1] == "/":
264
+ self._skip_comment()
265
+ continue
266
+
267
+ # 字符串
268
+ if ch == '"' or ch == "\u201c":
269
+ self._read_string()
270
+ continue
271
+
272
+ # 数字(包括 . 开头的小数)
273
+ if ch.isdigit() or ch == "." or (ch == "-" and self._peek_isdigit()):
274
+ self._read_number()
275
+ continue
276
+
277
+ # 中文标点
278
+ if ch in PUNCTUATION_MAP:
279
+ self._emit(PUNCTUATION_MAP[ch], ch)
280
+ self.pos += 1
281
+ self.column += 1
282
+ continue
283
+
284
+ # 赋值符号 = (ASCII)或 =(全角)
285
+ if ch == "=" or ch == "\uFF1D":
286
+ self._emit(TokenType.EQUALS, ch)
287
+ self.pos += 1
288
+ self.column += 1
289
+ continue
290
+
291
+ # 中文/字母/下划线 → 粗分 + 精分
292
+ if _is_cjk_or_alpha(ch):
293
+ self._read_and_segment()
294
+ continue
295
+
296
+ # 未知字符
297
+ self.errors.append(f"L{self.line}:C{self.column} 未知字符: '{ch}' (U+{ord(ch):04X})")
298
+ self.pos += 1
299
+ self.column += 1
300
+
301
+ self._emit(TokenType.EOF, "")
302
+ return self.tokens, self.errors
303
+
304
+ # ---- 阶段一:粗分 ----
305
+
306
+ def _read_and_segment(self):
307
+ """
308
+ 读取连续的中文/字母/数字/下划线,然后做关键字精分
309
+
310
+ 这是核心方法:
311
+ 1. 贪婪读取所有"词字符"(中文/字母/数字/下划线)
312
+ 2. 对结果做正向最大匹配切分
313
+ """
314
+ start_col = self.column
315
+ start_pos = self.pos
316
+
317
+ # 贪婪读取
318
+ while self.pos < len(self.source):
319
+ ch = self.source[self.pos]
320
+ if _is_cjk_or_alpha(ch) or ch.isdigit():
321
+ self.pos += 1
322
+ else:
323
+ break
324
+
325
+ text = self.source[start_pos:self.pos]
326
+ self.column += (self.pos - start_pos)
327
+
328
+ # 阶段二:精分
329
+ self._segment(text, start_col)
330
+
331
+ # ---- 阶段二:精分 ----
332
+
333
+ def _segment(self, text: str, start_col: int):
334
+ """
335
+ 正向最大匹配(Forward Maximum Matching)
336
+
337
+ 对 text 中的每个位置,找最长的匹配关键字。
338
+ 如果找不到关键字,发出单个字符作为 IDENTIFIER。
339
+
340
+ 关键改进:使用位置指针 i 遍历 text,
341
+ 每次从 i 开始找最长关键字。
342
+ 找到后 i 跳过该关键字长度。
343
+ 找不到时 i 前进 1(发出单个字符)。
344
+
345
+ 标识符规则:连续 CJK 串优先整体为标识符(如「阶乘」含关键词「乘」,
346
+ 但整体不是关键词 → 保持为标识符,避免误切分)。
347
+ """
348
+ i = 0
349
+ col = start_col
350
+
351
+ while i < len(text):
352
+ # 尝试从位置 i 找最长关键字
353
+ matched_kw = None
354
+ matched_len = 0
355
+
356
+ for kw in SORTED_KW:
357
+ if text.startswith(kw, i):
358
+ if len(kw) > matched_len:
359
+ matched_kw = kw
360
+ matched_len = len(kw)
361
+
362
+ if matched_kw is not None:
363
+ # 发出关键字 token
364
+ self._emit(KEYWORD_MAP[matched_kw], matched_kw, col)
365
+ i += matched_len
366
+ col += matched_len
367
+ else:
368
+ # 不是关键字 → 收集连续的非关键字字符作为标识符
369
+ # 中文规则:连续 CJK 串优先整体为标识符(如「阶乘」含关键词「乘」,
370
+ # 但整串「阶乘」非关键词 → 保持整体,避免误切分)
371
+ ident_start = i
372
+ ident_col = col
373
+
374
+ if _is_cjk(text[i]):
375
+ while i < len(text) and _is_cjk(text[i]):
376
+ i += 1
377
+ col += 1
378
+ else:
379
+ while i < len(text):
380
+ ch = text[i]
381
+ has_kw_ahead = any(text.startswith(kw, i) for kw in SORTED_KW)
382
+ if has_kw_ahead:
383
+ break
384
+ i += 1
385
+ col += 1
386
+
387
+ ident = text[ident_start:i]
388
+ if ident:
389
+ self._emit(TokenType.IDENTIFIER, ident, ident_col)
390
+
391
+ # ---- 数字读取 ----
392
+
393
+ def _peek_isdigit(self) -> bool:
394
+ """前瞻当前字符是否数字(多位数聚合判断)。"""
395
+ return self.pos + 1 < len(self.source) and self.source[self.pos + 1].isdigit()
396
+
397
+ def _read_number(self):
398
+ """读取数值(支持整数、小数、负数)"""
399
+ start_col = self.column
400
+ result = []
401
+
402
+ # 负号
403
+ if self.source[self.pos] == "-":
404
+ result.append("-")
405
+ self.pos += 1
406
+ self.column += 1
407
+
408
+ # 整数部分
409
+ while self.pos < len(self.source) and self.source[self.pos].isdigit():
410
+ result.append(self.source[self.pos])
411
+ self.pos += 1
412
+ self.column += 1
413
+
414
+ # 小数部分
415
+ if self.pos < len(self.source) and self.source[self.pos] == ".":
416
+ result.append(".")
417
+ self.pos += 1
418
+ self.column += 1
419
+ while self.pos < len(self.source) and self.source[self.pos].isdigit():
420
+ result.append(self.source[self.pos])
421
+ self.pos += 1
422
+ self.column += 1
423
+
424
+ num_str = "".join(result)
425
+ try:
426
+ float(num_str)
427
+ self._emit(TokenType.NUMBER, num_str)
428
+ except ValueError:
429
+ self.errors.append(f"L{self.line}:C{start_col} 非法数值: '{num_str}'")
430
+ self._emit(TokenType.NUMBER, num_str)
431
+
432
+ # ---- 字符串读取 ----
433
+
434
+ def _read_string(self):
435
+ """读取字符串"""
436
+ quote_char = self.source[self.pos]
437
+ end_quote = '"' if quote_char == '"' else "\u201d"
438
+
439
+ self.pos += 1
440
+ self.column += 1
441
+ result = []
442
+
443
+ while self.pos < len(self.source) and self.source[self.pos] != end_quote:
444
+ ch = self.source[self.pos]
445
+ if ch == "\n":
446
+ self.errors.append(f"L{self.line}:C{self.column} 字符串未闭合")
447
+ break
448
+ result.append(ch)
449
+ self.pos += 1
450
+ self.column += 1
451
+
452
+ if self.pos < len(self.source):
453
+ self.pos += 1 # 跳过结束引号
454
+ self.column += 1
455
+
456
+ self._emit(TokenType.STRING, "".join(result))
457
+
458
+ # ---- 注释跳过 ----
459
+
460
+ def _skip_comment(self):
461
+ """跳过注释至本行结束。"""
462
+ while self.pos < len(self.source) and self.source[self.pos] != "\n":
463
+ self.pos += 1
464
+
465
+ # ---- Token 输出 ----
466
+
467
+ def _emit(self, token_type: TokenType, value: str, col: int = None):
468
+ """输出一个 Token"""
469
+ c = col if col is not None else self.column
470
+ self.tokens.append(Token(token_type, value, self.line, c))
471
+
472
+
473
+ # =============================================================================
474
+ # 便捷函数
475
+ # =============================================================================
476
+
477
+ def tokenize(source: str) -> Tuple[List[Token], List[str]]:
478
+ """便捷函数"""
479
+ lexer = Lexer(source)
480
+ return lexer.tokenize()
481
+
482
+
483
+ # =============================================================================
484
+ # 测试
485
+ # =============================================================================
486
+
487
+ if __name__ == "__main__":
488
+ test_code = """若条件空间为伴侣,则止情感权重于0.15。
489
+ 道 新信任路径
490
+ 问曰:如何验证信任?
491
+ 答曰:信任值大于0.7。
492
+ 术曰:1。德 累积信任值;2。自然 恢复默认。"""
493
+
494
+ print("=" * 60)
495
+ print("词法分析器 v2.0 测试")
496
+ print("=" * 60)
497
+ print(f"源代码:\n{test_code}\n")
498
+
499
+ tokens, errors = tokenize(test_code)
500
+
501
+ print("Token 序列:")
502
+ for t in tokens:
503
+ if t.type != TokenType.EOF:
504
+ print(f" {t}")
505
+
506
+ if errors:
507
+ print(f"\n错误:")
508
+ for e in errors:
509
+ print(f" ❌ {e}")
510
+
511
+ valid = len([t for t in tokens if t.type != TokenType.EOF])
512
+ print(f"\n总计:{valid} 个 Token,{len(errors)} 个错误")