LangSC 2.0.14__tar.gz → 2.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Parser.lua +10 -513
  2. langsc-2.2.2/LangSC/__version__.py +7 -0
  3. {langsc-2.0.14 → langsc-2.2.2}/LangSC/_utils.py +11 -49
  4. {langsc-2.0.14 → langsc-2.2.2}/LangSC/bcc.py +5 -5
  5. langsc-2.2.2/LangSC/bcclib.dll +0 -0
  6. {langsc-2.0.14 → langsc-2.2.2}/LangSC/gpf.py +44 -21
  7. langsc-2.2.2/LangSC/gpflib.dll +0 -0
  8. {langsc-2.0.14 → langsc-2.2.2}/LangSC/jss.py +128 -55
  9. langsc-2.2.2/LangSC/jsslib.dll +0 -0
  10. langsc-2.2.2/LangSC/libbcclib.dylib +0 -0
  11. {langsc-2.0.14 → langsc-2.2.2}/LangSC/libgpflib.dylib +0 -0
  12. {langsc-2.0.14 → langsc-2.2.2}/LangSC/libgpflib.so +0 -0
  13. {langsc-2.0.14 → langsc-2.2.2}/LangSC/libjsslib.dylib +0 -0
  14. langsc-2.2.2/LangSC/libjsslib.so +0 -0
  15. {langsc-2.0.14 → langsc-2.2.2/LangSC.egg-info}/PKG-INFO +36 -1
  16. {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/SOURCES.txt +3 -3
  17. {langsc-2.0.14/LangSC.egg-info → langsc-2.2.2}/PKG-INFO +36 -1
  18. {langsc-2.0.14 → langsc-2.2.2}/README.md +35 -0
  19. {langsc-2.0.14 → langsc-2.2.2}/setup.cfg +1 -2
  20. langsc-2.2.2/tests/test_gpf_threadsafe.py +52 -0
  21. langsc-2.2.2/tests/test_win_smoke.py +138 -0
  22. langsc-2.0.14/LangSC/Conf.yaml +0 -11
  23. langsc-2.0.14/LangSC/__version__.py +0 -3
  24. langsc-2.0.14/LangSC/bcclib.dll +0 -0
  25. langsc-2.0.14/LangSC/gpflib.dll +0 -0
  26. langsc-2.0.14/LangSC/jsslib.dll +0 -0
  27. langsc-2.0.14/LangSC/libbcclib.dylib +0 -0
  28. langsc-2.0.14/LangSC/libjsslib.so +0 -0
  29. langsc-2.0.14/LangSC/test.py +0 -7
  30. {langsc-2.0.14 → langsc-2.2.2}/LICENSE +0 -0
  31. {langsc-2.0.14 → langsc-2.2.2}/LangSC/BCCconfig.txt +0 -0
  32. {langsc-2.0.14 → langsc-2.2.2}/LangSC/GPFconfig.txt +0 -0
  33. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/Pathplan.dll +0 -0
  34. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/cdt.dll +0 -0
  35. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/cgraph.dll +0 -0
  36. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/config6 +0 -0
  37. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/dot.exe +0 -0
  38. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/fontconfig.dll +0 -0
  39. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/fontconfig_fix.dll +0 -0
  40. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/freetype6.dll +0 -0
  41. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvc.dll +0 -0
  42. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_core.dll +0 -0
  43. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_dot_layout.dll +0 -0
  44. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_gd.dll +0 -0
  45. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_pango.dll +0 -0
  46. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/iconv.dll +0 -0
  47. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/intl.dll +0 -0
  48. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/jpeg62.dll +0 -0
  49. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libcairo-2.dll +0 -0
  50. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libexpat-1.dll +0 -0
  51. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libexpat.dll +0 -0
  52. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libfontconfig-1.dll +0 -0
  53. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libfreetype-6.dll +0 -0
  54. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgdk_pixbuf-2.0-0.dll +0 -0
  55. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libglade-2.0-0.dll +0 -0
  56. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libglib-2.0-0.dll +0 -0
  57. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgmodule-2.0-0.dll +0 -0
  58. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgobject-2.0-0.dll +0 -0
  59. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgthread-2.0-0.dll +0 -0
  60. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgtkglext-win32-1.0-0.dll +0 -0
  61. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpango-1.0-0.dll +0 -0
  62. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangocairo-1.0-0.dll +0 -0
  63. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangoft2-1.0-0.dll +0 -0
  64. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangowin32-1.0-0.dll +0 -0
  65. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpng14-14.dll +0 -0
  66. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/librsvg-2-2.dll +0 -0
  67. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libxml2.dll +0 -0
  68. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/ltdl.dll +0 -0
  69. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/zlib1.dll +0 -0
  70. {langsc-2.0.14 → langsc-2.2.2}/LangSC/Segment.dat +0 -0
  71. {langsc-2.0.14 → langsc-2.2.2}/LangSC/__init__.py +0 -0
  72. {langsc-2.0.14 → langsc-2.2.2}/LangSC/base.lex +0 -0
  73. {langsc-2.0.14 → langsc-2.2.2}/LangSC/idxPOS.dat +0 -0
  74. {langsc-2.0.14 → langsc-2.2.2}/LangSC/libbcclib.so +0 -0
  75. {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/dependency_links.txt +0 -0
  76. {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/not-zip-safe +0 -0
  77. {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/requires.txt +0 -0
  78. {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/top_level.txt +0 -0
  79. {langsc-2.0.14 → langsc-2.2.2}/MANIFEST.in +0 -0
  80. {langsc-2.0.14 → langsc-2.2.2}/pyproject.toml +0 -0
  81. {langsc-2.0.14 → langsc-2.2.2}/setup.py +0 -0
@@ -394,17 +394,15 @@ function GetASInfo(Query)
394
394
  local QueryEx=""
395
395
  local lFix=""
396
396
  local rFix=""
397
- local lFixInner=""
398
- local rFixInner=""
397
+ local lFixinn=""
398
+ local rFixinn=""
399
399
  local lBracket=""
400
400
  local rBracket=""
401
401
  local lBracketInner=""
402
402
  local rBracketInner=""
403
403
  QueryEx=string.gsub(Query,"[%.%(%)~]","")
404
404
  Re="^([%~%.%(%)]*)[^%~%.%(%)]+([%~%.%(%)]*)$"
405
- -- 去掉中段 ( ) 焦点标记再抽 lFix/rFix: 括号位置由 GetBracketInfo 单独处理,
406
- -- 否则像 (v)了个~ 这种中段含 ) 的查询会让单段正则整体失配, 丢掉尾部 ~/. 修饰
407
- B,E,lFix,rFix=string.find((string.gsub(Query,"[%(%)]","")),Re)
405
+ B,E,lFix,rFix=string.find(Query,Re)
408
406
  if B == nil then
409
407
  Re="^[a-zA-Z_-]+%[([%~%.%(%)]*)%*?([%~%.%(%)]*)[^%s\1-\127]*%]$"
410
408
  B,E,lFixInner,rFixInner=string.find(Query,Re)
@@ -459,11 +457,7 @@ function Replace(Inp)
459
457
  else
460
458
  l=string.sub(Query,1,B-1)
461
459
  r=string.sub(Query,E+1)
462
- local lTok=string.match(l,"(%S+)$")
463
- local rTok=string.match(r,"^(%S+)")
464
- local lHasSlash=lTok and string.find(lTok,"/")
465
- local rHasSlash=rTok and string.find(rTok,"/")
466
- if lHasSlash or rHasSlash or (string.find(l,"[a-zA-Z0-9]$") ~=nil and string.find(r,"^[a-zA-Z0-9]") ~=nil) then
460
+ if string.find(l,"[a-zA-Z0-9]$") ~=nil and string.find(r,"^[a-zA-Z0-9]") ~=nil then
467
461
  Query=l.." "..r
468
462
  else
469
463
  Query=l.."#"..r
@@ -679,168 +673,6 @@ function GeCLtInfo(Query)
679
673
 
680
674
  end
681
675
 
682
- -- word/POS 检索式扩展 ---------------------------------------------------------
683
- -- token 分类: slash (word/POS) | POS (纯字母) | HZs (汉字串, 可单/多词)
684
- -- 子表达式拼接后送 GetScriptAS, 由它走对应分支输出单个 GetAS:
685
- -- HZs → GetAS("<HZ", HZs, ...)
686
- -- POS → GetAS("|POS", "", ...)
687
- -- POS+HZs → GetAS("|POS_HZ", HZs, ...) [HZ = 首字]
688
- -- HZs+POS → GetAS("HZ_POS|", HZs, ...) [HZ = 末字]
689
- -- 注意: word 与 string 在引擎层面同质 (都是 HZs), Word+String 视为一个整串.
690
- --
691
- -- 形态:
692
- -- word/POS → word , POS ; SameBoundary
693
- -- word/POS1 POS2 → word+POS2 , POS1 ; SameLeft
694
- -- POS1 word/POS2 → POS1+word , POS2 ; SameRight
695
- -- word1/POS1 word2/POS2 → word1+POS2 , POS1+word2 ; SameBoundary
696
- -- POS1 word/POS2 POS3 → (POS1+word) ShareQuery (word+POS3) = R
697
- -- POS2 InChunk R
698
- -- word/POS1 string → word+string , POS1+string ; SameLeft
699
- -- string word/POS1 → string+word , string+POS1 ; SameRight
700
- -- word1/POS1 string word2/POS2 → POS1+string+word2 , word1+string+POS2 ; SameBoundary
701
- -- word1/POS1 word2/POS2 string → word1+POS2 , POS1+word2+string ; SameLeft
702
- -- string word1/POS1 word2/POS2 → (string+POS1) SameLeft (string+word1+POS2) = R
703
- -- R SameBoundary (string+word1+word2)
704
- -- 增量扩展: 上述 10 种形态作为查询的最左前缀(贪心最长匹配 3→2→1 token), 后面可
705
- -- 跟 0..N 个非 slash 原子, 每个用 GetScriptAS 编译 + Link / InChunk 连接.
706
- -- prefix atom → JoinAS(prefix, atom, "Link")
707
- -- prefix *atom → JoinAS(prefix, atom, "InChunk") -- atom 前缀 * 表 InChunk
708
- -- 多个 trailing 依次链接: ((prefix R1 t1) R2 t2) R3 t3 ...
709
- -- GetAS-JoinAS 交替, 不出现连续 JoinAS, 也不出现 ≥3 连续 GetAS.
710
- function EmitJoinAS(leftHandle,rightHandle,relation)
711
- local hOut=HandleNo
712
- HandleNo=HandleNo+1
713
- local s=string.format('Handle%d=JoinAS(%s,%s,"%s")\n',hOut,leftHandle,rightHandle,relation)
714
- return s,string.format('Handle%d',hOut)
715
- end
716
-
717
- local function classifyToken(tok)
718
- local _,_,w,p=string.find(tok,"^([^/]+)/([^/]+)$")
719
- if w then return "slash",w,p end
720
- if string.find(tok,"^[a-zA-Z_-]+$") then return "POS",tok,nil end
721
- if string.find(tok,"^[^%s\1-\127]+$") then return "HZs",tok,nil end
722
- return "unknown",tok,nil
723
- end
724
-
725
- local function matchPattern1(tokens)
726
- local k,w,p=classifyToken(tokens[1])
727
- if k=="slash" then
728
- local s1,h1=GetScriptAS(w)
729
- local s2,h2=GetScriptAS(p)
730
- local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
731
- return s1..s2..sj,hj
732
- end
733
- return nil,nil
734
- end
735
-
736
- local function matchPattern2(tokens)
737
- local k1,a1,b1=classifyToken(tokens[1])
738
- local k2,a2,b2=classifyToken(tokens[2])
739
- if k1=="slash" and k2=="POS" then
740
- local s1,h1=GetScriptAS(a1..a2)
741
- local s2,h2=GetScriptAS(b1)
742
- local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
743
- return s1..s2..sj,hj
744
- elseif k1=="POS" and k2=="slash" then
745
- local s1,h1=GetScriptAS(a1..a2)
746
- local s2,h2=GetScriptAS(b2)
747
- local sj,hj=EmitJoinAS(h1,h2,"SameRight")
748
- return s1..s2..sj,hj
749
- elseif k1=="slash" and k2=="slash" then
750
- local s1,h1=GetScriptAS(a1..b2)
751
- local s2,h2=GetScriptAS(b1..a2)
752
- local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
753
- return s1..s2..sj,hj
754
- elseif k1=="slash" and k2=="HZs" then
755
- local s1,h1=GetScriptAS(a1..a2)
756
- local s2,h2=GetScriptAS(b1..a2)
757
- local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
758
- return s1..s2..sj,hj
759
- elseif k1=="HZs" and k2=="slash" then
760
- local s1,h1=GetScriptAS(a1..a2)
761
- local s2,h2=GetScriptAS(a1..b2)
762
- local sj,hj=EmitJoinAS(h1,h2,"SameRight")
763
- return s1..s2..sj,hj
764
- end
765
- return nil,nil
766
- end
767
-
768
- local function matchPattern3(tokens)
769
- local k1,a1,b1=classifyToken(tokens[1])
770
- local k2,a2,b2=classifyToken(tokens[2])
771
- local k3,a3,b3=classifyToken(tokens[3])
772
- if k1=="POS" and k2=="slash" and k3=="POS" then
773
- local s1,h1=GetScriptAS(a1..a2)
774
- local s2,h2=GetScriptAS(a2..a3)
775
- local sShare,hShare=EmitJoinAS(h1,h2,"ShareQuery")
776
- local s3,h3=GetScriptAS(b2)
777
- local sIn,hIn=EmitJoinAS(h3,hShare,"InChunk")
778
- return s1..s2..sShare..s3..sIn,hIn
779
- elseif k1=="slash" and k2=="HZs" and k3=="slash" then
780
- local s1,h1=GetScriptAS(b1..a2..a3)
781
- local s2,h2=GetScriptAS(a1..a2..b3)
782
- local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
783
- return s1..s2..sj,hj
784
- elseif k1=="slash" and k2=="slash" and k3=="HZs" then
785
- local s1,h1=GetScriptAS(a1..b2)
786
- local s2,h2=GetScriptAS(b1..a2..a3)
787
- local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
788
- return s1..s2..sj,hj
789
- elseif k1=="HZs" and k2=="slash" and k3=="slash" then
790
- local s1,h1=GetScriptAS(a1..b2)
791
- local s2,h2=GetScriptAS(a1..a2..b3)
792
- local sR,hR=EmitJoinAS(h1,h2,"SameLeft")
793
- local s3,h3=GetScriptAS(a1..a2..a3)
794
- local sFinal,hFinal=EmitJoinAS(hR,h3,"SameBoundary")
795
- return s1..s2..sR..s3..sFinal,hFinal
796
- end
797
- return nil,nil
798
- end
799
-
800
- function BuildSlashScript(Query)
801
- -- Replace 会把 *X 前后的空格吃掉, 这里恢复一下 (slash 查询里不会有 q[*])
802
- Query=string.gsub(Query,"%*"," *")
803
- local tokens={}
804
- for tok in string.gmatch(Query,"%S+") do
805
- table.insert(tokens,tok)
806
- end
807
- local n=#tokens
808
- local prefixScript,prefixHandle,prefixLen=nil,nil,0
809
- if n>=3 then
810
- local s,h=matchPattern3(tokens)
811
- if s then prefixScript,prefixHandle,prefixLen=s,h,3 end
812
- end
813
- if not prefixScript and n>=2 then
814
- local s,h=matchPattern2(tokens)
815
- if s then prefixScript,prefixHandle,prefixLen=s,h,2 end
816
- end
817
- if not prefixScript and n>=1 then
818
- local s,h=matchPattern1(tokens)
819
- if s then prefixScript,prefixHandle,prefixLen=s,h,1 end
820
- end
821
- if not prefixScript then
822
- error("Slash query: no valid slash prefix in '"..Query.."'")
823
- end
824
- local script=prefixScript
825
- local running=prefixHandle
826
- for i=prefixLen+1,n do
827
- local tok=tokens[i]
828
- local relation="Link"
829
- if string.sub(tok,1,1)=="*" then
830
- relation="InChunk"
831
- tok=string.sub(tok,2)
832
- end
833
- if tok=="" or string.find(tok,"/") then
834
- error("Slash query: invalid trailing atom '"..tokens[i].."'")
835
- end
836
- local sa,ha=GetScriptAS(tok)
837
- local sj,hj=EmitJoinAS(running,ha,relation)
838
- script=script..sa..sj
839
- running=hj
840
- end
841
- return script
842
- end
843
-
844
676
  function Parser(Query,launguage,Option)
845
677
  Re="^Lua:(.+)"
846
678
  B,E,ReadlQuery=string.find(Query,Re)
@@ -908,17 +740,12 @@ function Parser(Query,launguage,Option)
908
740
  Query=Eng2Chin(Query)
909
741
  end
910
742
  end
911
- if string.find(Query,"/") then
912
- Script=Script..BuildSlashScript(Query)
913
- HANDLENO=HandleNo-1
914
- else
915
- local rTree =CreateTree(Query,1)
916
- CreateScriptR(rTree)
917
- HANDLENO=HandleNo-1
918
- if LEFT ~= nil and LEFT ~= "" then
919
- local lTree=CreateTree(LEFT.."@",0)
920
- CreateScriptL(lTree)
921
- end
743
+ local rTree =CreateTree(Query,1)
744
+ CreateScriptR(rTree)
745
+ HANDLENO=HandleNo-1
746
+ if LEFT ~= nil and LEFT ~= "" then
747
+ local lTree=CreateTree(LEFT.."@",0)
748
+ CreateScriptL(lTree)
922
749
  end
923
750
  Operation=GetOperation(Operation)
924
751
  Script=Script..Operation
@@ -1052,333 +879,3 @@ function Test()
1052
879
  end
1053
880
 
1054
881
  --Test()
1055
-
1056
- -- ═══════════════════════════════════════════════════════════════════
1057
- -- 词集功能(spec: Doc/superpowers/specs/2026-05-07-bcc-wordset-design.md)
1058
- -- ═══════════════════════════════════════════════════════════════════
1059
-
1060
- g_WordSets = {}
1061
- MAX_TOTAL_MS = 30000
1062
-
1063
- -- 解析受限 YAML
1064
- -- 支持: 顶格 key:、缩进(空格)词条、# 注释、空行、CRLF、UTF-8 BOM
1065
- -- 返回: (table sets, nil) 成功 / (nil, "Conf.yaml line N: ...") 失败
1066
- function ParseYaml(path)
1067
- local fp = io.open(path, "rb")
1068
- if not fp then return {}, nil end -- 不存在视为空字典(不算错误)
1069
- local sets = {}
1070
- local cur_key = nil
1071
- local lineNo = 0
1072
- local first = true
1073
- for line in fp:lines() do
1074
- lineNo = lineNo + 1
1075
- if first then
1076
- -- 去 UTF-8 BOM
1077
- if line:sub(1,3) == "\xEF\xBB\xBF" then line = line:sub(4) end
1078
- first = false
1079
- end
1080
- line = line:gsub("\r$", "") -- 去 CRLF
1081
- local stripped = line:gsub("^%s+",""):gsub("%s+$","")
1082
- if stripped == "" or stripped:sub(1,1) == "#" then
1083
- -- 空行 / 注释
1084
- elseif line:sub(1,1) ~= " " and line:sub(1,1) ~= "\t" then
1085
- -- 顶格行 = 词集名
1086
- local k = line:match("^([A-Za-z_][A-Za-z0-9_]*)%s*:%s*$")
1087
- if not k then
1088
- fp:close()
1089
- return nil, "Conf.yaml line "..lineNo..": bad key syntax"
1090
- end
1091
- cur_key = k
1092
- sets[k] = {}
1093
- elseif line:sub(1,1) == "\t" then
1094
- fp:close()
1095
- return nil, "Conf.yaml line "..lineNo..": tab indent not allowed"
1096
- else
1097
- -- 缩进行 = 词条
1098
- if not cur_key then
1099
- fp:close()
1100
- return nil, "Conf.yaml line "..lineNo..": value before any key"
1101
- end
1102
- local tok = line:match("^%s+(%S+)%s*$")
1103
- if not tok then
1104
- fp:close()
1105
- return nil, "Conf.yaml line "..lineNo..": bad value syntax"
1106
- end
1107
- table.insert(sets[cur_key], tok)
1108
- end
1109
- end
1110
- fp:close()
1111
- return sets, nil
1112
- end
1113
-
1114
- -- LoadConf:由 C 层 EnsureParserVM 在 mtime 变化时调用
1115
- -- 解析 Conf.yaml,把词条从 UTF-8(文件编码)转为 GBK(引擎内部编码)
1116
- function LoadConf(path)
1117
- local sets, err = ParseYaml(path)
1118
- if err then
1119
- print("[LoadConf] "..err)
1120
- g_WordSets = {}
1121
- return
1122
- end
1123
- -- 文件以 UTF-8 保存,词条需转为 GBK 才能拼到 GBK 查询里
1124
- for k, words in pairs(sets) do
1125
- for i = 1, #words do
1126
- words[i] = UTF82GB(words[i])
1127
- end
1128
- end
1129
- g_WordSets = sets
1130
- end
1131
-
1132
- -- 扫描查询里的 <NAME> 占位符
1133
- -- 返回 (key, words, nil) / (nil, nil, nil 无词集) / (nil, nil, errMsg)
1134
- function GetWordSet(query)
1135
- local hits = {}
1136
- for name in query:gmatch("<([A-Za-z_][A-Za-z0-9_]*)>") do
1137
- table.insert(hits, name)
1138
- end
1139
- if #hits == 0 then return nil, nil, nil end
1140
- if #hits > 1 then
1141
- return nil, nil,
1142
- "multiple word sets in one query: "..table.concat(hits, ",")
1143
- end
1144
- local key = hits[1]
1145
- if not g_WordSets[key] or #g_WordSets[key] == 0 then
1146
- return nil, nil, "unknown or empty word set: <"..key..">"
1147
- end
1148
- return key, g_WordSets[key], nil
1149
- end
1150
-
1151
- -- 把 query 里的 <key> 替换成 word(只替换一次)
1152
- -- 关键:gsub 替换串里的 % 是元字符,必须先转义
1153
- function ReplaceWS(query, key, word)
1154
- local safe = word:gsub("%%", "%%%%")
1155
- return (query:gsub("<"..key..">", safe, 1))
1156
- end
1157
-
1158
- -- 在 json 字符串里找 "fieldName": [...],返回数组体(不含外层 [])和结束位置
1159
- -- 处理嵌套 {}[] 与字符串内的转义引号
1160
- function ExtractArrayBody(json, field)
1161
- local _, e = json:find('"'..field..'"%s*:%s*%[')
1162
- if not e then return nil, nil end
1163
- local startPos = e + 1
1164
- local depth, i, len, inStr = 1, e + 1, #json, false
1165
- while i <= len and depth > 0 do
1166
- local c = json:sub(i, i)
1167
- if inStr then
1168
- if c == '"' and json:sub(i-1, i-1) ~= '\\' then inStr = false end
1169
- else
1170
- if c == '"' then inStr = true
1171
- elseif c == '[' or c == '{' then depth = depth + 1
1172
- elseif c == ']' or c == '}' then
1173
- depth = depth - 1
1174
- if depth == 0 and c == ']' then
1175
- return json:sub(startPos, i - 1), i
1176
- end
1177
- end
1178
- end
1179
- i = i + 1
1180
- end
1181
- return nil, nil
1182
- end
1183
-
1184
- -- 内部辅助:简单 JSON 字符串值转义(用于 WordSetErrors 字段)
1185
- local function _jsonEscape(s)
1186
- s = tostring(s or "")
1187
- s = s:gsub('\\', '\\\\'):gsub('"', '\\"')
1188
- s = s:gsub('\n', '\\n'):gsub('\r', '\\r'):gsub('\t', '\\t')
1189
- return s
1190
- end
1191
-
1192
- -- 错误体格式化:{"error":"msg","WordSetErrors":[...]}
1193
- function JsonError(msg, errs)
1194
- local s = '{"error":"'.._jsonEscape(msg)..'"'
1195
- if errs and #errs > 0 then
1196
- s = s..',"WordSetErrors":['
1197
- for i, e in ipairs(errs) do
1198
- if i > 1 then s = s..',' end
1199
- s = s..'{"Word":"'.._jsonEscape(e.Word)..
1200
- '","Error":"'.._jsonEscape(e.Error)..'"}'
1201
- end
1202
- s = s..']'
1203
- end
1204
- return s..'}'
1205
- end
1206
-
1207
- -- 内部:WordSetErrors 字段尾巴
1208
- local function _errorsTail(errors)
1209
- if not errors or #errors == 0 then return "" end
1210
- local s = ',"WordSetErrors":['
1211
- for i, e in ipairs(errors) do
1212
- if i > 1 then s = s..',' end
1213
- s = s..'{"Word":"'.._jsonEscape(e.Word)..
1214
- '","Error":"'.._jsonEscape(e.Error)..'"}'
1215
- end
1216
- return s..']'
1217
- end
1218
-
1219
- -- 创建空累加器
1220
- function NewAcc(typ)
1221
- if typ == "Context" then
1222
- return { type="Context", items={}, total=0, count=0, approx=false }
1223
- elseif typ == "Freq" then
1224
- return { type="Freq", map={}, total=0 }
1225
- elseif typ == "Count" then
1226
- return { type="Count", buckets={}, total=0 }
1227
- end
1228
- return nil
1229
- end
1230
-
1231
- -- ── Context ──
1232
- local function _mergeContext(acc, out, word)
1233
- local body, _ = ExtractArrayBody(out, "Context")
1234
- if not body then return false, "no Context array body" end
1235
- table.insert(acc.items, body)
1236
-
1237
- local total = tonumber(out:match('"Total"%s*:%s*(%-?%d+)')) or 0
1238
- local count = tonumber(out:match('"Count"%s*:%s*(%-?%d+)')) or 0
1239
- acc.total = acc.total + total
1240
- acc.count = acc.count + count
1241
-
1242
- if out:match('"Approximate"%s*:%s*"1"') then acc.approx = true end
1243
- return true, nil
1244
- end
1245
-
1246
- local function _emitContext(acc, errors)
1247
- local s = '{"Type":"Context",'
1248
- s = s..'"Count":'..acc.count..','
1249
- s = s..'"Total":'..acc.total..','
1250
- s = s..'"Approximate":"'..(acc.approx and "1" or "0")..'",'
1251
- s = s..'"Context":['..table.concat(acc.items, ',')..']'
1252
- s = s.._errorsTail(errors)
1253
- return s..'}'
1254
- end
1255
-
1256
- -- ── Freq ──
1257
- local function _mergeFreq(acc, out, word)
1258
- local body, _ = ExtractArrayBody(out, "Freq")
1259
- if not body then return false, "no Freq array body" end
1260
-
1261
- -- 解析 {"Word":"X","Freq":N} 格式条目(Word/Freq 来自索引,值可信)
1262
- for w, f in body:gmatch('{"Word"%s*:%s*"([^"]*)"%s*,%s*"Freq"%s*:%s*(%-?%d+)}') do
1263
- local n = tonumber(f) or 0
1264
- acc.map[w] = (acc.map[w] or 0) + n
1265
- acc.total = acc.total + n
1266
- end
1267
- return true, nil
1268
- end
1269
-
1270
- local function _emitFreq(acc, errors)
1271
- -- 按 Freq 降序排序
1272
- local arr = {}
1273
- for w, f in pairs(acc.map) do
1274
- table.insert(arr, { Word = w, Freq = f })
1275
- end
1276
- table.sort(arr, function(a, b) return a.Freq > b.Freq end)
1277
-
1278
- local pieces = {}
1279
- for _, e in ipairs(arr) do
1280
- table.insert(pieces, '{"Word":"'.._jsonEscape(e.Word)..
1281
- '","Freq":'..e.Freq..'}')
1282
- end
1283
-
1284
- local s = '{"Type":"Freq","Count":'..#arr..
1285
- ',"Total":'..acc.total..
1286
- ',"Freq":['..table.concat(pieces, ',')..']'
1287
- s = s.._errorsTail(errors)
1288
- return s..'}'
1289
- end
1290
-
1291
- -- ── Count ──
1292
- local function _mergeCount(acc, out, word)
1293
- local body, _ = ExtractArrayBody(out, "Freq")
1294
- if not body then return false, "no Count Freq array body" end
1295
- local i = 1
1296
- for n in body:gmatch("%-?%d+") do
1297
- local v = tonumber(n) or 0
1298
- acc.buckets[i] = (acc.buckets[i] or 0) + v
1299
- acc.total = acc.total + v
1300
- i = i + 1
1301
- end
1302
- return true, nil
1303
- end
1304
-
1305
- local function _emitCount(acc, errors)
1306
- local pieces = {}
1307
- for i = 1, #acc.buckets do
1308
- table.insert(pieces, tostring(acc.buckets[i]))
1309
- end
1310
- local s = '{"Type":"Count","Count":1,"Total":'..acc.total..
1311
- ',"Freq":['..table.concat(pieces, ',')..']'
1312
- s = s.._errorsTail(errors)
1313
- return s..'}'
1314
- end
1315
-
1316
- -- 分派器:把 out 合并到 acc
1317
- function MergeOne(acc, typ, out, word)
1318
- if typ == "Context" then return _mergeContext(acc, out, word) end
1319
- if typ == "Freq" then return _mergeFreq(acc, out, word) end
1320
- if typ == "Count" then return _mergeCount(acc, out, word) end
1321
- return false, "unknown type: "..tostring(typ)
1322
- end
1323
-
1324
- -- 分派器:序列化为最终 JSON
1325
- function EmitMerged(acc, errors)
1326
- if acc.type == "Context" then return _emitContext(acc, errors) end
1327
- if acc.type == "Freq" then return _emitFreq(acc, errors) end
1328
- if acc.type == "Count" then return _emitCount(acc, errors) end
1329
- return JsonError("unknown acc type")
1330
- end
1331
-
1332
- -- ParserNew 主入口:替代 Parser 处理含词集的查询
1333
- -- 由 C 层 Query2Result 调用;无词集时回退到原 Parser + BCCLua
1334
- function ParserNew(Query, language, Option)
1335
- local isLuaGen = (Query:sub(1,4) == "Lua:")
1336
-
1337
- local key, words, err = GetWordSet(Query)
1338
- if err then
1339
- return JsonError(err)
1340
- end
1341
-
1342
- -- 路径 ①:无词集 → 原 Parser + BCCLua
1343
- if key == nil then
1344
- local script = Parser(Query, language, Option)
1345
- if isLuaGen then return script end
1346
- return BCCLua(script)
1347
- end
1348
-
1349
- -- 路径 ②:Lua: + 词集 → 用第一个词展开成示例脚本
1350
- if isLuaGen then
1351
- local q1 = ReplaceWS(Query, key, words[1])
1352
- return Parser(q1, language, Option)
1353
- end
1354
-
1355
- -- 路径 ③:词集批处理
1356
- local errors = {}
1357
- local acc = nil
1358
- for i, w in ipairs(words) do
1359
- if (not g_skip_budget) and ElapsedMs() > MAX_TOTAL_MS then
1360
- table.insert(errors,
1361
- { Word = w, Error = "aborted: total time budget exceeded" })
1362
- break
1363
- end
1364
- local qx = ReplaceWS(Query, key, w)
1365
- local script = Parser(qx, language, Option)
1366
- local out = BCCLua(script)
1367
- local typ = out and out:match('"Type"%s*:%s*"([^"]+)"')
1368
- if not typ then
1369
- local msg = (out or ""):sub(1, 200)
1370
- table.insert(errors, { Word = w, Error = msg })
1371
- else
1372
- if not acc then acc = NewAcc(typ) end
1373
- local ok, msg = MergeOne(acc, typ, out, w)
1374
- if not ok then
1375
- table.insert(errors, { Word = w, Error = msg or "merge failed" })
1376
- end
1377
- end
1378
- end
1379
-
1380
- if not acc then
1381
- return JsonError("all sub-queries failed", errors)
1382
- end
1383
- return EmitMerged(acc, errors)
1384
- end
@@ -0,0 +1,7 @@
1
+ # Single source of truth for the package version.
2
+ # Keep __version__ a plain string literal: setup.cfg reads it statically via
3
+ # `version = attr: LangSC.__version__.__version__`, so it must stay AST-parseable
4
+ # (no computed expression) to avoid importing the package at build time.
5
+ __version__ = "2.2.2"
6
+
7
+ VERSION = tuple(int(x) for x in __version__.split("."))
@@ -2,7 +2,6 @@
2
2
 
3
3
  import os
4
4
  import re
5
- import json
6
5
  import chardet
7
6
 
8
7
 
@@ -141,54 +140,15 @@ def get_bcc_files(path, idxed_file2time, bcc_files):
141
140
  bcc_files.append(full_path)
142
141
 
143
142
 
144
- def is_json_data_file(path):
145
- """Content-based check: is this a JSON / JSONL data file, regardless of extension?
146
-
147
- Naming is intentionally NOT required — a user can drop in *.json, *.jsonl, *.txt
148
- or extension-less files and they all get picked up, as long as the content is JSON.
149
- A file qualifies when its first non-whitespace character is '{' or '[' (a JSON
150
- object or array, the only shapes JSS indexes). cfg_*.txt config files and any
151
- binary / non-JSON text are skipped.
152
- """
153
- name = os.path.basename(path)
154
- if name.startswith('cfg_'):
155
- return False
156
- try:
157
- encoding = detect_file_encoding(path)
158
- with open(path, 'r', encoding=encoding) as f:
159
- head = f.read(8192)
160
- except (OSError, UnicodeDecodeError, LookupError):
161
- return False # unreadable / binary -> not a data file
162
- head = head.lstrip()
163
- return bool(head) and head[0] in '{['
164
-
165
-
166
- def detect_record_format(path):
167
- """Decide record_format by content: 'json' (whole file is one JSON value) vs
168
- 'jsonl' (one JSON value per line). Falls back to 'json' on empty/unreadable."""
169
- try:
170
- encoding = detect_file_encoding(path)
171
- with open(path, 'r', encoding=encoding) as f:
172
- text = f.read()
173
- except (OSError, UnicodeDecodeError, LookupError):
174
- return 'json'
175
- stripped = text.strip()
176
- if not stripped:
177
- return 'json'
178
- try:
179
- json.loads(stripped) # the whole file parses as a single JSON value
180
- return 'json'
181
- except ValueError:
182
- return 'jsonl' # otherwise treat as one JSON object per line
183
-
184
-
185
143
  def get_jss_file_info(path, to_idx_file2time):
186
- """Walk path and collect modification times for JSON/JSONL data files (by content)."""
144
+ """Walk path and collect modification times for JSON/JSONL data files only."""
187
145
  for root, dirs, files in os.walk(path):
188
146
  for f in files:
189
- full_path = os.path.join(root, f)
190
- if not is_json_data_file(full_path):
147
+ if not (f.endswith('.json') or f.endswith('.jsonl')):
148
+ continue
149
+ if f.startswith('cfg_'):
191
150
  continue
151
+ full_path = os.path.join(root, f)
192
152
  time_id = os.path.getmtime(full_path)
193
153
  if not to_idx_file2time.get(f):
194
154
  to_idx_file2time[f] = {}
@@ -197,12 +157,14 @@ def get_jss_file_info(path, to_idx_file2time):
197
157
 
198
158
 
199
159
  def get_jss_files(path, idxed_file2time, jss_files):
200
- """Scan path for JSON/JSONL data files (by content) that need indexing."""
160
+ """Scan path for JSON/JSONL data files that need indexing."""
201
161
  for root, dirs, files in os.walk(path):
202
162
  for f in files:
203
- full_path = os.path.join(root, f)
204
- if not is_json_data_file(full_path):
163
+ if not (f.endswith('.json') or f.endswith('.jsonl')):
164
+ continue
165
+ if f.startswith('cfg_'):
205
166
  continue
167
+ full_path = os.path.join(root, f)
206
168
  if idxed_file2time.get(f):
207
169
  time_id = os.path.getmtime(full_path)
208
170
  if idxed_file2time[f].get(str(time_id)):
@@ -299,7 +261,7 @@ def check_words(all_words):
299
261
  def process_file(file_path, file_tmp, cmd, gpf=None):
300
262
  """Process a raw file into corpus format."""
301
263
  encoding = detect_file_encoding(file_path)
302
- with open(file_path, "r", encoding=encoding, errors="ignore") as f_in:
264
+ with open(file_path, "r", encoding=encoding) as f_in:
303
265
  with open(file_tmp, "w") as f_out:
304
266
  print("Doc {}".format(file_path), file=f_out)
305
267
  for line in f_in:
@@ -24,12 +24,12 @@ lock = threading.Lock()
24
24
  class BCC:
25
25
  def __init__(self, dataPath="./data"):
26
26
  dataPath = dataPath.replace("\\", "/")
27
- if dataPath[-1] == "/":
27
+ if len(dataPath) > 1 and dataPath.endswith("/"):
28
28
  dataPath = dataPath[:-1]
29
- if dataPath.find("./") != 0:
30
- dataPath = "./" + dataPath
31
- if dataPath.find(".") != 0:
32
- dataPath = "." + dataPath
29
+ # 相对路径规范为 "./" 前缀;绝对路径(/... 或 X:/...)原样保留
30
+ if not (os.path.isabs(dataPath) or (len(dataPath) >= 2 and dataPath[1] == ":")):
31
+ if not dataPath.startswith("./"):
32
+ dataPath = "./" + dataPath
33
33
 
34
34
  dll_name_bcc = ''
35
35
  self.g_IdxLog = "IdxLog_BCC.txt"
Binary file