LangSC 2.0.14__tar.gz → 2.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Parser.lua +10 -513
- langsc-2.2.2/LangSC/__version__.py +7 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/_utils.py +11 -49
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/bcc.py +5 -5
- langsc-2.2.2/LangSC/bcclib.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/gpf.py +44 -21
- langsc-2.2.2/LangSC/gpflib.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/jss.py +128 -55
- langsc-2.2.2/LangSC/jsslib.dll +0 -0
- langsc-2.2.2/LangSC/libbcclib.dylib +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/libgpflib.dylib +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/libgpflib.so +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/libjsslib.dylib +0 -0
- langsc-2.2.2/LangSC/libjsslib.so +0 -0
- {langsc-2.0.14 → langsc-2.2.2/LangSC.egg-info}/PKG-INFO +36 -1
- {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/SOURCES.txt +3 -3
- {langsc-2.0.14/LangSC.egg-info → langsc-2.2.2}/PKG-INFO +36 -1
- {langsc-2.0.14 → langsc-2.2.2}/README.md +35 -0
- {langsc-2.0.14 → langsc-2.2.2}/setup.cfg +1 -2
- langsc-2.2.2/tests/test_gpf_threadsafe.py +52 -0
- langsc-2.2.2/tests/test_win_smoke.py +138 -0
- langsc-2.0.14/LangSC/Conf.yaml +0 -11
- langsc-2.0.14/LangSC/__version__.py +0 -3
- langsc-2.0.14/LangSC/bcclib.dll +0 -0
- langsc-2.0.14/LangSC/gpflib.dll +0 -0
- langsc-2.0.14/LangSC/jsslib.dll +0 -0
- langsc-2.0.14/LangSC/libbcclib.dylib +0 -0
- langsc-2.0.14/LangSC/libjsslib.so +0 -0
- langsc-2.0.14/LangSC/test.py +0 -7
- {langsc-2.0.14 → langsc-2.2.2}/LICENSE +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/BCCconfig.txt +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/GPFconfig.txt +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/Pathplan.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/cdt.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/cgraph.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/config6 +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/dot.exe +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/fontconfig.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/fontconfig_fix.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/freetype6.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvc.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_core.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_dot_layout.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_gd.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/gvplugin_pango.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/iconv.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/intl.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/jpeg62.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libcairo-2.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libexpat-1.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libexpat.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libfontconfig-1.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libfreetype-6.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgdk_pixbuf-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libglade-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libglib-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgmodule-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgobject-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgthread-2.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libgtkglext-win32-1.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpango-1.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangocairo-1.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangoft2-1.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpangowin32-1.0-0.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libpng14-14.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/librsvg-2-2.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/libxml2.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/ltdl.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Graph/zlib1.dll +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/Segment.dat +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/__init__.py +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/base.lex +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/idxPOS.dat +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC/libbcclib.so +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/dependency_links.txt +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/not-zip-safe +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/requires.txt +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/LangSC.egg-info/top_level.txt +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/MANIFEST.in +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/pyproject.toml +0 -0
- {langsc-2.0.14 → langsc-2.2.2}/setup.py +0 -0
|
@@ -394,17 +394,15 @@ function GetASInfo(Query)
|
|
|
394
394
|
local QueryEx=""
|
|
395
395
|
local lFix=""
|
|
396
396
|
local rFix=""
|
|
397
|
-
local
|
|
398
|
-
local
|
|
397
|
+
local lFixinn=""
|
|
398
|
+
local rFixinn=""
|
|
399
399
|
local lBracket=""
|
|
400
400
|
local rBracket=""
|
|
401
401
|
local lBracketInner=""
|
|
402
402
|
local rBracketInner=""
|
|
403
403
|
QueryEx=string.gsub(Query,"[%.%(%)~]","")
|
|
404
404
|
Re="^([%~%.%(%)]*)[^%~%.%(%)]+([%~%.%(%)]*)$"
|
|
405
|
-
|
|
406
|
-
-- 否则像 (v)了个~ 这种中段含 ) 的查询会让单段正则整体失配, 丢掉尾部 ~/. 修饰
|
|
407
|
-
B,E,lFix,rFix=string.find((string.gsub(Query,"[%(%)]","")),Re)
|
|
405
|
+
B,E,lFix,rFix=string.find(Query,Re)
|
|
408
406
|
if B == nil then
|
|
409
407
|
Re="^[a-zA-Z_-]+%[([%~%.%(%)]*)%*?([%~%.%(%)]*)[^%s\1-\127]*%]$"
|
|
410
408
|
B,E,lFixInner,rFixInner=string.find(Query,Re)
|
|
@@ -459,11 +457,7 @@ function Replace(Inp)
|
|
|
459
457
|
else
|
|
460
458
|
l=string.sub(Query,1,B-1)
|
|
461
459
|
r=string.sub(Query,E+1)
|
|
462
|
-
|
|
463
|
-
local rTok=string.match(r,"^(%S+)")
|
|
464
|
-
local lHasSlash=lTok and string.find(lTok,"/")
|
|
465
|
-
local rHasSlash=rTok and string.find(rTok,"/")
|
|
466
|
-
if lHasSlash or rHasSlash or (string.find(l,"[a-zA-Z0-9]$") ~=nil and string.find(r,"^[a-zA-Z0-9]") ~=nil) then
|
|
460
|
+
if string.find(l,"[a-zA-Z0-9]$") ~=nil and string.find(r,"^[a-zA-Z0-9]") ~=nil then
|
|
467
461
|
Query=l.." "..r
|
|
468
462
|
else
|
|
469
463
|
Query=l.."#"..r
|
|
@@ -679,168 +673,6 @@ function GeCLtInfo(Query)
|
|
|
679
673
|
|
|
680
674
|
end
|
|
681
675
|
|
|
682
|
-
-- word/POS 检索式扩展 ---------------------------------------------------------
|
|
683
|
-
-- token 分类: slash (word/POS) | POS (纯字母) | HZs (汉字串, 可单/多词)
|
|
684
|
-
-- 子表达式拼接后送 GetScriptAS, 由它走对应分支输出单个 GetAS:
|
|
685
|
-
-- HZs → GetAS("<HZ", HZs, ...)
|
|
686
|
-
-- POS → GetAS("|POS", "", ...)
|
|
687
|
-
-- POS+HZs → GetAS("|POS_HZ", HZs, ...) [HZ = 首字]
|
|
688
|
-
-- HZs+POS → GetAS("HZ_POS|", HZs, ...) [HZ = 末字]
|
|
689
|
-
-- 注意: word 与 string 在引擎层面同质 (都是 HZs), Word+String 视为一个整串.
|
|
690
|
-
--
|
|
691
|
-
-- 形态:
|
|
692
|
-
-- word/POS → word , POS ; SameBoundary
|
|
693
|
-
-- word/POS1 POS2 → word+POS2 , POS1 ; SameLeft
|
|
694
|
-
-- POS1 word/POS2 → POS1+word , POS2 ; SameRight
|
|
695
|
-
-- word1/POS1 word2/POS2 → word1+POS2 , POS1+word2 ; SameBoundary
|
|
696
|
-
-- POS1 word/POS2 POS3 → (POS1+word) ShareQuery (word+POS3) = R
|
|
697
|
-
-- POS2 InChunk R
|
|
698
|
-
-- word/POS1 string → word+string , POS1+string ; SameLeft
|
|
699
|
-
-- string word/POS1 → string+word , string+POS1 ; SameRight
|
|
700
|
-
-- word1/POS1 string word2/POS2 → POS1+string+word2 , word1+string+POS2 ; SameBoundary
|
|
701
|
-
-- word1/POS1 word2/POS2 string → word1+POS2 , POS1+word2+string ; SameLeft
|
|
702
|
-
-- string word1/POS1 word2/POS2 → (string+POS1) SameLeft (string+word1+POS2) = R
|
|
703
|
-
-- R SameBoundary (string+word1+word2)
|
|
704
|
-
-- 增量扩展: 上述 10 种形态作为查询的最左前缀(贪心最长匹配 3→2→1 token), 后面可
|
|
705
|
-
-- 跟 0..N 个非 slash 原子, 每个用 GetScriptAS 编译 + Link / InChunk 连接.
|
|
706
|
-
-- prefix atom → JoinAS(prefix, atom, "Link")
|
|
707
|
-
-- prefix *atom → JoinAS(prefix, atom, "InChunk") -- atom 前缀 * 表 InChunk
|
|
708
|
-
-- 多个 trailing 依次链接: ((prefix R1 t1) R2 t2) R3 t3 ...
|
|
709
|
-
-- GetAS-JoinAS 交替, 不出现连续 JoinAS, 也不出现 ≥3 连续 GetAS.
|
|
710
|
-
function EmitJoinAS(leftHandle,rightHandle,relation)
|
|
711
|
-
local hOut=HandleNo
|
|
712
|
-
HandleNo=HandleNo+1
|
|
713
|
-
local s=string.format('Handle%d=JoinAS(%s,%s,"%s")\n',hOut,leftHandle,rightHandle,relation)
|
|
714
|
-
return s,string.format('Handle%d',hOut)
|
|
715
|
-
end
|
|
716
|
-
|
|
717
|
-
local function classifyToken(tok)
|
|
718
|
-
local _,_,w,p=string.find(tok,"^([^/]+)/([^/]+)$")
|
|
719
|
-
if w then return "slash",w,p end
|
|
720
|
-
if string.find(tok,"^[a-zA-Z_-]+$") then return "POS",tok,nil end
|
|
721
|
-
if string.find(tok,"^[^%s\1-\127]+$") then return "HZs",tok,nil end
|
|
722
|
-
return "unknown",tok,nil
|
|
723
|
-
end
|
|
724
|
-
|
|
725
|
-
local function matchPattern1(tokens)
|
|
726
|
-
local k,w,p=classifyToken(tokens[1])
|
|
727
|
-
if k=="slash" then
|
|
728
|
-
local s1,h1=GetScriptAS(w)
|
|
729
|
-
local s2,h2=GetScriptAS(p)
|
|
730
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
|
|
731
|
-
return s1..s2..sj,hj
|
|
732
|
-
end
|
|
733
|
-
return nil,nil
|
|
734
|
-
end
|
|
735
|
-
|
|
736
|
-
local function matchPattern2(tokens)
|
|
737
|
-
local k1,a1,b1=classifyToken(tokens[1])
|
|
738
|
-
local k2,a2,b2=classifyToken(tokens[2])
|
|
739
|
-
if k1=="slash" and k2=="POS" then
|
|
740
|
-
local s1,h1=GetScriptAS(a1..a2)
|
|
741
|
-
local s2,h2=GetScriptAS(b1)
|
|
742
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
|
|
743
|
-
return s1..s2..sj,hj
|
|
744
|
-
elseif k1=="POS" and k2=="slash" then
|
|
745
|
-
local s1,h1=GetScriptAS(a1..a2)
|
|
746
|
-
local s2,h2=GetScriptAS(b2)
|
|
747
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameRight")
|
|
748
|
-
return s1..s2..sj,hj
|
|
749
|
-
elseif k1=="slash" and k2=="slash" then
|
|
750
|
-
local s1,h1=GetScriptAS(a1..b2)
|
|
751
|
-
local s2,h2=GetScriptAS(b1..a2)
|
|
752
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
|
|
753
|
-
return s1..s2..sj,hj
|
|
754
|
-
elseif k1=="slash" and k2=="HZs" then
|
|
755
|
-
local s1,h1=GetScriptAS(a1..a2)
|
|
756
|
-
local s2,h2=GetScriptAS(b1..a2)
|
|
757
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
|
|
758
|
-
return s1..s2..sj,hj
|
|
759
|
-
elseif k1=="HZs" and k2=="slash" then
|
|
760
|
-
local s1,h1=GetScriptAS(a1..a2)
|
|
761
|
-
local s2,h2=GetScriptAS(a1..b2)
|
|
762
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameRight")
|
|
763
|
-
return s1..s2..sj,hj
|
|
764
|
-
end
|
|
765
|
-
return nil,nil
|
|
766
|
-
end
|
|
767
|
-
|
|
768
|
-
local function matchPattern3(tokens)
|
|
769
|
-
local k1,a1,b1=classifyToken(tokens[1])
|
|
770
|
-
local k2,a2,b2=classifyToken(tokens[2])
|
|
771
|
-
local k3,a3,b3=classifyToken(tokens[3])
|
|
772
|
-
if k1=="POS" and k2=="slash" and k3=="POS" then
|
|
773
|
-
local s1,h1=GetScriptAS(a1..a2)
|
|
774
|
-
local s2,h2=GetScriptAS(a2..a3)
|
|
775
|
-
local sShare,hShare=EmitJoinAS(h1,h2,"ShareQuery")
|
|
776
|
-
local s3,h3=GetScriptAS(b2)
|
|
777
|
-
local sIn,hIn=EmitJoinAS(h3,hShare,"InChunk")
|
|
778
|
-
return s1..s2..sShare..s3..sIn,hIn
|
|
779
|
-
elseif k1=="slash" and k2=="HZs" and k3=="slash" then
|
|
780
|
-
local s1,h1=GetScriptAS(b1..a2..a3)
|
|
781
|
-
local s2,h2=GetScriptAS(a1..a2..b3)
|
|
782
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameBoundary")
|
|
783
|
-
return s1..s2..sj,hj
|
|
784
|
-
elseif k1=="slash" and k2=="slash" and k3=="HZs" then
|
|
785
|
-
local s1,h1=GetScriptAS(a1..b2)
|
|
786
|
-
local s2,h2=GetScriptAS(b1..a2..a3)
|
|
787
|
-
local sj,hj=EmitJoinAS(h1,h2,"SameLeft")
|
|
788
|
-
return s1..s2..sj,hj
|
|
789
|
-
elseif k1=="HZs" and k2=="slash" and k3=="slash" then
|
|
790
|
-
local s1,h1=GetScriptAS(a1..b2)
|
|
791
|
-
local s2,h2=GetScriptAS(a1..a2..b3)
|
|
792
|
-
local sR,hR=EmitJoinAS(h1,h2,"SameLeft")
|
|
793
|
-
local s3,h3=GetScriptAS(a1..a2..a3)
|
|
794
|
-
local sFinal,hFinal=EmitJoinAS(hR,h3,"SameBoundary")
|
|
795
|
-
return s1..s2..sR..s3..sFinal,hFinal
|
|
796
|
-
end
|
|
797
|
-
return nil,nil
|
|
798
|
-
end
|
|
799
|
-
|
|
800
|
-
function BuildSlashScript(Query)
|
|
801
|
-
-- Replace 会把 *X 前后的空格吃掉, 这里恢复一下 (slash 查询里不会有 q[*])
|
|
802
|
-
Query=string.gsub(Query,"%*"," *")
|
|
803
|
-
local tokens={}
|
|
804
|
-
for tok in string.gmatch(Query,"%S+") do
|
|
805
|
-
table.insert(tokens,tok)
|
|
806
|
-
end
|
|
807
|
-
local n=#tokens
|
|
808
|
-
local prefixScript,prefixHandle,prefixLen=nil,nil,0
|
|
809
|
-
if n>=3 then
|
|
810
|
-
local s,h=matchPattern3(tokens)
|
|
811
|
-
if s then prefixScript,prefixHandle,prefixLen=s,h,3 end
|
|
812
|
-
end
|
|
813
|
-
if not prefixScript and n>=2 then
|
|
814
|
-
local s,h=matchPattern2(tokens)
|
|
815
|
-
if s then prefixScript,prefixHandle,prefixLen=s,h,2 end
|
|
816
|
-
end
|
|
817
|
-
if not prefixScript and n>=1 then
|
|
818
|
-
local s,h=matchPattern1(tokens)
|
|
819
|
-
if s then prefixScript,prefixHandle,prefixLen=s,h,1 end
|
|
820
|
-
end
|
|
821
|
-
if not prefixScript then
|
|
822
|
-
error("Slash query: no valid slash prefix in '"..Query.."'")
|
|
823
|
-
end
|
|
824
|
-
local script=prefixScript
|
|
825
|
-
local running=prefixHandle
|
|
826
|
-
for i=prefixLen+1,n do
|
|
827
|
-
local tok=tokens[i]
|
|
828
|
-
local relation="Link"
|
|
829
|
-
if string.sub(tok,1,1)=="*" then
|
|
830
|
-
relation="InChunk"
|
|
831
|
-
tok=string.sub(tok,2)
|
|
832
|
-
end
|
|
833
|
-
if tok=="" or string.find(tok,"/") then
|
|
834
|
-
error("Slash query: invalid trailing atom '"..tokens[i].."'")
|
|
835
|
-
end
|
|
836
|
-
local sa,ha=GetScriptAS(tok)
|
|
837
|
-
local sj,hj=EmitJoinAS(running,ha,relation)
|
|
838
|
-
script=script..sa..sj
|
|
839
|
-
running=hj
|
|
840
|
-
end
|
|
841
|
-
return script
|
|
842
|
-
end
|
|
843
|
-
|
|
844
676
|
function Parser(Query,launguage,Option)
|
|
845
677
|
Re="^Lua:(.+)"
|
|
846
678
|
B,E,ReadlQuery=string.find(Query,Re)
|
|
@@ -908,17 +740,12 @@ function Parser(Query,launguage,Option)
|
|
|
908
740
|
Query=Eng2Chin(Query)
|
|
909
741
|
end
|
|
910
742
|
end
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
local
|
|
916
|
-
|
|
917
|
-
HANDLENO=HandleNo-1
|
|
918
|
-
if LEFT ~= nil and LEFT ~= "" then
|
|
919
|
-
local lTree=CreateTree(LEFT.."@",0)
|
|
920
|
-
CreateScriptL(lTree)
|
|
921
|
-
end
|
|
743
|
+
local rTree =CreateTree(Query,1)
|
|
744
|
+
CreateScriptR(rTree)
|
|
745
|
+
HANDLENO=HandleNo-1
|
|
746
|
+
if LEFT ~= nil and LEFT ~= "" then
|
|
747
|
+
local lTree=CreateTree(LEFT.."@",0)
|
|
748
|
+
CreateScriptL(lTree)
|
|
922
749
|
end
|
|
923
750
|
Operation=GetOperation(Operation)
|
|
924
751
|
Script=Script..Operation
|
|
@@ -1052,333 +879,3 @@ function Test()
|
|
|
1052
879
|
end
|
|
1053
880
|
|
|
1054
881
|
--Test()
|
|
1055
|
-
|
|
1056
|
-
-- ═══════════════════════════════════════════════════════════════════
|
|
1057
|
-
-- 词集功能(spec: Doc/superpowers/specs/2026-05-07-bcc-wordset-design.md)
|
|
1058
|
-
-- ═══════════════════════════════════════════════════════════════════
|
|
1059
|
-
|
|
1060
|
-
g_WordSets = {}
|
|
1061
|
-
MAX_TOTAL_MS = 30000
|
|
1062
|
-
|
|
1063
|
-
-- 解析受限 YAML
|
|
1064
|
-
-- 支持: 顶格 key:、缩进(空格)词条、# 注释、空行、CRLF、UTF-8 BOM
|
|
1065
|
-
-- 返回: (table sets, nil) 成功 / (nil, "Conf.yaml line N: ...") 失败
|
|
1066
|
-
function ParseYaml(path)
|
|
1067
|
-
local fp = io.open(path, "rb")
|
|
1068
|
-
if not fp then return {}, nil end -- 不存在视为空字典(不算错误)
|
|
1069
|
-
local sets = {}
|
|
1070
|
-
local cur_key = nil
|
|
1071
|
-
local lineNo = 0
|
|
1072
|
-
local first = true
|
|
1073
|
-
for line in fp:lines() do
|
|
1074
|
-
lineNo = lineNo + 1
|
|
1075
|
-
if first then
|
|
1076
|
-
-- 去 UTF-8 BOM
|
|
1077
|
-
if line:sub(1,3) == "\xEF\xBB\xBF" then line = line:sub(4) end
|
|
1078
|
-
first = false
|
|
1079
|
-
end
|
|
1080
|
-
line = line:gsub("\r$", "") -- 去 CRLF
|
|
1081
|
-
local stripped = line:gsub("^%s+",""):gsub("%s+$","")
|
|
1082
|
-
if stripped == "" or stripped:sub(1,1) == "#" then
|
|
1083
|
-
-- 空行 / 注释
|
|
1084
|
-
elseif line:sub(1,1) ~= " " and line:sub(1,1) ~= "\t" then
|
|
1085
|
-
-- 顶格行 = 词集名
|
|
1086
|
-
local k = line:match("^([A-Za-z_][A-Za-z0-9_]*)%s*:%s*$")
|
|
1087
|
-
if not k then
|
|
1088
|
-
fp:close()
|
|
1089
|
-
return nil, "Conf.yaml line "..lineNo..": bad key syntax"
|
|
1090
|
-
end
|
|
1091
|
-
cur_key = k
|
|
1092
|
-
sets[k] = {}
|
|
1093
|
-
elseif line:sub(1,1) == "\t" then
|
|
1094
|
-
fp:close()
|
|
1095
|
-
return nil, "Conf.yaml line "..lineNo..": tab indent not allowed"
|
|
1096
|
-
else
|
|
1097
|
-
-- 缩进行 = 词条
|
|
1098
|
-
if not cur_key then
|
|
1099
|
-
fp:close()
|
|
1100
|
-
return nil, "Conf.yaml line "..lineNo..": value before any key"
|
|
1101
|
-
end
|
|
1102
|
-
local tok = line:match("^%s+(%S+)%s*$")
|
|
1103
|
-
if not tok then
|
|
1104
|
-
fp:close()
|
|
1105
|
-
return nil, "Conf.yaml line "..lineNo..": bad value syntax"
|
|
1106
|
-
end
|
|
1107
|
-
table.insert(sets[cur_key], tok)
|
|
1108
|
-
end
|
|
1109
|
-
end
|
|
1110
|
-
fp:close()
|
|
1111
|
-
return sets, nil
|
|
1112
|
-
end
|
|
1113
|
-
|
|
1114
|
-
-- LoadConf:由 C 层 EnsureParserVM 在 mtime 变化时调用
|
|
1115
|
-
-- 解析 Conf.yaml,把词条从 UTF-8(文件编码)转为 GBK(引擎内部编码)
|
|
1116
|
-
function LoadConf(path)
|
|
1117
|
-
local sets, err = ParseYaml(path)
|
|
1118
|
-
if err then
|
|
1119
|
-
print("[LoadConf] "..err)
|
|
1120
|
-
g_WordSets = {}
|
|
1121
|
-
return
|
|
1122
|
-
end
|
|
1123
|
-
-- 文件以 UTF-8 保存,词条需转为 GBK 才能拼到 GBK 查询里
|
|
1124
|
-
for k, words in pairs(sets) do
|
|
1125
|
-
for i = 1, #words do
|
|
1126
|
-
words[i] = UTF82GB(words[i])
|
|
1127
|
-
end
|
|
1128
|
-
end
|
|
1129
|
-
g_WordSets = sets
|
|
1130
|
-
end
|
|
1131
|
-
|
|
1132
|
-
-- 扫描查询里的 <NAME> 占位符
|
|
1133
|
-
-- 返回 (key, words, nil) / (nil, nil, nil 无词集) / (nil, nil, errMsg)
|
|
1134
|
-
function GetWordSet(query)
|
|
1135
|
-
local hits = {}
|
|
1136
|
-
for name in query:gmatch("<([A-Za-z_][A-Za-z0-9_]*)>") do
|
|
1137
|
-
table.insert(hits, name)
|
|
1138
|
-
end
|
|
1139
|
-
if #hits == 0 then return nil, nil, nil end
|
|
1140
|
-
if #hits > 1 then
|
|
1141
|
-
return nil, nil,
|
|
1142
|
-
"multiple word sets in one query: "..table.concat(hits, ",")
|
|
1143
|
-
end
|
|
1144
|
-
local key = hits[1]
|
|
1145
|
-
if not g_WordSets[key] or #g_WordSets[key] == 0 then
|
|
1146
|
-
return nil, nil, "unknown or empty word set: <"..key..">"
|
|
1147
|
-
end
|
|
1148
|
-
return key, g_WordSets[key], nil
|
|
1149
|
-
end
|
|
1150
|
-
|
|
1151
|
-
-- 把 query 里的 <key> 替换成 word(只替换一次)
|
|
1152
|
-
-- 关键:gsub 替换串里的 % 是元字符,必须先转义
|
|
1153
|
-
function ReplaceWS(query, key, word)
|
|
1154
|
-
local safe = word:gsub("%%", "%%%%")
|
|
1155
|
-
return (query:gsub("<"..key..">", safe, 1))
|
|
1156
|
-
end
|
|
1157
|
-
|
|
1158
|
-
-- 在 json 字符串里找 "fieldName": [...],返回数组体(不含外层 [])和结束位置
|
|
1159
|
-
-- 处理嵌套 {}[] 与字符串内的转义引号
|
|
1160
|
-
function ExtractArrayBody(json, field)
|
|
1161
|
-
local _, e = json:find('"'..field..'"%s*:%s*%[')
|
|
1162
|
-
if not e then return nil, nil end
|
|
1163
|
-
local startPos = e + 1
|
|
1164
|
-
local depth, i, len, inStr = 1, e + 1, #json, false
|
|
1165
|
-
while i <= len and depth > 0 do
|
|
1166
|
-
local c = json:sub(i, i)
|
|
1167
|
-
if inStr then
|
|
1168
|
-
if c == '"' and json:sub(i-1, i-1) ~= '\\' then inStr = false end
|
|
1169
|
-
else
|
|
1170
|
-
if c == '"' then inStr = true
|
|
1171
|
-
elseif c == '[' or c == '{' then depth = depth + 1
|
|
1172
|
-
elseif c == ']' or c == '}' then
|
|
1173
|
-
depth = depth - 1
|
|
1174
|
-
if depth == 0 and c == ']' then
|
|
1175
|
-
return json:sub(startPos, i - 1), i
|
|
1176
|
-
end
|
|
1177
|
-
end
|
|
1178
|
-
end
|
|
1179
|
-
i = i + 1
|
|
1180
|
-
end
|
|
1181
|
-
return nil, nil
|
|
1182
|
-
end
|
|
1183
|
-
|
|
1184
|
-
-- 内部辅助:简单 JSON 字符串值转义(用于 WordSetErrors 字段)
|
|
1185
|
-
local function _jsonEscape(s)
|
|
1186
|
-
s = tostring(s or "")
|
|
1187
|
-
s = s:gsub('\\', '\\\\'):gsub('"', '\\"')
|
|
1188
|
-
s = s:gsub('\n', '\\n'):gsub('\r', '\\r'):gsub('\t', '\\t')
|
|
1189
|
-
return s
|
|
1190
|
-
end
|
|
1191
|
-
|
|
1192
|
-
-- 错误体格式化:{"error":"msg","WordSetErrors":[...]}
|
|
1193
|
-
function JsonError(msg, errs)
|
|
1194
|
-
local s = '{"error":"'.._jsonEscape(msg)..'"'
|
|
1195
|
-
if errs and #errs > 0 then
|
|
1196
|
-
s = s..',"WordSetErrors":['
|
|
1197
|
-
for i, e in ipairs(errs) do
|
|
1198
|
-
if i > 1 then s = s..',' end
|
|
1199
|
-
s = s..'{"Word":"'.._jsonEscape(e.Word)..
|
|
1200
|
-
'","Error":"'.._jsonEscape(e.Error)..'"}'
|
|
1201
|
-
end
|
|
1202
|
-
s = s..']'
|
|
1203
|
-
end
|
|
1204
|
-
return s..'}'
|
|
1205
|
-
end
|
|
1206
|
-
|
|
1207
|
-
-- 内部:WordSetErrors 字段尾巴
|
|
1208
|
-
local function _errorsTail(errors)
|
|
1209
|
-
if not errors or #errors == 0 then return "" end
|
|
1210
|
-
local s = ',"WordSetErrors":['
|
|
1211
|
-
for i, e in ipairs(errors) do
|
|
1212
|
-
if i > 1 then s = s..',' end
|
|
1213
|
-
s = s..'{"Word":"'.._jsonEscape(e.Word)..
|
|
1214
|
-
'","Error":"'.._jsonEscape(e.Error)..'"}'
|
|
1215
|
-
end
|
|
1216
|
-
return s..']'
|
|
1217
|
-
end
|
|
1218
|
-
|
|
1219
|
-
-- 创建空累加器
|
|
1220
|
-
function NewAcc(typ)
|
|
1221
|
-
if typ == "Context" then
|
|
1222
|
-
return { type="Context", items={}, total=0, count=0, approx=false }
|
|
1223
|
-
elseif typ == "Freq" then
|
|
1224
|
-
return { type="Freq", map={}, total=0 }
|
|
1225
|
-
elseif typ == "Count" then
|
|
1226
|
-
return { type="Count", buckets={}, total=0 }
|
|
1227
|
-
end
|
|
1228
|
-
return nil
|
|
1229
|
-
end
|
|
1230
|
-
|
|
1231
|
-
-- ── Context ──
|
|
1232
|
-
local function _mergeContext(acc, out, word)
|
|
1233
|
-
local body, _ = ExtractArrayBody(out, "Context")
|
|
1234
|
-
if not body then return false, "no Context array body" end
|
|
1235
|
-
table.insert(acc.items, body)
|
|
1236
|
-
|
|
1237
|
-
local total = tonumber(out:match('"Total"%s*:%s*(%-?%d+)')) or 0
|
|
1238
|
-
local count = tonumber(out:match('"Count"%s*:%s*(%-?%d+)')) or 0
|
|
1239
|
-
acc.total = acc.total + total
|
|
1240
|
-
acc.count = acc.count + count
|
|
1241
|
-
|
|
1242
|
-
if out:match('"Approximate"%s*:%s*"1"') then acc.approx = true end
|
|
1243
|
-
return true, nil
|
|
1244
|
-
end
|
|
1245
|
-
|
|
1246
|
-
local function _emitContext(acc, errors)
|
|
1247
|
-
local s = '{"Type":"Context",'
|
|
1248
|
-
s = s..'"Count":'..acc.count..','
|
|
1249
|
-
s = s..'"Total":'..acc.total..','
|
|
1250
|
-
s = s..'"Approximate":"'..(acc.approx and "1" or "0")..'",'
|
|
1251
|
-
s = s..'"Context":['..table.concat(acc.items, ',')..']'
|
|
1252
|
-
s = s.._errorsTail(errors)
|
|
1253
|
-
return s..'}'
|
|
1254
|
-
end
|
|
1255
|
-
|
|
1256
|
-
-- ── Freq ──
|
|
1257
|
-
local function _mergeFreq(acc, out, word)
|
|
1258
|
-
local body, _ = ExtractArrayBody(out, "Freq")
|
|
1259
|
-
if not body then return false, "no Freq array body" end
|
|
1260
|
-
|
|
1261
|
-
-- 解析 {"Word":"X","Freq":N} 格式条目(Word/Freq 来自索引,值可信)
|
|
1262
|
-
for w, f in body:gmatch('{"Word"%s*:%s*"([^"]*)"%s*,%s*"Freq"%s*:%s*(%-?%d+)}') do
|
|
1263
|
-
local n = tonumber(f) or 0
|
|
1264
|
-
acc.map[w] = (acc.map[w] or 0) + n
|
|
1265
|
-
acc.total = acc.total + n
|
|
1266
|
-
end
|
|
1267
|
-
return true, nil
|
|
1268
|
-
end
|
|
1269
|
-
|
|
1270
|
-
local function _emitFreq(acc, errors)
|
|
1271
|
-
-- 按 Freq 降序排序
|
|
1272
|
-
local arr = {}
|
|
1273
|
-
for w, f in pairs(acc.map) do
|
|
1274
|
-
table.insert(arr, { Word = w, Freq = f })
|
|
1275
|
-
end
|
|
1276
|
-
table.sort(arr, function(a, b) return a.Freq > b.Freq end)
|
|
1277
|
-
|
|
1278
|
-
local pieces = {}
|
|
1279
|
-
for _, e in ipairs(arr) do
|
|
1280
|
-
table.insert(pieces, '{"Word":"'.._jsonEscape(e.Word)..
|
|
1281
|
-
'","Freq":'..e.Freq..'}')
|
|
1282
|
-
end
|
|
1283
|
-
|
|
1284
|
-
local s = '{"Type":"Freq","Count":'..#arr..
|
|
1285
|
-
',"Total":'..acc.total..
|
|
1286
|
-
',"Freq":['..table.concat(pieces, ',')..']'
|
|
1287
|
-
s = s.._errorsTail(errors)
|
|
1288
|
-
return s..'}'
|
|
1289
|
-
end
|
|
1290
|
-
|
|
1291
|
-
-- ── Count ──
|
|
1292
|
-
local function _mergeCount(acc, out, word)
|
|
1293
|
-
local body, _ = ExtractArrayBody(out, "Freq")
|
|
1294
|
-
if not body then return false, "no Count Freq array body" end
|
|
1295
|
-
local i = 1
|
|
1296
|
-
for n in body:gmatch("%-?%d+") do
|
|
1297
|
-
local v = tonumber(n) or 0
|
|
1298
|
-
acc.buckets[i] = (acc.buckets[i] or 0) + v
|
|
1299
|
-
acc.total = acc.total + v
|
|
1300
|
-
i = i + 1
|
|
1301
|
-
end
|
|
1302
|
-
return true, nil
|
|
1303
|
-
end
|
|
1304
|
-
|
|
1305
|
-
local function _emitCount(acc, errors)
|
|
1306
|
-
local pieces = {}
|
|
1307
|
-
for i = 1, #acc.buckets do
|
|
1308
|
-
table.insert(pieces, tostring(acc.buckets[i]))
|
|
1309
|
-
end
|
|
1310
|
-
local s = '{"Type":"Count","Count":1,"Total":'..acc.total..
|
|
1311
|
-
',"Freq":['..table.concat(pieces, ',')..']'
|
|
1312
|
-
s = s.._errorsTail(errors)
|
|
1313
|
-
return s..'}'
|
|
1314
|
-
end
|
|
1315
|
-
|
|
1316
|
-
-- 分派器:把 out 合并到 acc
|
|
1317
|
-
function MergeOne(acc, typ, out, word)
|
|
1318
|
-
if typ == "Context" then return _mergeContext(acc, out, word) end
|
|
1319
|
-
if typ == "Freq" then return _mergeFreq(acc, out, word) end
|
|
1320
|
-
if typ == "Count" then return _mergeCount(acc, out, word) end
|
|
1321
|
-
return false, "unknown type: "..tostring(typ)
|
|
1322
|
-
end
|
|
1323
|
-
|
|
1324
|
-
-- 分派器:序列化为最终 JSON
|
|
1325
|
-
function EmitMerged(acc, errors)
|
|
1326
|
-
if acc.type == "Context" then return _emitContext(acc, errors) end
|
|
1327
|
-
if acc.type == "Freq" then return _emitFreq(acc, errors) end
|
|
1328
|
-
if acc.type == "Count" then return _emitCount(acc, errors) end
|
|
1329
|
-
return JsonError("unknown acc type")
|
|
1330
|
-
end
|
|
1331
|
-
|
|
1332
|
-
-- ParserNew 主入口:替代 Parser 处理含词集的查询
|
|
1333
|
-
-- 由 C 层 Query2Result 调用;无词集时回退到原 Parser + BCCLua
|
|
1334
|
-
function ParserNew(Query, language, Option)
|
|
1335
|
-
local isLuaGen = (Query:sub(1,4) == "Lua:")
|
|
1336
|
-
|
|
1337
|
-
local key, words, err = GetWordSet(Query)
|
|
1338
|
-
if err then
|
|
1339
|
-
return JsonError(err)
|
|
1340
|
-
end
|
|
1341
|
-
|
|
1342
|
-
-- 路径 ①:无词集 → 原 Parser + BCCLua
|
|
1343
|
-
if key == nil then
|
|
1344
|
-
local script = Parser(Query, language, Option)
|
|
1345
|
-
if isLuaGen then return script end
|
|
1346
|
-
return BCCLua(script)
|
|
1347
|
-
end
|
|
1348
|
-
|
|
1349
|
-
-- 路径 ②:Lua: + 词集 → 用第一个词展开成示例脚本
|
|
1350
|
-
if isLuaGen then
|
|
1351
|
-
local q1 = ReplaceWS(Query, key, words[1])
|
|
1352
|
-
return Parser(q1, language, Option)
|
|
1353
|
-
end
|
|
1354
|
-
|
|
1355
|
-
-- 路径 ③:词集批处理
|
|
1356
|
-
local errors = {}
|
|
1357
|
-
local acc = nil
|
|
1358
|
-
for i, w in ipairs(words) do
|
|
1359
|
-
if (not g_skip_budget) and ElapsedMs() > MAX_TOTAL_MS then
|
|
1360
|
-
table.insert(errors,
|
|
1361
|
-
{ Word = w, Error = "aborted: total time budget exceeded" })
|
|
1362
|
-
break
|
|
1363
|
-
end
|
|
1364
|
-
local qx = ReplaceWS(Query, key, w)
|
|
1365
|
-
local script = Parser(qx, language, Option)
|
|
1366
|
-
local out = BCCLua(script)
|
|
1367
|
-
local typ = out and out:match('"Type"%s*:%s*"([^"]+)"')
|
|
1368
|
-
if not typ then
|
|
1369
|
-
local msg = (out or ""):sub(1, 200)
|
|
1370
|
-
table.insert(errors, { Word = w, Error = msg })
|
|
1371
|
-
else
|
|
1372
|
-
if not acc then acc = NewAcc(typ) end
|
|
1373
|
-
local ok, msg = MergeOne(acc, typ, out, w)
|
|
1374
|
-
if not ok then
|
|
1375
|
-
table.insert(errors, { Word = w, Error = msg or "merge failed" })
|
|
1376
|
-
end
|
|
1377
|
-
end
|
|
1378
|
-
end
|
|
1379
|
-
|
|
1380
|
-
if not acc then
|
|
1381
|
-
return JsonError("all sub-queries failed", errors)
|
|
1382
|
-
end
|
|
1383
|
-
return EmitMerged(acc, errors)
|
|
1384
|
-
end
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Single source of truth for the package version.
|
|
2
|
+
# Keep __version__ a plain string literal: setup.cfg reads it statically via
|
|
3
|
+
# `version = attr: LangSC.__version__.__version__`, so it must stay AST-parseable
|
|
4
|
+
# (no computed expression) to avoid importing the package at build time.
|
|
5
|
+
__version__ = "2.2.2"
|
|
6
|
+
|
|
7
|
+
VERSION = tuple(int(x) for x in __version__.split("."))
|
|
@@ -2,7 +2,6 @@
|
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
4
|
import re
|
|
5
|
-
import json
|
|
6
5
|
import chardet
|
|
7
6
|
|
|
8
7
|
|
|
@@ -141,54 +140,15 @@ def get_bcc_files(path, idxed_file2time, bcc_files):
|
|
|
141
140
|
bcc_files.append(full_path)
|
|
142
141
|
|
|
143
142
|
|
|
144
|
-
def is_json_data_file(path):
|
|
145
|
-
"""Content-based check: is this a JSON / JSONL data file, regardless of extension?
|
|
146
|
-
|
|
147
|
-
Naming is intentionally NOT required — a user can drop in *.json, *.jsonl, *.txt
|
|
148
|
-
or extension-less files and they all get picked up, as long as the content is JSON.
|
|
149
|
-
A file qualifies when its first non-whitespace character is '{' or '[' (a JSON
|
|
150
|
-
object or array, the only shapes JSS indexes). cfg_*.txt config files and any
|
|
151
|
-
binary / non-JSON text are skipped.
|
|
152
|
-
"""
|
|
153
|
-
name = os.path.basename(path)
|
|
154
|
-
if name.startswith('cfg_'):
|
|
155
|
-
return False
|
|
156
|
-
try:
|
|
157
|
-
encoding = detect_file_encoding(path)
|
|
158
|
-
with open(path, 'r', encoding=encoding) as f:
|
|
159
|
-
head = f.read(8192)
|
|
160
|
-
except (OSError, UnicodeDecodeError, LookupError):
|
|
161
|
-
return False # unreadable / binary -> not a data file
|
|
162
|
-
head = head.lstrip()
|
|
163
|
-
return bool(head) and head[0] in '{['
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
def detect_record_format(path):
|
|
167
|
-
"""Decide record_format by content: 'json' (whole file is one JSON value) vs
|
|
168
|
-
'jsonl' (one JSON value per line). Falls back to 'json' on empty/unreadable."""
|
|
169
|
-
try:
|
|
170
|
-
encoding = detect_file_encoding(path)
|
|
171
|
-
with open(path, 'r', encoding=encoding) as f:
|
|
172
|
-
text = f.read()
|
|
173
|
-
except (OSError, UnicodeDecodeError, LookupError):
|
|
174
|
-
return 'json'
|
|
175
|
-
stripped = text.strip()
|
|
176
|
-
if not stripped:
|
|
177
|
-
return 'json'
|
|
178
|
-
try:
|
|
179
|
-
json.loads(stripped) # the whole file parses as a single JSON value
|
|
180
|
-
return 'json'
|
|
181
|
-
except ValueError:
|
|
182
|
-
return 'jsonl' # otherwise treat as one JSON object per line
|
|
183
|
-
|
|
184
|
-
|
|
185
143
|
def get_jss_file_info(path, to_idx_file2time):
|
|
186
|
-
"""Walk path and collect modification times for JSON/JSONL data files
|
|
144
|
+
"""Walk path and collect modification times for JSON/JSONL data files only."""
|
|
187
145
|
for root, dirs, files in os.walk(path):
|
|
188
146
|
for f in files:
|
|
189
|
-
|
|
190
|
-
|
|
147
|
+
if not (f.endswith('.json') or f.endswith('.jsonl')):
|
|
148
|
+
continue
|
|
149
|
+
if f.startswith('cfg_'):
|
|
191
150
|
continue
|
|
151
|
+
full_path = os.path.join(root, f)
|
|
192
152
|
time_id = os.path.getmtime(full_path)
|
|
193
153
|
if not to_idx_file2time.get(f):
|
|
194
154
|
to_idx_file2time[f] = {}
|
|
@@ -197,12 +157,14 @@ def get_jss_file_info(path, to_idx_file2time):
|
|
|
197
157
|
|
|
198
158
|
|
|
199
159
|
def get_jss_files(path, idxed_file2time, jss_files):
|
|
200
|
-
"""Scan path for JSON/JSONL data files
|
|
160
|
+
"""Scan path for JSON/JSONL data files that need indexing."""
|
|
201
161
|
for root, dirs, files in os.walk(path):
|
|
202
162
|
for f in files:
|
|
203
|
-
|
|
204
|
-
|
|
163
|
+
if not (f.endswith('.json') or f.endswith('.jsonl')):
|
|
164
|
+
continue
|
|
165
|
+
if f.startswith('cfg_'):
|
|
205
166
|
continue
|
|
167
|
+
full_path = os.path.join(root, f)
|
|
206
168
|
if idxed_file2time.get(f):
|
|
207
169
|
time_id = os.path.getmtime(full_path)
|
|
208
170
|
if idxed_file2time[f].get(str(time_id)):
|
|
@@ -299,7 +261,7 @@ def check_words(all_words):
|
|
|
299
261
|
def process_file(file_path, file_tmp, cmd, gpf=None):
|
|
300
262
|
"""Process a raw file into corpus format."""
|
|
301
263
|
encoding = detect_file_encoding(file_path)
|
|
302
|
-
with open(file_path, "r", encoding=encoding
|
|
264
|
+
with open(file_path, "r", encoding=encoding) as f_in:
|
|
303
265
|
with open(file_tmp, "w") as f_out:
|
|
304
266
|
print("Doc {}".format(file_path), file=f_out)
|
|
305
267
|
for line in f_in:
|
|
@@ -24,12 +24,12 @@ lock = threading.Lock()
|
|
|
24
24
|
class BCC:
|
|
25
25
|
def __init__(self, dataPath="./data"):
|
|
26
26
|
dataPath = dataPath.replace("\\", "/")
|
|
27
|
-
if dataPath
|
|
27
|
+
if len(dataPath) > 1 and dataPath.endswith("/"):
|
|
28
28
|
dataPath = dataPath[:-1]
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
29
|
+
# 相对路径规范为 "./" 前缀;绝对路径(/... 或 X:/...)原样保留
|
|
30
|
+
if not (os.path.isabs(dataPath) or (len(dataPath) >= 2 and dataPath[1] == ":")):
|
|
31
|
+
if not dataPath.startswith("./"):
|
|
32
|
+
dataPath = "./" + dataPath
|
|
33
33
|
|
|
34
34
|
dll_name_bcc = ''
|
|
35
35
|
self.g_IdxLog = "IdxLog_BCC.txt"
|
|
Binary file
|