LangSC 2.2.2__tar.gz → 2.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. {langsc-2.2.2 → langsc-2.5.0}/LangSC/__version__.py +1 -1
  2. {langsc-2.2.2 → langsc-2.5.0}/LangSC/_utils.py +108 -22
  3. {langsc-2.2.2 → langsc-2.5.0}/LangSC/bcc.py +35 -9
  4. {langsc-2.2.2 → langsc-2.5.0}/LangSC/gpf.py +265 -51
  5. {langsc-2.2.2 → langsc-2.5.0}/LangSC/jss.py +74 -22
  6. {langsc-2.2.2 → langsc-2.5.0}/LangSC.egg-info/PKG-INFO +32 -6
  7. langsc-2.5.0/LangSC.egg-info/SOURCES.txt +35 -0
  8. {langsc-2.2.2 → langsc-2.5.0}/LangSC.egg-info/requires.txt +1 -0
  9. langsc-2.2.2/README.md → langsc-2.5.0/PKG-INFO +53 -4
  10. langsc-2.2.2/PKG-INFO → langsc-2.5.0/README.md +29 -27
  11. {langsc-2.2.2 → langsc-2.5.0}/setup.cfg +2 -3
  12. langsc-2.2.2/LangSC/Graph/Pathplan.dll +0 -0
  13. langsc-2.2.2/LangSC/Graph/cdt.dll +0 -0
  14. langsc-2.2.2/LangSC/Graph/cgraph.dll +0 -0
  15. langsc-2.2.2/LangSC/Graph/config6 +0 -273
  16. langsc-2.2.2/LangSC/Graph/dot.exe +0 -0
  17. langsc-2.2.2/LangSC/Graph/fontconfig.dll +0 -0
  18. langsc-2.2.2/LangSC/Graph/fontconfig_fix.dll +0 -0
  19. langsc-2.2.2/LangSC/Graph/freetype6.dll +0 -0
  20. langsc-2.2.2/LangSC/Graph/gvc.dll +0 -0
  21. langsc-2.2.2/LangSC/Graph/gvplugin_core.dll +0 -0
  22. langsc-2.2.2/LangSC/Graph/gvplugin_dot_layout.dll +0 -0
  23. langsc-2.2.2/LangSC/Graph/gvplugin_gd.dll +0 -0
  24. langsc-2.2.2/LangSC/Graph/gvplugin_pango.dll +0 -0
  25. langsc-2.2.2/LangSC/Graph/iconv.dll +0 -0
  26. langsc-2.2.2/LangSC/Graph/intl.dll +0 -0
  27. langsc-2.2.2/LangSC/Graph/jpeg62.dll +0 -0
  28. langsc-2.2.2/LangSC/Graph/libcairo-2.dll +0 -0
  29. langsc-2.2.2/LangSC/Graph/libexpat-1.dll +0 -0
  30. langsc-2.2.2/LangSC/Graph/libexpat.dll +0 -0
  31. langsc-2.2.2/LangSC/Graph/libfontconfig-1.dll +0 -0
  32. langsc-2.2.2/LangSC/Graph/libfreetype-6.dll +0 -0
  33. langsc-2.2.2/LangSC/Graph/libgdk_pixbuf-2.0-0.dll +0 -0
  34. langsc-2.2.2/LangSC/Graph/libglade-2.0-0.dll +0 -0
  35. langsc-2.2.2/LangSC/Graph/libglib-2.0-0.dll +0 -0
  36. langsc-2.2.2/LangSC/Graph/libgmodule-2.0-0.dll +0 -0
  37. langsc-2.2.2/LangSC/Graph/libgobject-2.0-0.dll +0 -0
  38. langsc-2.2.2/LangSC/Graph/libgthread-2.0-0.dll +0 -0
  39. langsc-2.2.2/LangSC/Graph/libgtkglext-win32-1.0-0.dll +0 -0
  40. langsc-2.2.2/LangSC/Graph/libpango-1.0-0.dll +0 -0
  41. langsc-2.2.2/LangSC/Graph/libpangocairo-1.0-0.dll +0 -0
  42. langsc-2.2.2/LangSC/Graph/libpangoft2-1.0-0.dll +0 -0
  43. langsc-2.2.2/LangSC/Graph/libpangowin32-1.0-0.dll +0 -0
  44. langsc-2.2.2/LangSC/Graph/libpng14-14.dll +0 -0
  45. langsc-2.2.2/LangSC/Graph/librsvg-2-2.dll +0 -0
  46. langsc-2.2.2/LangSC/Graph/libxml2.dll +0 -0
  47. langsc-2.2.2/LangSC/Graph/ltdl.dll +0 -0
  48. langsc-2.2.2/LangSC/Graph/zlib1.dll +0 -0
  49. langsc-2.2.2/LangSC.egg-info/SOURCES.txt +0 -72
  50. {langsc-2.2.2 → langsc-2.5.0}/LICENSE +0 -0
  51. {langsc-2.2.2 → langsc-2.5.0}/LangSC/BCCconfig.txt +0 -0
  52. {langsc-2.2.2 → langsc-2.5.0}/LangSC/GPFconfig.txt +0 -0
  53. {langsc-2.2.2 → langsc-2.5.0}/LangSC/Parser.lua +0 -0
  54. {langsc-2.2.2 → langsc-2.5.0}/LangSC/Segment.dat +0 -0
  55. {langsc-2.2.2 → langsc-2.5.0}/LangSC/__init__.py +0 -0
  56. {langsc-2.2.2 → langsc-2.5.0}/LangSC/base.lex +0 -0
  57. {langsc-2.2.2 → langsc-2.5.0}/LangSC/bcclib.dll +0 -0
  58. {langsc-2.2.2 → langsc-2.5.0}/LangSC/gpflib.dll +0 -0
  59. {langsc-2.2.2 → langsc-2.5.0}/LangSC/idxPOS.dat +0 -0
  60. {langsc-2.2.2 → langsc-2.5.0}/LangSC/jsslib.dll +0 -0
  61. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libbcclib.dylib +0 -0
  62. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libbcclib.so +0 -0
  63. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libgpflib.dylib +0 -0
  64. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libgpflib.so +0 -0
  65. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libjsslib.dylib +0 -0
  66. {langsc-2.2.2 → langsc-2.5.0}/LangSC/libjsslib.so +0 -0
  67. {langsc-2.2.2 → langsc-2.5.0}/LangSC.egg-info/dependency_links.txt +0 -0
  68. {langsc-2.2.2 → langsc-2.5.0}/LangSC.egg-info/not-zip-safe +0 -0
  69. {langsc-2.2.2 → langsc-2.5.0}/LangSC.egg-info/top_level.txt +0 -0
  70. {langsc-2.2.2 → langsc-2.5.0}/MANIFEST.in +0 -0
  71. {langsc-2.2.2 → langsc-2.5.0}/pyproject.toml +0 -0
  72. {langsc-2.2.2 → langsc-2.5.0}/setup.py +0 -0
  73. {langsc-2.2.2 → langsc-2.5.0}/tests/test_gpf_threadsafe.py +0 -0
  74. {langsc-2.2.2 → langsc-2.5.0}/tests/test_win_smoke.py +0 -0
@@ -2,6 +2,6 @@
2
2
  # Keep __version__ a plain string literal: setup.cfg reads it statically via
3
3
  # `version = attr: LangSC.__version__.__version__`, so it must stay AST-parseable
4
4
  # (no computed expression) to avoid importing the package at build time.
5
- __version__ = "2.2.2"
5
+ __version__ = "2.5.0"
6
6
 
7
7
  VERSION = tuple(int(x) for x in __version__.split("."))
@@ -2,9 +2,66 @@
2
2
 
3
3
  import os
4
4
  import re
5
+ import struct
6
+ import platform
5
7
  import chardet
6
8
 
7
9
 
10
+ def _win_short_path(path):
11
+ """返回 Windows 8.3 短路径(纯 ASCII);路径不存在或该卷禁用 8.3 名称时返回 None。"""
12
+ import ctypes
13
+ GetShortPathNameW = ctypes.windll.kernel32.GetShortPathNameW
14
+ GetShortPathNameW.restype = ctypes.c_uint
15
+ buf = ctypes.create_unicode_buffer(32768)
16
+ n = GetShortPathNameW(str(path), buf, len(buf))
17
+ if n == 0 or n >= len(buf):
18
+ return None
19
+ short = buf.value
20
+ try:
21
+ short.encode("ascii") # 8.3 被禁用时会原样返回长路径(仍含非 ASCII)→ 视为失败
22
+ except UnicodeEncodeError:
23
+ return None
24
+ return short
25
+
26
+
27
+ def native_path(path):
28
+ """把文件系统路径编码为原生 C 库 fopen() 所需的字节(跨平台)。
29
+
30
+ - POSIX(Linux/macOS):UTF-8 字节。POSIX fopen 直接按字节打开,文件系统即
31
+ UTF-8,故中文路径可用。
32
+ - Windows:按当前 ANSI 代码页(中文系统通常是 GBK)编码;若路径含该代码页
33
+ 无法表示的字符,则退化为其 8.3 短路径(纯 ASCII)。两者都能被引擎编译期
34
+ 绑定的 ANSI fopen 打开。
35
+
36
+ 解决:BCC/GPF 原先在 Linux/macOS 也发 GBK 路径 → 打不开中文目录;JSS 原先在
37
+ Windows 发 UTF-8 路径 → 打不开中文目录。统一由本函数按平台给出正确字节。
38
+ """
39
+ if platform.system() != "Windows":
40
+ return os.fsencode(path) # Linux/macOS:UTF-8(含 surrogateescape)
41
+ try:
42
+ return path.encode("mbcs") # Windows:当前 ANSI 代码页(GBK),与现状一致
43
+ except UnicodeEncodeError:
44
+ short = _win_short_path(path)
45
+ if short is not None:
46
+ return short.encode("mbcs")
47
+ raise ValueError(
48
+ "路径含当前 ANSI 代码页无法表示的字符,且无法获取 8.3 短路径"
49
+ "(该卷可能禁用了 8.3 名称):%r。请改用 ASCII 路径,或在该卷启用 8.3 名称。"
50
+ % path
51
+ )
52
+
53
+
54
+ def format_native_load_error(lib_name, err):
55
+ """构造原生库(.dll/.so/.dylib)加载失败时的可诊断错误信息。"""
56
+ bits = 8 * struct.calcsize("P")
57
+ return (
58
+ "LangSC 无法加载原生库 {lib}(系统={sys} 架构={mach} Python={bits}位):{err}\n"
59
+ "请确认:1) 安装的 LangSC 与当前系统/架构匹配(建议从 PyPI 重新安装以获取对应平台的包);"
60
+ "2) Windows 需已安装 VC++ 运行库;3) Linux/macOS 需确保该库的依赖项齐全。"
61
+ ).format(lib=lib_name, sys=platform.system(), mach=platform.machine(),
62
+ bits=bits, err=err)
63
+
64
+
8
65
  def detect_file_encoding(file_path, sample_size=1024*10):
9
66
  """Detect file encoding by BOM or chardet."""
10
67
  with open(file_path, 'rb') as f:
@@ -26,20 +83,30 @@ def detect_file_encoding(file_path, sample_size=1024*10):
26
83
 
27
84
  result = chardet.detect(raw_sample)
28
85
  encoding = result['encoding']
29
-
30
- if encoding is None:
31
- return 'ascii'
32
- elif encoding.lower() == 'gb2312':
86
+ confidence = result.get('confidence') or 0
87
+
88
+ # 检测不出(空/过短/二进制)时回退 utf-8 而非 ascii:ascii 对任何非 ASCII 字节
89
+ # 都会在后续 open 时抛行级 decode 错;utf-8 是 ascii 的超集,更安全的默认。
90
+ if encoding is None or confidence < 0.3:
91
+ return 'utf-8'
92
+ enc = encoding.lower()
93
+ # gb2312/gb18030 统一按 gbk 家族处理(python 的 gbk/gb18030 均可解码)
94
+ if enc == 'gb2312':
33
95
  return 'gbk'
34
- else:
35
- return encoding
96
+ return encoding
36
97
 
37
98
 
38
99
  def is_file_format(file_path):
39
- """Determine file type: 'Table', 'FSA', or 'BCC'."""
100
+ """Determine file type: 'Table', 'FSA', or 'BCC'.
101
+
102
+ 用探测到的编码 + errors='ignore' 打开:避免 GBK 的 Table/FSA 文件在 UTF-8
103
+ locale 系统上因默认编码解码失败而被裸 except 吞掉、误判成 'BCC'。判别前缀
104
+ ('FSA '/'Table ') 均为 ASCII,errors='ignore' 不影响匹配。
105
+ """
40
106
  ret = "BCC"
41
107
  try:
42
- with open(file_path, "rt") as f:
108
+ encoding = detect_file_encoding(file_path)
109
+ with open(file_path, "rt", encoding=encoding, errors="ignore") as f:
43
110
  no = 0
44
111
  for line in f:
45
112
  line = line.strip()
@@ -52,7 +119,7 @@ def is_file_format(file_path):
52
119
  if re.search('^Table ', line):
53
120
  ret = "Table"
54
121
  break
55
- except:
122
+ except Exception:
56
123
  return ret
57
124
  return ret
58
125
 
@@ -189,18 +256,29 @@ def write_idx_log(path, file_path, idx_log):
189
256
 
190
257
 
191
258
  def is_file_list(filelistname):
192
- """Check if a file contains a list of file paths."""
193
- with open(filelistname, "r") as f:
259
+ """Check if a file contains a list of file paths.
260
+
261
+ 用探测到的编码打开(GBK 清单在 UTF-8 locale 下不再解码失败),空文件判为非清单。
262
+ """
263
+ try:
264
+ encoding = detect_file_encoding(filelistname)
194
265
  is_file = True
195
266
  is_possible = True
196
- for line in f:
197
- if not os.path.isfile(line.strip()):
198
- is_file = False
199
- if line.find("\\") == -1 and line.find("/") == -1:
200
- is_possible = False
201
- if is_file or is_possible:
202
- return True
203
- return False
267
+ has_line = False
268
+ with open(filelistname, "r", encoding=encoding, errors="ignore") as f:
269
+ for line in f:
270
+ if line.strip() == "":
271
+ continue
272
+ has_line = True
273
+ if not os.path.isfile(line.strip()):
274
+ is_file = False
275
+ if line.find("\\") == -1 and line.find("/") == -1:
276
+ is_possible = False
277
+ except OSError:
278
+ return False
279
+ if not has_line:
280
+ return False
281
+ return bool(is_file or is_possible)
204
282
 
205
283
 
206
284
  def is_raw(file_path):
@@ -259,10 +337,18 @@ def check_words(all_words):
259
337
 
260
338
 
261
339
  def process_file(file_path, file_tmp, cmd, gpf=None):
262
- """Process a raw file into corpus format."""
340
+ """Process a raw file into corpus format.
341
+
342
+ 输出必须用 GBK 写:BCC 引擎按原始字节(二进制)读取语料并以 GBK 存储/匹配
343
+ (F2019.cpp 中有 GBK 字节哨兵,查询在 BCC_RunBCC 里由 UTF-8 转 GBK 后匹配)。
344
+ Windows 的默认文本编码恰为 GBK(cp936),故原先隐式可用;但 Linux/macOS 默认
345
+ UTF-8 会导致中文语料按 UTF-8 存入、与 GBK 查询无法匹配。显式 GBK 让各平台一致。
346
+ errors='replace':个别超出 GBK 的字符以占位符代替,避免整次索引崩溃(引擎本就
347
+ 只支持 GBK)。
348
+ """
263
349
  encoding = detect_file_encoding(file_path)
264
- with open(file_path, "r", encoding=encoding) as f_in:
265
- with open(file_tmp, "w") as f_out:
350
+ with open(file_path, "r", encoding=encoding, errors="replace") as f_in:
351
+ with open(file_tmp, "w", encoding="gbk", errors="replace") as f_out:
266
352
  print("Doc {}".format(file_path), file=f_out)
267
353
  for line in f_in:
268
354
  line = line.strip()
@@ -10,7 +10,7 @@ from ctypes import *
10
10
  from ._utils import (
11
11
  get_idx_info, get_file_info, is_same, get_bcc_files,
12
12
  write_idx_log, is_file_list, is_raw, process_file,
13
- file2corpus, corpus,
13
+ file2corpus, corpus, format_native_load_error, native_path,
14
14
  )
15
15
 
16
16
  OS = platform.system()
@@ -45,20 +45,26 @@ class BCC:
45
45
  cfg_file_bcc = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'BCCconfig.txt')
46
46
  cfg_file_parser = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'Parser.lua')
47
47
 
48
- self.library_bcc = cdll.LoadLibrary(dll_file_bcc)
48
+ try:
49
+ self.library_bcc = cdll.LoadLibrary(dll_file_bcc)
50
+ except OSError as e:
51
+ raise OSError(format_native_load_error(dll_name_bcc, e)) from e
49
52
 
50
53
  self.ParserBCC = cfg_file_parser
51
54
  self.ConfigBCC = cfg_file_bcc
52
55
  self.dataPath = dataPath
53
56
  self.RetBuff = create_string_buffer(''.encode(), self.buf_max_size)
54
57
 
58
+ self.dll_close = None
55
59
  if OS == "Windows":
56
60
  self.dll_close = win32api.FreeLibrary
57
- elif OS == "Linux":
61
+ else:
62
+ # Linux / macOS:通过 libc 的 dlclose 释放句柄(原先缺 macOS 分支)
58
63
  try:
59
64
  stdlib = CDLL("")
60
65
  except OSError:
61
- stdlib = CDLL("libc.so")
66
+ # libc.so 在部分发行版是链接脚本不可 dlopen,用 libc.so.6
67
+ stdlib = CDLL("libc.so.6" if OS == "Linux" else "libc.dylib")
62
68
  self.dll_close = stdlib.dlclose
63
69
  self.dll_close.argtypes = [c_void_p]
64
70
 
@@ -75,6 +81,22 @@ class BCC:
75
81
  def __del__(self):
76
82
  return
77
83
 
84
+ def _enc_path(self, p):
85
+ """把路径编码为原生 C API 所需字节:POSIX→UTF-8,Windows→ANSI/短路径。
86
+
87
+ 统一走 _utils.native_path,解决中文路径在 Linux/macOS 打不开的问题。
88
+ """
89
+ return native_path(p)
90
+
91
+ def _check_trunc(self, str_len):
92
+ """定长缓冲区被写满 → 结果很可能被截断,报错而非返回残缺数据(须在 string_at 前调用)。"""
93
+ if str_len >= self.buf_max_size:
94
+ raise RuntimeError(
95
+ "BCC 结果超出缓冲区上限(%d 字节)被截断。请缩小检索范围或分页(Number/PageNo)。"
96
+ % self.buf_max_size
97
+ )
98
+ return str_len
99
+
78
100
  def _init_bcc_data(self, path):
79
101
  """Initialize BCC data by scanning and indexing BCC corpus files.
80
102
 
@@ -132,14 +154,18 @@ class BCC:
132
154
  if IsBCCInit == 0:
133
155
  self.library_bcc.BCC_Init.argtypes = [c_char_p]
134
156
  self.library_bcc.BCC_Init.restype = c_int
135
- IsBCCInit = self.library_bcc.BCC_Init(self.dataPath.encode('gbk', errors='strict'))
157
+ IsBCCInit = self.library_bcc.BCC_Init(self._enc_path(self.dataPath))
136
158
  lock.release()
137
159
 
138
160
  if IsBCCInit == 0 and query.find("Lua") == -1:
139
- return json.loads("{}")
161
+ # 返回字符串 "{}"(而非 dict),保证 Run/_normalize 始终拿到字符串、
162
+ # 产出统一信封(records:[]);否则空结果会让 Run 返回 dict,调用方
163
+ # json.loads(Ret) 直接 TypeError。
164
+ return "{}"
140
165
  self.library_bcc.BCC_RunBCC.argtypes = [c_char_p, c_char_p, c_char_p, c_char_p]
141
166
  self.library_bcc.BCC_RunBCC.restype = c_int
142
- str_len = self.library_bcc.BCC_RunBCC(self.ParserBCC.encode(), self.dataPath.encode('gbk', errors='strict'), query.encode(), self.RetBuff)
167
+ str_len = self.library_bcc.BCC_RunBCC(self.ParserBCC.encode(), self._enc_path(self.dataPath), query.encode(), self.RetBuff)
168
+ self._check_trunc(str_len)
143
169
  ret = string_at(self.RetBuff, str_len)
144
170
  return ret.decode()
145
171
 
@@ -364,7 +390,7 @@ class BCC:
364
390
  filelist_ex = os.path.join(self.dataPath, "indexlistEx.tmp")
365
391
  gpf = self._get_gpf() if Param[0] == "Segment" else None
366
392
  file2corpus(filelist, filelist_ex, self.dataPath, Param[0], gpf=gpf)
367
- ret = self.library_bcc.BCC_IndexBCC(self.ConfigBCC.encode(), filelist_ex.encode('gbk', errors='strict'), self.dataPath.encode('gbk', errors='strict'))
393
+ ret = self.library_bcc.BCC_IndexBCC(self.ConfigBCC.encode(), self._enc_path(filelist_ex), self._enc_path(self.dataPath))
368
394
  os.remove(filelist_ex)
369
395
  if filelist.find("indexlist.tmp") != -1:
370
396
  os.remove(filelist)
@@ -421,7 +447,7 @@ class BCC:
421
447
  elif Command == "Freq":
422
448
  Operation = 'Freq({},{},{})'.format(Number, Target, ContextNum)
423
449
  elif Command == "Count":
424
- Operation = "Count()".format()
450
+ Operation = "Count()"
425
451
  else:
426
452
  Operation = "{}".format(Command)
427
453
  if not re.search(r"[\{\}]", Query):