PyUMSN 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pyumsn/lexer.py ADDED
@@ -0,0 +1,422 @@
1
+ """엄슨/파이썬 공통 렉서.
2
+
3
+ 소스를 토큰으로 나누되, 공백·주석·문자열은 원문 그대로 보존한다.
4
+ 토큰 텍스트를 순서대로 이어 붙이면 항상 원래 소스와 같다.
5
+
6
+ 토큰 종류(kind)
7
+ ws 공백, 줄바꿈, 줄 잇기(``\\`` + 줄바꿈)
8
+ comment ``#`` 주석
9
+ string 일반 문자열 (접두사 포함) .prefix
10
+ fstart f-string 시작 (접두사 + 여는 따옴표) .prefix
11
+ fmiddle f-string 의 글자 부분, 변환(!r), 서식 지정자, 디버그 ``=``
12
+ fopen f-string 치환 필드 ``{``
13
+ fclose f-string 치환 필드 ``}``
14
+ fend f-string 닫는 따옴표
15
+ name 이름(식별자)
16
+ word ``엄!`` / ``엄?`` 처럼 느낌표·물음표로 끝나는 엄슨 단어
17
+ dollar ``$이름`` (엄슨 탈출 이름)
18
+ number 숫자
19
+ sigil ``..한`` 같은 엄슨 기호어
20
+ op 연산자·문장부호 (ASCII)
21
+ error 알 수 없는 기호어 (관대 모드에서만)
22
+
23
+ ``fdepth`` 가 0 보다 크면 f-string 치환 필드 안의 식이다.
24
+ """
25
+
26
+ from bisect import bisect_right
27
+
28
+ from . import vocab
29
+ from .errors import UmsnSyntaxError
30
+
31
+ MODE_UMSN = "umsn"
32
+ MODE_PY = "py"
33
+
34
+ _OPERATORS = sorted([
35
+ "**=", "//=", ">>=", "<<=", "...", "->", ":=", "**", "//", ">>", "<<",
36
+ "<=", ">=", "==", "!=", "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "@=",
37
+ "+", "-", "*", "/", "%", "@", "&", "|", "^", "~", "<", ">",
38
+ "(", ")", "[", "]", "{", "}", ",", ":", ".", ";", "=", "!",
39
+ ], key=len, reverse=True)
40
+
41
+ _OPEN = {"(": 1, "[": 1, "{": 1, ")": -1, "]": -1, "}": -1}
42
+ _DIGITS = "0123456789"
43
+ _PY_PREFIXES = {"r", "u", "f", "b", "t", "br", "rb", "fr", "rf", "tr", "rt"}
44
+
45
+
46
+ class Token(object):
47
+ __slots__ = ("kind", "text", "start", "end", "prefix", "fdepth")
48
+
49
+ def __init__(self, kind, text, start, end, prefix=None, fdepth=0):
50
+ self.kind = kind
51
+ self.text = text
52
+ self.start = start
53
+ self.end = end
54
+ self.prefix = prefix
55
+ self.fdepth = fdepth
56
+
57
+ def __repr__(self):
58
+ return "Token(%s, %r)" % (self.kind, self.text)
59
+
60
+
61
+ def is_hangul(ch):
62
+ o = ord(ch)
63
+ return (0xAC00 <= o <= 0xD7A3 or 0x1100 <= o <= 0x11FF or 0x3131 <= o <= 0x318E
64
+ or 0xA960 <= o <= 0xA97F or 0xD7B0 <= o <= 0xD7FF)
65
+
66
+
67
+ def is_hangul_name(text):
68
+ """한글(음절·자모) + 숫자 + ``_`` 로만 된 이름인가."""
69
+ return bool(text) and all(is_hangul(ch) or ch == "_" or ch in _DIGITS for ch in text) \
70
+ and text[0] not in _DIGITS
71
+
72
+
73
+ def _id_start(ch):
74
+ return ch == "_" or ch.isidentifier()
75
+
76
+
77
+ def _id_char(ch):
78
+ return ch == "_" or ("a" + ch).isidentifier()
79
+
80
+
81
+ # ---------------------------------------------------------------------------
82
+ # 문자열 접두사 변환
83
+ # ---------------------------------------------------------------------------
84
+ def umsn_prefix_to_py(prefix):
85
+ """``형날`` → ``fr``, ``대형`` → ``F``. 올바르지 않으면 None."""
86
+ out = []
87
+ i = 0
88
+ while i < len(prefix):
89
+ upper = False
90
+ if prefix[i] == vocab.UPPER_MARK:
91
+ upper = True
92
+ i += 1
93
+ if i >= len(prefix):
94
+ return None
95
+ ch = vocab.PREFIX_REVERSE.get(prefix[i])
96
+ if ch is None:
97
+ return None
98
+ out.append(ch.upper() if upper else ch)
99
+ i += 1
100
+ result = "".join(out)
101
+ if result.lower() not in _PY_PREFIXES:
102
+ return None
103
+ return result
104
+
105
+
106
+ def py_prefix_to_umsn(prefix):
107
+ out = []
108
+ for ch in prefix:
109
+ syl = vocab.STRING_PREFIX_CHARS.get(ch.lower())
110
+ if syl is None:
111
+ return None
112
+ out.append(vocab.UPPER_MARK + syl if ch.isupper() else syl)
113
+ return "".join(out)
114
+
115
+
116
+ class Lexer(object):
117
+ def __init__(self, src, mode, tolerant=False):
118
+ self.src = src
119
+ self.n = len(src)
120
+ self.mode = mode
121
+ self.tolerant = tolerant
122
+ self.tokens = []
123
+ self._line_starts = None
124
+
125
+ # -- 위치 계산 --------------------------------------------------------
126
+ def line_col(self, pos):
127
+ if self._line_starts is None:
128
+ starts = [0]
129
+ src = self.src
130
+ i = src.find("\n")
131
+ while i != -1:
132
+ starts.append(i + 1)
133
+ i = src.find("\n", i + 1)
134
+ self._line_starts = starts
135
+ idx = bisect_right(self._line_starts, pos) - 1
136
+ return idx + 1, pos - self._line_starts[idx]
137
+
138
+ def fail(self, msg, pos, length=1, incomplete=False):
139
+ lineno, col = self.line_col(pos)
140
+ raise UmsnSyntaxError([(lineno, col, length, msg)], source=self.src,
141
+ incomplete=incomplete)
142
+
143
+ def emit(self, kind, start, end, prefix=None, fdepth=0):
144
+ self.tokens.append(Token(kind, self.src[start:end], start, end, prefix, fdepth))
145
+
146
+ # -- 진입점 -----------------------------------------------------------
147
+ def run(self):
148
+ self._code(0, 0, None)
149
+ return self.tokens
150
+
151
+ # -- 코드 -------------------------------------------------------------
152
+ def _code(self, pos, fdepth, fquote):
153
+ """코드를 토큰으로 나눈다. f-string 식 안이면 종결 문자에서 멈추고 위치를 돌려준다."""
154
+ src, n = self.src, self.n
155
+ depth = 0
156
+ umsn = self.mode == MODE_UMSN
157
+ while pos < n:
158
+ c = src[pos]
159
+ if fdepth and depth <= 0:
160
+ # PEP 701: 필드 안의 따옴표는 (바깥과 같은 따옴표라도) 중첩 문자열의 시작이다.
161
+ if c in "}:" or (c == "!" and src[pos + 1:pos + 2] != "="):
162
+ return pos
163
+ # 공백
164
+ if c in " \t\f\r\n":
165
+ end = pos + 1
166
+ while end < n and src[end] in " \t\f\r\n":
167
+ end += 1
168
+ self.emit("ws", pos, end, fdepth=fdepth)
169
+ pos = end
170
+ continue
171
+ if c == "\\" and src[pos + 1:pos + 2] in ("\n", "\r"):
172
+ end = pos + 2
173
+ if src[pos + 1:pos + 3] == "\r\n":
174
+ end = pos + 3
175
+ self.emit("ws", pos, end, fdepth=fdepth)
176
+ pos = end
177
+ continue
178
+ # 주석
179
+ if c == "#":
180
+ end = pos
181
+ while end < n and src[end] not in "\r\n":
182
+ end += 1
183
+ self.emit("comment", pos, end, fdepth=fdepth)
184
+ pos = end
185
+ continue
186
+ # 문자열 (접두사 없음)
187
+ if c in "\"'":
188
+ pos = self._string(pos, pos, "", fdepth)
189
+ continue
190
+ # 탈출 이름 $이름
191
+ if c == "$" and umsn:
192
+ end = pos + 1
193
+ if end < n and _id_start(src[end]):
194
+ end += 1
195
+ while end < n and _id_char(src[end]):
196
+ end += 1
197
+ self.emit("dollar", pos, end, fdepth=fdepth)
198
+ pos = end
199
+ continue
200
+ if self.tolerant:
201
+ self.emit("error", pos, pos + 1, fdepth=fdepth)
202
+ pos += 1
203
+ continue
204
+ self.fail("'$' 뒤에는 한글 이름이 와야 합니다.", pos)
205
+ # 이름 / 단어 / 접두사 문자열
206
+ if _id_start(c):
207
+ end = pos + 1
208
+ while end < n and _id_char(src[end]):
209
+ end += 1
210
+ name = src[pos:end]
211
+ nxt = src[end:end + 1]
212
+ if nxt and nxt in "\"'":
213
+ if umsn:
214
+ if umsn_prefix_to_py(name) is not None:
215
+ pos = self._string(pos, end, name, fdepth)
216
+ continue
217
+ if name.lower() in _PY_PREFIXES:
218
+ if not self.tolerant:
219
+ self.fail("영문 문자열 접두사 '%s' 는 쓸 수 없슨! %s 처럼 한글 접두사를 쓰세요."
220
+ % (name, (py_prefix_to_umsn(name) or "형") + nxt + "..." + nxt),
221
+ pos, len(name))
222
+ pos = self._string(pos, end, name, fdepth, py_prefix=name)
223
+ continue
224
+ elif name.lower() in _PY_PREFIXES:
225
+ pos = self._string(pos, end, name, fdepth)
226
+ continue
227
+ if umsn and nxt and nxt in "!?":
228
+ word = name + nxt
229
+ if word in vocab.UMSN2PY and not (nxt == "!" and src[end + 1:end + 2] == "="):
230
+ self.emit("word", pos, end + 1, fdepth=fdepth)
231
+ pos = end + 1
232
+ continue
233
+ self.emit("name", pos, end, fdepth=fdepth)
234
+ pos = end
235
+ continue
236
+ # 숫자
237
+ if c in _DIGITS or (c == "." and pos + 1 < n and src[pos + 1] in _DIGITS):
238
+ end = self._number(pos)
239
+ self.emit("number", pos, end, fdepth=fdepth)
240
+ pos = end
241
+ continue
242
+ # 엄슨 기호어 ..한
243
+ if umsn and c == "." and src[pos + 1:pos + 2] == "." and pos + 2 < n and is_hangul(src[pos + 2]):
244
+ for word in vocab.SYMBOL_WORDS_BY_LENGTH:
245
+ if src.startswith(word, pos):
246
+ self.emit("sigil", pos, pos + len(word), fdepth=fdepth)
247
+ depth += _OPEN.get(vocab.UMSN2SYMBOL[word], 0)
248
+ pos += len(word)
249
+ break
250
+ else:
251
+ end = pos + 2
252
+ while end < n and _id_char(src[end]):
253
+ end += 1
254
+ if self.tolerant:
255
+ self.emit("error", pos, end, fdepth=fdepth)
256
+ pos = end
257
+ continue
258
+ self.fail("알 수 없는 엄슨 기호어 '%s' 입니다." % src[pos:end], pos, end - pos)
259
+ continue
260
+ # 연산자
261
+ for op in _OPERATORS:
262
+ if src.startswith(op, pos):
263
+ break
264
+ else:
265
+ op = c
266
+ if fdepth and depth <= 0 and op == "=":
267
+ # f"{x=}" 디버그 표시
268
+ look = pos + 1
269
+ while look < n and src[look] in " \t":
270
+ look += 1
271
+ if src[look:look + 1] in ("}", "!", ":") and src[look:look + 1]:
272
+ return pos
273
+ self.emit("op", pos, pos + len(op), fdepth=fdepth)
274
+ depth += _OPEN.get(op, 0)
275
+ pos += len(op)
276
+ if fdepth and not self.tolerant:
277
+ self.fail("f-string 의 중괄호 '{' 가 닫히지 않았슨!", pos, incomplete=True)
278
+ return pos
279
+
280
+ # -- 숫자 -------------------------------------------------------------
281
+ def _number(self, pos):
282
+ src, n = self.src, self.n
283
+ i = pos
284
+ if src[i] == "0" and src[i + 1:i + 2] in ("x", "X", "o", "O", "b", "B") and src[i + 1:i + 2]:
285
+ i += 2
286
+ while i < n and (src[i] in _DIGITS or src[i] == "_" or ("a" <= src[i].lower() <= "f")):
287
+ i += 1
288
+ return i
289
+ while i < n and (src[i] in _DIGITS or src[i] == "_"):
290
+ i += 1
291
+ if i < n and src[i] == ".":
292
+ if not (self.mode == MODE_UMSN and src[i + 1:i + 2] == "."):
293
+ i += 1
294
+ while i < n and (src[i] in _DIGITS or src[i] == "_"):
295
+ i += 1
296
+ if i < n and src[i] in "eE":
297
+ j = i + 1
298
+ if j < n and src[j] in "+-":
299
+ j += 1
300
+ if j < n and src[j] in _DIGITS:
301
+ i = j
302
+ while i < n and (src[i] in _DIGITS or src[i] == "_"):
303
+ i += 1
304
+ if i < n and src[i] in "jJ":
305
+ i += 1
306
+ return i
307
+
308
+ # -- 문자열 -----------------------------------------------------------
309
+ def _string(self, start, qpos, prefix, fdepth, py_prefix=None):
310
+ """start: 접두사 시작, qpos: 여는 따옴표 위치. 끝난 위치를 돌려준다."""
311
+ src, n = self.src, self.n
312
+ if py_prefix is None:
313
+ py_prefix = umsn_prefix_to_py(prefix) if (self.mode == MODE_UMSN and prefix) else prefix
314
+ low = (py_prefix or "").lower()
315
+ raw = "r" in low
316
+ is_f = "f" in low or "t" in low
317
+ q = src[qpos]
318
+ quote = q * 3 if src.startswith(q * 3, qpos) else q
319
+ triple = len(quote) == 3
320
+ pos = qpos + len(quote)
321
+ if not is_f:
322
+ while True:
323
+ if pos >= n:
324
+ self._unterminated(start, triple)
325
+ self.emit("string", start, n, prefix=prefix, fdepth=fdepth)
326
+ return n
327
+ c = src[pos]
328
+ if c == "\\":
329
+ pos += 2
330
+ continue
331
+ if src.startswith(quote, pos):
332
+ pos += len(quote)
333
+ self.emit("string", start, pos, prefix=prefix, fdepth=fdepth)
334
+ return pos
335
+ if c in "\r\n" and not triple:
336
+ self._unterminated(start, False)
337
+ self.emit("string", start, pos, prefix=prefix, fdepth=fdepth)
338
+ return pos
339
+ pos += 1
340
+ # f-string
341
+ self.emit("fstart", start, pos, prefix=prefix, fdepth=fdepth)
342
+ return self._fbody(pos, quote, raw, fdepth, start, in_spec=False)
343
+
344
+ def _unterminated(self, start, triple):
345
+ if not self.tolerant:
346
+ self.fail("문자열이 닫히지 않았슨!", start, 1, incomplete=triple)
347
+
348
+ def _fbody(self, pos, quote, raw, fdepth, start, in_spec):
349
+ """f-string 본문(또는 서식 지정자)을 처리한다."""
350
+ src, n = self.src, self.n
351
+ triple = len(quote) == 3
352
+ lit = pos
353
+
354
+ def flush(upto):
355
+ if upto > lit:
356
+ self.emit("fmiddle", lit, upto, fdepth=fdepth)
357
+
358
+ while True:
359
+ if pos >= n:
360
+ flush(n)
361
+ if not self.tolerant:
362
+ self.fail("문자열이 닫히지 않았슨!", start, 1, incomplete=triple)
363
+ return n
364
+ c = src[pos]
365
+ if c == "\\":
366
+ if not raw and src.startswith("N{", pos + 1):
367
+ close = src.find("}", pos)
368
+ pos = close + 1 if close != -1 else n
369
+ else:
370
+ pos += 2
371
+ continue
372
+ if in_spec and c == "}":
373
+ flush(pos)
374
+ return pos
375
+ if not in_spec and src.startswith(quote, pos):
376
+ flush(pos)
377
+ self.emit("fend", pos, pos + len(quote), fdepth=fdepth)
378
+ return pos + len(quote)
379
+ if c in "\r\n" and not triple:
380
+ flush(pos)
381
+ if not self.tolerant:
382
+ self.fail("문자열이 닫히지 않았슨!", start)
383
+ return pos
384
+ if not in_spec and c in "{}" and src[pos + 1:pos + 2] == c:
385
+ pos += 2
386
+ continue
387
+ if c == "{":
388
+ flush(pos)
389
+ pos = self._field(pos, quote, raw, fdepth, start)
390
+ lit = pos
391
+ continue
392
+ pos += 1
393
+
394
+ def _field(self, pos, quote, raw, fdepth, start):
395
+ """``{`` 부터 짝이 맞는 ``}`` 까지."""
396
+ src, n = self.src, self.n
397
+ self.emit("fopen", pos, pos + 1, fdepth=fdepth)
398
+ pos = self._code(pos + 1, fdepth + 1, quote)
399
+ lit = pos
400
+ if pos < n and src[pos] == "=":
401
+ pos += 1
402
+ while pos < n and src[pos] in " \t":
403
+ pos += 1
404
+ if pos < n and src[pos] == "!":
405
+ pos += 1
406
+ while pos < n and src[pos].isalpha():
407
+ pos += 1
408
+ if pos > lit:
409
+ self.emit("fmiddle", lit, pos, fdepth=fdepth)
410
+ if pos < n and src[pos] == ":":
411
+ pos = self._fbody(pos, quote, raw, fdepth, start, in_spec=True)
412
+ if pos < n and src[pos] == "}":
413
+ self.emit("fclose", pos, pos + 1, fdepth=fdepth)
414
+ return pos + 1
415
+ if not self.tolerant:
416
+ self.fail("f-string 의 중괄호 '{' 가 닫히지 않았슨!", pos, incomplete=pos >= n)
417
+ return pos
418
+
419
+
420
+ def tokenize(src, mode=MODE_UMSN, tolerant=False):
421
+ """소스를 토큰 목록으로 나눈다."""
422
+ return Lexer(src, mode, tolerant).run()
pyumsn/repl.py ADDED
@@ -0,0 +1,32 @@
1
+ """엄슨 대화형 셸."""
2
+
3
+ import code
4
+ import sys
5
+
6
+ from . import __version__
7
+ from .errors import UmsnSyntaxError
8
+ from .importer import install
9
+ from .translator import umsn_to_py
10
+
11
+
12
+ class UmsnConsole(code.InteractiveConsole):
13
+ def runsource(self, source, filename="<엄슨>", symbol="single"):
14
+ try:
15
+ py = umsn_to_py(source)
16
+ except UmsnSyntaxError as exc:
17
+ if exc.incomplete:
18
+ return True
19
+ self.write(exc.report() + "\n")
20
+ return False
21
+ return code.InteractiveConsole.runsource(self, py, filename, symbol)
22
+
23
+
24
+ def main():
25
+ install()
26
+ sys.ps1 = "엄>>> "
27
+ sys.ps2 = "엄... "
28
+ banner = ("엄슨 %s (파이썬 %s)\n"
29
+ "끝내려면 엄나가..하다 또는 Ctrl+D (윈도우: Ctrl+Z 후 Enter)"
30
+ % (__version__, sys.version.split()[0]))
31
+ UmsnConsole(locals={"__name__": "__main__"}).interact(banner=banner, exitmsg="엄슨 안녕!")
32
+ return 0
pyumsn/runner.py ADDED
@@ -0,0 +1,118 @@
1
+ """엄슨 파일 실행기: 임시 파이썬 파일을 만들어 하위 프로세스로 실행한다."""
2
+
3
+ import os
4
+ import signal
5
+ import subprocess
6
+ import sys
7
+ import tempfile
8
+ from pathlib import Path
9
+
10
+ from .sourceio import read_source, write_source
11
+ from .translator import umsn_to_py
12
+
13
+ _PACKAGE_PARENT = str(Path(__file__).resolve().parent.parent)
14
+
15
+
16
+ KEEP_ENV = "PYUMSN_KEEP_TEMP"
17
+
18
+
19
+ def child_env(keep=False):
20
+ """하위 프로세스 환경: UTF-8 강제 + pyumsn 을 찾을 수 있게.
21
+
22
+ keep 이 거짓이면 launcher 가 임시 .py 를 읽자마자 지운다.
23
+ """
24
+ env = os.environ.copy()
25
+ if keep:
26
+ env[KEEP_ENV] = "1"
27
+ else:
28
+ env.pop(KEEP_ENV, None)
29
+ env["PYTHONUTF8"] = "1"
30
+ env["PYTHONIOENCODING"] = "utf-8"
31
+ old = env.get("PYTHONPATH")
32
+ env["PYTHONPATH"] = _PACKAGE_PARENT + (os.pathsep + old if old else "")
33
+ return env
34
+
35
+
36
+ def build_command(py_path, umsn_path, args=(), unbuffered=False):
37
+ """임시 .py 를 원본 .umsn 이름으로 실행하는 명령줄."""
38
+ cmd = [sys.executable, "-X", "utf8"]
39
+ if unbuffered:
40
+ cmd.append("-u")
41
+ cmd += ["-m", "pyumsn.launcher", str(py_path), str(umsn_path)]
42
+ cmd += [str(a) for a in args]
43
+ return cmd
44
+
45
+
46
+ def make_temp_py(py_source):
47
+ """UTF-8 임시 파이썬 파일을 만들고 경로를 돌려준다 (윈도우 잠금을 피하려고 바로 닫음)."""
48
+ fd, tmp = tempfile.mkstemp(prefix="umsn_", suffix=".py")
49
+ os.close(fd)
50
+ write_source(tmp, py_source)
51
+ return tmp
52
+
53
+
54
+ def prepare(umsn_path):
55
+ """엄슨 파일을 변환해 임시 .py 를 만든다. 문법 오류면 파일을 만들지 않고 예외."""
56
+ umsn_path = Path(umsn_path)
57
+ source = read_source(umsn_path)
58
+ py = umsn_to_py(source, filename=str(umsn_path))
59
+ return make_temp_py(py)
60
+
61
+
62
+ def remove_quietly(path):
63
+ try:
64
+ os.remove(path)
65
+ except OSError:
66
+ pass
67
+
68
+
69
+ def _catch_sigterm():
70
+ """SIGTERM 을 SystemExit 로 바꿔 finally 정리(임시 파일 삭제)가 되게 한다."""
71
+ if not hasattr(signal, "SIGTERM"):
72
+ return None
73
+ def handler(signum, frame):
74
+ raise SystemExit(128 + signum)
75
+ try:
76
+ return signal.signal(signal.SIGTERM, handler)
77
+ except ValueError: # 메인 스레드가 아님
78
+ return None
79
+
80
+
81
+ def _restore_sigterm(old):
82
+ if old is not None:
83
+ try:
84
+ signal.signal(signal.SIGTERM, old)
85
+ except ValueError:
86
+ pass
87
+
88
+
89
+ def run_file(path, args=(), keep=False, show_py=False):
90
+ """엄슨 파일을 실행하고 종료 코드를 돌려준다."""
91
+ path = Path(path).resolve()
92
+ tmp = prepare(path)
93
+ old_handler = _catch_sigterm()
94
+ try:
95
+ if show_py:
96
+ sys.stderr.write("---- 변환된 파이썬 (%s) ----\n" % tmp)
97
+ sys.stderr.write(read_source(tmp))
98
+ sys.stderr.write("\n---- 실행 ----\n")
99
+ sys.stderr.flush()
100
+ elif keep:
101
+ sys.stderr.write("임시 파이썬 파일: %s\n" % tmp)
102
+ sys.stderr.flush()
103
+ proc = subprocess.Popen(build_command(tmp, path, args), env=child_env(keep))
104
+ try:
105
+ return proc.wait()
106
+ except KeyboardInterrupt:
107
+ try:
108
+ return proc.wait(timeout=5)
109
+ except (subprocess.TimeoutExpired, KeyboardInterrupt):
110
+ proc.kill()
111
+ return 130
112
+ except SystemExit:
113
+ proc.terminate()
114
+ raise
115
+ finally:
116
+ _restore_sigterm(old_handler)
117
+ if not keep:
118
+ remove_quietly(tmp)
pyumsn/sourceio.py ADDED
@@ -0,0 +1,53 @@
1
+ """UTF-8 전용 소스 입출력.
2
+
3
+ 엄슨은 내부에서 무조건 UTF-8 만 쓴다. 다른 인코딩은 추측하거나 대체하지 않고 거부한다.
4
+ """
5
+
6
+ import codecs
7
+ import re
8
+ from pathlib import Path
9
+
10
+ from .errors import UmsnEncodingError
11
+
12
+ _CODING_RE = re.compile(r"^[ \t\f]*#.*?coding[:=][ \t]*([-\w.]+)")
13
+ _UTF8_NAMES = {"utf-8", "utf8", "utf_8", "utf-8-sig", "utf8-sig", "utf_8_sig", "u8"}
14
+
15
+
16
+ def decode_source(data, filename="<소스>"):
17
+ """바이트를 UTF-8 로 엄격하게 디코딩한다 (BOM 은 허용하고 제거)."""
18
+ if data.startswith(codecs.BOM_UTF8):
19
+ data = data[len(codecs.BOM_UTF8):]
20
+ elif data.startswith((codecs.BOM_UTF16_LE, codecs.BOM_UTF16_BE,
21
+ codecs.BOM_UTF32_LE, codecs.BOM_UTF32_BE)):
22
+ raise UmsnEncodingError(
23
+ "%s: UTF-16/UTF-32 파일은 쓸 수 없슨! UTF-8 로 저장하세요." % filename)
24
+ try:
25
+ text = data.decode("utf-8", "strict")
26
+ except UnicodeDecodeError as exc:
27
+ lineno = data.count(b"\n", 0, exc.start) + 1
28
+ raise UmsnEncodingError(
29
+ "%s: UTF-8 이 아닌 파일은 쓸 수 없슨! (%d줄, %d번째 바이트) "
30
+ "파일을 UTF-8 로 저장하세요." % (filename, lineno, exc.start)) from None
31
+ for line in text.splitlines()[:2]:
32
+ match = _CODING_RE.match(line)
33
+ if match:
34
+ name = match.group(1).lower()
35
+ if name not in _UTF8_NAMES:
36
+ raise UmsnEncodingError(
37
+ "%s: 인코딩 선언 '%s' 는 쓸 수 없슨! 엄슨은 UTF-8 만 씁니다."
38
+ % (filename, match.group(1)))
39
+ break
40
+ return text
41
+
42
+
43
+ def read_source(path):
44
+ """파일을 UTF-8 로 읽는다. 줄바꿈(CRLF 등)은 그대로 보존한다."""
45
+ path = Path(path)
46
+ return decode_source(path.read_bytes(), str(path))
47
+
48
+
49
+ def write_source(path, text):
50
+ """BOM 없는 UTF-8 로 쓴다. 줄바꿈은 주어진 그대로."""
51
+ path = Path(path)
52
+ path.write_bytes(text.encode("utf-8"))
53
+ return path