echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/text/sniff.py
ADDED
|
@@ -0,0 +1,1202 @@
|
|
|
1
|
+
"""Whether a file may be read as text at all, and in which encoding.
|
|
2
|
+
|
|
3
|
+
F-32 forbids trusting the extension, so every decision here is taken from the
|
|
4
|
+
bytes; a name, when one is known, only decides whether F-35 wants the user to
|
|
5
|
+
confirm before the content is accepted. F-33 forbids two things this module
|
|
6
|
+
therefore does not contain: any suggestion that renaming the file will help,
|
|
7
|
+
and any automatic upload or online conversion. There is deliberately no
|
|
8
|
+
network import anywhere in this module, and ``tests/test_sniff.py`` asserts
|
|
9
|
+
that rather than trusting review to notice one being added later.
|
|
10
|
+
|
|
11
|
+
F-34 fixes the encoding order. UTF-8 is tried first and CP949 only as a
|
|
12
|
+
fallback, because the asymmetry runs one way: CP949 text is rarely valid
|
|
13
|
+
UTF-8 by accident, while UTF-8 text is routinely valid CP949. When neither
|
|
14
|
+
is confident the file is *not* decoded with ``errors="replace"`` and handed
|
|
15
|
+
on -- F-34 forbids exactly that -- the verdict carries the candidate list and
|
|
16
|
+
a preview instead, so a person can choose (F-34) or an automated caller
|
|
17
|
+
receives a correctable error (F-37).
|
|
18
|
+
|
|
19
|
+
The module is pure: it opens nothing, writes nothing, and keeps no state.
|
|
20
|
+
``echoact.text.loader`` owns the filesystem. That split is what makes F-32's
|
|
21
|
+
"on failure the existing input, documents, and playback job are unchanged"
|
|
22
|
+
true by construction rather than by discipline.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import io
|
|
28
|
+
import re
|
|
29
|
+
import struct
|
|
30
|
+
import unicodedata
|
|
31
|
+
import zipfile
|
|
32
|
+
import zlib
|
|
33
|
+
from dataclasses import dataclass, field
|
|
34
|
+
from enum import StrEnum
|
|
35
|
+
from typing import Any, Final
|
|
36
|
+
|
|
37
|
+
from ..errors import Code, EchoActError, Problem
|
|
38
|
+
|
|
39
|
+
#: Bytes the *density* checks look at. Magic numbers sit in the first few
|
|
40
|
+
#: dozen; control-character density is sampled rather than counted over the
|
|
41
|
+
#: whole file, because a file whose first 8 KiB are clean prose is prose and
|
|
42
|
+
#: the cost of deciding then stays flat at the 2,000,000-byte import ceiling.
|
|
43
|
+
#: A NUL byte is not sampled this way -- it is a single decisive byte rather
|
|
44
|
+
#: than a proportion, one that must not reach the source text from anywhere
|
|
45
|
+
#: in the file, and finding one is a memchr over 2,000,000 bytes at worst.
|
|
46
|
+
SNIFF_WINDOW_BYTES: Final = 8192
|
|
47
|
+
|
|
48
|
+
#: Bytes of a ZIP ``mimetype`` entry that are read to identify the package.
|
|
49
|
+
#: The longest name matched is 33 bytes; the rest of the allowance is for a
|
|
50
|
+
#: trailing newline and for telling one OpenDocument subtype from another.
|
|
51
|
+
MIMETYPE_PROBE_BYTES: Final = 128
|
|
52
|
+
|
|
53
|
+
#: Code points shown to the user so an encoding can be judged (F-34, F-35).
|
|
54
|
+
PREVIEW_CODEPOINTS: Final = 400
|
|
55
|
+
|
|
56
|
+
#: Share of C0 control characters above which a window is not prose. Real
|
|
57
|
+
#: text carries tab, newline, carriage return, form feed and vertical tab and
|
|
58
|
+
#: essentially nothing else; a handful of stray escapes in a long log must
|
|
59
|
+
#: not condemn it, hence the absolute floor as well.
|
|
60
|
+
MAX_CONTROL_RATIO: Final = 0.05
|
|
61
|
+
MIN_CONTROL_COUNT: Final = 3
|
|
62
|
+
|
|
63
|
+
#: What F-02 supports, named the way F-33 requires the notice to name it.
|
|
64
|
+
SUPPORTED_FORMATS: Final = ("TXT", "Markdown")
|
|
65
|
+
|
|
66
|
+
#: F-34's automatic pair, in the order they are attempted.
|
|
67
|
+
AUTO_ENCODINGS: Final = ("utf-8", "cp949")
|
|
68
|
+
|
|
69
|
+
#: What the encoding selector may offer once automatic reading has failed.
|
|
70
|
+
#: Deliberately excludes any encoding that cannot fail -- latin-1 decodes
|
|
71
|
+
#: every byte sequence, so offering it would be offering mojibake with no
|
|
72
|
+
#: signal that anything went wrong, which is F-34's silent corruption under
|
|
73
|
+
#: another name.
|
|
74
|
+
SELECTABLE_ENCODINGS: Final = ("utf-8", "cp949", "utf-16", "utf-16-le", "utf-16-be")
|
|
75
|
+
|
|
76
|
+
#: The selector entries whose *valid* text is full of NUL bytes. A file in
|
|
77
|
+
#: one of them written without a byte-order mark is precisely the case F-34's
|
|
78
|
+
#: override exists for, and it arrives at the NUL rule looking exactly like a
|
|
79
|
+
#: binary file. These two are therefore tried before that conclusion is
|
|
80
|
+
#: drawn: otherwise the one failure the override was designed for is the one
|
|
81
|
+
#: failure whose verdict never mentions it, and F-37's "correctable error"
|
|
82
|
+
#: carries nothing an automated caller could correct.
|
|
83
|
+
_BOM_LESS_WIDE_ENCODINGS: Final = ("utf-16-le", "utf-16-be")
|
|
84
|
+
|
|
85
|
+
#: Code-point ranges EchoAct actually reads aloud: F-04's Korean and English,
|
|
86
|
+
#: the punctuation and full-width forms that travel with them, and the Latin
|
|
87
|
+
#: letters of a European name in an otherwise Korean document.
|
|
88
|
+
_APP_SCRIPT_RANGES: Final = (
|
|
89
|
+
(0x0009, 0x000D),
|
|
90
|
+
(0x0020, 0x007E),
|
|
91
|
+
(0x00A0, 0x024F),
|
|
92
|
+
(0x1100, 0x11FF),
|
|
93
|
+
(0x2000, 0x206F),
|
|
94
|
+
(0x3000, 0x303F),
|
|
95
|
+
(0x3130, 0x318F),
|
|
96
|
+
(0xAC00, 0xD7A3),
|
|
97
|
+
(0xFF01, 0xFF60),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
#: Share of a decode that must fall in those ranges before the encoding is
|
|
101
|
+
#: offered as a candidate for bytes that otherwise look binary. Every
|
|
102
|
+
#: even-length byte string decodes as UTF-16 into *something*: twenty-two
|
|
103
|
+
#: bytes of ASCII prose with one NUL in them become eleven CJK ideographs,
|
|
104
|
+
#: which is a reading rather than a file. Without this the NUL rule would
|
|
105
|
+
#: hand a preview of ideograph soup to every binary file of even length.
|
|
106
|
+
MIN_APP_SCRIPT_RATIO: Final = 0.8
|
|
107
|
+
|
|
108
|
+
ENCODING_LABELS: Final = {
|
|
109
|
+
"utf-8": "UTF-8",
|
|
110
|
+
"utf-8-sig": "UTF-8 with byte-order mark",
|
|
111
|
+
"cp949": "CP949 (Korean, Windows ANSI)",
|
|
112
|
+
"utf-16": "UTF-16",
|
|
113
|
+
"utf-16-le": "UTF-16 little-endian",
|
|
114
|
+
"utf-16-be": "UTF-16 big-endian",
|
|
115
|
+
"utf-32": "UTF-32",
|
|
116
|
+
"utf-32-le": "UTF-32 little-endian",
|
|
117
|
+
"utf-32-be": "UTF-32 big-endian",
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
# Remedy sentences. F-33 requires the reason, the supported formats, and how
|
|
121
|
+
# to copy the text in or convert to TXT. It also forbids suggesting that a
|
|
122
|
+
# rename fixes anything and forbids offering an online conversion, so no
|
|
123
|
+
# string here may mention either; the tests scan these constants for that.
|
|
124
|
+
_COPY_IN: Final = (
|
|
125
|
+
"Open the file in an application that can display it, select the text, and paste it "
|
|
126
|
+
"into EchoAct."
|
|
127
|
+
)
|
|
128
|
+
_SAVE_AS_TXT: Final = (
|
|
129
|
+
"Save or export the content as a plain-text file (.txt) encoded in UTF-8, then open that file."
|
|
130
|
+
)
|
|
131
|
+
_TYPE_INSTEAD: Final = "Type or paste the text you want spoken directly into EchoAct."
|
|
132
|
+
|
|
133
|
+
_REMEDIES_DOCUMENT: Final = (_COPY_IN, _SAVE_AS_TXT)
|
|
134
|
+
_REMEDIES_IMAGE: Final = (
|
|
135
|
+
"EchoAct does not read text inside pictures; it has no text recognition.",
|
|
136
|
+
_TYPE_INSTEAD,
|
|
137
|
+
)
|
|
138
|
+
_REMEDIES_MEDIA: Final = (
|
|
139
|
+
"EchoAct does not transcribe audio or video.",
|
|
140
|
+
_TYPE_INSTEAD,
|
|
141
|
+
)
|
|
142
|
+
_REMEDIES_ARCHIVE: Final = (
|
|
143
|
+
"Unpack the archive with your own tool, then open a text file from it.",
|
|
144
|
+
_TYPE_INSTEAD,
|
|
145
|
+
)
|
|
146
|
+
_REMEDIES_EXECUTABLE: Final = (
|
|
147
|
+
"EchoAct never runs a file you open, and it will not read a program as text.",
|
|
148
|
+
_TYPE_INSTEAD,
|
|
149
|
+
)
|
|
150
|
+
_REMEDIES_ENCRYPTED: Final = (
|
|
151
|
+
"Open it in the application that created it, supply the password, and save an unprotected "
|
|
152
|
+
"plain-text copy.",
|
|
153
|
+
_COPY_IN,
|
|
154
|
+
)
|
|
155
|
+
_REMEDIES_CORRUPT: Final = (
|
|
156
|
+
"Try another copy of the file, or the original it was made from.",
|
|
157
|
+
_SAVE_AS_TXT,
|
|
158
|
+
)
|
|
159
|
+
_REMEDIES_BINARY: Final = (
|
|
160
|
+
"Choose a plain-text file (.txt) or a Markdown file (.md).",
|
|
161
|
+
_TYPE_INSTEAD,
|
|
162
|
+
)
|
|
163
|
+
_REMEDIES_ENCODING: Final = (
|
|
164
|
+
"Check the preview, then choose the encoding the file was written in.",
|
|
165
|
+
"Or re-save the file as UTF-8 in a text editor and open it again.",
|
|
166
|
+
)
|
|
167
|
+
_REMEDIES_EMPTY: Final = (
|
|
168
|
+
"Choose a file that contains text.",
|
|
169
|
+
_TYPE_INSTEAD,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class FileKind(StrEnum):
|
|
174
|
+
"""What the bytes turned out to be. ``UNKNOWN`` covers a file the module
|
|
175
|
+
never got to look at, such as one refused on size or on permissions."""
|
|
176
|
+
|
|
177
|
+
TEXT = "text"
|
|
178
|
+
EMPTY = "empty"
|
|
179
|
+
HTML = "html"
|
|
180
|
+
XML = "xml"
|
|
181
|
+
RTF = "rtf"
|
|
182
|
+
PDF = "pdf"
|
|
183
|
+
DOC = "doc"
|
|
184
|
+
DOCX = "docx"
|
|
185
|
+
XLS = "xls"
|
|
186
|
+
XLSX = "xlsx"
|
|
187
|
+
PPT = "ppt"
|
|
188
|
+
PPTX = "pptx"
|
|
189
|
+
HWP = "hwp"
|
|
190
|
+
HWPX = "hwpx"
|
|
191
|
+
EPUB = "epub"
|
|
192
|
+
OPENDOCUMENT = "opendocument"
|
|
193
|
+
OLE_DOCUMENT = "ole_document"
|
|
194
|
+
ARCHIVE = "archive"
|
|
195
|
+
IMAGE = "image"
|
|
196
|
+
AUDIO = "audio"
|
|
197
|
+
VIDEO = "video"
|
|
198
|
+
EXECUTABLE = "executable"
|
|
199
|
+
DATABASE = "database"
|
|
200
|
+
BINARY = "binary"
|
|
201
|
+
UNKNOWN = "unknown"
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
class Confidence(StrEnum):
|
|
205
|
+
"""How firmly the encoding was established.
|
|
206
|
+
|
|
207
|
+
``CERTAIN`` means the file said so itself with a byte-order mark, or the
|
|
208
|
+
bytes admit only one reading. ``LIKELY`` means one candidate decoded
|
|
209
|
+
strictly and the others did not. ``UNCERTAIN`` means nothing decoded,
|
|
210
|
+
which F-34 turns into a question rather than into a guess.
|
|
211
|
+
"""
|
|
212
|
+
|
|
213
|
+
CERTAIN = "certain"
|
|
214
|
+
LIKELY = "likely"
|
|
215
|
+
UNCERTAIN = "uncertain"
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
@dataclass(frozen=True, slots=True)
|
|
219
|
+
class EncodingCandidate:
|
|
220
|
+
"""One entry in F-34's encoding selector.
|
|
221
|
+
|
|
222
|
+
The preview of a candidate that does not decode cleanly is produced with
|
|
223
|
+
replacement characters *on purpose*: seeing the damage is how a person
|
|
224
|
+
rules the choice out. It is never a source of loaded text -- the loader
|
|
225
|
+
decodes strictly and only strictly.
|
|
226
|
+
"""
|
|
227
|
+
|
|
228
|
+
name: str
|
|
229
|
+
label: str
|
|
230
|
+
decodes_cleanly: bool
|
|
231
|
+
preview: str
|
|
232
|
+
failure_offset: int | None = None
|
|
233
|
+
failure_reason: str | None = None
|
|
234
|
+
|
|
235
|
+
def to_dict(self) -> dict[str, Any]:
|
|
236
|
+
return {
|
|
237
|
+
"encoding": self.name,
|
|
238
|
+
"label": self.label,
|
|
239
|
+
"decodes_cleanly": self.decodes_cleanly,
|
|
240
|
+
"failure_offset": self.failure_offset,
|
|
241
|
+
"failure_reason": self.failure_reason,
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
@dataclass(frozen=True, slots=True)
|
|
246
|
+
class Verdict:
|
|
247
|
+
"""The answer to "may this be read as text, and how".
|
|
248
|
+
|
|
249
|
+
One object carries every case F-32 requires to be distinguished --
|
|
250
|
+
unsupported, corrupt, encrypted, permission-denied, oversized, encoding
|
|
251
|
+
error -- because a caller that must distinguish them should not have to
|
|
252
|
+
catch a different exception for each. ``problem`` is ``None`` exactly
|
|
253
|
+
when the bytes may become text.
|
|
254
|
+
"""
|
|
255
|
+
|
|
256
|
+
kind: FileKind
|
|
257
|
+
readable_as_text: bool
|
|
258
|
+
encoding: str | None = None
|
|
259
|
+
confidence: Confidence = Confidence.UNCERTAIN
|
|
260
|
+
candidates: tuple[EncodingCandidate, ...] = ()
|
|
261
|
+
problem: Problem | None = None
|
|
262
|
+
#: F-35: an unknown or absent extension means the user sees the preview
|
|
263
|
+
#: and says yes before the text is accepted.
|
|
264
|
+
needs_confirmation: bool = False
|
|
265
|
+
preview: str = ""
|
|
266
|
+
#: True when ``preview`` contains replacement characters and therefore
|
|
267
|
+
#: shows damage rather than content.
|
|
268
|
+
preview_lossy: bool = False
|
|
269
|
+
byte_size: int = 0
|
|
270
|
+
#: The specific format behind a family kind, e.g. "png" inside IMAGE.
|
|
271
|
+
format_name: str = ""
|
|
272
|
+
detail: dict[str, Any] = field(default_factory=dict)
|
|
273
|
+
|
|
274
|
+
@property
|
|
275
|
+
def ok(self) -> bool:
|
|
276
|
+
return self.problem is None and self.readable_as_text
|
|
277
|
+
|
|
278
|
+
def to_dict(self) -> dict[str, Any]:
|
|
279
|
+
"""What a surface may show or send. Carries no path and no filename:
|
|
280
|
+
F-57 keeps internal paths out of error payloads and N-20 keeps them
|
|
281
|
+
out of logs, and this object reaches both."""
|
|
282
|
+
body: dict[str, Any] = {
|
|
283
|
+
"kind": self.kind.value,
|
|
284
|
+
"format": self.format_name,
|
|
285
|
+
"readable_as_text": self.readable_as_text,
|
|
286
|
+
"encoding": self.encoding,
|
|
287
|
+
"confidence": self.confidence.value,
|
|
288
|
+
"needs_confirmation": self.needs_confirmation,
|
|
289
|
+
"byte_size": self.byte_size,
|
|
290
|
+
"supported_formats": list(SUPPORTED_FORMATS),
|
|
291
|
+
}
|
|
292
|
+
if self.candidates:
|
|
293
|
+
body["encoding_candidates"] = [c.to_dict() for c in self.candidates]
|
|
294
|
+
if self.problem is not None:
|
|
295
|
+
body["code"] = self.problem.code.value
|
|
296
|
+
body["message"] = self.problem.message
|
|
297
|
+
body["remedies"] = list(self.problem.remedies)
|
|
298
|
+
if self.detail:
|
|
299
|
+
body["detail"] = dict(self.detail)
|
|
300
|
+
return body
|
|
301
|
+
|
|
302
|
+
def as_error(self) -> EchoActError:
|
|
303
|
+
"""The rejection as the one exception type that crosses a boundary.
|
|
304
|
+
|
|
305
|
+
F-33's remedies travel in ``detail`` so REST and MCP deliver the same
|
|
306
|
+
guidance the GUI shows; N-24 wants one set of error semantics, not a
|
|
307
|
+
richer one for the surface that happens to be in-process.
|
|
308
|
+
"""
|
|
309
|
+
if self.problem is None:
|
|
310
|
+
raise AssertionError("as_error() on a verdict with no problem")
|
|
311
|
+
detail: dict[str, Any] = {
|
|
312
|
+
"kind": self.kind.value,
|
|
313
|
+
"remedies": list(self.problem.remedies),
|
|
314
|
+
"supported_formats": list(SUPPORTED_FORMATS),
|
|
315
|
+
}
|
|
316
|
+
if self.format_name:
|
|
317
|
+
detail["format"] = self.format_name
|
|
318
|
+
if self.candidates:
|
|
319
|
+
detail["encoding_candidates"] = [c.to_dict() for c in self.candidates]
|
|
320
|
+
if self.needs_confirmation:
|
|
321
|
+
detail["needs_confirmation"] = True
|
|
322
|
+
detail.update(self.detail)
|
|
323
|
+
return EchoActError(self.problem.code, self.problem.message, detail=detail)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
# --------------------------------------------------------------- helpers ---
|
|
327
|
+
|
|
328
|
+
_OLE_MAGIC: Final = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
|
|
329
|
+
|
|
330
|
+
#: Extensions F-02 names. A name outside this set is not a refusal: it only
|
|
331
|
+
#: means F-35's confirmation is required, because the content still decides.
|
|
332
|
+
TEXT_EXTENSIONS: Final = frozenset({".txt", ".md", ".markdown", ".mkd", ".mdown", ".text"})
|
|
333
|
+
|
|
334
|
+
#: Names that claim to be something this app cannot read. Used only to say
|
|
335
|
+
#: "the name and the content disagree" in a verdict's detail; the bytes are
|
|
336
|
+
#: what decide, per F-32.
|
|
337
|
+
_NON_TEXT_EXTENSIONS: Final = frozenset(
|
|
338
|
+
{
|
|
339
|
+
".pdf", ".doc", ".docx", ".hwp", ".hwpx", ".epub", ".odt", ".rtf",
|
|
340
|
+
".xls", ".xlsx", ".ppt", ".pptx", ".zip", ".rar", ".7z", ".gz", ".tar",
|
|
341
|
+
".png", ".jpg", ".jpeg", ".gif", ".bmp", ".webp", ".tif", ".tiff",
|
|
342
|
+
".wav", ".mp3", ".mp4", ".m4a", ".mov", ".ogg", ".flac", ".avi", ".mkv",
|
|
343
|
+
".exe", ".dll", ".so", ".dylib", ".bin", ".db", ".sqlite", ".sqlite3",
|
|
344
|
+
}
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
_HTML_STARTS: Final = ("<!doctype html", "<html", "<head", "<body")
|
|
348
|
+
_PDF_ENCRYPT_RE: Final = re.compile(rb"/Encrypt\s*\d+\s+\d+\s+R")
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _u16(name: str) -> bytes:
|
|
352
|
+
"""A compound-file stream name as it appears in the raw bytes.
|
|
353
|
+
|
|
354
|
+
Directory entries in an OLE compound file store their names in UTF-16LE,
|
|
355
|
+
so searching for the ASCII spelling finds nothing. Reading the directory
|
|
356
|
+
properly would mean parsing the allocation tables of a format the app
|
|
357
|
+
refuses either way.
|
|
358
|
+
"""
|
|
359
|
+
return name.encode("utf-16-le")
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _extension(filename: str | None) -> str:
|
|
363
|
+
if not filename:
|
|
364
|
+
return ""
|
|
365
|
+
_, dot, ext = filename.rpartition(".")
|
|
366
|
+
return f".{ext.lower()}" if dot and ext else ""
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _count_control_bytes(sample: bytes) -> int:
|
|
370
|
+
printable_controls = (0x09, 0x0A, 0x0B, 0x0C, 0x0D)
|
|
371
|
+
controls = sum(1 for b in sample if b < 0x20 and b not in printable_controls)
|
|
372
|
+
return controls + sample.count(0x7F)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _byte_window_is_control_dense(sample: bytes) -> bool:
|
|
376
|
+
if not sample:
|
|
377
|
+
return False
|
|
378
|
+
controls = _count_control_bytes(sample)
|
|
379
|
+
return controls >= MIN_CONTROL_COUNT and controls / len(sample) > MAX_CONTROL_RATIO
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _text_is_control_dense(text: str) -> bool:
|
|
383
|
+
if not text:
|
|
384
|
+
return False
|
|
385
|
+
controls = sum(
|
|
386
|
+
1
|
|
387
|
+
for ch in text
|
|
388
|
+
if ch == "\ufffd" or (unicodedata.category(ch) == "Cc" and ch not in "\t\n\r\v\f")
|
|
389
|
+
)
|
|
390
|
+
return controls >= MIN_CONTROL_COUNT and controls / len(text) > MAX_CONTROL_RATIO
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _in_app_scripts(ch: str) -> bool:
|
|
394
|
+
point = ord(ch)
|
|
395
|
+
return any(low <= point <= high for low, high in _APP_SCRIPT_RANGES)
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _reads_as_app_script(text: str) -> bool:
|
|
399
|
+
"""Whether a decode produced something EchoAct could plausibly speak.
|
|
400
|
+
|
|
401
|
+
Used only to decide whether an encoding is worth *offering* for bytes
|
|
402
|
+
that otherwise look binary; it never admits text on its own, and a file
|
|
403
|
+
it rejects can still be opened by naming the encoding explicitly.
|
|
404
|
+
"""
|
|
405
|
+
if not text:
|
|
406
|
+
return False
|
|
407
|
+
return sum(1 for ch in text if _in_app_scripts(ch)) / len(text) >= MIN_APP_SCRIPT_RATIO
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _preview_of(text: str) -> str:
|
|
411
|
+
return text[:PREVIEW_CODEPOINTS]
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _problem_verdict(
|
|
415
|
+
kind: FileKind,
|
|
416
|
+
code: Code,
|
|
417
|
+
message: str,
|
|
418
|
+
remedies: tuple[str, ...],
|
|
419
|
+
*,
|
|
420
|
+
size: int,
|
|
421
|
+
format_name: str = "",
|
|
422
|
+
detail: dict[str, Any] | None = None,
|
|
423
|
+
needs_confirmation: bool = False,
|
|
424
|
+
candidates: tuple[EncodingCandidate, ...] = (),
|
|
425
|
+
preview: str = "",
|
|
426
|
+
preview_lossy: bool = False,
|
|
427
|
+
) -> Verdict:
|
|
428
|
+
return Verdict(
|
|
429
|
+
kind=kind,
|
|
430
|
+
readable_as_text=False,
|
|
431
|
+
problem=Problem(code=code, message=message, remedies=remedies),
|
|
432
|
+
byte_size=size,
|
|
433
|
+
format_name=format_name,
|
|
434
|
+
detail=detail or {},
|
|
435
|
+
needs_confirmation=needs_confirmation,
|
|
436
|
+
candidates=candidates,
|
|
437
|
+
preview=preview,
|
|
438
|
+
preview_lossy=preview_lossy,
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _unsupported(
|
|
443
|
+
kind: FileKind, format_name: str, size: int, remedies: tuple[str, ...], reason: str
|
|
444
|
+
) -> Verdict:
|
|
445
|
+
return _problem_verdict(
|
|
446
|
+
kind, Code.FILE_UNSUPPORTED, reason, remedies, size=size, format_name=format_name
|
|
447
|
+
)
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
# ------------------------------------------------------- format families ---
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _image_format(data: bytes) -> str:
|
|
454
|
+
if data.startswith(b"\x89PNG\r\n\x1a\n"):
|
|
455
|
+
return "png"
|
|
456
|
+
if data.startswith(b"\xff\xd8\xff"):
|
|
457
|
+
return "jpeg"
|
|
458
|
+
if data.startswith((b"GIF87a", b"GIF89a")):
|
|
459
|
+
return "gif"
|
|
460
|
+
if data.startswith(b"RIFF") and data[8:12] == b"WEBP":
|
|
461
|
+
return "webp"
|
|
462
|
+
if data.startswith((b"II*\x00", b"MM\x00*")):
|
|
463
|
+
return "tiff"
|
|
464
|
+
if data.startswith(b"\x00\x00\x01\x00"):
|
|
465
|
+
return "ico"
|
|
466
|
+
# "BM" is two printable letters, so a bitmap has to prove itself or a
|
|
467
|
+
# sentence beginning "BMW..." becomes an image: the reserved words must
|
|
468
|
+
# be zero and either the declared size or the pixel offset must fit.
|
|
469
|
+
if data.startswith(b"BM") and len(data) >= 14 and data[6:10] == b"\x00\x00\x00\x00":
|
|
470
|
+
declared = int.from_bytes(data[2:6], "little")
|
|
471
|
+
pixel_offset = int.from_bytes(data[10:14], "little")
|
|
472
|
+
if declared == len(data) or 0 < pixel_offset <= len(data):
|
|
473
|
+
return "bmp"
|
|
474
|
+
return ""
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
def _media_format(data: bytes) -> tuple[str, FileKind]:
|
|
478
|
+
if data.startswith(b"RIFF") and data[8:12] == b"WAVE":
|
|
479
|
+
return "wav", FileKind.AUDIO
|
|
480
|
+
if data.startswith(b"RIFF") and data[8:12] == b"AVI ":
|
|
481
|
+
return "avi", FileKind.VIDEO
|
|
482
|
+
if data.startswith(b"ID3"):
|
|
483
|
+
return "mp3", FileKind.AUDIO
|
|
484
|
+
if data.startswith((b"\xff\xfb", b"\xff\xf3", b"\xff\xf2")):
|
|
485
|
+
return "mp3", FileKind.AUDIO
|
|
486
|
+
if data.startswith(b"OggS"):
|
|
487
|
+
return "ogg", FileKind.AUDIO
|
|
488
|
+
if data.startswith(b"fLaC"):
|
|
489
|
+
return "flac", FileKind.AUDIO
|
|
490
|
+
if data[4:8] == b"ftyp":
|
|
491
|
+
brand = data[8:12].decode("ascii", "replace").strip()
|
|
492
|
+
kind = FileKind.AUDIO if brand in {"M4A", "M4B", "M4P"} else FileKind.VIDEO
|
|
493
|
+
return (f"iso-bmff ({brand})" if brand else "iso-bmff"), kind
|
|
494
|
+
if data.startswith(b"\x1aE\xdf\xa3"):
|
|
495
|
+
return "matroska", FileKind.VIDEO
|
|
496
|
+
return "", FileKind.BINARY
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
def _archive_format(data: bytes) -> str:
|
|
500
|
+
if data.startswith(b"Rar!\x1a\x07"):
|
|
501
|
+
return "rar"
|
|
502
|
+
if data.startswith(b"7z\xbc\xaf\x27\x1c"):
|
|
503
|
+
return "7z"
|
|
504
|
+
if data.startswith(b"\x1f\x8b"):
|
|
505
|
+
return "gzip"
|
|
506
|
+
if data.startswith(b"BZh") and data[3:4].isdigit():
|
|
507
|
+
return "bzip2"
|
|
508
|
+
if data.startswith(b"\xfd7zXZ\x00"):
|
|
509
|
+
return "xz"
|
|
510
|
+
if data.startswith(b"\x28\xb5\x2f\xfd"):
|
|
511
|
+
return "zstandard"
|
|
512
|
+
if data[257:262] == b"ustar":
|
|
513
|
+
return "tar"
|
|
514
|
+
return ""
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def _executable_format(data: bytes) -> str:
|
|
518
|
+
if data.startswith(b"\x7fELF"):
|
|
519
|
+
return "elf"
|
|
520
|
+
if data.startswith(
|
|
521
|
+
(b"\xfe\xed\xfa\xce", b"\xfe\xed\xfa\xcf", b"\xce\xfa\xed\xfe", b"\xcf\xfa\xed\xfe")
|
|
522
|
+
):
|
|
523
|
+
return "mach-o"
|
|
524
|
+
if data.startswith(b"\xca\xfe\xba\xbe"):
|
|
525
|
+
# Shared by a Mach-O universal binary and a Java class file; either
|
|
526
|
+
# way it is a program, and the notice says the same of both.
|
|
527
|
+
return "mach-o universal or java class"
|
|
528
|
+
if data.startswith(b"MZ"):
|
|
529
|
+
if len(data) >= 0x40:
|
|
530
|
+
pe_offset = int.from_bytes(data[0x3C:0x40], "little")
|
|
531
|
+
if 0 < pe_offset < len(data) - 4 and data[pe_offset : pe_offset + 4] == b"PE\x00\x00":
|
|
532
|
+
return "windows pe"
|
|
533
|
+
return "dos executable"
|
|
534
|
+
return ""
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def _pdf_verdict(data: bytes, size: int) -> Verdict:
|
|
538
|
+
"""PDF, and which of F-32's cases it is.
|
|
539
|
+
|
|
540
|
+
Encryption is read from the trailer's ``/Encrypt`` reference rather than
|
|
541
|
+
from any attempt to open the document, because the distinction F-32 wants
|
|
542
|
+
-- encrypted as against merely corrupt -- has to survive the file being
|
|
543
|
+
unreadable for either reason.
|
|
544
|
+
"""
|
|
545
|
+
tail = data[-8192:]
|
|
546
|
+
if _PDF_ENCRYPT_RE.search(data) or b"/Encrypt" in tail:
|
|
547
|
+
return _problem_verdict(
|
|
548
|
+
FileKind.PDF,
|
|
549
|
+
Code.FILE_ENCRYPTED,
|
|
550
|
+
"This is a password-protected PDF, so its content cannot be read.",
|
|
551
|
+
_REMEDIES_ENCRYPTED,
|
|
552
|
+
size=size,
|
|
553
|
+
format_name="pdf",
|
|
554
|
+
)
|
|
555
|
+
if b"%%EOF" not in tail and b"startxref" not in tail:
|
|
556
|
+
return _problem_verdict(
|
|
557
|
+
FileKind.PDF,
|
|
558
|
+
Code.FILE_CORRUPT,
|
|
559
|
+
"This PDF is missing its end marker, so it was truncated or damaged.",
|
|
560
|
+
_REMEDIES_CORRUPT,
|
|
561
|
+
size=size,
|
|
562
|
+
format_name="pdf",
|
|
563
|
+
)
|
|
564
|
+
return _unsupported(
|
|
565
|
+
FileKind.PDF,
|
|
566
|
+
"pdf",
|
|
567
|
+
size,
|
|
568
|
+
_REMEDIES_DOCUMENT,
|
|
569
|
+
"This is a PDF. EchoAct reads plain text only and does not extract text from a page "
|
|
570
|
+
"layout.",
|
|
571
|
+
)
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
_ZIP_CONTENT_RULES: Final = (
|
|
575
|
+
("word/document.xml", FileKind.DOCX, "docx", "Word document"),
|
|
576
|
+
("xl/workbook.xml", FileKind.XLSX, "xlsx", "Excel workbook"),
|
|
577
|
+
("ppt/presentation.xml", FileKind.PPTX, "pptx", "PowerPoint presentation"),
|
|
578
|
+
("Contents/content.hpf", FileKind.HWPX, "hwpx", "HWPX document"),
|
|
579
|
+
("Contents/section0.xml", FileKind.HWPX, "hwpx", "HWPX document"),
|
|
580
|
+
("META-INF/container.xml", FileKind.EPUB, "epub", "EPUB book"),
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
_MIMETYPE_RULES: Final = (
|
|
584
|
+
(b"application/epub+zip", FileKind.EPUB, "epub", "EPUB book"),
|
|
585
|
+
(b"application/hwp+zip", FileKind.HWPX, "hwpx", "HWPX document"),
|
|
586
|
+
(
|
|
587
|
+
b"application/vnd.oasis.opendocument",
|
|
588
|
+
FileKind.OPENDOCUMENT,
|
|
589
|
+
"opendocument",
|
|
590
|
+
"OpenDocument file",
|
|
591
|
+
),
|
|
592
|
+
)
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
#: Everything ``zipfile`` throws at a file that is damaged, hostile, or
|
|
596
|
+
#: merely in a corner of the format this app does not implement. It is a
|
|
597
|
+
#: named tuple rather than two inline ones because the list is not
|
|
598
|
+
#: guessable: the constructor raises ``NotImplementedError`` for a central
|
|
599
|
+
#: directory whose "version needed to extract" exceeds 63, reading raises
|
|
600
|
+
#: ``zlib.error`` for a damaged deflate stream and ``NotImplementedError``
|
|
601
|
+
#: for a compression method the standard library does not carry, and either
|
|
602
|
+
#: escaping is a raw non-``EchoActError`` out of ``sniff`` -- which CLAUDE.md
|
|
603
|
+
#: rule 3 forbids, and which F-32 needs told apart as FILE_CORRUPT instead.
|
|
604
|
+
#: ``MemoryError`` is deliberately absent: it is not a fact about the file.
|
|
605
|
+
_ZIP_FAILURES: Final = (
|
|
606
|
+
zipfile.BadZipFile,
|
|
607
|
+
zipfile.LargeZipFile,
|
|
608
|
+
NotImplementedError,
|
|
609
|
+
RuntimeError,
|
|
610
|
+
zlib.error,
|
|
611
|
+
struct.error,
|
|
612
|
+
OSError,
|
|
613
|
+
ValueError,
|
|
614
|
+
EOFError,
|
|
615
|
+
IndexError,
|
|
616
|
+
KeyError,
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def _zip_verdict(data: bytes, size: int) -> Verdict:
|
|
621
|
+
"""A ZIP container, and which document format it holds.
|
|
622
|
+
|
|
623
|
+
Section 2.7 lists DOCX, HWPX and EPUB as separate formats and all three
|
|
624
|
+
are ZIP files, so the signature alone cannot tell them apart and the
|
|
625
|
+
entry names have to be read. Only the central directory and, at most,
|
|
626
|
+
the first ``MIMETYPE_PROBE_BYTES`` of the ``mimetype`` entry are touched:
|
|
627
|
+
nothing else is extracted, because Section 2.7 puts decompression out of
|
|
628
|
+
scope and a file that cannot be read must stay unread.
|
|
629
|
+
|
|
630
|
+
That entry is read through ``open(...).read(n)`` rather than ``read()``.
|
|
631
|
+
The difference is not style: ``read()`` decompresses the whole entry
|
|
632
|
+
before anything can slice it, so a 400 KB file whose ``mimetype`` inflates
|
|
633
|
+
to 400 MB would allocate all of it inside a sniffer whose input is capped
|
|
634
|
+
at 2,000,000 bytes -- N-21's "no unbounded in-memory loading" and N-23's
|
|
635
|
+
"limits actually bound the work" both fail there, and A-05's *safe*
|
|
636
|
+
rejection becomes an out-of-memory kill. The bounded form stops the
|
|
637
|
+
decompressor at the length asked for. The entry's declared size is not
|
|
638
|
+
consulted, because it is written by whoever wrote the file.
|
|
639
|
+
"""
|
|
640
|
+
try:
|
|
641
|
+
with zipfile.ZipFile(io.BytesIO(data)) as zf:
|
|
642
|
+
infos = zf.infolist()
|
|
643
|
+
names = {info.filename for info in infos}
|
|
644
|
+
encrypted = any(info.flag_bits & 0x1 for info in infos)
|
|
645
|
+
mimetype = b""
|
|
646
|
+
if "mimetype" in names:
|
|
647
|
+
try:
|
|
648
|
+
with zf.open("mimetype") as entry:
|
|
649
|
+
mimetype = entry.read(MIMETYPE_PROBE_BYTES)
|
|
650
|
+
except _ZIP_FAILURES:
|
|
651
|
+
# A package whose declaration cannot be read is still
|
|
652
|
+
# identifiable from its entry names, and one that is not
|
|
653
|
+
# is still a ZIP: neither is a reason to raise.
|
|
654
|
+
mimetype = b""
|
|
655
|
+
except _ZIP_FAILURES:
|
|
656
|
+
return _problem_verdict(
|
|
657
|
+
FileKind.ARCHIVE,
|
|
658
|
+
Code.FILE_CORRUPT,
|
|
659
|
+
"This file begins like a ZIP container, but its index could not be read, so it is "
|
|
660
|
+
"damaged or incomplete.",
|
|
661
|
+
_REMEDIES_CORRUPT,
|
|
662
|
+
size=size,
|
|
663
|
+
format_name="zip",
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
kind, format_name, label = FileKind.ARCHIVE, "zip", "ZIP archive"
|
|
667
|
+
for prefix, mime_kind, mime_format, mime_label in _MIMETYPE_RULES:
|
|
668
|
+
if mimetype.startswith(prefix):
|
|
669
|
+
kind, format_name, label = mime_kind, mime_format, mime_label
|
|
670
|
+
break
|
|
671
|
+
else:
|
|
672
|
+
for entry, rule_kind, rule_format, rule_label in _ZIP_CONTENT_RULES:
|
|
673
|
+
if entry in names:
|
|
674
|
+
kind, format_name, label = rule_kind, rule_format, rule_label
|
|
675
|
+
break
|
|
676
|
+
else:
|
|
677
|
+
if "META-INF/MANIFEST.MF" in names:
|
|
678
|
+
kind, format_name, label = FileKind.EXECUTABLE, "jar", "Java archive"
|
|
679
|
+
|
|
680
|
+
if encrypted:
|
|
681
|
+
return _problem_verdict(
|
|
682
|
+
kind,
|
|
683
|
+
Code.FILE_ENCRYPTED,
|
|
684
|
+
f"This {label} has password-protected contents, so they cannot be read.",
|
|
685
|
+
_REMEDIES_ENCRYPTED,
|
|
686
|
+
size=size,
|
|
687
|
+
format_name=format_name,
|
|
688
|
+
)
|
|
689
|
+
if kind is FileKind.ARCHIVE:
|
|
690
|
+
return _unsupported(
|
|
691
|
+
kind,
|
|
692
|
+
format_name,
|
|
693
|
+
size,
|
|
694
|
+
_REMEDIES_ARCHIVE,
|
|
695
|
+
"This is a ZIP archive. EchoAct does not unpack archives.",
|
|
696
|
+
)
|
|
697
|
+
if kind is FileKind.EXECUTABLE:
|
|
698
|
+
return _unsupported(
|
|
699
|
+
kind,
|
|
700
|
+
format_name,
|
|
701
|
+
size,
|
|
702
|
+
_REMEDIES_EXECUTABLE,
|
|
703
|
+
"This is a Java archive, which is a program rather than a document.",
|
|
704
|
+
)
|
|
705
|
+
return _unsupported(
|
|
706
|
+
kind,
|
|
707
|
+
format_name,
|
|
708
|
+
size,
|
|
709
|
+
_REMEDIES_DOCUMENT,
|
|
710
|
+
f"This is a {label}. Its text is stored inside a document package that this version "
|
|
711
|
+
"does not open.",
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
_OLE_STREAM_RULES: Final = (
|
|
716
|
+
("WordDocument", FileKind.DOC, "doc", "Word 97-2003 document"),
|
|
717
|
+
("Workbook", FileKind.XLS, "xls", "Excel 97-2003 workbook"),
|
|
718
|
+
("PowerPoint Document", FileKind.PPT, "ppt", "PowerPoint 97-2003 presentation"),
|
|
719
|
+
)
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def _ole_verdict(data: bytes, size: int) -> Verdict:
|
|
723
|
+
"""A legacy compound-file document: DOC, XLS, PPT, HWP, or an encrypted
|
|
724
|
+
OOXML file, which Office stores in this container rather than as a ZIP.
|
|
725
|
+
|
|
726
|
+
HWP declares itself in its ``FileHeader`` stream, whose 32-byte signature
|
|
727
|
+
is followed by a version word and a property word; bit 1 of that property
|
|
728
|
+
word is the password flag, which is how an encrypted HWP is told from an
|
|
729
|
+
ordinary one without parsing the container's allocation tables.
|
|
730
|
+
"""
|
|
731
|
+
if _u16("EncryptedPackage") in data:
|
|
732
|
+
return _problem_verdict(
|
|
733
|
+
FileKind.OLE_DOCUMENT,
|
|
734
|
+
Code.FILE_ENCRYPTED,
|
|
735
|
+
"This Office document is password-protected, so its text cannot be read.",
|
|
736
|
+
_REMEDIES_ENCRYPTED,
|
|
737
|
+
size=size,
|
|
738
|
+
format_name="ooxml-encrypted",
|
|
739
|
+
)
|
|
740
|
+
|
|
741
|
+
hwp_at = data.find(b"HWP Document File")
|
|
742
|
+
if hwp_at >= 0:
|
|
743
|
+
properties_at = hwp_at + 36
|
|
744
|
+
properties = (
|
|
745
|
+
int.from_bytes(data[properties_at : properties_at + 4], "little")
|
|
746
|
+
if properties_at + 4 <= len(data)
|
|
747
|
+
else 0
|
|
748
|
+
)
|
|
749
|
+
if properties & 0x02:
|
|
750
|
+
return _problem_verdict(
|
|
751
|
+
FileKind.HWP,
|
|
752
|
+
Code.FILE_ENCRYPTED,
|
|
753
|
+
"This HWP document is password-protected, so its text cannot be read.",
|
|
754
|
+
_REMEDIES_ENCRYPTED,
|
|
755
|
+
size=size,
|
|
756
|
+
format_name="hwp",
|
|
757
|
+
)
|
|
758
|
+
return _unsupported(
|
|
759
|
+
FileKind.HWP,
|
|
760
|
+
"hwp",
|
|
761
|
+
size,
|
|
762
|
+
_REMEDIES_DOCUMENT,
|
|
763
|
+
"This is an HWP document. Its text is stored in a binary document format that this "
|
|
764
|
+
"version does not open.",
|
|
765
|
+
)
|
|
766
|
+
|
|
767
|
+
for stream, kind, format_name, label in _OLE_STREAM_RULES:
|
|
768
|
+
if _u16(stream) in data:
|
|
769
|
+
return _unsupported(
|
|
770
|
+
kind,
|
|
771
|
+
format_name,
|
|
772
|
+
size,
|
|
773
|
+
_REMEDIES_DOCUMENT,
|
|
774
|
+
f"This is a {label}. Its text is stored in a binary document format that this "
|
|
775
|
+
"version does not open.",
|
|
776
|
+
)
|
|
777
|
+
return _unsupported(
|
|
778
|
+
FileKind.OLE_DOCUMENT,
|
|
779
|
+
"ole",
|
|
780
|
+
size,
|
|
781
|
+
_REMEDIES_DOCUMENT,
|
|
782
|
+
"This is a legacy compound-file document. Its text is stored in a binary format that "
|
|
783
|
+
"this version does not open.",
|
|
784
|
+
)
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
def _markup_verdict(text: str, size: int) -> Verdict | None:
|
|
788
|
+
"""HTML and XML decode perfectly and are still refused.
|
|
789
|
+
|
|
790
|
+
Section 2.7 puts web pages out of scope, and markup read aloud is tag
|
|
791
|
+
names rather than prose. Only a document that *begins* as markup is
|
|
792
|
+
caught, so Markdown with an inline tag part-way through stays readable.
|
|
793
|
+
"""
|
|
794
|
+
head = text[:1024].lstrip("\ufeff \t\r\n")
|
|
795
|
+
lowered = head.lower()
|
|
796
|
+
if lowered.startswith("{\\rtf"):
|
|
797
|
+
return _unsupported(
|
|
798
|
+
FileKind.RTF,
|
|
799
|
+
"rtf",
|
|
800
|
+
size,
|
|
801
|
+
_REMEDIES_DOCUMENT,
|
|
802
|
+
"This is a Rich Text Format document, which stores its text among formatting "
|
|
803
|
+
"commands that this version does not interpret.",
|
|
804
|
+
)
|
|
805
|
+
if lowered.startswith(_HTML_STARTS):
|
|
806
|
+
return _unsupported(
|
|
807
|
+
FileKind.HTML,
|
|
808
|
+
"html",
|
|
809
|
+
size,
|
|
810
|
+
_REMEDIES_DOCUMENT,
|
|
811
|
+
"This is an HTML page. EchoAct does not fetch or interpret web pages.",
|
|
812
|
+
)
|
|
813
|
+
if lowered.startswith("<?xml"):
|
|
814
|
+
is_html = "<html" in lowered
|
|
815
|
+
return _unsupported(
|
|
816
|
+
FileKind.HTML if is_html else FileKind.XML,
|
|
817
|
+
"xhtml" if is_html else "xml",
|
|
818
|
+
size,
|
|
819
|
+
_REMEDIES_DOCUMENT,
|
|
820
|
+
"This is a markup document. Reading it aloud would speak its tags rather than its "
|
|
821
|
+
"text.",
|
|
822
|
+
)
|
|
823
|
+
return None
|
|
824
|
+
|
|
825
|
+
|
|
826
|
+
# -------------------------------------------------------------- encoding ---
|
|
827
|
+
|
|
828
|
+
_BOMS: Final = (
|
|
829
|
+
(b"\x00\x00\xfe\xff", "utf-32"),
|
|
830
|
+
(b"\xff\xfe\x00\x00", "utf-32"),
|
|
831
|
+
(b"\xef\xbb\xbf", "utf-8-sig"),
|
|
832
|
+
(b"\xff\xfe", "utf-16"),
|
|
833
|
+
(b"\xfe\xff", "utf-16"),
|
|
834
|
+
)
|
|
835
|
+
|
|
836
|
+
|
|
837
|
+
def bom_encoding(data: bytes) -> str | None:
|
|
838
|
+
"""The codec a file's byte-order mark calls for, if it has one.
|
|
839
|
+
|
|
840
|
+
Two details are deliberate. UTF-32's little-endian mark begins with
|
|
841
|
+
UTF-16's, so the four-byte marks are tested first; the other order turns
|
|
842
|
+
every UTF-32 file into UTF-16 text full of NUL characters. And the codec
|
|
843
|
+
named is the endianness-detecting one rather than the explicit ``-le`` or
|
|
844
|
+
``-be`` form, because only the former consumes the mark: the explicit
|
|
845
|
+
codecs leave it in the text as a zero-width character at offset 0, which
|
|
846
|
+
would silently shift every 4.2 offset by one.
|
|
847
|
+
"""
|
|
848
|
+
for mark, encoding in _BOMS:
|
|
849
|
+
if data.startswith(mark):
|
|
850
|
+
return encoding
|
|
851
|
+
return None
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def _try_decode(data: bytes, encoding: str) -> tuple[str | None, int | None, str | None]:
|
|
855
|
+
try:
|
|
856
|
+
return data.decode(encoding), None, None
|
|
857
|
+
except UnicodeDecodeError as exc:
|
|
858
|
+
return None, exc.start, exc.reason
|
|
859
|
+
except LookupError:
|
|
860
|
+
return None, None, "unknown encoding"
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
def _candidate(data: bytes, encoding: str) -> EncodingCandidate:
|
|
864
|
+
text, offset, reason = _try_decode(data, encoding)
|
|
865
|
+
label = ENCODING_LABELS.get(encoding, encoding.upper())
|
|
866
|
+
if text is not None:
|
|
867
|
+
return EncodingCandidate(
|
|
868
|
+
name=encoding, label=label, decodes_cleanly=True, preview=_preview_of(text)
|
|
869
|
+
)
|
|
870
|
+
lossy = data[: PREVIEW_CODEPOINTS * 4].decode(encoding, errors="replace")
|
|
871
|
+
return EncodingCandidate(
|
|
872
|
+
name=encoding,
|
|
873
|
+
label=label,
|
|
874
|
+
decodes_cleanly=False,
|
|
875
|
+
preview=_preview_of(lossy),
|
|
876
|
+
failure_offset=offset,
|
|
877
|
+
failure_reason=reason,
|
|
878
|
+
)
|
|
879
|
+
|
|
880
|
+
|
|
881
|
+
def verdict_for_text(
|
|
882
|
+
text: str,
|
|
883
|
+
encoding: str,
|
|
884
|
+
*,
|
|
885
|
+
byte_size: int,
|
|
886
|
+
filename: str | None = None,
|
|
887
|
+
confidence: Confidence = Confidence.CERTAIN,
|
|
888
|
+
) -> Verdict:
|
|
889
|
+
"""Judge text that has already been decoded.
|
|
890
|
+
|
|
891
|
+
Public because F-34's confirmed encoding arrives after this module has
|
|
892
|
+
already given up: the loader decodes with the encoding the user chose and
|
|
893
|
+
the result still has to face the checks every readable file faces, or
|
|
894
|
+
"choose CP949" would become a way past F-35's refusal to interpret a
|
|
895
|
+
binary file as text.
|
|
896
|
+
"""
|
|
897
|
+
nul_at = text.find("\x00")
|
|
898
|
+
if nul_at >= 0:
|
|
899
|
+
# A single NUL character condemns the whole text, wherever it sits.
|
|
900
|
+
# The density rule below cannot do this job: 200 NUL bytes in 23,000
|
|
901
|
+
# code points of prose are 0.9%, far under MAX_CONTROL_RATIO, so a
|
|
902
|
+
# ratio test accepts them -- and then F-35's "binary is not
|
|
903
|
+
# force-interpreted as text" has been broken by a file that is
|
|
904
|
+
# binary only after the first 8 KiB. Every one of those NULs would
|
|
905
|
+
# go to the segmenter, to the engine, and into the 4.2 offsets.
|
|
906
|
+
return _problem_verdict(
|
|
907
|
+
FileKind.BINARY,
|
|
908
|
+
Code.FILE_NOT_TEXT,
|
|
909
|
+
"This file contains binary data rather than text: it holds NUL characters, which "
|
|
910
|
+
"readable text does not.",
|
|
911
|
+
_REMEDIES_BINARY,
|
|
912
|
+
size=byte_size,
|
|
913
|
+
format_name=encoding,
|
|
914
|
+
detail={"first_nul_codepoint": nul_at},
|
|
915
|
+
)
|
|
916
|
+
if _text_is_control_dense(text):
|
|
917
|
+
return _problem_verdict(
|
|
918
|
+
FileKind.BINARY,
|
|
919
|
+
Code.FILE_NOT_TEXT,
|
|
920
|
+
"This file decodes into control characters rather than readable text, so it is not "
|
|
921
|
+
"a text file.",
|
|
922
|
+
_REMEDIES_BINARY,
|
|
923
|
+
size=byte_size,
|
|
924
|
+
format_name=encoding,
|
|
925
|
+
)
|
|
926
|
+
markup = _markup_verdict(text, byte_size)
|
|
927
|
+
if markup is not None:
|
|
928
|
+
return markup
|
|
929
|
+
|
|
930
|
+
extension = _extension(filename)
|
|
931
|
+
detail: dict[str, Any] = {}
|
|
932
|
+
if extension and extension in _NON_TEXT_EXTENSIONS:
|
|
933
|
+
# F-32's disguise case in reverse: the content is text but the name
|
|
934
|
+
# claims otherwise. The content decides; the disagreement is only
|
|
935
|
+
# recorded so F-35's confirmation can say why it is being asked.
|
|
936
|
+
detail["extension_mismatch"] = True
|
|
937
|
+
return Verdict(
|
|
938
|
+
kind=FileKind.TEXT,
|
|
939
|
+
readable_as_text=True,
|
|
940
|
+
encoding=encoding,
|
|
941
|
+
confidence=confidence,
|
|
942
|
+
needs_confirmation=extension not in TEXT_EXTENSIONS,
|
|
943
|
+
preview=_preview_of(text),
|
|
944
|
+
byte_size=byte_size,
|
|
945
|
+
format_name="text",
|
|
946
|
+
detail=detail,
|
|
947
|
+
)
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def _sniff_text(data: bytes, filename: str | None, size: int) -> Verdict:
|
|
951
|
+
"""F-34's order: UTF-8, then CP949, then ask.
|
|
952
|
+
|
|
953
|
+
Trying UTF-8 first is not a preference. A CP949 file that happens to be
|
|
954
|
+
valid UTF-8 is a curiosity; a UTF-8 file that happens to be valid CP949
|
|
955
|
+
is routine, because CP949 accepts nearly every high-byte pair. Reversing
|
|
956
|
+
the order would read ordinary Korean UTF-8 as hanja soup.
|
|
957
|
+
"""
|
|
958
|
+
utf8_text, utf8_offset, utf8_reason = _try_decode(data, "utf-8")
|
|
959
|
+
if utf8_text is not None:
|
|
960
|
+
# Pure ASCII is the one case with nothing to be wrong about: CP949
|
|
961
|
+
# agrees with UTF-8 byte for byte below 0x80.
|
|
962
|
+
confidence = Confidence.CERTAIN if data.isascii() else Confidence.LIKELY
|
|
963
|
+
return verdict_for_text(
|
|
964
|
+
utf8_text, "utf-8", byte_size=size, filename=filename, confidence=confidence
|
|
965
|
+
)
|
|
966
|
+
|
|
967
|
+
cp949_text, _offset, _reason = _try_decode(data, "cp949")
|
|
968
|
+
if cp949_text is not None:
|
|
969
|
+
return verdict_for_text(
|
|
970
|
+
cp949_text, "cp949", byte_size=size, filename=filename, confidence=Confidence.LIKELY
|
|
971
|
+
)
|
|
972
|
+
|
|
973
|
+
candidates = tuple(_candidate(data, name) for name in AUTO_ENCODINGS)
|
|
974
|
+
return _problem_verdict(
|
|
975
|
+
FileKind.TEXT,
|
|
976
|
+
Code.FILE_ENCODING,
|
|
977
|
+
"This file's text encoding could not be determined: it is neither valid UTF-8 nor valid "
|
|
978
|
+
"CP949, so it is in some other encoding or partly damaged.",
|
|
979
|
+
_REMEDIES_ENCODING,
|
|
980
|
+
size=size,
|
|
981
|
+
format_name="text",
|
|
982
|
+
candidates=candidates,
|
|
983
|
+
# F-35 still applies once an encoding is chosen, so the flag is
|
|
984
|
+
# computed here too rather than being lost with the failed decode.
|
|
985
|
+
needs_confirmation=_extension(filename) not in TEXT_EXTENSIONS,
|
|
986
|
+
preview=candidates[0].preview,
|
|
987
|
+
preview_lossy=True,
|
|
988
|
+
detail={
|
|
989
|
+
"utf8_failure_offset": utf8_offset,
|
|
990
|
+
"utf8_failure_reason": utf8_reason,
|
|
991
|
+
"selectable_encodings": list(SELECTABLE_ENCODINGS),
|
|
992
|
+
},
|
|
993
|
+
)
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
# ----------------------------------------------------------- entry point ---
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def sniff(data: bytes, *, filename: str | None = None) -> Verdict:
|
|
1000
|
+
"""Decide what ``data`` is and whether it may become text (F-32, F-34, F-35).
|
|
1001
|
+
|
|
1002
|
+
``filename`` is advisory only. It never makes a file readable and never
|
|
1003
|
+
makes one unreadable; it decides nothing but whether F-35 asks the user to
|
|
1004
|
+
confirm first, which is the only role F-32 leaves to an extension.
|
|
1005
|
+
"""
|
|
1006
|
+
size = len(data)
|
|
1007
|
+
if size == 0:
|
|
1008
|
+
return _problem_verdict(
|
|
1009
|
+
FileKind.EMPTY,
|
|
1010
|
+
Code.INPUT_EMPTY,
|
|
1011
|
+
"This file is empty.",
|
|
1012
|
+
_REMEDIES_EMPTY,
|
|
1013
|
+
size=size,
|
|
1014
|
+
)
|
|
1015
|
+
|
|
1016
|
+
declared = bom_encoding(data)
|
|
1017
|
+
if declared is not None:
|
|
1018
|
+
# A byte-order mark is the file speaking for itself, so this is the
|
|
1019
|
+
# one place an encoding is accepted without trial -- and the one case
|
|
1020
|
+
# where NUL bytes do not mean "binary", UTF-16 text being full of
|
|
1021
|
+
# them. F-37 still holds: nothing was chosen on the caller's behalf.
|
|
1022
|
+
text, offset, reason = _try_decode(data, declared)
|
|
1023
|
+
if text is None:
|
|
1024
|
+
return _problem_verdict(
|
|
1025
|
+
FileKind.TEXT,
|
|
1026
|
+
Code.FILE_CORRUPT,
|
|
1027
|
+
f"This file declares {ENCODING_LABELS.get(declared, declared)} but does not "
|
|
1028
|
+
"decode as it, so it is damaged or was cut short.",
|
|
1029
|
+
_REMEDIES_CORRUPT,
|
|
1030
|
+
size=size,
|
|
1031
|
+
format_name=declared,
|
|
1032
|
+
detail={"failure_offset": offset, "failure_reason": reason},
|
|
1033
|
+
)
|
|
1034
|
+
return verdict_for_text(
|
|
1035
|
+
text, declared, byte_size=size, filename=filename, confidence=Confidence.CERTAIN
|
|
1036
|
+
)
|
|
1037
|
+
|
|
1038
|
+
structural = _detect_binary(data, filename, size)
|
|
1039
|
+
if structural is not None:
|
|
1040
|
+
return structural
|
|
1041
|
+
return _sniff_text(data, filename, size)
|
|
1042
|
+
|
|
1043
|
+
|
|
1044
|
+
def _wide_encoding_candidates(data: bytes) -> tuple[EncodingCandidate, ...]:
|
|
1045
|
+
"""The byte-order-mark-less UTF-16 readings that would yield real text.
|
|
1046
|
+
|
|
1047
|
+
Each returned candidate decodes the whole file strictly, has no NUL and
|
|
1048
|
+
no control soup in it, and is written in a script F-04 covers. Anything
|
|
1049
|
+
weaker would offer a preview of ideograph soup for every binary file of
|
|
1050
|
+
even length; anything stronger would drop the one case F-34's selector
|
|
1051
|
+
exists for back into a bare "not a text file".
|
|
1052
|
+
"""
|
|
1053
|
+
found: list[EncodingCandidate] = []
|
|
1054
|
+
for name in _BOM_LESS_WIDE_ENCODINGS:
|
|
1055
|
+
text, _offset, _reason = _try_decode(data, name)
|
|
1056
|
+
if text is None or "\x00" in text or _text_is_control_dense(text):
|
|
1057
|
+
continue
|
|
1058
|
+
if not _reads_as_app_script(text):
|
|
1059
|
+
continue
|
|
1060
|
+
found.append(
|
|
1061
|
+
EncodingCandidate(
|
|
1062
|
+
name=name,
|
|
1063
|
+
label=ENCODING_LABELS.get(name, name.upper()),
|
|
1064
|
+
decodes_cleanly=True,
|
|
1065
|
+
preview=_preview_of(text),
|
|
1066
|
+
)
|
|
1067
|
+
)
|
|
1068
|
+
return tuple(found)
|
|
1069
|
+
|
|
1070
|
+
|
|
1071
|
+
def _detect_binary(data: bytes, filename: str | None, size: int) -> Verdict | None:
|
|
1072
|
+
"""Everything decided by the bytes' shape rather than by decoding them.
|
|
1073
|
+
|
|
1074
|
+
Returns ``None`` when the file is still a candidate for being text.
|
|
1075
|
+
"""
|
|
1076
|
+
if data.startswith(b"%PDF-"):
|
|
1077
|
+
return _pdf_verdict(data, size)
|
|
1078
|
+
if data[:4] in (b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"):
|
|
1079
|
+
return _zip_verdict(data, size)
|
|
1080
|
+
if data.startswith(_OLE_MAGIC):
|
|
1081
|
+
return _ole_verdict(data, size)
|
|
1082
|
+
if data.startswith(b"SQLite format 3\x00"):
|
|
1083
|
+
return _unsupported(
|
|
1084
|
+
FileKind.DATABASE,
|
|
1085
|
+
"sqlite",
|
|
1086
|
+
size,
|
|
1087
|
+
_REMEDIES_BINARY,
|
|
1088
|
+
"This is a database file, not a document.",
|
|
1089
|
+
)
|
|
1090
|
+
|
|
1091
|
+
image = _image_format(data)
|
|
1092
|
+
if image:
|
|
1093
|
+
return _unsupported(
|
|
1094
|
+
FileKind.IMAGE,
|
|
1095
|
+
image,
|
|
1096
|
+
size,
|
|
1097
|
+
_REMEDIES_IMAGE,
|
|
1098
|
+
f"This is a {image.upper()} image. EchoAct cannot read text that is part of a "
|
|
1099
|
+
"picture or a scan.",
|
|
1100
|
+
)
|
|
1101
|
+
|
|
1102
|
+
executable = _executable_format(data)
|
|
1103
|
+
if executable:
|
|
1104
|
+
return _unsupported(
|
|
1105
|
+
FileKind.EXECUTABLE,
|
|
1106
|
+
executable,
|
|
1107
|
+
size,
|
|
1108
|
+
_REMEDIES_EXECUTABLE,
|
|
1109
|
+
"This is a program file, not a document.",
|
|
1110
|
+
)
|
|
1111
|
+
|
|
1112
|
+
media_format, media_kind = _media_format(data)
|
|
1113
|
+
if media_format:
|
|
1114
|
+
noun = "an audio file" if media_kind is FileKind.AUDIO else "a video file"
|
|
1115
|
+
return _unsupported(
|
|
1116
|
+
media_kind,
|
|
1117
|
+
media_format,
|
|
1118
|
+
size,
|
|
1119
|
+
_REMEDIES_MEDIA,
|
|
1120
|
+
f"This is {noun} ({media_format}). EchoAct generates speech but does not listen to "
|
|
1121
|
+
"it.",
|
|
1122
|
+
)
|
|
1123
|
+
|
|
1124
|
+
archive = _archive_format(data)
|
|
1125
|
+
if archive:
|
|
1126
|
+
return _unsupported(
|
|
1127
|
+
FileKind.ARCHIVE,
|
|
1128
|
+
archive,
|
|
1129
|
+
size,
|
|
1130
|
+
_REMEDIES_ARCHIVE,
|
|
1131
|
+
f"This is a {archive} archive. EchoAct does not unpack archives.",
|
|
1132
|
+
)
|
|
1133
|
+
|
|
1134
|
+
# The whole file, not the window: a NUL is one decisive byte rather than
|
|
1135
|
+
# a proportion, and one at offset 11,000 puts a NUL character into the
|
|
1136
|
+
# source text just as surely as one at offset 10 (F-35).
|
|
1137
|
+
nul_at = data.find(b"\x00")
|
|
1138
|
+
if nul_at >= 0:
|
|
1139
|
+
candidates = _wide_encoding_candidates(data)
|
|
1140
|
+
if candidates:
|
|
1141
|
+
# F-34: this is the shape a UTF-16 file with no byte-order mark
|
|
1142
|
+
# arrives in, and it is the very case the encoding override was
|
|
1143
|
+
# added for. Refusing it with the binary remedies would tell
|
|
1144
|
+
# the user to abandon a file that is only one selector click
|
|
1145
|
+
# from readable, and would leave F-37's caller with a payload
|
|
1146
|
+
# naming no correction.
|
|
1147
|
+
return _problem_verdict(
|
|
1148
|
+
FileKind.TEXT,
|
|
1149
|
+
Code.FILE_ENCODING,
|
|
1150
|
+
"This file's NUL bytes fall in the pattern of UTF-16 text saved without a "
|
|
1151
|
+
"byte-order mark, so its encoding cannot be settled from the bytes alone.",
|
|
1152
|
+
_REMEDIES_ENCODING,
|
|
1153
|
+
size=size,
|
|
1154
|
+
format_name="text",
|
|
1155
|
+
candidates=candidates,
|
|
1156
|
+
needs_confirmation=_extension(filename) not in TEXT_EXTENSIONS,
|
|
1157
|
+
preview=candidates[0].preview,
|
|
1158
|
+
detail={
|
|
1159
|
+
"first_nul_offset": nul_at,
|
|
1160
|
+
"selectable_encodings": list(SELECTABLE_ENCODINGS),
|
|
1161
|
+
},
|
|
1162
|
+
)
|
|
1163
|
+
return _problem_verdict(
|
|
1164
|
+
FileKind.BINARY,
|
|
1165
|
+
Code.FILE_NOT_TEXT,
|
|
1166
|
+
"This file contains binary data rather than text.",
|
|
1167
|
+
_REMEDIES_BINARY,
|
|
1168
|
+
size=size,
|
|
1169
|
+
format_name="binary",
|
|
1170
|
+
detail={"first_nul_offset": nul_at},
|
|
1171
|
+
)
|
|
1172
|
+
window = data[:SNIFF_WINDOW_BYTES]
|
|
1173
|
+
if _byte_window_is_control_dense(window):
|
|
1174
|
+
return _problem_verdict(
|
|
1175
|
+
FileKind.BINARY,
|
|
1176
|
+
Code.FILE_NOT_TEXT,
|
|
1177
|
+
"This file is mostly control characters, so it is not a text file.",
|
|
1178
|
+
_REMEDIES_BINARY,
|
|
1179
|
+
size=size,
|
|
1180
|
+
format_name="binary",
|
|
1181
|
+
)
|
|
1182
|
+
return None
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
__all__ = [
|
|
1186
|
+
"AUTO_ENCODINGS",
|
|
1187
|
+
"ENCODING_LABELS",
|
|
1188
|
+
"MIMETYPE_PROBE_BYTES",
|
|
1189
|
+
"MIN_APP_SCRIPT_RATIO",
|
|
1190
|
+
"PREVIEW_CODEPOINTS",
|
|
1191
|
+
"SELECTABLE_ENCODINGS",
|
|
1192
|
+
"SNIFF_WINDOW_BYTES",
|
|
1193
|
+
"SUPPORTED_FORMATS",
|
|
1194
|
+
"TEXT_EXTENSIONS",
|
|
1195
|
+
"Confidence",
|
|
1196
|
+
"EncodingCandidate",
|
|
1197
|
+
"FileKind",
|
|
1198
|
+
"Verdict",
|
|
1199
|
+
"bom_encoding",
|
|
1200
|
+
"sniff",
|
|
1201
|
+
"verdict_for_text",
|
|
1202
|
+
]
|