echoact 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. echoact/__init__.py +3 -0
  2. echoact/__main__.py +117 -0
  3. echoact/app.py +315 -0
  4. echoact/audio/__init__.py +0 -0
  5. echoact/audio/devices.py +192 -0
  6. echoact/audio/player.py +611 -0
  7. echoact/audio/wav.py +854 -0
  8. echoact/config/__init__.py +0 -0
  9. echoact/config/budget.py +370 -0
  10. echoact/config/settings.py +1244 -0
  11. echoact/db/__init__.py +0 -0
  12. echoact/db/backup.py +2429 -0
  13. echoact/db/migrations.py +434 -0
  14. echoact/db/schema.sql +214 -0
  15. echoact/db/store.py +2062 -0
  16. echoact/diagnostics.py +902 -0
  17. echoact/domain.py +487 -0
  18. echoact/engine/__init__.py +0 -0
  19. echoact/engine/container.py +843 -0
  20. echoact/engine/protocol.py +241 -0
  21. echoact/engine/runtime.py +324 -0
  22. echoact/engine/supervisor.py +961 -0
  23. echoact/engine/worker.py +659 -0
  24. echoact/errors.py +281 -0
  25. echoact/instance.py +172 -0
  26. echoact/jobs/__init__.py +0 -0
  27. echoact/jobs/engine.py +776 -0
  28. echoact/jobs/request.py +300 -0
  29. echoact/mcp/__init__.py +0 -0
  30. echoact/mcp/__main__.py +50 -0
  31. echoact/mcp/client.py +202 -0
  32. echoact/mcp/config.py +112 -0
  33. echoact/mcp/server.py +340 -0
  34. echoact/models/__init__.py +0 -0
  35. echoact/models/catalog.py +273 -0
  36. echoact/models/manifest.py +278 -0
  37. echoact/models/registry.py +1551 -0
  38. echoact/paths.py +93 -0
  39. echoact/policy.py +189 -0
  40. echoact/security/__init__.py +0 -0
  41. echoact/security/credentials.py +930 -0
  42. echoact/security/ratelimit.py +534 -0
  43. echoact/service/__init__.py +20 -0
  44. echoact/service/app.py +182 -0
  45. echoact/service/deps.py +563 -0
  46. echoact/service/errors.py +241 -0
  47. echoact/service/routes.py +1125 -0
  48. echoact/service/schemas.py +509 -0
  49. echoact/service/server.py +270 -0
  50. echoact/text/__init__.py +0 -0
  51. echoact/text/language.py +44 -0
  52. echoact/text/loader.py +577 -0
  53. echoact/text/normalize.py +924 -0
  54. echoact/text/segment.py +499 -0
  55. echoact/text/sniff.py +1202 -0
  56. echoact/ui/__init__.py +0 -0
  57. echoact/ui/bridge.py +50 -0
  58. echoact/ui/controls.py +360 -0
  59. echoact/ui/credential_dialog.py +131 -0
  60. echoact/ui/fonts.py +94 -0
  61. echoact/ui/i18n.py +260 -0
  62. echoact/ui/icons.py +440 -0
  63. echoact/ui/library.py +1642 -0
  64. echoact/ui/licence.py +162 -0
  65. echoact/ui/main_window.py +1202 -0
  66. echoact/ui/mcp_setup.py +494 -0
  67. echoact/ui/models_view.py +1142 -0
  68. echoact/ui/notifications.py +202 -0
  69. echoact/ui/reading.py +494 -0
  70. echoact/ui/settings_view.py +2258 -0
  71. echoact/ui/status_view.py +1193 -0
  72. echoact/ui/theme.py +579 -0
  73. echoact/util/__init__.py +0 -0
  74. echoact/util/ids.py +62 -0
  75. echoact/util/logging.py +127 -0
  76. echoact-0.1.0.dist-info/METADATA +162 -0
  77. echoact-0.1.0.dist-info/RECORD +80 -0
  78. echoact-0.1.0.dist-info/WHEEL +4 -0
  79. echoact-0.1.0.dist-info/entry_points.txt +3 -0
  80. echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/text/sniff.py ADDED
@@ -0,0 +1,1202 @@
1
+ """Whether a file may be read as text at all, and in which encoding.
2
+
3
+ F-32 forbids trusting the extension, so every decision here is taken from the
4
+ bytes; a name, when one is known, only decides whether F-35 wants the user to
5
+ confirm before the content is accepted. F-33 forbids two things this module
6
+ therefore does not contain: any suggestion that renaming the file will help,
7
+ and any automatic upload or online conversion. There is deliberately no
8
+ network import anywhere in this module, and ``tests/test_sniff.py`` asserts
9
+ that rather than trusting review to notice one being added later.
10
+
11
+ F-34 fixes the encoding order. UTF-8 is tried first and CP949 only as a
12
+ fallback, because the asymmetry runs one way: CP949 text is rarely valid
13
+ UTF-8 by accident, while UTF-8 text is routinely valid CP949. When neither
14
+ is confident the file is *not* decoded with ``errors="replace"`` and handed
15
+ on -- F-34 forbids exactly that -- the verdict carries the candidate list and
16
+ a preview instead, so a person can choose (F-34) or an automated caller
17
+ receives a correctable error (F-37).
18
+
19
+ The module is pure: it opens nothing, writes nothing, and keeps no state.
20
+ ``echoact.text.loader`` owns the filesystem. That split is what makes F-32's
21
+ "on failure the existing input, documents, and playback job are unchanged"
22
+ true by construction rather than by discipline.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import io
28
+ import re
29
+ import struct
30
+ import unicodedata
31
+ import zipfile
32
+ import zlib
33
+ from dataclasses import dataclass, field
34
+ from enum import StrEnum
35
+ from typing import Any, Final
36
+
37
+ from ..errors import Code, EchoActError, Problem
38
+
39
+ #: Bytes the *density* checks look at. Magic numbers sit in the first few
40
+ #: dozen; control-character density is sampled rather than counted over the
41
+ #: whole file, because a file whose first 8 KiB are clean prose is prose and
42
+ #: the cost of deciding then stays flat at the 2,000,000-byte import ceiling.
43
+ #: A NUL byte is not sampled this way -- it is a single decisive byte rather
44
+ #: than a proportion, one that must not reach the source text from anywhere
45
+ #: in the file, and finding one is a memchr over 2,000,000 bytes at worst.
46
+ SNIFF_WINDOW_BYTES: Final = 8192
47
+
48
+ #: Bytes of a ZIP ``mimetype`` entry that are read to identify the package.
49
+ #: The longest name matched is 33 bytes; the rest of the allowance is for a
50
+ #: trailing newline and for telling one OpenDocument subtype from another.
51
+ MIMETYPE_PROBE_BYTES: Final = 128
52
+
53
+ #: Code points shown to the user so an encoding can be judged (F-34, F-35).
54
+ PREVIEW_CODEPOINTS: Final = 400
55
+
56
+ #: Share of C0 control characters above which a window is not prose. Real
57
+ #: text carries tab, newline, carriage return, form feed and vertical tab and
58
+ #: essentially nothing else; a handful of stray escapes in a long log must
59
+ #: not condemn it, hence the absolute floor as well.
60
+ MAX_CONTROL_RATIO: Final = 0.05
61
+ MIN_CONTROL_COUNT: Final = 3
62
+
63
+ #: What F-02 supports, named the way F-33 requires the notice to name it.
64
+ SUPPORTED_FORMATS: Final = ("TXT", "Markdown")
65
+
66
+ #: F-34's automatic pair, in the order they are attempted.
67
+ AUTO_ENCODINGS: Final = ("utf-8", "cp949")
68
+
69
+ #: What the encoding selector may offer once automatic reading has failed.
70
+ #: Deliberately excludes any encoding that cannot fail -- latin-1 decodes
71
+ #: every byte sequence, so offering it would be offering mojibake with no
72
+ #: signal that anything went wrong, which is F-34's silent corruption under
73
+ #: another name.
74
+ SELECTABLE_ENCODINGS: Final = ("utf-8", "cp949", "utf-16", "utf-16-le", "utf-16-be")
75
+
76
+ #: The selector entries whose *valid* text is full of NUL bytes. A file in
77
+ #: one of them written without a byte-order mark is precisely the case F-34's
78
+ #: override exists for, and it arrives at the NUL rule looking exactly like a
79
+ #: binary file. These two are therefore tried before that conclusion is
80
+ #: drawn: otherwise the one failure the override was designed for is the one
81
+ #: failure whose verdict never mentions it, and F-37's "correctable error"
82
+ #: carries nothing an automated caller could correct.
83
+ _BOM_LESS_WIDE_ENCODINGS: Final = ("utf-16-le", "utf-16-be")
84
+
85
+ #: Code-point ranges EchoAct actually reads aloud: F-04's Korean and English,
86
+ #: the punctuation and full-width forms that travel with them, and the Latin
87
+ #: letters of a European name in an otherwise Korean document.
88
+ _APP_SCRIPT_RANGES: Final = (
89
+ (0x0009, 0x000D),
90
+ (0x0020, 0x007E),
91
+ (0x00A0, 0x024F),
92
+ (0x1100, 0x11FF),
93
+ (0x2000, 0x206F),
94
+ (0x3000, 0x303F),
95
+ (0x3130, 0x318F),
96
+ (0xAC00, 0xD7A3),
97
+ (0xFF01, 0xFF60),
98
+ )
99
+
100
+ #: Share of a decode that must fall in those ranges before the encoding is
101
+ #: offered as a candidate for bytes that otherwise look binary. Every
102
+ #: even-length byte string decodes as UTF-16 into *something*: twenty-two
103
+ #: bytes of ASCII prose with one NUL in them become eleven CJK ideographs,
104
+ #: which is a reading rather than a file. Without this the NUL rule would
105
+ #: hand a preview of ideograph soup to every binary file of even length.
106
+ MIN_APP_SCRIPT_RATIO: Final = 0.8
107
+
108
+ ENCODING_LABELS: Final = {
109
+ "utf-8": "UTF-8",
110
+ "utf-8-sig": "UTF-8 with byte-order mark",
111
+ "cp949": "CP949 (Korean, Windows ANSI)",
112
+ "utf-16": "UTF-16",
113
+ "utf-16-le": "UTF-16 little-endian",
114
+ "utf-16-be": "UTF-16 big-endian",
115
+ "utf-32": "UTF-32",
116
+ "utf-32-le": "UTF-32 little-endian",
117
+ "utf-32-be": "UTF-32 big-endian",
118
+ }
119
+
120
+ # Remedy sentences. F-33 requires the reason, the supported formats, and how
121
+ # to copy the text in or convert to TXT. It also forbids suggesting that a
122
+ # rename fixes anything and forbids offering an online conversion, so no
123
+ # string here may mention either; the tests scan these constants for that.
124
+ _COPY_IN: Final = (
125
+ "Open the file in an application that can display it, select the text, and paste it "
126
+ "into EchoAct."
127
+ )
128
+ _SAVE_AS_TXT: Final = (
129
+ "Save or export the content as a plain-text file (.txt) encoded in UTF-8, then open that file."
130
+ )
131
+ _TYPE_INSTEAD: Final = "Type or paste the text you want spoken directly into EchoAct."
132
+
133
+ _REMEDIES_DOCUMENT: Final = (_COPY_IN, _SAVE_AS_TXT)
134
+ _REMEDIES_IMAGE: Final = (
135
+ "EchoAct does not read text inside pictures; it has no text recognition.",
136
+ _TYPE_INSTEAD,
137
+ )
138
+ _REMEDIES_MEDIA: Final = (
139
+ "EchoAct does not transcribe audio or video.",
140
+ _TYPE_INSTEAD,
141
+ )
142
+ _REMEDIES_ARCHIVE: Final = (
143
+ "Unpack the archive with your own tool, then open a text file from it.",
144
+ _TYPE_INSTEAD,
145
+ )
146
+ _REMEDIES_EXECUTABLE: Final = (
147
+ "EchoAct never runs a file you open, and it will not read a program as text.",
148
+ _TYPE_INSTEAD,
149
+ )
150
+ _REMEDIES_ENCRYPTED: Final = (
151
+ "Open it in the application that created it, supply the password, and save an unprotected "
152
+ "plain-text copy.",
153
+ _COPY_IN,
154
+ )
155
+ _REMEDIES_CORRUPT: Final = (
156
+ "Try another copy of the file, or the original it was made from.",
157
+ _SAVE_AS_TXT,
158
+ )
159
+ _REMEDIES_BINARY: Final = (
160
+ "Choose a plain-text file (.txt) or a Markdown file (.md).",
161
+ _TYPE_INSTEAD,
162
+ )
163
+ _REMEDIES_ENCODING: Final = (
164
+ "Check the preview, then choose the encoding the file was written in.",
165
+ "Or re-save the file as UTF-8 in a text editor and open it again.",
166
+ )
167
+ _REMEDIES_EMPTY: Final = (
168
+ "Choose a file that contains text.",
169
+ _TYPE_INSTEAD,
170
+ )
171
+
172
+
173
+ class FileKind(StrEnum):
174
+ """What the bytes turned out to be. ``UNKNOWN`` covers a file the module
175
+ never got to look at, such as one refused on size or on permissions."""
176
+
177
+ TEXT = "text"
178
+ EMPTY = "empty"
179
+ HTML = "html"
180
+ XML = "xml"
181
+ RTF = "rtf"
182
+ PDF = "pdf"
183
+ DOC = "doc"
184
+ DOCX = "docx"
185
+ XLS = "xls"
186
+ XLSX = "xlsx"
187
+ PPT = "ppt"
188
+ PPTX = "pptx"
189
+ HWP = "hwp"
190
+ HWPX = "hwpx"
191
+ EPUB = "epub"
192
+ OPENDOCUMENT = "opendocument"
193
+ OLE_DOCUMENT = "ole_document"
194
+ ARCHIVE = "archive"
195
+ IMAGE = "image"
196
+ AUDIO = "audio"
197
+ VIDEO = "video"
198
+ EXECUTABLE = "executable"
199
+ DATABASE = "database"
200
+ BINARY = "binary"
201
+ UNKNOWN = "unknown"
202
+
203
+
204
+ class Confidence(StrEnum):
205
+ """How firmly the encoding was established.
206
+
207
+ ``CERTAIN`` means the file said so itself with a byte-order mark, or the
208
+ bytes admit only one reading. ``LIKELY`` means one candidate decoded
209
+ strictly and the others did not. ``UNCERTAIN`` means nothing decoded,
210
+ which F-34 turns into a question rather than into a guess.
211
+ """
212
+
213
+ CERTAIN = "certain"
214
+ LIKELY = "likely"
215
+ UNCERTAIN = "uncertain"
216
+
217
+
218
+ @dataclass(frozen=True, slots=True)
219
+ class EncodingCandidate:
220
+ """One entry in F-34's encoding selector.
221
+
222
+ The preview of a candidate that does not decode cleanly is produced with
223
+ replacement characters *on purpose*: seeing the damage is how a person
224
+ rules the choice out. It is never a source of loaded text -- the loader
225
+ decodes strictly and only strictly.
226
+ """
227
+
228
+ name: str
229
+ label: str
230
+ decodes_cleanly: bool
231
+ preview: str
232
+ failure_offset: int | None = None
233
+ failure_reason: str | None = None
234
+
235
+ def to_dict(self) -> dict[str, Any]:
236
+ return {
237
+ "encoding": self.name,
238
+ "label": self.label,
239
+ "decodes_cleanly": self.decodes_cleanly,
240
+ "failure_offset": self.failure_offset,
241
+ "failure_reason": self.failure_reason,
242
+ }
243
+
244
+
245
+ @dataclass(frozen=True, slots=True)
246
+ class Verdict:
247
+ """The answer to "may this be read as text, and how".
248
+
249
+ One object carries every case F-32 requires to be distinguished --
250
+ unsupported, corrupt, encrypted, permission-denied, oversized, encoding
251
+ error -- because a caller that must distinguish them should not have to
252
+ catch a different exception for each. ``problem`` is ``None`` exactly
253
+ when the bytes may become text.
254
+ """
255
+
256
+ kind: FileKind
257
+ readable_as_text: bool
258
+ encoding: str | None = None
259
+ confidence: Confidence = Confidence.UNCERTAIN
260
+ candidates: tuple[EncodingCandidate, ...] = ()
261
+ problem: Problem | None = None
262
+ #: F-35: an unknown or absent extension means the user sees the preview
263
+ #: and says yes before the text is accepted.
264
+ needs_confirmation: bool = False
265
+ preview: str = ""
266
+ #: True when ``preview`` contains replacement characters and therefore
267
+ #: shows damage rather than content.
268
+ preview_lossy: bool = False
269
+ byte_size: int = 0
270
+ #: The specific format behind a family kind, e.g. "png" inside IMAGE.
271
+ format_name: str = ""
272
+ detail: dict[str, Any] = field(default_factory=dict)
273
+
274
+ @property
275
+ def ok(self) -> bool:
276
+ return self.problem is None and self.readable_as_text
277
+
278
+ def to_dict(self) -> dict[str, Any]:
279
+ """What a surface may show or send. Carries no path and no filename:
280
+ F-57 keeps internal paths out of error payloads and N-20 keeps them
281
+ out of logs, and this object reaches both."""
282
+ body: dict[str, Any] = {
283
+ "kind": self.kind.value,
284
+ "format": self.format_name,
285
+ "readable_as_text": self.readable_as_text,
286
+ "encoding": self.encoding,
287
+ "confidence": self.confidence.value,
288
+ "needs_confirmation": self.needs_confirmation,
289
+ "byte_size": self.byte_size,
290
+ "supported_formats": list(SUPPORTED_FORMATS),
291
+ }
292
+ if self.candidates:
293
+ body["encoding_candidates"] = [c.to_dict() for c in self.candidates]
294
+ if self.problem is not None:
295
+ body["code"] = self.problem.code.value
296
+ body["message"] = self.problem.message
297
+ body["remedies"] = list(self.problem.remedies)
298
+ if self.detail:
299
+ body["detail"] = dict(self.detail)
300
+ return body
301
+
302
+ def as_error(self) -> EchoActError:
303
+ """The rejection as the one exception type that crosses a boundary.
304
+
305
+ F-33's remedies travel in ``detail`` so REST and MCP deliver the same
306
+ guidance the GUI shows; N-24 wants one set of error semantics, not a
307
+ richer one for the surface that happens to be in-process.
308
+ """
309
+ if self.problem is None:
310
+ raise AssertionError("as_error() on a verdict with no problem")
311
+ detail: dict[str, Any] = {
312
+ "kind": self.kind.value,
313
+ "remedies": list(self.problem.remedies),
314
+ "supported_formats": list(SUPPORTED_FORMATS),
315
+ }
316
+ if self.format_name:
317
+ detail["format"] = self.format_name
318
+ if self.candidates:
319
+ detail["encoding_candidates"] = [c.to_dict() for c in self.candidates]
320
+ if self.needs_confirmation:
321
+ detail["needs_confirmation"] = True
322
+ detail.update(self.detail)
323
+ return EchoActError(self.problem.code, self.problem.message, detail=detail)
324
+
325
+
326
+ # --------------------------------------------------------------- helpers ---
327
+
328
+ _OLE_MAGIC: Final = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
329
+
330
+ #: Extensions F-02 names. A name outside this set is not a refusal: it only
331
+ #: means F-35's confirmation is required, because the content still decides.
332
+ TEXT_EXTENSIONS: Final = frozenset({".txt", ".md", ".markdown", ".mkd", ".mdown", ".text"})
333
+
334
+ #: Names that claim to be something this app cannot read. Used only to say
335
+ #: "the name and the content disagree" in a verdict's detail; the bytes are
336
+ #: what decide, per F-32.
337
+ _NON_TEXT_EXTENSIONS: Final = frozenset(
338
+ {
339
+ ".pdf", ".doc", ".docx", ".hwp", ".hwpx", ".epub", ".odt", ".rtf",
340
+ ".xls", ".xlsx", ".ppt", ".pptx", ".zip", ".rar", ".7z", ".gz", ".tar",
341
+ ".png", ".jpg", ".jpeg", ".gif", ".bmp", ".webp", ".tif", ".tiff",
342
+ ".wav", ".mp3", ".mp4", ".m4a", ".mov", ".ogg", ".flac", ".avi", ".mkv",
343
+ ".exe", ".dll", ".so", ".dylib", ".bin", ".db", ".sqlite", ".sqlite3",
344
+ }
345
+ )
346
+
347
+ _HTML_STARTS: Final = ("<!doctype html", "<html", "<head", "<body")
348
+ _PDF_ENCRYPT_RE: Final = re.compile(rb"/Encrypt\s*\d+\s+\d+\s+R")
349
+
350
+
351
+ def _u16(name: str) -> bytes:
352
+ """A compound-file stream name as it appears in the raw bytes.
353
+
354
+ Directory entries in an OLE compound file store their names in UTF-16LE,
355
+ so searching for the ASCII spelling finds nothing. Reading the directory
356
+ properly would mean parsing the allocation tables of a format the app
357
+ refuses either way.
358
+ """
359
+ return name.encode("utf-16-le")
360
+
361
+
362
+ def _extension(filename: str | None) -> str:
363
+ if not filename:
364
+ return ""
365
+ _, dot, ext = filename.rpartition(".")
366
+ return f".{ext.lower()}" if dot and ext else ""
367
+
368
+
369
+ def _count_control_bytes(sample: bytes) -> int:
370
+ printable_controls = (0x09, 0x0A, 0x0B, 0x0C, 0x0D)
371
+ controls = sum(1 for b in sample if b < 0x20 and b not in printable_controls)
372
+ return controls + sample.count(0x7F)
373
+
374
+
375
+ def _byte_window_is_control_dense(sample: bytes) -> bool:
376
+ if not sample:
377
+ return False
378
+ controls = _count_control_bytes(sample)
379
+ return controls >= MIN_CONTROL_COUNT and controls / len(sample) > MAX_CONTROL_RATIO
380
+
381
+
382
+ def _text_is_control_dense(text: str) -> bool:
383
+ if not text:
384
+ return False
385
+ controls = sum(
386
+ 1
387
+ for ch in text
388
+ if ch == "\ufffd" or (unicodedata.category(ch) == "Cc" and ch not in "\t\n\r\v\f")
389
+ )
390
+ return controls >= MIN_CONTROL_COUNT and controls / len(text) > MAX_CONTROL_RATIO
391
+
392
+
393
+ def _in_app_scripts(ch: str) -> bool:
394
+ point = ord(ch)
395
+ return any(low <= point <= high for low, high in _APP_SCRIPT_RANGES)
396
+
397
+
398
+ def _reads_as_app_script(text: str) -> bool:
399
+ """Whether a decode produced something EchoAct could plausibly speak.
400
+
401
+ Used only to decide whether an encoding is worth *offering* for bytes
402
+ that otherwise look binary; it never admits text on its own, and a file
403
+ it rejects can still be opened by naming the encoding explicitly.
404
+ """
405
+ if not text:
406
+ return False
407
+ return sum(1 for ch in text if _in_app_scripts(ch)) / len(text) >= MIN_APP_SCRIPT_RATIO
408
+
409
+
410
+ def _preview_of(text: str) -> str:
411
+ return text[:PREVIEW_CODEPOINTS]
412
+
413
+
414
+ def _problem_verdict(
415
+ kind: FileKind,
416
+ code: Code,
417
+ message: str,
418
+ remedies: tuple[str, ...],
419
+ *,
420
+ size: int,
421
+ format_name: str = "",
422
+ detail: dict[str, Any] | None = None,
423
+ needs_confirmation: bool = False,
424
+ candidates: tuple[EncodingCandidate, ...] = (),
425
+ preview: str = "",
426
+ preview_lossy: bool = False,
427
+ ) -> Verdict:
428
+ return Verdict(
429
+ kind=kind,
430
+ readable_as_text=False,
431
+ problem=Problem(code=code, message=message, remedies=remedies),
432
+ byte_size=size,
433
+ format_name=format_name,
434
+ detail=detail or {},
435
+ needs_confirmation=needs_confirmation,
436
+ candidates=candidates,
437
+ preview=preview,
438
+ preview_lossy=preview_lossy,
439
+ )
440
+
441
+
442
+ def _unsupported(
443
+ kind: FileKind, format_name: str, size: int, remedies: tuple[str, ...], reason: str
444
+ ) -> Verdict:
445
+ return _problem_verdict(
446
+ kind, Code.FILE_UNSUPPORTED, reason, remedies, size=size, format_name=format_name
447
+ )
448
+
449
+
450
+ # ------------------------------------------------------- format families ---
451
+
452
+
453
+ def _image_format(data: bytes) -> str:
454
+ if data.startswith(b"\x89PNG\r\n\x1a\n"):
455
+ return "png"
456
+ if data.startswith(b"\xff\xd8\xff"):
457
+ return "jpeg"
458
+ if data.startswith((b"GIF87a", b"GIF89a")):
459
+ return "gif"
460
+ if data.startswith(b"RIFF") and data[8:12] == b"WEBP":
461
+ return "webp"
462
+ if data.startswith((b"II*\x00", b"MM\x00*")):
463
+ return "tiff"
464
+ if data.startswith(b"\x00\x00\x01\x00"):
465
+ return "ico"
466
+ # "BM" is two printable letters, so a bitmap has to prove itself or a
467
+ # sentence beginning "BMW..." becomes an image: the reserved words must
468
+ # be zero and either the declared size or the pixel offset must fit.
469
+ if data.startswith(b"BM") and len(data) >= 14 and data[6:10] == b"\x00\x00\x00\x00":
470
+ declared = int.from_bytes(data[2:6], "little")
471
+ pixel_offset = int.from_bytes(data[10:14], "little")
472
+ if declared == len(data) or 0 < pixel_offset <= len(data):
473
+ return "bmp"
474
+ return ""
475
+
476
+
477
+ def _media_format(data: bytes) -> tuple[str, FileKind]:
478
+ if data.startswith(b"RIFF") and data[8:12] == b"WAVE":
479
+ return "wav", FileKind.AUDIO
480
+ if data.startswith(b"RIFF") and data[8:12] == b"AVI ":
481
+ return "avi", FileKind.VIDEO
482
+ if data.startswith(b"ID3"):
483
+ return "mp3", FileKind.AUDIO
484
+ if data.startswith((b"\xff\xfb", b"\xff\xf3", b"\xff\xf2")):
485
+ return "mp3", FileKind.AUDIO
486
+ if data.startswith(b"OggS"):
487
+ return "ogg", FileKind.AUDIO
488
+ if data.startswith(b"fLaC"):
489
+ return "flac", FileKind.AUDIO
490
+ if data[4:8] == b"ftyp":
491
+ brand = data[8:12].decode("ascii", "replace").strip()
492
+ kind = FileKind.AUDIO if brand in {"M4A", "M4B", "M4P"} else FileKind.VIDEO
493
+ return (f"iso-bmff ({brand})" if brand else "iso-bmff"), kind
494
+ if data.startswith(b"\x1aE\xdf\xa3"):
495
+ return "matroska", FileKind.VIDEO
496
+ return "", FileKind.BINARY
497
+
498
+
499
+ def _archive_format(data: bytes) -> str:
500
+ if data.startswith(b"Rar!\x1a\x07"):
501
+ return "rar"
502
+ if data.startswith(b"7z\xbc\xaf\x27\x1c"):
503
+ return "7z"
504
+ if data.startswith(b"\x1f\x8b"):
505
+ return "gzip"
506
+ if data.startswith(b"BZh") and data[3:4].isdigit():
507
+ return "bzip2"
508
+ if data.startswith(b"\xfd7zXZ\x00"):
509
+ return "xz"
510
+ if data.startswith(b"\x28\xb5\x2f\xfd"):
511
+ return "zstandard"
512
+ if data[257:262] == b"ustar":
513
+ return "tar"
514
+ return ""
515
+
516
+
517
+ def _executable_format(data: bytes) -> str:
518
+ if data.startswith(b"\x7fELF"):
519
+ return "elf"
520
+ if data.startswith(
521
+ (b"\xfe\xed\xfa\xce", b"\xfe\xed\xfa\xcf", b"\xce\xfa\xed\xfe", b"\xcf\xfa\xed\xfe")
522
+ ):
523
+ return "mach-o"
524
+ if data.startswith(b"\xca\xfe\xba\xbe"):
525
+ # Shared by a Mach-O universal binary and a Java class file; either
526
+ # way it is a program, and the notice says the same of both.
527
+ return "mach-o universal or java class"
528
+ if data.startswith(b"MZ"):
529
+ if len(data) >= 0x40:
530
+ pe_offset = int.from_bytes(data[0x3C:0x40], "little")
531
+ if 0 < pe_offset < len(data) - 4 and data[pe_offset : pe_offset + 4] == b"PE\x00\x00":
532
+ return "windows pe"
533
+ return "dos executable"
534
+ return ""
535
+
536
+
537
+ def _pdf_verdict(data: bytes, size: int) -> Verdict:
538
+ """PDF, and which of F-32's cases it is.
539
+
540
+ Encryption is read from the trailer's ``/Encrypt`` reference rather than
541
+ from any attempt to open the document, because the distinction F-32 wants
542
+ -- encrypted as against merely corrupt -- has to survive the file being
543
+ unreadable for either reason.
544
+ """
545
+ tail = data[-8192:]
546
+ if _PDF_ENCRYPT_RE.search(data) or b"/Encrypt" in tail:
547
+ return _problem_verdict(
548
+ FileKind.PDF,
549
+ Code.FILE_ENCRYPTED,
550
+ "This is a password-protected PDF, so its content cannot be read.",
551
+ _REMEDIES_ENCRYPTED,
552
+ size=size,
553
+ format_name="pdf",
554
+ )
555
+ if b"%%EOF" not in tail and b"startxref" not in tail:
556
+ return _problem_verdict(
557
+ FileKind.PDF,
558
+ Code.FILE_CORRUPT,
559
+ "This PDF is missing its end marker, so it was truncated or damaged.",
560
+ _REMEDIES_CORRUPT,
561
+ size=size,
562
+ format_name="pdf",
563
+ )
564
+ return _unsupported(
565
+ FileKind.PDF,
566
+ "pdf",
567
+ size,
568
+ _REMEDIES_DOCUMENT,
569
+ "This is a PDF. EchoAct reads plain text only and does not extract text from a page "
570
+ "layout.",
571
+ )
572
+
573
+
574
+ _ZIP_CONTENT_RULES: Final = (
575
+ ("word/document.xml", FileKind.DOCX, "docx", "Word document"),
576
+ ("xl/workbook.xml", FileKind.XLSX, "xlsx", "Excel workbook"),
577
+ ("ppt/presentation.xml", FileKind.PPTX, "pptx", "PowerPoint presentation"),
578
+ ("Contents/content.hpf", FileKind.HWPX, "hwpx", "HWPX document"),
579
+ ("Contents/section0.xml", FileKind.HWPX, "hwpx", "HWPX document"),
580
+ ("META-INF/container.xml", FileKind.EPUB, "epub", "EPUB book"),
581
+ )
582
+
583
+ _MIMETYPE_RULES: Final = (
584
+ (b"application/epub+zip", FileKind.EPUB, "epub", "EPUB book"),
585
+ (b"application/hwp+zip", FileKind.HWPX, "hwpx", "HWPX document"),
586
+ (
587
+ b"application/vnd.oasis.opendocument",
588
+ FileKind.OPENDOCUMENT,
589
+ "opendocument",
590
+ "OpenDocument file",
591
+ ),
592
+ )
593
+
594
+
595
+ #: Everything ``zipfile`` throws at a file that is damaged, hostile, or
596
+ #: merely in a corner of the format this app does not implement. It is a
597
+ #: named tuple rather than two inline ones because the list is not
598
+ #: guessable: the constructor raises ``NotImplementedError`` for a central
599
+ #: directory whose "version needed to extract" exceeds 63, reading raises
600
+ #: ``zlib.error`` for a damaged deflate stream and ``NotImplementedError``
601
+ #: for a compression method the standard library does not carry, and either
602
+ #: escaping is a raw non-``EchoActError`` out of ``sniff`` -- which CLAUDE.md
603
+ #: rule 3 forbids, and which F-32 needs told apart as FILE_CORRUPT instead.
604
+ #: ``MemoryError`` is deliberately absent: it is not a fact about the file.
605
+ _ZIP_FAILURES: Final = (
606
+ zipfile.BadZipFile,
607
+ zipfile.LargeZipFile,
608
+ NotImplementedError,
609
+ RuntimeError,
610
+ zlib.error,
611
+ struct.error,
612
+ OSError,
613
+ ValueError,
614
+ EOFError,
615
+ IndexError,
616
+ KeyError,
617
+ )
618
+
619
+
620
+ def _zip_verdict(data: bytes, size: int) -> Verdict:
621
+ """A ZIP container, and which document format it holds.
622
+
623
+ Section 2.7 lists DOCX, HWPX and EPUB as separate formats and all three
624
+ are ZIP files, so the signature alone cannot tell them apart and the
625
+ entry names have to be read. Only the central directory and, at most,
626
+ the first ``MIMETYPE_PROBE_BYTES`` of the ``mimetype`` entry are touched:
627
+ nothing else is extracted, because Section 2.7 puts decompression out of
628
+ scope and a file that cannot be read must stay unread.
629
+
630
+ That entry is read through ``open(...).read(n)`` rather than ``read()``.
631
+ The difference is not style: ``read()`` decompresses the whole entry
632
+ before anything can slice it, so a 400 KB file whose ``mimetype`` inflates
633
+ to 400 MB would allocate all of it inside a sniffer whose input is capped
634
+ at 2,000,000 bytes -- N-21's "no unbounded in-memory loading" and N-23's
635
+ "limits actually bound the work" both fail there, and A-05's *safe*
636
+ rejection becomes an out-of-memory kill. The bounded form stops the
637
+ decompressor at the length asked for. The entry's declared size is not
638
+ consulted, because it is written by whoever wrote the file.
639
+ """
640
+ try:
641
+ with zipfile.ZipFile(io.BytesIO(data)) as zf:
642
+ infos = zf.infolist()
643
+ names = {info.filename for info in infos}
644
+ encrypted = any(info.flag_bits & 0x1 for info in infos)
645
+ mimetype = b""
646
+ if "mimetype" in names:
647
+ try:
648
+ with zf.open("mimetype") as entry:
649
+ mimetype = entry.read(MIMETYPE_PROBE_BYTES)
650
+ except _ZIP_FAILURES:
651
+ # A package whose declaration cannot be read is still
652
+ # identifiable from its entry names, and one that is not
653
+ # is still a ZIP: neither is a reason to raise.
654
+ mimetype = b""
655
+ except _ZIP_FAILURES:
656
+ return _problem_verdict(
657
+ FileKind.ARCHIVE,
658
+ Code.FILE_CORRUPT,
659
+ "This file begins like a ZIP container, but its index could not be read, so it is "
660
+ "damaged or incomplete.",
661
+ _REMEDIES_CORRUPT,
662
+ size=size,
663
+ format_name="zip",
664
+ )
665
+
666
+ kind, format_name, label = FileKind.ARCHIVE, "zip", "ZIP archive"
667
+ for prefix, mime_kind, mime_format, mime_label in _MIMETYPE_RULES:
668
+ if mimetype.startswith(prefix):
669
+ kind, format_name, label = mime_kind, mime_format, mime_label
670
+ break
671
+ else:
672
+ for entry, rule_kind, rule_format, rule_label in _ZIP_CONTENT_RULES:
673
+ if entry in names:
674
+ kind, format_name, label = rule_kind, rule_format, rule_label
675
+ break
676
+ else:
677
+ if "META-INF/MANIFEST.MF" in names:
678
+ kind, format_name, label = FileKind.EXECUTABLE, "jar", "Java archive"
679
+
680
+ if encrypted:
681
+ return _problem_verdict(
682
+ kind,
683
+ Code.FILE_ENCRYPTED,
684
+ f"This {label} has password-protected contents, so they cannot be read.",
685
+ _REMEDIES_ENCRYPTED,
686
+ size=size,
687
+ format_name=format_name,
688
+ )
689
+ if kind is FileKind.ARCHIVE:
690
+ return _unsupported(
691
+ kind,
692
+ format_name,
693
+ size,
694
+ _REMEDIES_ARCHIVE,
695
+ "This is a ZIP archive. EchoAct does not unpack archives.",
696
+ )
697
+ if kind is FileKind.EXECUTABLE:
698
+ return _unsupported(
699
+ kind,
700
+ format_name,
701
+ size,
702
+ _REMEDIES_EXECUTABLE,
703
+ "This is a Java archive, which is a program rather than a document.",
704
+ )
705
+ return _unsupported(
706
+ kind,
707
+ format_name,
708
+ size,
709
+ _REMEDIES_DOCUMENT,
710
+ f"This is a {label}. Its text is stored inside a document package that this version "
711
+ "does not open.",
712
+ )
713
+
714
+
715
+ _OLE_STREAM_RULES: Final = (
716
+ ("WordDocument", FileKind.DOC, "doc", "Word 97-2003 document"),
717
+ ("Workbook", FileKind.XLS, "xls", "Excel 97-2003 workbook"),
718
+ ("PowerPoint Document", FileKind.PPT, "ppt", "PowerPoint 97-2003 presentation"),
719
+ )
720
+
721
+
722
+ def _ole_verdict(data: bytes, size: int) -> Verdict:
723
+ """A legacy compound-file document: DOC, XLS, PPT, HWP, or an encrypted
724
+ OOXML file, which Office stores in this container rather than as a ZIP.
725
+
726
+ HWP declares itself in its ``FileHeader`` stream, whose 32-byte signature
727
+ is followed by a version word and a property word; bit 1 of that property
728
+ word is the password flag, which is how an encrypted HWP is told from an
729
+ ordinary one without parsing the container's allocation tables.
730
+ """
731
+ if _u16("EncryptedPackage") in data:
732
+ return _problem_verdict(
733
+ FileKind.OLE_DOCUMENT,
734
+ Code.FILE_ENCRYPTED,
735
+ "This Office document is password-protected, so its text cannot be read.",
736
+ _REMEDIES_ENCRYPTED,
737
+ size=size,
738
+ format_name="ooxml-encrypted",
739
+ )
740
+
741
+ hwp_at = data.find(b"HWP Document File")
742
+ if hwp_at >= 0:
743
+ properties_at = hwp_at + 36
744
+ properties = (
745
+ int.from_bytes(data[properties_at : properties_at + 4], "little")
746
+ if properties_at + 4 <= len(data)
747
+ else 0
748
+ )
749
+ if properties & 0x02:
750
+ return _problem_verdict(
751
+ FileKind.HWP,
752
+ Code.FILE_ENCRYPTED,
753
+ "This HWP document is password-protected, so its text cannot be read.",
754
+ _REMEDIES_ENCRYPTED,
755
+ size=size,
756
+ format_name="hwp",
757
+ )
758
+ return _unsupported(
759
+ FileKind.HWP,
760
+ "hwp",
761
+ size,
762
+ _REMEDIES_DOCUMENT,
763
+ "This is an HWP document. Its text is stored in a binary document format that this "
764
+ "version does not open.",
765
+ )
766
+
767
+ for stream, kind, format_name, label in _OLE_STREAM_RULES:
768
+ if _u16(stream) in data:
769
+ return _unsupported(
770
+ kind,
771
+ format_name,
772
+ size,
773
+ _REMEDIES_DOCUMENT,
774
+ f"This is a {label}. Its text is stored in a binary document format that this "
775
+ "version does not open.",
776
+ )
777
+ return _unsupported(
778
+ FileKind.OLE_DOCUMENT,
779
+ "ole",
780
+ size,
781
+ _REMEDIES_DOCUMENT,
782
+ "This is a legacy compound-file document. Its text is stored in a binary format that "
783
+ "this version does not open.",
784
+ )
785
+
786
+
787
+ def _markup_verdict(text: str, size: int) -> Verdict | None:
788
+ """HTML and XML decode perfectly and are still refused.
789
+
790
+ Section 2.7 puts web pages out of scope, and markup read aloud is tag
791
+ names rather than prose. Only a document that *begins* as markup is
792
+ caught, so Markdown with an inline tag part-way through stays readable.
793
+ """
794
+ head = text[:1024].lstrip("\ufeff \t\r\n")
795
+ lowered = head.lower()
796
+ if lowered.startswith("{\\rtf"):
797
+ return _unsupported(
798
+ FileKind.RTF,
799
+ "rtf",
800
+ size,
801
+ _REMEDIES_DOCUMENT,
802
+ "This is a Rich Text Format document, which stores its text among formatting "
803
+ "commands that this version does not interpret.",
804
+ )
805
+ if lowered.startswith(_HTML_STARTS):
806
+ return _unsupported(
807
+ FileKind.HTML,
808
+ "html",
809
+ size,
810
+ _REMEDIES_DOCUMENT,
811
+ "This is an HTML page. EchoAct does not fetch or interpret web pages.",
812
+ )
813
+ if lowered.startswith("<?xml"):
814
+ is_html = "<html" in lowered
815
+ return _unsupported(
816
+ FileKind.HTML if is_html else FileKind.XML,
817
+ "xhtml" if is_html else "xml",
818
+ size,
819
+ _REMEDIES_DOCUMENT,
820
+ "This is a markup document. Reading it aloud would speak its tags rather than its "
821
+ "text.",
822
+ )
823
+ return None
824
+
825
+
826
+ # -------------------------------------------------------------- encoding ---
827
+
828
+ _BOMS: Final = (
829
+ (b"\x00\x00\xfe\xff", "utf-32"),
830
+ (b"\xff\xfe\x00\x00", "utf-32"),
831
+ (b"\xef\xbb\xbf", "utf-8-sig"),
832
+ (b"\xff\xfe", "utf-16"),
833
+ (b"\xfe\xff", "utf-16"),
834
+ )
835
+
836
+
837
+ def bom_encoding(data: bytes) -> str | None:
838
+ """The codec a file's byte-order mark calls for, if it has one.
839
+
840
+ Two details are deliberate. UTF-32's little-endian mark begins with
841
+ UTF-16's, so the four-byte marks are tested first; the other order turns
842
+ every UTF-32 file into UTF-16 text full of NUL characters. And the codec
843
+ named is the endianness-detecting one rather than the explicit ``-le`` or
844
+ ``-be`` form, because only the former consumes the mark: the explicit
845
+ codecs leave it in the text as a zero-width character at offset 0, which
846
+ would silently shift every 4.2 offset by one.
847
+ """
848
+ for mark, encoding in _BOMS:
849
+ if data.startswith(mark):
850
+ return encoding
851
+ return None
852
+
853
+
854
+ def _try_decode(data: bytes, encoding: str) -> tuple[str | None, int | None, str | None]:
855
+ try:
856
+ return data.decode(encoding), None, None
857
+ except UnicodeDecodeError as exc:
858
+ return None, exc.start, exc.reason
859
+ except LookupError:
860
+ return None, None, "unknown encoding"
861
+
862
+
863
+ def _candidate(data: bytes, encoding: str) -> EncodingCandidate:
864
+ text, offset, reason = _try_decode(data, encoding)
865
+ label = ENCODING_LABELS.get(encoding, encoding.upper())
866
+ if text is not None:
867
+ return EncodingCandidate(
868
+ name=encoding, label=label, decodes_cleanly=True, preview=_preview_of(text)
869
+ )
870
+ lossy = data[: PREVIEW_CODEPOINTS * 4].decode(encoding, errors="replace")
871
+ return EncodingCandidate(
872
+ name=encoding,
873
+ label=label,
874
+ decodes_cleanly=False,
875
+ preview=_preview_of(lossy),
876
+ failure_offset=offset,
877
+ failure_reason=reason,
878
+ )
879
+
880
+
881
+ def verdict_for_text(
882
+ text: str,
883
+ encoding: str,
884
+ *,
885
+ byte_size: int,
886
+ filename: str | None = None,
887
+ confidence: Confidence = Confidence.CERTAIN,
888
+ ) -> Verdict:
889
+ """Judge text that has already been decoded.
890
+
891
+ Public because F-34's confirmed encoding arrives after this module has
892
+ already given up: the loader decodes with the encoding the user chose and
893
+ the result still has to face the checks every readable file faces, or
894
+ "choose CP949" would become a way past F-35's refusal to interpret a
895
+ binary file as text.
896
+ """
897
+ nul_at = text.find("\x00")
898
+ if nul_at >= 0:
899
+ # A single NUL character condemns the whole text, wherever it sits.
900
+ # The density rule below cannot do this job: 200 NUL bytes in 23,000
901
+ # code points of prose are 0.9%, far under MAX_CONTROL_RATIO, so a
902
+ # ratio test accepts them -- and then F-35's "binary is not
903
+ # force-interpreted as text" has been broken by a file that is
904
+ # binary only after the first 8 KiB. Every one of those NULs would
905
+ # go to the segmenter, to the engine, and into the 4.2 offsets.
906
+ return _problem_verdict(
907
+ FileKind.BINARY,
908
+ Code.FILE_NOT_TEXT,
909
+ "This file contains binary data rather than text: it holds NUL characters, which "
910
+ "readable text does not.",
911
+ _REMEDIES_BINARY,
912
+ size=byte_size,
913
+ format_name=encoding,
914
+ detail={"first_nul_codepoint": nul_at},
915
+ )
916
+ if _text_is_control_dense(text):
917
+ return _problem_verdict(
918
+ FileKind.BINARY,
919
+ Code.FILE_NOT_TEXT,
920
+ "This file decodes into control characters rather than readable text, so it is not "
921
+ "a text file.",
922
+ _REMEDIES_BINARY,
923
+ size=byte_size,
924
+ format_name=encoding,
925
+ )
926
+ markup = _markup_verdict(text, byte_size)
927
+ if markup is not None:
928
+ return markup
929
+
930
+ extension = _extension(filename)
931
+ detail: dict[str, Any] = {}
932
+ if extension and extension in _NON_TEXT_EXTENSIONS:
933
+ # F-32's disguise case in reverse: the content is text but the name
934
+ # claims otherwise. The content decides; the disagreement is only
935
+ # recorded so F-35's confirmation can say why it is being asked.
936
+ detail["extension_mismatch"] = True
937
+ return Verdict(
938
+ kind=FileKind.TEXT,
939
+ readable_as_text=True,
940
+ encoding=encoding,
941
+ confidence=confidence,
942
+ needs_confirmation=extension not in TEXT_EXTENSIONS,
943
+ preview=_preview_of(text),
944
+ byte_size=byte_size,
945
+ format_name="text",
946
+ detail=detail,
947
+ )
948
+
949
+
950
+ def _sniff_text(data: bytes, filename: str | None, size: int) -> Verdict:
951
+ """F-34's order: UTF-8, then CP949, then ask.
952
+
953
+ Trying UTF-8 first is not a preference. A CP949 file that happens to be
954
+ valid UTF-8 is a curiosity; a UTF-8 file that happens to be valid CP949
955
+ is routine, because CP949 accepts nearly every high-byte pair. Reversing
956
+ the order would read ordinary Korean UTF-8 as hanja soup.
957
+ """
958
+ utf8_text, utf8_offset, utf8_reason = _try_decode(data, "utf-8")
959
+ if utf8_text is not None:
960
+ # Pure ASCII is the one case with nothing to be wrong about: CP949
961
+ # agrees with UTF-8 byte for byte below 0x80.
962
+ confidence = Confidence.CERTAIN if data.isascii() else Confidence.LIKELY
963
+ return verdict_for_text(
964
+ utf8_text, "utf-8", byte_size=size, filename=filename, confidence=confidence
965
+ )
966
+
967
+ cp949_text, _offset, _reason = _try_decode(data, "cp949")
968
+ if cp949_text is not None:
969
+ return verdict_for_text(
970
+ cp949_text, "cp949", byte_size=size, filename=filename, confidence=Confidence.LIKELY
971
+ )
972
+
973
+ candidates = tuple(_candidate(data, name) for name in AUTO_ENCODINGS)
974
+ return _problem_verdict(
975
+ FileKind.TEXT,
976
+ Code.FILE_ENCODING,
977
+ "This file's text encoding could not be determined: it is neither valid UTF-8 nor valid "
978
+ "CP949, so it is in some other encoding or partly damaged.",
979
+ _REMEDIES_ENCODING,
980
+ size=size,
981
+ format_name="text",
982
+ candidates=candidates,
983
+ # F-35 still applies once an encoding is chosen, so the flag is
984
+ # computed here too rather than being lost with the failed decode.
985
+ needs_confirmation=_extension(filename) not in TEXT_EXTENSIONS,
986
+ preview=candidates[0].preview,
987
+ preview_lossy=True,
988
+ detail={
989
+ "utf8_failure_offset": utf8_offset,
990
+ "utf8_failure_reason": utf8_reason,
991
+ "selectable_encodings": list(SELECTABLE_ENCODINGS),
992
+ },
993
+ )
994
+
995
+
996
+ # ----------------------------------------------------------- entry point ---
997
+
998
+
999
+ def sniff(data: bytes, *, filename: str | None = None) -> Verdict:
1000
+ """Decide what ``data`` is and whether it may become text (F-32, F-34, F-35).
1001
+
1002
+ ``filename`` is advisory only. It never makes a file readable and never
1003
+ makes one unreadable; it decides nothing but whether F-35 asks the user to
1004
+ confirm first, which is the only role F-32 leaves to an extension.
1005
+ """
1006
+ size = len(data)
1007
+ if size == 0:
1008
+ return _problem_verdict(
1009
+ FileKind.EMPTY,
1010
+ Code.INPUT_EMPTY,
1011
+ "This file is empty.",
1012
+ _REMEDIES_EMPTY,
1013
+ size=size,
1014
+ )
1015
+
1016
+ declared = bom_encoding(data)
1017
+ if declared is not None:
1018
+ # A byte-order mark is the file speaking for itself, so this is the
1019
+ # one place an encoding is accepted without trial -- and the one case
1020
+ # where NUL bytes do not mean "binary", UTF-16 text being full of
1021
+ # them. F-37 still holds: nothing was chosen on the caller's behalf.
1022
+ text, offset, reason = _try_decode(data, declared)
1023
+ if text is None:
1024
+ return _problem_verdict(
1025
+ FileKind.TEXT,
1026
+ Code.FILE_CORRUPT,
1027
+ f"This file declares {ENCODING_LABELS.get(declared, declared)} but does not "
1028
+ "decode as it, so it is damaged or was cut short.",
1029
+ _REMEDIES_CORRUPT,
1030
+ size=size,
1031
+ format_name=declared,
1032
+ detail={"failure_offset": offset, "failure_reason": reason},
1033
+ )
1034
+ return verdict_for_text(
1035
+ text, declared, byte_size=size, filename=filename, confidence=Confidence.CERTAIN
1036
+ )
1037
+
1038
+ structural = _detect_binary(data, filename, size)
1039
+ if structural is not None:
1040
+ return structural
1041
+ return _sniff_text(data, filename, size)
1042
+
1043
+
1044
+ def _wide_encoding_candidates(data: bytes) -> tuple[EncodingCandidate, ...]:
1045
+ """The byte-order-mark-less UTF-16 readings that would yield real text.
1046
+
1047
+ Each returned candidate decodes the whole file strictly, has no NUL and
1048
+ no control soup in it, and is written in a script F-04 covers. Anything
1049
+ weaker would offer a preview of ideograph soup for every binary file of
1050
+ even length; anything stronger would drop the one case F-34's selector
1051
+ exists for back into a bare "not a text file".
1052
+ """
1053
+ found: list[EncodingCandidate] = []
1054
+ for name in _BOM_LESS_WIDE_ENCODINGS:
1055
+ text, _offset, _reason = _try_decode(data, name)
1056
+ if text is None or "\x00" in text or _text_is_control_dense(text):
1057
+ continue
1058
+ if not _reads_as_app_script(text):
1059
+ continue
1060
+ found.append(
1061
+ EncodingCandidate(
1062
+ name=name,
1063
+ label=ENCODING_LABELS.get(name, name.upper()),
1064
+ decodes_cleanly=True,
1065
+ preview=_preview_of(text),
1066
+ )
1067
+ )
1068
+ return tuple(found)
1069
+
1070
+
1071
+ def _detect_binary(data: bytes, filename: str | None, size: int) -> Verdict | None:
1072
+ """Everything decided by the bytes' shape rather than by decoding them.
1073
+
1074
+ Returns ``None`` when the file is still a candidate for being text.
1075
+ """
1076
+ if data.startswith(b"%PDF-"):
1077
+ return _pdf_verdict(data, size)
1078
+ if data[:4] in (b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"):
1079
+ return _zip_verdict(data, size)
1080
+ if data.startswith(_OLE_MAGIC):
1081
+ return _ole_verdict(data, size)
1082
+ if data.startswith(b"SQLite format 3\x00"):
1083
+ return _unsupported(
1084
+ FileKind.DATABASE,
1085
+ "sqlite",
1086
+ size,
1087
+ _REMEDIES_BINARY,
1088
+ "This is a database file, not a document.",
1089
+ )
1090
+
1091
+ image = _image_format(data)
1092
+ if image:
1093
+ return _unsupported(
1094
+ FileKind.IMAGE,
1095
+ image,
1096
+ size,
1097
+ _REMEDIES_IMAGE,
1098
+ f"This is a {image.upper()} image. EchoAct cannot read text that is part of a "
1099
+ "picture or a scan.",
1100
+ )
1101
+
1102
+ executable = _executable_format(data)
1103
+ if executable:
1104
+ return _unsupported(
1105
+ FileKind.EXECUTABLE,
1106
+ executable,
1107
+ size,
1108
+ _REMEDIES_EXECUTABLE,
1109
+ "This is a program file, not a document.",
1110
+ )
1111
+
1112
+ media_format, media_kind = _media_format(data)
1113
+ if media_format:
1114
+ noun = "an audio file" if media_kind is FileKind.AUDIO else "a video file"
1115
+ return _unsupported(
1116
+ media_kind,
1117
+ media_format,
1118
+ size,
1119
+ _REMEDIES_MEDIA,
1120
+ f"This is {noun} ({media_format}). EchoAct generates speech but does not listen to "
1121
+ "it.",
1122
+ )
1123
+
1124
+ archive = _archive_format(data)
1125
+ if archive:
1126
+ return _unsupported(
1127
+ FileKind.ARCHIVE,
1128
+ archive,
1129
+ size,
1130
+ _REMEDIES_ARCHIVE,
1131
+ f"This is a {archive} archive. EchoAct does not unpack archives.",
1132
+ )
1133
+
1134
+ # The whole file, not the window: a NUL is one decisive byte rather than
1135
+ # a proportion, and one at offset 11,000 puts a NUL character into the
1136
+ # source text just as surely as one at offset 10 (F-35).
1137
+ nul_at = data.find(b"\x00")
1138
+ if nul_at >= 0:
1139
+ candidates = _wide_encoding_candidates(data)
1140
+ if candidates:
1141
+ # F-34: this is the shape a UTF-16 file with no byte-order mark
1142
+ # arrives in, and it is the very case the encoding override was
1143
+ # added for. Refusing it with the binary remedies would tell
1144
+ # the user to abandon a file that is only one selector click
1145
+ # from readable, and would leave F-37's caller with a payload
1146
+ # naming no correction.
1147
+ return _problem_verdict(
1148
+ FileKind.TEXT,
1149
+ Code.FILE_ENCODING,
1150
+ "This file's NUL bytes fall in the pattern of UTF-16 text saved without a "
1151
+ "byte-order mark, so its encoding cannot be settled from the bytes alone.",
1152
+ _REMEDIES_ENCODING,
1153
+ size=size,
1154
+ format_name="text",
1155
+ candidates=candidates,
1156
+ needs_confirmation=_extension(filename) not in TEXT_EXTENSIONS,
1157
+ preview=candidates[0].preview,
1158
+ detail={
1159
+ "first_nul_offset": nul_at,
1160
+ "selectable_encodings": list(SELECTABLE_ENCODINGS),
1161
+ },
1162
+ )
1163
+ return _problem_verdict(
1164
+ FileKind.BINARY,
1165
+ Code.FILE_NOT_TEXT,
1166
+ "This file contains binary data rather than text.",
1167
+ _REMEDIES_BINARY,
1168
+ size=size,
1169
+ format_name="binary",
1170
+ detail={"first_nul_offset": nul_at},
1171
+ )
1172
+ window = data[:SNIFF_WINDOW_BYTES]
1173
+ if _byte_window_is_control_dense(window):
1174
+ return _problem_verdict(
1175
+ FileKind.BINARY,
1176
+ Code.FILE_NOT_TEXT,
1177
+ "This file is mostly control characters, so it is not a text file.",
1178
+ _REMEDIES_BINARY,
1179
+ size=size,
1180
+ format_name="binary",
1181
+ )
1182
+ return None
1183
+
1184
+
1185
+ __all__ = [
1186
+ "AUTO_ENCODINGS",
1187
+ "ENCODING_LABELS",
1188
+ "MIMETYPE_PROBE_BYTES",
1189
+ "MIN_APP_SCRIPT_RATIO",
1190
+ "PREVIEW_CODEPOINTS",
1191
+ "SELECTABLE_ENCODINGS",
1192
+ "SNIFF_WINDOW_BYTES",
1193
+ "SUPPORTED_FORMATS",
1194
+ "TEXT_EXTENSIONS",
1195
+ "Confidence",
1196
+ "EncodingCandidate",
1197
+ "FileKind",
1198
+ "Verdict",
1199
+ "bom_encoding",
1200
+ "sniff",
1201
+ "verdict_for_text",
1202
+ ]