dsh-ops 0.0.0-stage → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +202 -0
  2. package/LICENSE +30 -0
  3. package/NOTICE +106 -0
  4. package/PROVENANCE.md +435 -0
  5. package/README.en.md +126 -0
  6. package/README.md +115 -2
  7. package/README.zh.md +116 -0
  8. package/bin/dsh-ops.mjs +1216 -0
  9. package/cordis.patch.yml +160 -0
  10. package/docs/manual-validation.md +53 -0
  11. package/docs/release-0.2.1.md +72 -0
  12. package/docs/schema-baseline.json +64 -0
  13. package/docs/schema-current.json +84 -0
  14. package/docs/schema-measurement.md +17 -0
  15. package/dsh-plugin.json +88 -0
  16. package/icon.svg +12 -0
  17. package/lib/binary.js +409 -0
  18. package/lib/config.js +198 -0
  19. package/lib/handshake.js +252 -0
  20. package/lib/index.js +108 -0
  21. package/lib/jobs.js +42 -0
  22. package/lib/policy.js +64 -0
  23. package/lib/presentation.js +63 -0
  24. package/lib/profile-install.js +61 -0
  25. package/lib/rust.js +194 -0
  26. package/lib/session-shells.js +78 -0
  27. package/lib/shells.js +998 -0
  28. package/lib/tools.js +657 -0
  29. package/locale/en.json +6 -0
  30. package/locale/zh.json +6 -0
  31. package/package.json +114 -4
  32. package/vendor/fastctx/Cargo.lock +3210 -0
  33. package/vendor/fastctx/Cargo.toml +94 -0
  34. package/vendor/fastctx/FORK.md +119 -0
  35. package/vendor/fastctx/LICENSE-APACHE +201 -0
  36. package/vendor/fastctx/NOTICE +40 -0
  37. package/vendor/fastctx/README.md +439 -0
  38. package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
  39. package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
  40. package/vendor/fastctx/UPSTREAM.md +49 -0
  41. package/vendor/fastctx/build.rs +413 -0
  42. package/vendor/fastctx/src/background_status.rs +403 -0
  43. package/vendor/fastctx/src/binary.rs +75 -0
  44. package/vendor/fastctx/src/bounded_sort.rs +500 -0
  45. package/vendor/fastctx/src/budget.rs +781 -0
  46. package/vendor/fastctx/src/cli/mod.rs +110 -0
  47. package/vendor/fastctx/src/context_guard.rs +289 -0
  48. package/vendor/fastctx/src/control/mod.rs +6 -0
  49. package/vendor/fastctx/src/control/paths.rs +49 -0
  50. package/vendor/fastctx/src/control/settings.rs +753 -0
  51. package/vendor/fastctx/src/control/transaction.rs +531 -0
  52. package/vendor/fastctx/src/edit/document.rs +535 -0
  53. package/vendor/fastctx/src/edit/locks.rs +371 -0
  54. package/vendor/fastctx/src/edit/mod.rs +213 -0
  55. package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
  56. package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
  57. package/vendor/fastctx/src/edit/private_storage.rs +234 -0
  58. package/vendor/fastctx/src/edit/replace.rs +1030 -0
  59. package/vendor/fastctx/src/edit_server.rs +53 -0
  60. package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
  61. package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
  62. package/vendor/fastctx/src/encoding.rs +1118 -0
  63. package/vendor/fastctx/src/file_executor.rs +1151 -0
  64. package/vendor/fastctx/src/file_snapshot.rs +1491 -0
  65. package/vendor/fastctx/src/glob_filter.rs +98 -0
  66. package/vendor/fastctx/src/glob_tool.rs +653 -0
  67. package/vendor/fastctx/src/grep_sink.rs +1162 -0
  68. package/vendor/fastctx/src/grep_tool.rs +2449 -0
  69. package/vendor/fastctx/src/lib.rs +45 -0
  70. package/vendor/fastctx/src/main.rs +15 -0
  71. package/vendor/fastctx/src/model.rs +51 -0
  72. package/vendor/fastctx/src/model_guidance.rs +62 -0
  73. package/vendor/fastctx/src/operation.rs +356 -0
  74. package/vendor/fastctx/src/ordered_window.rs +1235 -0
  75. package/vendor/fastctx/src/os_environment.rs +414 -0
  76. package/vendor/fastctx/src/path_codec.rs +850 -0
  77. package/vendor/fastctx/src/paths.rs +244 -0
  78. package/vendor/fastctx/src/process_identity.rs +763 -0
  79. package/vendor/fastctx/src/process_policy.rs +74 -0
  80. package/vendor/fastctx/src/read_tool/batch.rs +496 -0
  81. package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
  82. package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
  83. package/vendor/fastctx/src/read_tool/mod.rs +245 -0
  84. package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
  85. package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
  86. package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
  87. package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
  88. package/vendor/fastctx/src/render_plan.rs +468 -0
  89. package/vendor/fastctx/src/runtime/activity.rs +159 -0
  90. package/vendor/fastctx/src/runtime/hosts.rs +99 -0
  91. package/vendor/fastctx/src/runtime/journal.rs +556 -0
  92. package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
  93. package/vendor/fastctx/src/runtime/mod.rs +746 -0
  94. package/vendor/fastctx/src/runtime/protocol.rs +296 -0
  95. package/vendor/fastctx/src/runtime/session.rs +536 -0
  96. package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
  97. package/vendor/fastctx/src/search_parallelism.rs +106 -0
  98. package/vendor/fastctx/src/search_text.rs +227 -0
  99. package/vendor/fastctx/src/server.rs +359 -0
  100. package/vendor/fastctx/src/server_manifest.rs +468 -0
  101. package/vendor/fastctx/src/server_support.rs +826 -0
  102. package/vendor/fastctx/src/session.rs +629 -0
  103. package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
  104. package/vendor/fastctx/src/shell/bash.rs +263 -0
  105. package/vendor/fastctx/src/shell/buffer.rs +108 -0
  106. package/vendor/fastctx/src/shell/encoding.rs +403 -0
  107. package/vendor/fastctx/src/shell/foreground.rs +115 -0
  108. package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
  109. package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
  110. package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
  111. package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
  112. package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
  113. package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
  114. package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
  115. package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
  116. package/vendor/fastctx/src/shell/mod.rs +345 -0
  117. package/vendor/fastctx/src/shell/normalize.rs +389 -0
  118. package/vendor/fastctx/src/shell/output.rs +406 -0
  119. package/vendor/fastctx/src/shell/process.rs +493 -0
  120. package/vendor/fastctx/src/shell_server.rs +156 -0
  121. package/vendor/fastctx/src/skip_report.rs +83 -0
  122. package/vendor/fastctx/src/stdio_transport.rs +177 -0
  123. package/vendor/fastctx/src/tool_schema.rs +204 -0
  124. package/vendor/fastctx/src/traversal.rs +846 -0
  125. package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
  126. package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
  127. package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
  128. package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
  129. package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
  130. package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
  131. package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
  132. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
  133. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
  134. package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
  135. package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
  136. package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
  137. package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
  138. package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
  139. package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
  140. package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
@@ -0,0 +1,1118 @@
1
+ //! Binary detection, trusted encoding classification, and UTF-8 transcoding.
2
+
3
+ use crate::file_snapshot::{SealedSnapshot, SnapshotReader};
4
+ use encoding_rs::{
5
+ BIG5, DecoderResult, EUC_KR, EncoderResult, Encoding, GBK, SHIFT_JIS, UTF_8, UTF_16BE,
6
+ UTF_16LE, WINDOWS_1252,
7
+ };
8
+ use std::fs::File;
9
+ use std::io::{self, BufReader, Read, Seek, SeekFrom};
10
+ use std::path::Path;
11
+
12
+ #[cfg(test)]
13
+ mod reference_v011;
14
+ mod snapshot_pipeline;
15
+
16
+ pub(crate) use snapshot_pipeline::{EncodingPipelineFailure, validate_snapshot_encoding};
17
+
18
+ const BINARY_PROBE_BYTES: usize = 8 * 1024;
19
+ const DECODE_CHUNK_BYTES: usize = 64 * 1024;
20
+ const LEGACY_EVIDENCE_BYTES: usize = 32;
21
+ // Segments buffer at most 4 KiB; a valid UTF-8 run of 32 bytes with 8 non-ASCII bytes catches one-line mixtures while limiting accidental matches in pure GBK (2026-07-13).
22
+ const LEGACY_SEGMENT_MAX_BYTES: usize = 4 * 1024;
23
+ const UTF8_SEGMENT_MIN_BYTES: usize = 32;
24
+ const UTF8_SEGMENT_MIN_NON_ASCII_BYTES: usize = 8;
25
+
26
+ const FIXED_LEGACY_ENCODINGS: [(&str, &Encoding); 5] = [
27
+ ("windows-1252", WINDOWS_1252),
28
+ ("gbk", GBK),
29
+ ("shift_jis", SHIFT_JIS),
30
+ ("big5", BIG5),
31
+ ("euc-kr", EUC_KR),
32
+ ];
33
+
34
+ /// One validation input: an on-disk file or an immutable byte snapshot.
35
+ ///
36
+ /// Both variants drive the exact same chunked validation machinery, so the
37
+ /// trust hierarchy has a single implementation. Grep uses `Snapshot` so every
38
+ /// pass and the regex search observe the same single-open capture.
39
+ #[derive(Clone, Copy)]
40
+ pub(crate) enum ByteSource<'a> {
41
+ File(&'a Path),
42
+ Bytes(&'a [u8]),
43
+ Snapshot(&'a SealedSnapshot),
44
+ }
45
+
46
+ enum SourceReader<'a> {
47
+ File(BufReader<File>),
48
+ Bytes(io::Cursor<&'a [u8]>),
49
+ Snapshot(SnapshotReader<'a>),
50
+ }
51
+
52
+ impl<'a> SourceReader<'a> {
53
+ fn open(source: ByteSource<'a>, start: u64) -> io::Result<Self> {
54
+ match source {
55
+ ByteSource::File(path) => Ok(Self::open_file(path, start)?),
56
+ ByteSource::Bytes(bytes) => {
57
+ let start = usize::try_from(start)
58
+ .unwrap_or(usize::MAX)
59
+ .min(bytes.len());
60
+ Ok(SourceReader::Bytes(io::Cursor::new(&bytes[start..])))
61
+ }
62
+ ByteSource::Snapshot(snapshot) => {
63
+ snapshot.open_reader(start).map(SourceReader::Snapshot)
64
+ }
65
+ }
66
+ }
67
+
68
+ fn open_file(path: &Path, start: u64) -> io::Result<SourceReader<'static>> {
69
+ let mut reader = BufReader::new(File::open(path)?);
70
+ if start > 0 {
71
+ reader.seek(SeekFrom::Start(start))?;
72
+ }
73
+ Ok(SourceReader::File(reader))
74
+ }
75
+ }
76
+
77
+ impl Read for SourceReader<'_> {
78
+ fn read(&mut self, output: &mut [u8]) -> io::Result<usize> {
79
+ match self {
80
+ SourceReader::File(reader) => reader.read(output),
81
+ SourceReader::Bytes(cursor) => cursor.read(output),
82
+ SourceReader::Snapshot(reader) => reader.read(output),
83
+ }
84
+ }
85
+ }
86
+
87
+ #[derive(Clone, Copy, Debug, Eq, PartialEq)]
88
+ enum EncodingKind {
89
+ EncodingRs(&'static Encoding),
90
+ Utf32Le,
91
+ Utf32Be,
92
+ }
93
+
94
+ #[derive(Clone, Debug, Eq, PartialEq)]
95
+ enum EncodingOrigin {
96
+ Explicit(String),
97
+ Bom(&'static str),
98
+ Automatic,
99
+ }
100
+
101
+ /// Encoding, BOM length, and public source label required for incremental decoding.
102
+ #[derive(Clone, Debug)]
103
+ pub(crate) struct DetectedEncoding {
104
+ kind: EncodingKind,
105
+ bom_len: u64,
106
+ /// Canonical encoding name for non-UTF-8 input; UTF-8 uses None.
107
+ pub(crate) source_encoding: Option<&'static str>,
108
+ origin: EncodingOrigin,
109
+ }
110
+
111
+ /// Strictly validated file encoding and line-ending metadata shared by read and grep.
112
+ #[derive(Clone, Debug)]
113
+ pub(crate) struct ValidatedFileEncoding {
114
+ pub(crate) detected: DetectedEncoding,
115
+ pub(crate) total_lines: usize,
116
+ pub(crate) has_trailing_newline: bool,
117
+ explicit_utf8_warning: bool,
118
+ }
119
+
120
+ impl ValidatedFileEncoding {
121
+ /// Returns the raw snapshot offset where decoded UTF-8 begins when no transcoding is needed.
122
+ pub(crate) fn utf8_snapshot_start(&self) -> Option<u64> {
123
+ (self.detected.kind == EncodingKind::EncodingRs(UTF_8)).then_some(self.detected.bom_len)
124
+ }
125
+
126
+ /// Opens a streaming UTF-8 reader over the exact source that was validated.
127
+ pub(crate) fn open_source_reader<'a>(
128
+ &self,
129
+ source: ByteSource<'a>,
130
+ ) -> io::Result<Utf8Reader<'a>> {
131
+ Utf8Reader::open_source(source, self.detected.clone())
132
+ }
133
+
134
+ /// Decodes an already-validated snapshot for in-memory search.
135
+ ///
136
+ /// UTF-8 content is borrowed without copying; every other source encoding
137
+ /// is decoded to owned UTF-8 bytes. Returns `None` only if the bytes no
138
+ /// longer decode under the validated encoding.
139
+ pub(crate) fn decode_for_search<'a>(
140
+ &self,
141
+ raw: &'a [u8],
142
+ ) -> Option<std::borrow::Cow<'a, [u8]>> {
143
+ let content = raw.get(self.detected.bom_len as usize..)?;
144
+ if self.detected.kind == EncodingKind::EncodingRs(UTF_8) {
145
+ return Some(std::borrow::Cow::Borrowed(content));
146
+ }
147
+ decode_bytes(raw, &self.detected).map(|text| std::borrow::Cow::Owned(text.into_bytes()))
148
+ }
149
+
150
+ /// Streams UTF-8 text on character boundaries and stops immediately when the callback returns false.
151
+ pub(crate) fn stream_text(
152
+ &self,
153
+ path: &Path,
154
+ mut on_text: impl FnMut(&str) -> bool,
155
+ ) -> Result<bool, StreamDecodeFailure> {
156
+ let mut decoder = DecodedChunkReader::open(ByteSource::File(path), self.detected.clone())?;
157
+ loop {
158
+ match decoder.next_chunk()? {
159
+ Some(chunk) if !on_text(&chunk) => return Ok(false),
160
+ Some(_) => {}
161
+ None => return Ok(true),
162
+ }
163
+ }
164
+ }
165
+
166
+ /// Preserves an accurate failure channel from the original trust source when the file changes after validation.
167
+ pub(crate) fn malformed_rejection(&self) -> EncodingRejection {
168
+ match &self.detected.origin {
169
+ EncodingOrigin::Explicit(encoding) => EncodingRejection::ExplicitMalformed {
170
+ encoding: encoding.clone(),
171
+ },
172
+ EncodingOrigin::Bom(encoding) => EncodingRejection::BomMismatch { encoding },
173
+ EncodingOrigin::Automatic => EncodingRejection::Undecodable,
174
+ }
175
+ }
176
+
177
+ /// Returns the transcoding audit note published after the body; original UTF-8 produces no note.
178
+ pub(crate) fn transcoding_note(&self) -> Option<String> {
179
+ let encoding = self.detected.source_encoding?;
180
+ match &self.detected.origin {
181
+ EncodingOrigin::Explicit(_) if self.explicit_utf8_warning => Some(format!(
182
+ "(Note: decoded from {encoding} as requested; output is UTF-8. Warning: the raw bytes are also valid UTF-8 — if this looks garbled, retry with encoding=\"utf-8\" or omit encoding.)"
183
+ )),
184
+ EncodingOrigin::Explicit(_) => Some(format!(
185
+ "(Note: decoded from {encoding} as requested; output is UTF-8.)"
186
+ )),
187
+ EncodingOrigin::Bom(_) | EncodingOrigin::Automatic => {
188
+ Some(format!("(Note: decoded from {encoding}; output is UTF-8.)"))
189
+ }
190
+ }
191
+ }
192
+
193
+ /// Decodes an already-read snapshot and proves that stateless re-encoding reproduces it exactly.
194
+ pub(crate) fn decode_editable_snapshot(&self, raw: &[u8]) -> Result<String, String> {
195
+ let text = decode_bytes(raw, &self.detected)
196
+ .ok_or_else(|| "the file changed after encoding validation".to_string())?;
197
+ let bom_len = self.detected.bom_len as usize;
198
+ let mut text_offset = 0_usize;
199
+ let mut raw_offset = bom_len;
200
+ while text_offset < text.len() {
201
+ let end = next_char_boundary(&text, text_offset, DECODE_CHUNK_BYTES);
202
+ let fragment = self.encode_fragment(&text[text_offset..end]).ok_or_else(|| {
203
+ format!(
204
+ "stateful source encoding {} cannot provide byte-preserving edit boundaries",
205
+ self.encoding_label()
206
+ )
207
+ })?;
208
+ let Some(raw_end) = raw_offset.checked_add(fragment.len()) else {
209
+ return Err(format!(
210
+ "source encoding {} did not reproduce the original bytes exactly",
211
+ self.encoding_label()
212
+ ));
213
+ };
214
+ if raw.get(raw_offset..raw_end) != Some(fragment.as_slice()) {
215
+ return Err(format!(
216
+ "source encoding {} did not reproduce the original bytes exactly",
217
+ self.encoding_label()
218
+ ));
219
+ }
220
+ text_offset = end;
221
+ raw_offset = raw_end;
222
+ }
223
+ if raw_offset != raw.len() {
224
+ return Err(format!(
225
+ "source encoding {} did not reproduce the original bytes exactly",
226
+ self.encoding_label()
227
+ ));
228
+ }
229
+ Ok(text)
230
+ }
231
+
232
+ /// Encodes newly inserted text without adding a BOM; `None` means unmappable or stateful.
233
+ pub(crate) fn encode_fragment(&self, text: &str) -> Option<Vec<u8>> {
234
+ match self.detected.kind {
235
+ EncodingKind::Utf32Le => Some(
236
+ text.chars()
237
+ .flat_map(|character| (character as u32).to_le_bytes())
238
+ .collect(),
239
+ ),
240
+ EncodingKind::Utf32Be => Some(
241
+ text.chars()
242
+ .flat_map(|character| (character as u32).to_be_bytes())
243
+ .collect(),
244
+ ),
245
+ EncodingKind::EncodingRs(encoding) if encoding == UTF_8 => {
246
+ Some(text.as_bytes().to_vec())
247
+ }
248
+ EncodingKind::EncodingRs(encoding) if encoding == UTF_16LE => Some(
249
+ text.encode_utf16()
250
+ .flat_map(u16::to_le_bytes)
251
+ .collect::<Vec<_>>(),
252
+ ),
253
+ EncodingKind::EncodingRs(encoding) if encoding == UTF_16BE => Some(
254
+ text.encode_utf16()
255
+ .flat_map(u16::to_be_bytes)
256
+ .collect::<Vec<_>>(),
257
+ ),
258
+ EncodingKind::EncodingRs(encoding) if is_editable_stateless_encoding(encoding) => {
259
+ let mut encoder = encoding.new_encoder();
260
+ let capacity = encoder
261
+ .max_buffer_length_from_utf8_without_replacement(text.len())
262
+ .unwrap_or(text.len().saturating_mul(4).saturating_add(16))
263
+ .saturating_add(16)
264
+ .max(16);
265
+ let mut output = Vec::with_capacity(capacity);
266
+ let (result, read) =
267
+ encoder.encode_from_utf8_to_vec_without_replacement(text, &mut output, true);
268
+ (result == EncoderResult::InputEmpty && read == text.len()).then_some(output)
269
+ }
270
+ EncodingKind::EncodingRs(_) => None,
271
+ }
272
+ }
273
+
274
+ /// Returns the canonical file encoding for diagnostics and write-back metadata.
275
+ pub(crate) fn encoding_label(&self) -> &'static str {
276
+ self.detected.source_encoding.unwrap_or("UTF-8")
277
+ }
278
+
279
+ /// Returns the first raw byte after a preserved byte-order mark.
280
+ pub(crate) fn editable_raw_start(&self) -> usize {
281
+ self.detected.bom_len as usize
282
+ }
283
+ }
284
+
285
+ fn next_char_boundary(text: &str, start: usize, maximum_bytes: usize) -> usize {
286
+ let mut end = start.saturating_add(maximum_bytes).min(text.len());
287
+ while end > start && !text.is_char_boundary(end) {
288
+ end -= 1;
289
+ }
290
+ if end == start && start < text.len() {
291
+ start
292
+ + text[start..]
293
+ .chars()
294
+ .next()
295
+ .expect("start is before the end of text")
296
+ .len_utf8()
297
+ } else {
298
+ end
299
+ }
300
+ }
301
+
302
+ fn is_editable_stateless_encoding(encoding: &'static Encoding) -> bool {
303
+ encoding == GBK
304
+ || encoding == SHIFT_JIS
305
+ || encoding == BIG5
306
+ || encoding == EUC_KR
307
+ || encoding == WINDOWS_1252
308
+ }
309
+
310
+ /// Rejection reason from automatic or explicit encoding selection, each mapping to a frozen error message.
311
+ #[derive(Clone, Debug, Eq, PartialEq)]
312
+ pub(crate) enum EncodingRejection {
313
+ Ambiguous { candidates: Vec<&'static str> },
314
+ MixedOrInconsistent { conflict_hex_offset: Option<usize> },
315
+ Iso2022JpSignature,
316
+ Undecodable,
317
+ BomMismatch { encoding: &'static str },
318
+ ExplicitMalformed { encoding: String },
319
+ InvalidLabel { value: String },
320
+ }
321
+
322
+ impl EncodingRejection {
323
+ /// Translates an internal rejection into the model-visible error shared by read and grep.
324
+ pub(crate) fn message(&self, path_display: &str) -> String {
325
+ match self {
326
+ Self::Ambiguous { candidates } => format!(
327
+ "Cannot determine the text encoding of {path_display} with confidence: the bytes decode cleanly as {}. Retry with encoding=\"...\" if the context tells you which one, or use view=\"hex\".",
328
+ candidates.join(", ")
329
+ ),
330
+ Self::MixedOrInconsistent {
331
+ conflict_hex_offset,
332
+ } => {
333
+ let conflict = conflict_hex_offset.map_or_else(String::new, |offset| {
334
+ format!(" The first conflicting bytes are at hex-view offset {offset}.")
335
+ });
336
+ format!(
337
+ "Cannot decode {path_display} as text: it appears to contain mixed or inconsistent encodings — no single encoding explains the whole file.{conflict} Use view=\"hex\" to inspect the raw bytes, or split/normalize the file to a single encoding externally."
338
+ )
339
+ }
340
+ Self::Iso2022JpSignature => format!(
341
+ "Cannot decode {path_display} as text with confidence: the bytes are valid UTF-8 but contain ISO-2022 escape sequences (a stateful encoding such as ISO-2022-JP). Retry with encoding=\"iso-2022-jp\", or encoding=\"utf-8\" to force the raw UTF-8 reading, or use view=\"hex\"."
342
+ ),
343
+ Self::Undecodable => format!(
344
+ "Cannot decode {path_display} as text: no supported encoding decodes it cleanly. Use view=\"hex\" to inspect its raw bytes."
345
+ ),
346
+ Self::BomMismatch { encoding } => format!(
347
+ "Cannot decode {path_display}: it has a {encoding} byte order mark but the content is not valid {encoding}. Use view=\"hex\" to inspect its raw bytes."
348
+ ),
349
+ Self::ExplicitMalformed { encoding } => format!(
350
+ "Cannot decode {path_display} as {encoding}: the content is not valid {encoding}. Try another encoding or view=\"hex\"."
351
+ ),
352
+ Self::InvalidLabel { value } => format!(
353
+ "Invalid encoding value \"{value}\". Use a WHATWG encoding label such as \"gbk\", \"shift_jis\", \"big5\", \"euc-kr\", \"windows-1252\", \"utf-16le\", or \"utf-32le\"."
354
+ ),
355
+ }
356
+ }
357
+
358
+ /// Stable short reason used by grep directory skip reports.
359
+ pub(crate) fn skip_reason(&self) -> String {
360
+ match self {
361
+ Self::Ambiguous { candidates } => format!("ambiguous: {}", candidates.join(", ")),
362
+ Self::Iso2022JpSignature => "ambiguous: iso-2022-jp".to_string(),
363
+ Self::MixedOrInconsistent { .. } => "mixed or inconsistent encodings".to_string(),
364
+ Self::Undecodable
365
+ | Self::BomMismatch { .. }
366
+ | Self::ExplicitMalformed { .. }
367
+ | Self::InvalidLabel { .. } => "undecodable".to_string(),
368
+ }
369
+ }
370
+ }
371
+
372
+ /// Three-state encoding decision: trusted text, binary, or reasoned rejection.
373
+ pub(crate) enum EncodingDecision {
374
+ Text(ValidatedFileEncoding),
375
+ Binary,
376
+ Rejected(EncodingRejection),
377
+ }
378
+
379
+ /// I/O or content-change failure while rereading a file after initial validation.
380
+ pub(crate) enum StreamDecodeFailure {
381
+ Io(io::Error),
382
+ Malformed,
383
+ }
384
+
385
+ impl From<io::Error> for StreamDecodeFailure {
386
+ fn from(error: io::Error) -> Self {
387
+ Self::Io(error)
388
+ }
389
+ }
390
+
391
+ impl From<DecodeFailure> for StreamDecodeFailure {
392
+ fn from(error: DecodeFailure) -> Self {
393
+ match error {
394
+ DecodeFailure::Io(error) => Self::Io(error),
395
+ DecodeFailure::Malformed => Self::Malformed,
396
+ }
397
+ }
398
+ }
399
+
400
+ /// Decoded UTF-8 text and source encoding that must be disclosed to the model.
401
+ #[derive(Clone, Debug, Eq, PartialEq)]
402
+ pub struct DecodedText {
403
+ /// Strictly decoded UTF-8 content.
404
+ pub text: String,
405
+ /// Canonical encoding name for non-UTF-8 input; UTF-8 uses None.
406
+ pub source_encoding: Option<String>,
407
+ }
408
+
409
+ /// Contract NUL-based binary detection examines only the first 8 KiB.
410
+ pub fn has_binary_nul(bytes: &[u8]) -> bool {
411
+ bytes.iter().take(BINARY_PROBE_BYTES).any(|byte| *byte == 0)
412
+ }
413
+
414
+ /// Validates the full file in order: explicit intent, BOM, strict UTF-8, then trusted legacy encoding.
415
+ pub(crate) fn validate_file_encoding(
416
+ path: &Path,
417
+ explicit_encoding: Option<&str>,
418
+ ) -> io::Result<EncodingDecision> {
419
+ validate_source_encoding(ByteSource::File(path), explicit_encoding)
420
+ }
421
+
422
+ /// Runs the full trust hierarchy over a file or an in-memory snapshot.
423
+ pub(crate) fn validate_source_encoding(
424
+ source: ByteSource<'_>,
425
+ explicit_encoding: Option<&str>,
426
+ ) -> io::Result<EncodingDecision> {
427
+ match snapshot_pipeline::validate_source(source, explicit_encoding, None) {
428
+ Ok(decision) => Ok(decision),
429
+ Err(EncodingPipelineFailure::Io(error)) => Err(error),
430
+ Err(EncodingPipelineFailure::Stopped(_)) => {
431
+ unreachable!("shared validation without a work checkpoint cannot stop")
432
+ }
433
+ }
434
+ }
435
+
436
+ /// Applies the same trust classification to in-memory bytes, returning None for ambiguous, binary, or malformed input.
437
+ pub fn decode_text(bytes: &[u8]) -> Option<DecodedText> {
438
+ let EncodingDecision::Text(validated) =
439
+ validate_source_encoding(ByteSource::Bytes(bytes), None).ok()?
440
+ else {
441
+ return None;
442
+ };
443
+ decode_bytes(bytes, &validated.detected).map(|text| DecodedText {
444
+ text,
445
+ source_encoding: validated.detected.source_encoding.map(str::to_string),
446
+ })
447
+ }
448
+
449
+ fn validated(
450
+ detected: DetectedEncoding,
451
+ stats: ValidationStats,
452
+ explicit_utf8_warning: bool,
453
+ ) -> ValidatedFileEncoding {
454
+ ValidatedFileEncoding {
455
+ detected,
456
+ total_lines: stats.total_lines(),
457
+ has_trailing_newline: stats.has_trailing_newline,
458
+ explicit_utf8_warning,
459
+ }
460
+ }
461
+
462
+ /// Validates a WHATWG label and returns the canonical encoding name exposed to the model.
463
+ pub(crate) fn canonical_encoding_label(value: &str) -> Result<&'static str, EncodingRejection> {
464
+ let detected = explicit_detected_encoding(value, &[])?;
465
+ Ok(detected.source_encoding.unwrap_or("UTF-8"))
466
+ }
467
+
468
+ fn bom_detected_encoding(bytes: &[u8]) -> Option<DetectedEncoding> {
469
+ if bytes.starts_with(b"\x00\x00\xFE\xFF") {
470
+ return Some(DetectedEncoding {
471
+ kind: EncodingKind::Utf32Be,
472
+ bom_len: 4,
473
+ source_encoding: Some("UTF-32BE"),
474
+ origin: EncodingOrigin::Bom("UTF-32BE"),
475
+ });
476
+ }
477
+ if bytes.starts_with(b"\xFF\xFE\x00\x00") {
478
+ return Some(DetectedEncoding {
479
+ kind: EncodingKind::Utf32Le,
480
+ bom_len: 4,
481
+ source_encoding: Some("UTF-32LE"),
482
+ origin: EncodingOrigin::Bom("UTF-32LE"),
483
+ });
484
+ }
485
+ if bytes.starts_with(b"\xEF\xBB\xBF") {
486
+ return Some(DetectedEncoding {
487
+ kind: EncodingKind::EncodingRs(UTF_8),
488
+ bom_len: 3,
489
+ source_encoding: None,
490
+ origin: EncodingOrigin::Bom("UTF-8"),
491
+ });
492
+ }
493
+ if bytes.starts_with(b"\xFF\xFE") {
494
+ return Some(DetectedEncoding {
495
+ kind: EncodingKind::EncodingRs(UTF_16LE),
496
+ bom_len: 2,
497
+ source_encoding: Some("UTF-16LE"),
498
+ origin: EncodingOrigin::Bom("UTF-16LE"),
499
+ });
500
+ }
501
+ if bytes.starts_with(b"\xFE\xFF") {
502
+ return Some(DetectedEncoding {
503
+ kind: EncodingKind::EncodingRs(UTF_16BE),
504
+ bom_len: 2,
505
+ source_encoding: Some("UTF-16BE"),
506
+ origin: EncodingOrigin::Bom("UTF-16BE"),
507
+ });
508
+ }
509
+ None
510
+ }
511
+
512
+ fn explicit_detected_encoding(
513
+ value: &str,
514
+ prefix: &[u8],
515
+ ) -> Result<DetectedEncoding, EncodingRejection> {
516
+ let label = value.trim_matches(|character: char| character.is_ascii_whitespace());
517
+ let (kind, source_encoding) = if label.eq_ignore_ascii_case("utf-32le") {
518
+ (EncodingKind::Utf32Le, Some("UTF-32LE"))
519
+ } else if label.eq_ignore_ascii_case("utf-32be") {
520
+ (EncodingKind::Utf32Be, Some("UTF-32BE"))
521
+ } else {
522
+ let Some(encoding) = Encoding::for_label_no_replacement(label.as_bytes()) else {
523
+ return Err(EncodingRejection::InvalidLabel {
524
+ value: value.to_string(),
525
+ });
526
+ };
527
+ (
528
+ EncodingKind::EncodingRs(encoding),
529
+ (encoding != UTF_8).then_some(encoding.name()),
530
+ )
531
+ };
532
+ let bom_len = matching_bom_len(kind, prefix);
533
+ Ok(DetectedEncoding {
534
+ kind,
535
+ bom_len,
536
+ source_encoding,
537
+ origin: EncodingOrigin::Explicit(value.to_string()),
538
+ })
539
+ }
540
+
541
+ fn matching_bom_len(kind: EncodingKind, bytes: &[u8]) -> u64 {
542
+ match kind {
543
+ EncodingKind::Utf32Le if bytes.starts_with(b"\xFF\xFE\x00\x00") => 4,
544
+ EncodingKind::Utf32Be if bytes.starts_with(b"\x00\x00\xFE\xFF") => 4,
545
+ EncodingKind::EncodingRs(encoding)
546
+ if encoding == UTF_8 && bytes.starts_with(b"\xEF\xBB\xBF") =>
547
+ {
548
+ 3
549
+ }
550
+ EncodingKind::EncodingRs(encoding)
551
+ if encoding == UTF_16LE && bytes.starts_with(b"\xFF\xFE") =>
552
+ {
553
+ 2
554
+ }
555
+ EncodingKind::EncodingRs(encoding)
556
+ if encoding == UTF_16BE && bytes.starts_with(b"\xFE\xFF") =>
557
+ {
558
+ 2
559
+ }
560
+ _ => 0,
561
+ }
562
+ }
563
+
564
+ fn is_legacy_encoding(kind: EncodingKind) -> bool {
565
+ matches!(
566
+ kind,
567
+ EncodingKind::EncodingRs(encoding)
568
+ if encoding != UTF_8 && encoding != UTF_16LE && encoding != UTF_16BE
569
+ )
570
+ }
571
+
572
+ /// Incremental form of the v0.1.1 strong-UTF-8-prefix conflict detector.
573
+ pub(crate) struct Utf8ConflictProbe {
574
+ scanner: Utf8PrefixScanner,
575
+ }
576
+
577
+ impl Utf8ConflictProbe {
578
+ pub(crate) fn new() -> Self {
579
+ Self {
580
+ scanner: Utf8PrefixScanner::default(),
581
+ }
582
+ }
583
+
584
+ pub(crate) fn push(&mut self, bytes: &[u8]) -> Option<usize> {
585
+ self.scanner.push(bytes)
586
+ }
587
+ }
588
+
589
+ #[derive(Default)]
590
+ struct Utf8PrefixScanner {
591
+ absolute_base: usize,
592
+ valid_bytes: usize,
593
+ non_ascii_bytes: usize,
594
+ carry: Vec<u8>,
595
+ abandoned: bool,
596
+ }
597
+
598
+ impl Utf8PrefixScanner {
599
+ fn push(&mut self, input: &[u8]) -> Option<usize> {
600
+ if self.abandoned {
601
+ return None;
602
+ }
603
+ let mut combined = std::mem::take(&mut self.carry);
604
+ combined.extend_from_slice(input);
605
+ match std::str::from_utf8(&combined) {
606
+ Ok(_) => {
607
+ self.observe_valid(&combined);
608
+ self.absolute_base = self.absolute_base.saturating_add(combined.len());
609
+ None
610
+ }
611
+ Err(error) => {
612
+ let valid = &combined[..error.valid_up_to()];
613
+ self.observe_valid(valid);
614
+ if error.error_len().is_some() {
615
+ let conflict = self.absolute_base.saturating_add(error.valid_up_to());
616
+ if self.clear_prefix() {
617
+ return Some(conflict / 16 + 1);
618
+ }
619
+ self.abandoned = true;
620
+ return None;
621
+ }
622
+ self.absolute_base = self.absolute_base.saturating_add(error.valid_up_to());
623
+ self.carry
624
+ .extend_from_slice(&combined[error.valid_up_to()..]);
625
+ None
626
+ }
627
+ }
628
+ }
629
+
630
+ fn finish(mut self) -> Option<usize> {
631
+ if self.abandoned || self.carry.is_empty() {
632
+ return None;
633
+ }
634
+ let carry = std::mem::take(&mut self.carry);
635
+ match std::str::from_utf8(&carry) {
636
+ Ok(_) => None,
637
+ Err(error) => {
638
+ self.observe_valid(&carry[..error.valid_up_to()]);
639
+ self.clear_prefix()
640
+ .then_some((self.absolute_base + error.valid_up_to()) / 16 + 1)
641
+ }
642
+ }
643
+ }
644
+
645
+ fn observe_valid(&mut self, bytes: &[u8]) {
646
+ self.valid_bytes = self.valid_bytes.saturating_add(bytes.len());
647
+ self.non_ascii_bytes = self
648
+ .non_ascii_bytes
649
+ .saturating_add(bytes.iter().filter(|byte| !byte.is_ascii()).count());
650
+ }
651
+
652
+ fn clear_prefix(&self) -> bool {
653
+ self.valid_bytes >= UTF8_SEGMENT_MIN_BYTES
654
+ && self.non_ascii_bytes >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
655
+ }
656
+ }
657
+
658
+ fn is_disallowed_legacy_character(character: char) -> bool {
659
+ let value = character as u32;
660
+ (value <= 0x1F && !matches!(character, '\t' | '\n' | '\r'))
661
+ || (0x80..=0x9F).contains(&value)
662
+ || (0xFDD0..=0xFDEF).contains(&value)
663
+ || value & 0xFFFF == 0xFFFE
664
+ || value & 0xFFFF == 0xFFFF
665
+ }
666
+
667
+ #[derive(Clone, Default)]
668
+ struct ValidationStats {
669
+ decoded_any: bool,
670
+ newline_count: usize,
671
+ has_trailing_newline: bool,
672
+ has_non_ascii: bool,
673
+ has_iso_2022_escape: bool,
674
+ previous_was_escape: bool,
675
+ }
676
+
677
+ impl ValidationStats {
678
+ fn observe(&mut self, text: &str) {
679
+ if text.is_empty() {
680
+ return;
681
+ }
682
+ self.decoded_any = true;
683
+ self.has_non_ascii |= !text.is_ascii();
684
+ for byte in text.bytes() {
685
+ if self.previous_was_escape && matches!(byte, b'(' | b'$') {
686
+ self.has_iso_2022_escape = true;
687
+ }
688
+ self.previous_was_escape = byte == 0x1B;
689
+ }
690
+ self.newline_count = self
691
+ .newline_count
692
+ .saturating_add(text.bytes().filter(|byte| *byte == b'\n').count());
693
+ self.has_trailing_newline = text.ends_with('\n');
694
+ }
695
+
696
+ fn total_lines(&self) -> usize {
697
+ if self.decoded_any {
698
+ self.newline_count.saturating_add(1)
699
+ } else {
700
+ 0
701
+ }
702
+ }
703
+ }
704
+
705
+ enum DecodeFailure {
706
+ Io(io::Error),
707
+ Malformed,
708
+ }
709
+
710
+ /// A strict decoder used only to establish irreversible capture-time failures.
711
+ pub(crate) struct StrictStreamingValidator {
712
+ decoder: ChunkDecoder,
713
+ malformed: bool,
714
+ }
715
+
716
+ impl StrictStreamingValidator {
717
+ fn new(detected: &DetectedEncoding) -> Self {
718
+ Self {
719
+ decoder: chunk_decoder_for(detected),
720
+ malformed: false,
721
+ }
722
+ }
723
+
724
+ /// Returns true once the already-observed prefix can never decode successfully.
725
+ pub(crate) fn feed(&mut self, input: &[u8]) -> bool {
726
+ if self.malformed {
727
+ return true;
728
+ }
729
+ let result = match &mut self.decoder {
730
+ ChunkDecoder::Utf8 { carry } => decode_utf8_chunk(carry, input, false),
731
+ ChunkDecoder::EncodingRs(decoder) => decode_encoding_rs_chunk(decoder, input, false),
732
+ ChunkDecoder::Utf32 {
733
+ little_endian,
734
+ carry,
735
+ } => decode_utf32_chunk(carry, input, false, *little_endian),
736
+ };
737
+ self.malformed = result.is_err();
738
+ self.malformed
739
+ }
740
+ }
741
+
742
+ /// Capture-time decoder selected by an authoritative BOM.
743
+ pub(crate) struct BomStreamingValidator {
744
+ pub(crate) validator: StrictStreamingValidator,
745
+ pub(crate) bom_len: usize,
746
+ pub(crate) encoding: &'static str,
747
+ pub(crate) is_utf8: bool,
748
+ }
749
+
750
+ /// Builds the same strict decoder and matching-BOM offset as explicit validation.
751
+ pub(crate) fn explicit_streaming_validator(
752
+ value: &str,
753
+ prefix: &[u8],
754
+ ) -> Result<(StrictStreamingValidator, usize), EncodingRejection> {
755
+ let detected = explicit_detected_encoding(value, prefix)?;
756
+ Ok((
757
+ StrictStreamingValidator::new(&detected),
758
+ detected.bom_len as usize,
759
+ ))
760
+ }
761
+
762
+ /// Builds the same strict decoder selected by the normal BOM precedence.
763
+ pub(crate) fn bom_streaming_validator(prefix: &[u8]) -> Option<BomStreamingValidator> {
764
+ let detected = bom_detected_encoding(prefix)?;
765
+ let encoding = match detected.origin {
766
+ EncodingOrigin::Bom(encoding) => encoding,
767
+ _ => unreachable!("BOM detection always records a BOM origin"),
768
+ };
769
+ Some(BomStreamingValidator {
770
+ validator: StrictStreamingValidator::new(&detected),
771
+ bom_len: detected.bom_len as usize,
772
+ encoding,
773
+ is_utf8: detected.kind == EncodingKind::EncodingRs(UTF_8),
774
+ })
775
+ }
776
+
777
+ impl From<io::Error> for DecodeFailure {
778
+ fn from(error: io::Error) -> Self {
779
+ Self::Io(error)
780
+ }
781
+ }
782
+
783
+ enum ChunkDecoder {
784
+ Utf8 { carry: Vec<u8> },
785
+ EncodingRs(encoding_rs::Decoder),
786
+ Utf32 { little_endian: bool, carry: Vec<u8> },
787
+ }
788
+
789
+ struct DecodedChunkReader<'a> {
790
+ source: SourceReader<'a>,
791
+ decoder: ChunkDecoder,
792
+ finished: bool,
793
+ }
794
+
795
+ impl<'a> DecodedChunkReader<'a> {
796
+ fn open(source: ByteSource<'a>, detected: DetectedEncoding) -> io::Result<Self> {
797
+ Ok(Self {
798
+ source: SourceReader::open(source, detected.bom_len)?,
799
+ decoder: chunk_decoder_for(&detected),
800
+ finished: false,
801
+ })
802
+ }
803
+
804
+ fn next_chunk(&mut self) -> Result<Option<String>, DecodeFailure> {
805
+ if self.finished {
806
+ return Ok(None);
807
+ }
808
+ loop {
809
+ let mut input = [0_u8; DECODE_CHUNK_BYTES];
810
+ let count = self.source.read(&mut input)?;
811
+ let is_last = count == 0;
812
+ let decoded = match &mut self.decoder {
813
+ ChunkDecoder::Utf8 { carry } => decode_utf8_chunk(carry, &input[..count], is_last)?,
814
+ ChunkDecoder::EncodingRs(decoder) => {
815
+ decode_encoding_rs_chunk(decoder, &input[..count], is_last)?
816
+ }
817
+ ChunkDecoder::Utf32 {
818
+ little_endian,
819
+ carry,
820
+ } => decode_utf32_chunk(carry, &input[..count], is_last, *little_endian)?,
821
+ };
822
+ if is_last {
823
+ self.finished = true;
824
+ }
825
+ if !decoded.is_empty() {
826
+ return Ok(Some(decoded));
827
+ }
828
+ if self.finished {
829
+ return Ok(None);
830
+ }
831
+ }
832
+ }
833
+ }
834
+
835
+ fn chunk_decoder_for(detected: &DetectedEncoding) -> ChunkDecoder {
836
+ match detected.kind {
837
+ EncodingKind::EncodingRs(encoding) if encoding == UTF_8 => {
838
+ ChunkDecoder::Utf8 { carry: Vec::new() }
839
+ }
840
+ EncodingKind::EncodingRs(encoding) => {
841
+ ChunkDecoder::EncodingRs(encoding.new_decoder_without_bom_handling())
842
+ }
843
+ EncodingKind::Utf32Le => ChunkDecoder::Utf32 {
844
+ little_endian: true,
845
+ carry: Vec::new(),
846
+ },
847
+ EncodingKind::Utf32Be => ChunkDecoder::Utf32 {
848
+ little_endian: false,
849
+ carry: Vec::new(),
850
+ },
851
+ }
852
+ }
853
+
854
+ fn decode_utf8_chunk(
855
+ carry: &mut Vec<u8>,
856
+ input: &[u8],
857
+ is_last: bool,
858
+ ) -> Result<String, DecodeFailure> {
859
+ let mut bytes = std::mem::take(carry);
860
+ bytes.extend_from_slice(input);
861
+ match std::str::from_utf8(&bytes) {
862
+ Ok(text) => Ok(text.to_string()),
863
+ Err(error) if error.error_len().is_none() && !is_last => {
864
+ let valid_up_to = error.valid_up_to();
865
+ let tail = bytes.split_off(valid_up_to);
866
+ *carry = tail;
867
+ std::str::from_utf8(&bytes)
868
+ .map(str::to_string)
869
+ .map_err(|_| DecodeFailure::Malformed)
870
+ }
871
+ Err(_) => Err(DecodeFailure::Malformed),
872
+ }
873
+ }
874
+
875
+ fn decode_encoding_rs_chunk(
876
+ decoder: &mut encoding_rs::Decoder,
877
+ input: &[u8],
878
+ is_last: bool,
879
+ ) -> Result<String, DecodeFailure> {
880
+ let mut consumed = 0_usize;
881
+ let mut output = String::new();
882
+ loop {
883
+ let remaining = &input[consumed..];
884
+ let capacity = decoder
885
+ .max_utf8_buffer_length_without_replacement(remaining.len())
886
+ .ok_or(DecodeFailure::Malformed)?
887
+ .max(4);
888
+ let mut decoded = String::with_capacity(capacity);
889
+ let (result, read) =
890
+ decoder.decode_to_string_without_replacement(remaining, &mut decoded, is_last);
891
+ consumed = consumed.saturating_add(read);
892
+ output.push_str(&decoded);
893
+ match result {
894
+ DecoderResult::InputEmpty => return Ok(output),
895
+ DecoderResult::OutputFull => continue,
896
+ DecoderResult::Malformed(_, _) => return Err(DecodeFailure::Malformed),
897
+ }
898
+ }
899
+ }
900
+
901
+ fn decode_utf32_chunk(
902
+ carry: &mut Vec<u8>,
903
+ input: &[u8],
904
+ is_last: bool,
905
+ little_endian: bool,
906
+ ) -> Result<String, DecodeFailure> {
907
+ let mut bytes = std::mem::take(carry);
908
+ bytes.extend_from_slice(input);
909
+ if is_last && !bytes.len().is_multiple_of(4) {
910
+ return Err(DecodeFailure::Malformed);
911
+ }
912
+ let complete_len = bytes.len() / 4 * 4;
913
+ if !is_last {
914
+ *carry = bytes.split_off(complete_len);
915
+ }
916
+ let mut output = String::with_capacity(complete_len);
917
+ for raw in bytes[..complete_len].as_chunks::<4>().0 {
918
+ let unit = if little_endian {
919
+ u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]])
920
+ } else {
921
+ u32::from_be_bytes([raw[0], raw[1], raw[2], raw[3]])
922
+ };
923
+ let Some(character) = char::from_u32(unit) else {
924
+ return Err(DecodeFailure::Malformed);
925
+ };
926
+ output.push(character);
927
+ }
928
+ Ok(output)
929
+ }
930
+
931
+ /// UTF-8 streaming reader for validated text that never emits replacement characters.
932
+ pub(crate) struct Utf8Reader<'a> {
933
+ decoder: DecodedChunkReader<'a>,
934
+ pending: Vec<u8>,
935
+ pending_offset: usize,
936
+ }
937
+
938
+ impl<'a> Utf8Reader<'a> {
939
+ fn open_source(source: ByteSource<'a>, detected: DetectedEncoding) -> io::Result<Self> {
940
+ Ok(Self {
941
+ decoder: DecodedChunkReader::open(source, detected)?,
942
+ pending: Vec::new(),
943
+ pending_offset: 0,
944
+ })
945
+ }
946
+ }
947
+
948
+ impl Read for Utf8Reader<'_> {
949
+ fn read(&mut self, output: &mut [u8]) -> io::Result<usize> {
950
+ if output.is_empty() {
951
+ return Ok(0);
952
+ }
953
+ loop {
954
+ if self.pending_offset < self.pending.len() {
955
+ let available = &self.pending[self.pending_offset..];
956
+ let count = available.len().min(output.len());
957
+ output[..count].copy_from_slice(&available[..count]);
958
+ self.pending_offset += count;
959
+ return Ok(count);
960
+ }
961
+ match self.decoder.next_chunk() {
962
+ Ok(Some(chunk)) => {
963
+ self.pending = chunk.into_bytes();
964
+ self.pending_offset = 0;
965
+ }
966
+ Ok(None) => return Ok(0),
967
+ Err(DecodeFailure::Io(error)) => return Err(error),
968
+ Err(DecodeFailure::Malformed) => {
969
+ return Err(io::Error::new(
970
+ io::ErrorKind::InvalidData,
971
+ "file changed after encoding validation",
972
+ ));
973
+ }
974
+ }
975
+ }
976
+ }
977
+ }
978
+
979
+ fn decode_bytes(bytes: &[u8], detected: &DetectedEncoding) -> Option<String> {
980
+ let content = bytes.get(detected.bom_len as usize..)?;
981
+ match detected.kind {
982
+ EncodingKind::EncodingRs(encoding) => encoding
983
+ .decode_without_bom_handling_and_without_replacement(content)
984
+ .map(|text| text.into_owned()),
985
+ EncodingKind::Utf32Le => decode_utf32_bytes(content, true),
986
+ EncodingKind::Utf32Be => decode_utf32_bytes(content, false),
987
+ }
988
+ }
989
+
990
+ fn decode_utf32_bytes(bytes: &[u8], little_endian: bool) -> Option<String> {
991
+ if !bytes.len().is_multiple_of(4) {
992
+ return None;
993
+ }
994
+ let mut output = String::with_capacity(bytes.len());
995
+ for raw in bytes.as_chunks::<4>().0 {
996
+ let unit = if little_endian {
997
+ u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]])
998
+ } else {
999
+ u32::from_be_bytes([raw[0], raw[1], raw[2], raw[3]])
1000
+ };
1001
+ output.push(char::from_u32(unit)?);
1002
+ }
1003
+ Some(output)
1004
+ }
1005
+
1006
+ #[cfg(test)]
1007
+ mod tests {
1008
+ use super::{
1009
+ DECODE_CHUNK_BYTES, EncodingDecision, decode_text, has_binary_nul,
1010
+ is_disallowed_legacy_character, validate_file_encoding,
1011
+ };
1012
+ use std::io::Read;
1013
+
1014
+ #[test]
1015
+ fn nul_is_binary_but_unicode_boms_take_precedence() {
1016
+ assert!(decode_text(b"text\0binary").is_none());
1017
+ let decoded = decode_text(&[0xFF, 0xFE, b'A', 0, b'B', 0]).unwrap();
1018
+ assert_eq!(decoded.text, "AB");
1019
+ assert_eq!(decoded.source_encoding.as_deref(), Some("UTF-16LE"));
1020
+
1021
+ let decoded = decode_text(&[0x00, 0x00, 0xFE, 0xFF, 0x00, 0x00, 0x00, b'A']).unwrap();
1022
+ assert_eq!(decoded.text, "A");
1023
+ assert_eq!(decoded.source_encoding.as_deref(), Some("UTF-32BE"));
1024
+ }
1025
+
1026
+ #[test]
1027
+ fn low_evidence_legacy_bytes_are_never_returned_as_guessed_text() {
1028
+ assert!(decode_text(b"valid\xFFtail").is_none());
1029
+ }
1030
+
1031
+ #[test]
1032
+ fn binary_probe_stops_after_exactly_eight_kibibytes() {
1033
+ let mut inside = vec![b'a'; 8 * 1024];
1034
+ inside[8 * 1024 - 1] = 0;
1035
+ assert!(has_binary_nul(&inside));
1036
+
1037
+ let mut outside = vec![b'a'; 8 * 1024];
1038
+ outside.push(0);
1039
+ assert!(!has_binary_nul(&outside));
1040
+ }
1041
+
1042
+ #[test]
1043
+ fn arbitrary_byte_fuzz_corpus_never_panics() {
1044
+ let mut state = 0xA5A5_5A5A_1234_5678_u64;
1045
+ for length in 0..=256 {
1046
+ let mut bytes = Vec::with_capacity(length);
1047
+ for _ in 0..length {
1048
+ state = state
1049
+ .wrapping_mul(6_364_136_223_846_793_005)
1050
+ .wrapping_add(1_442_695_040_888_963_407);
1051
+ bytes.push((state >> 32) as u8);
1052
+ }
1053
+ let _ = decode_text(&bytes);
1054
+ }
1055
+ for byte in 0_u8..=u8::MAX {
1056
+ let _ = decode_text(&[byte]);
1057
+ let _ = decode_text(&[0xA1, byte]);
1058
+ let _ = decode_text(&[byte, 0xFF, 0x00, 0x7F]);
1059
+ }
1060
+ }
1061
+
1062
+ #[test]
1063
+ fn legacy_hard_filter_rejects_controls_and_unicode_noncharacters() {
1064
+ for character in [
1065
+ '\0',
1066
+ '\u{0001}',
1067
+ '\u{0080}',
1068
+ '\u{009F}',
1069
+ '\u{FDD0}',
1070
+ '\u{10FFFF}',
1071
+ ] {
1072
+ assert!(is_disallowed_legacy_character(character));
1073
+ }
1074
+ for character in ['\t', '\n', '\r', 'a', '界'] {
1075
+ assert!(!is_disallowed_legacy_character(character));
1076
+ }
1077
+ }
1078
+
1079
+ #[test]
1080
+ fn strict_decoders_preserve_characters_split_at_internal_chunk_boundaries() {
1081
+ let temp = tempfile::tempdir().unwrap();
1082
+
1083
+ let utf8_path = temp.path().join("utf8-boundary.txt");
1084
+ let utf8_text = format!("{}界", "a".repeat(DECODE_CHUNK_BYTES - 1));
1085
+ std::fs::write(&utf8_path, utf8_text.as_bytes()).unwrap();
1086
+ assert_decoded_file(&utf8_path, &utf8_text);
1087
+
1088
+ let utf16_path = temp.path().join("utf16-boundary.txt");
1089
+ let utf16_text = format!("{}😀Z", "a".repeat(DECODE_CHUNK_BYTES / 2 - 1));
1090
+ let mut utf16_bytes = vec![0xFF, 0xFE];
1091
+ for unit in utf16_text.encode_utf16() {
1092
+ utf16_bytes.extend(unit.to_le_bytes());
1093
+ }
1094
+ std::fs::write(&utf16_path, utf16_bytes).unwrap();
1095
+ assert_decoded_file(&utf16_path, &utf16_text);
1096
+
1097
+ let gbk_path = temp.path().join("gbk-boundary.txt");
1098
+ let mut gbk_bytes = vec![b'a'];
1099
+ for _ in 0..=DECODE_CHUNK_BYTES / 2 {
1100
+ gbk_bytes.extend([0xD6, 0xD0]);
1101
+ }
1102
+ std::fs::write(&gbk_path, gbk_bytes).unwrap();
1103
+ let gbk_text = format!("a{}", "中".repeat(DECODE_CHUNK_BYTES / 2 + 1));
1104
+ assert_decoded_file(&gbk_path, &gbk_text);
1105
+ }
1106
+
1107
+ fn assert_decoded_file(path: &std::path::Path, expected: &str) {
1108
+ let EncodingDecision::Text(validated) = validate_file_encoding(path, None).unwrap() else {
1109
+ panic!("expected trusted text");
1110
+ };
1111
+ let mut reader = validated
1112
+ .open_source_reader(super::ByteSource::File(path))
1113
+ .unwrap();
1114
+ let mut actual = String::new();
1115
+ reader.read_to_string(&mut actual).unwrap();
1116
+ assert_eq!(actual, expected);
1117
+ }
1118
+ }