dsh-ops 0.0.0-stage → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +202 -0
  2. package/LICENSE +30 -0
  3. package/NOTICE +106 -0
  4. package/PROVENANCE.md +435 -0
  5. package/README.en.md +126 -0
  6. package/README.md +115 -2
  7. package/README.zh.md +116 -0
  8. package/bin/dsh-ops.mjs +1216 -0
  9. package/cordis.patch.yml +160 -0
  10. package/docs/manual-validation.md +53 -0
  11. package/docs/release-0.2.1.md +72 -0
  12. package/docs/schema-baseline.json +64 -0
  13. package/docs/schema-current.json +84 -0
  14. package/docs/schema-measurement.md +17 -0
  15. package/dsh-plugin.json +88 -0
  16. package/icon.svg +12 -0
  17. package/lib/binary.js +409 -0
  18. package/lib/config.js +198 -0
  19. package/lib/handshake.js +252 -0
  20. package/lib/index.js +108 -0
  21. package/lib/jobs.js +42 -0
  22. package/lib/policy.js +64 -0
  23. package/lib/presentation.js +63 -0
  24. package/lib/profile-install.js +61 -0
  25. package/lib/rust.js +194 -0
  26. package/lib/session-shells.js +78 -0
  27. package/lib/shells.js +998 -0
  28. package/lib/tools.js +657 -0
  29. package/locale/en.json +6 -0
  30. package/locale/zh.json +6 -0
  31. package/package.json +114 -4
  32. package/vendor/fastctx/Cargo.lock +3210 -0
  33. package/vendor/fastctx/Cargo.toml +94 -0
  34. package/vendor/fastctx/FORK.md +119 -0
  35. package/vendor/fastctx/LICENSE-APACHE +201 -0
  36. package/vendor/fastctx/NOTICE +40 -0
  37. package/vendor/fastctx/README.md +439 -0
  38. package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
  39. package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
  40. package/vendor/fastctx/UPSTREAM.md +49 -0
  41. package/vendor/fastctx/build.rs +413 -0
  42. package/vendor/fastctx/src/background_status.rs +403 -0
  43. package/vendor/fastctx/src/binary.rs +75 -0
  44. package/vendor/fastctx/src/bounded_sort.rs +500 -0
  45. package/vendor/fastctx/src/budget.rs +781 -0
  46. package/vendor/fastctx/src/cli/mod.rs +110 -0
  47. package/vendor/fastctx/src/context_guard.rs +289 -0
  48. package/vendor/fastctx/src/control/mod.rs +6 -0
  49. package/vendor/fastctx/src/control/paths.rs +49 -0
  50. package/vendor/fastctx/src/control/settings.rs +753 -0
  51. package/vendor/fastctx/src/control/transaction.rs +531 -0
  52. package/vendor/fastctx/src/edit/document.rs +535 -0
  53. package/vendor/fastctx/src/edit/locks.rs +371 -0
  54. package/vendor/fastctx/src/edit/mod.rs +213 -0
  55. package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
  56. package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
  57. package/vendor/fastctx/src/edit/private_storage.rs +234 -0
  58. package/vendor/fastctx/src/edit/replace.rs +1030 -0
  59. package/vendor/fastctx/src/edit_server.rs +53 -0
  60. package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
  61. package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
  62. package/vendor/fastctx/src/encoding.rs +1118 -0
  63. package/vendor/fastctx/src/file_executor.rs +1151 -0
  64. package/vendor/fastctx/src/file_snapshot.rs +1491 -0
  65. package/vendor/fastctx/src/glob_filter.rs +98 -0
  66. package/vendor/fastctx/src/glob_tool.rs +653 -0
  67. package/vendor/fastctx/src/grep_sink.rs +1162 -0
  68. package/vendor/fastctx/src/grep_tool.rs +2449 -0
  69. package/vendor/fastctx/src/lib.rs +45 -0
  70. package/vendor/fastctx/src/main.rs +15 -0
  71. package/vendor/fastctx/src/model.rs +51 -0
  72. package/vendor/fastctx/src/model_guidance.rs +62 -0
  73. package/vendor/fastctx/src/operation.rs +356 -0
  74. package/vendor/fastctx/src/ordered_window.rs +1235 -0
  75. package/vendor/fastctx/src/os_environment.rs +414 -0
  76. package/vendor/fastctx/src/path_codec.rs +850 -0
  77. package/vendor/fastctx/src/paths.rs +244 -0
  78. package/vendor/fastctx/src/process_identity.rs +763 -0
  79. package/vendor/fastctx/src/process_policy.rs +74 -0
  80. package/vendor/fastctx/src/read_tool/batch.rs +496 -0
  81. package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
  82. package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
  83. package/vendor/fastctx/src/read_tool/mod.rs +245 -0
  84. package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
  85. package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
  86. package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
  87. package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
  88. package/vendor/fastctx/src/render_plan.rs +468 -0
  89. package/vendor/fastctx/src/runtime/activity.rs +159 -0
  90. package/vendor/fastctx/src/runtime/hosts.rs +99 -0
  91. package/vendor/fastctx/src/runtime/journal.rs +556 -0
  92. package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
  93. package/vendor/fastctx/src/runtime/mod.rs +746 -0
  94. package/vendor/fastctx/src/runtime/protocol.rs +296 -0
  95. package/vendor/fastctx/src/runtime/session.rs +536 -0
  96. package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
  97. package/vendor/fastctx/src/search_parallelism.rs +106 -0
  98. package/vendor/fastctx/src/search_text.rs +227 -0
  99. package/vendor/fastctx/src/server.rs +359 -0
  100. package/vendor/fastctx/src/server_manifest.rs +468 -0
  101. package/vendor/fastctx/src/server_support.rs +826 -0
  102. package/vendor/fastctx/src/session.rs +629 -0
  103. package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
  104. package/vendor/fastctx/src/shell/bash.rs +263 -0
  105. package/vendor/fastctx/src/shell/buffer.rs +108 -0
  106. package/vendor/fastctx/src/shell/encoding.rs +403 -0
  107. package/vendor/fastctx/src/shell/foreground.rs +115 -0
  108. package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
  109. package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
  110. package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
  111. package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
  112. package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
  113. package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
  114. package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
  115. package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
  116. package/vendor/fastctx/src/shell/mod.rs +345 -0
  117. package/vendor/fastctx/src/shell/normalize.rs +389 -0
  118. package/vendor/fastctx/src/shell/output.rs +406 -0
  119. package/vendor/fastctx/src/shell/process.rs +493 -0
  120. package/vendor/fastctx/src/shell_server.rs +156 -0
  121. package/vendor/fastctx/src/skip_report.rs +83 -0
  122. package/vendor/fastctx/src/stdio_transport.rs +177 -0
  123. package/vendor/fastctx/src/tool_schema.rs +204 -0
  124. package/vendor/fastctx/src/traversal.rs +846 -0
  125. package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
  126. package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
  127. package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
  128. package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
  129. package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
  130. package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
  131. package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
  132. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
  133. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
  134. package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
  135. package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
  136. package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
  137. package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
  138. package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
  139. package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
  140. package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
@@ -0,0 +1,1678 @@
1
+ //! Decision-equivalent encoding validation over immutable snapshots.
2
+
3
+ use super::{
4
+ BINARY_PROBE_BYTES, ByteSource, DECODE_CHUNK_BYTES, DetectedEncoding, EncodingDecision,
5
+ EncodingKind, EncodingOrigin, EncodingRejection, FIXED_LEGACY_ENCODINGS, LEGACY_EVIDENCE_BYTES,
6
+ LEGACY_SEGMENT_MAX_BYTES, SourceReader, UTF8_SEGMENT_MIN_BYTES,
7
+ UTF8_SEGMENT_MIN_NON_ASCII_BYTES, Utf8PrefixScanner, ValidationStats, bom_detected_encoding,
8
+ explicit_detected_encoding, has_binary_nul, is_disallowed_legacy_character, is_legacy_encoding,
9
+ validated,
10
+ };
11
+ use crate::binary::detect_binary_type;
12
+ use crate::file_snapshot::SealedSnapshot;
13
+ #[cfg(test)]
14
+ use crate::operation::TestStage;
15
+ use crate::operation::{WorkCheckpoint, WorkStop};
16
+ use chardetng::{EncodingDetector, Iso2022JpDetection, Utf8Detection};
17
+ use encoding_rs::{DecoderResult, EncoderResult, Encoding, UTF_8};
18
+ use std::collections::VecDeque;
19
+ use std::io::{self, Read};
20
+
21
+ const SEGMENT_CACHE_CAPACITY: usize = 256;
22
+ const SEGMENT_COMPACT_THRESHOLD: usize = 16 * 1024;
23
+
24
+ /// Failure channels that must remain distinct from an encoding rejection.
25
+ #[derive(Debug)]
26
+ pub(crate) enum EncodingPipelineFailure {
27
+ Io(io::Error),
28
+ Stopped(WorkStop),
29
+ }
30
+
31
+ impl From<io::Error> for EncodingPipelineFailure {
32
+ fn from(error: io::Error) -> Self {
33
+ Self::Io(error)
34
+ }
35
+ }
36
+
37
+ /// Validates one immutable capture while honoring request cancellation and epoch retirement.
38
+ pub(crate) fn validate_snapshot_encoding(
39
+ snapshot: &SealedSnapshot,
40
+ explicit_encoding: Option<&str>,
41
+ operation: Option<&dyn WorkCheckpoint>,
42
+ ) -> Result<EncodingDecision, EncodingPipelineFailure> {
43
+ validate_source(ByteSource::Snapshot(snapshot), explicit_encoding, operation)
44
+ }
45
+
46
+ /// Runs the replacement pipeline for shared read/edit callers and test byte sources.
47
+ pub(super) fn validate_source(
48
+ source: ByteSource<'_>,
49
+ explicit_encoding: Option<&str>,
50
+ operation: Option<&dyn WorkCheckpoint>,
51
+ ) -> Result<EncodingDecision, EncodingPipelineFailure> {
52
+ checkpoint(operation)?;
53
+ let result = match source {
54
+ ByteSource::Bytes(bytes) => validate_with_prefix(
55
+ source,
56
+ &bytes[..bytes.len().min(BINARY_PROBE_BYTES)],
57
+ explicit_encoding,
58
+ operation,
59
+ ),
60
+ ByteSource::Snapshot(snapshot) => {
61
+ if let Some(prefix) = snapshot.memory_prefix(BINARY_PROBE_BYTES) {
62
+ validate_with_prefix(source, prefix, explicit_encoding, operation)
63
+ } else {
64
+ read_prefix(source, operation).and_then(|prefix| {
65
+ validate_with_prefix(source, &prefix, explicit_encoding, operation)
66
+ })
67
+ }
68
+ }
69
+ ByteSource::File(_) => read_prefix(source, operation)
70
+ .and_then(|prefix| validate_with_prefix(source, &prefix, explicit_encoding, operation)),
71
+ };
72
+ prefer_stop(operation, result)
73
+ }
74
+
75
+ fn validate_with_prefix(
76
+ source: ByteSource<'_>,
77
+ prefix: &[u8],
78
+ explicit_encoding: Option<&str>,
79
+ operation: Option<&dyn WorkCheckpoint>,
80
+ ) -> Result<EncodingDecision, EncodingPipelineFailure> {
81
+ let pipeline = EncodingPipeline {
82
+ source,
83
+ operation,
84
+ chunk_bytes: DECODE_CHUNK_BYTES,
85
+ };
86
+ pipeline.validate(prefix, explicit_encoding)
87
+ }
88
+
89
+ fn read_prefix(
90
+ source: ByteSource<'_>,
91
+ operation: Option<&dyn WorkCheckpoint>,
92
+ ) -> Result<Vec<u8>, EncodingPipelineFailure> {
93
+ let mut reader = SourceReader::open(source, 0)?;
94
+ let mut prefix = Vec::with_capacity(BINARY_PROBE_BYTES);
95
+ let mut buffer = [0_u8; BINARY_PROBE_BYTES];
96
+ while prefix.len() < BINARY_PROBE_BYTES {
97
+ encoding_chunk_checkpoint(operation)?;
98
+ let count = reader.read(&mut buffer[..BINARY_PROBE_BYTES - prefix.len()])?;
99
+ checkpoint(operation)?;
100
+ if count == 0 {
101
+ break;
102
+ }
103
+ prefix.extend_from_slice(&buffer[..count]);
104
+ }
105
+ Ok(prefix)
106
+ }
107
+
108
+ struct EncodingPipeline<'a> {
109
+ source: ByteSource<'a>,
110
+ operation: Option<&'a dyn WorkCheckpoint>,
111
+ chunk_bytes: usize,
112
+ }
113
+
114
+ impl EncodingPipeline<'_> {
115
+ fn validate(
116
+ &self,
117
+ prefix: &[u8],
118
+ explicit_encoding: Option<&str>,
119
+ ) -> Result<EncodingDecision, EncodingPipelineFailure> {
120
+ checkpoint(self.operation)?;
121
+ if let Some(value) = explicit_encoding {
122
+ let detected = match explicit_detected_encoding(value, prefix) {
123
+ Ok(detected) => detected,
124
+ Err(rejection) => return Ok(EncodingDecision::Rejected(rejection)),
125
+ };
126
+ let options = PassOptions {
127
+ enforce_legacy_hard_checks: false,
128
+ scan_legacy_segments: None,
129
+ probe_raw_utf8: is_legacy_encoding(detected.kind),
130
+ };
131
+ let result = self.validate_selected(&detected, options)?;
132
+ return Ok(match result.stats {
133
+ Some(stats) => {
134
+ EncodingDecision::Text(validated(detected, stats, result.raw_is_multibyte_utf8))
135
+ }
136
+ None => EncodingDecision::Rejected(EncodingRejection::ExplicitMalformed {
137
+ encoding: value.to_string(),
138
+ }),
139
+ });
140
+ }
141
+
142
+ if let Some(detected) = bom_detected_encoding(prefix) {
143
+ if detected.kind == EncodingKind::EncodingRs(UTF_8) && has_binary_nul(prefix) {
144
+ return Ok(EncodingDecision::Binary);
145
+ }
146
+ let result = self.validate_selected(&detected, PassOptions::plain())?;
147
+ return Ok(match result.stats {
148
+ Some(stats) => EncodingDecision::Text(validated(detected, stats, false)),
149
+ None => EncodingDecision::Rejected(EncodingRejection::BomMismatch {
150
+ encoding: detected
151
+ .source_encoding
152
+ .or(match detected.origin {
153
+ EncodingOrigin::Bom(encoding) => Some(encoding),
154
+ _ => None,
155
+ })
156
+ .unwrap_or("UTF-8"),
157
+ }),
158
+ });
159
+ }
160
+
161
+ if has_binary_nul(prefix) {
162
+ return Ok(EncodingDecision::Binary);
163
+ }
164
+
165
+ let utf8 = DetectedEncoding {
166
+ kind: EncodingKind::EncodingRs(UTF_8),
167
+ bom_len: 0,
168
+ source_encoding: None,
169
+ origin: EncodingOrigin::Automatic,
170
+ };
171
+ if let Some(stats) = self.validate_selected(&utf8, PassOptions::plain())?.stats {
172
+ if stats.has_iso_2022_escape {
173
+ return Ok(EncodingDecision::Rejected(
174
+ EncodingRejection::Iso2022JpSignature,
175
+ ));
176
+ }
177
+ return Ok(EncodingDecision::Text(validated(utf8, stats, false)));
178
+ }
179
+ if detect_binary_type(prefix).is_some() {
180
+ return Ok(EncodingDecision::Binary);
181
+ }
182
+
183
+ let nomination = self.scan_nomination()?;
184
+ if let Some(conflict_hex_offset) = nomination.conflict_hex_offset {
185
+ return Ok(EncodingDecision::Rejected(
186
+ EncodingRejection::MixedOrInconsistent {
187
+ conflict_hex_offset: Some(conflict_hex_offset),
188
+ },
189
+ ));
190
+ }
191
+
192
+ let candidate = nomination.encoding;
193
+ let candidate_detected = automatic_detected(candidate);
194
+ let mut memo = ValidationMemo::default();
195
+ if nomination.non_ascii_bytes >= LEGACY_EVIDENCE_BYTES {
196
+ let result = self.validate_selected(
197
+ &candidate_detected,
198
+ PassOptions {
199
+ enforce_legacy_hard_checks: true,
200
+ scan_legacy_segments: Some(candidate),
201
+ probe_raw_utf8: false,
202
+ },
203
+ )?;
204
+ memo.insert(candidate, result.stats.clone());
205
+ if let Some(stats) = result.stats {
206
+ if result.segments_inconsistent {
207
+ return Ok(EncodingDecision::Rejected(
208
+ EncodingRejection::MixedOrInconsistent {
209
+ conflict_hex_offset: None,
210
+ },
211
+ ));
212
+ }
213
+ if !candidate.is_single_byte() {
214
+ return Ok(EncodingDecision::Text(validated(
215
+ candidate_detected,
216
+ stats,
217
+ false,
218
+ )));
219
+ }
220
+ }
221
+ }
222
+
223
+ let mut candidates = Vec::new();
224
+ for (label, encoding) in FIXED_LEGACY_ENCODINGS {
225
+ let stats = match memo.get(encoding) {
226
+ Some(cached) => {
227
+ candidate_checkpoint(self.operation)?;
228
+ cached
229
+ }
230
+ None => {
231
+ let detected = automatic_detected(encoding);
232
+ let result = self.validate_selected(
233
+ &detected,
234
+ PassOptions {
235
+ enforce_legacy_hard_checks: true,
236
+ scan_legacy_segments: None,
237
+ probe_raw_utf8: false,
238
+ },
239
+ )?;
240
+ memo.insert(encoding, result.stats.clone());
241
+ result.stats
242
+ }
243
+ };
244
+ if stats.is_some() {
245
+ candidates.push(label);
246
+ }
247
+ }
248
+ Ok(if candidates.is_empty() {
249
+ EncodingDecision::Rejected(EncodingRejection::Undecodable)
250
+ } else {
251
+ EncodingDecision::Rejected(EncodingRejection::Ambiguous { candidates })
252
+ })
253
+ }
254
+
255
+ fn scan_nomination(&self) -> Result<Nomination, EncodingPipelineFailure> {
256
+ let mut reader = SourceReader::open(self.source, 0)?;
257
+ let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
258
+ let mut conflict_scanner = Some(Utf8PrefixScanner::default());
259
+ let mut conflict_hex_offset = None;
260
+ let mut non_ascii_bytes = 0_usize;
261
+ let mut input = [0_u8; DECODE_CHUNK_BYTES];
262
+ loop {
263
+ encoding_chunk_checkpoint(self.operation)?;
264
+ let count = reader.read(&mut input[..self.chunk_bytes])?;
265
+ checkpoint(self.operation)?;
266
+ if count == 0 {
267
+ detector.feed(&[], true);
268
+ if conflict_hex_offset.is_none()
269
+ && let Some(scanner) = conflict_scanner.take()
270
+ {
271
+ conflict_hex_offset = scanner.finish();
272
+ }
273
+ break;
274
+ }
275
+ let bytes = &input[..count];
276
+ non_ascii_bytes = non_ascii_bytes
277
+ .saturating_add(bytes.iter().filter(|byte| !byte.is_ascii()).count());
278
+ detector.feed(bytes, false);
279
+ if conflict_hex_offset.is_none()
280
+ && let Some(scanner) = conflict_scanner.as_mut()
281
+ {
282
+ conflict_hex_offset = scanner.push(bytes);
283
+ }
284
+ }
285
+ Ok(Nomination {
286
+ encoding: detector.guess(None, Utf8Detection::Deny),
287
+ non_ascii_bytes,
288
+ conflict_hex_offset,
289
+ })
290
+ }
291
+
292
+ fn validate_selected(
293
+ &self,
294
+ detected: &DetectedEncoding,
295
+ options: PassOptions,
296
+ ) -> Result<PassResult, EncodingPipelineFailure> {
297
+ candidate_checkpoint(self.operation)?;
298
+ if options.enforce_legacy_hard_checks
299
+ && !matches!(detected.kind, EncodingKind::EncodingRs(_))
300
+ {
301
+ return Ok(PassResult::invalid());
302
+ }
303
+
304
+ let mut reader = SourceReader::open(self.source, 0)?;
305
+ let mut decoder = StrictDecoderScratch::new(detected.kind);
306
+ let mut raw_utf8 = options.probe_raw_utf8.then(RawUtf8Probe::new);
307
+ let mut roundtrip = if options.enforce_legacy_hard_checks {
308
+ let EncodingKind::EncodingRs(encoding) = detected.kind else {
309
+ unreachable!("legacy hard checks reject non-encoding_rs kinds above");
310
+ };
311
+ Some(RoundTripVerifier::new(encoding))
312
+ } else {
313
+ None
314
+ };
315
+ let mut segments = options
316
+ .scan_legacy_segments
317
+ .map(LegacySegmentScannerV2::new);
318
+ let mut segments_inconsistent = false;
319
+ let mut stats = ValidationStats::default();
320
+ let mut skip_bom = detected.bom_len as usize;
321
+ let mut input = [0_u8; DECODE_CHUNK_BYTES];
322
+
323
+ loop {
324
+ encoding_chunk_checkpoint(self.operation)?;
325
+ let count = reader.read(&mut input[..self.chunk_bytes])?;
326
+ checkpoint(self.operation)?;
327
+ let is_last = count == 0;
328
+ let raw = &input[..count];
329
+ if let Some(probe) = raw_utf8.as_mut() {
330
+ probe.push(raw, is_last);
331
+ }
332
+ let skipped = skip_bom.min(raw.len());
333
+ skip_bom -= skipped;
334
+ let content = &raw[skipped..];
335
+
336
+ if let Some(verifier) = roundtrip.as_mut() {
337
+ verifier.observe_source(content);
338
+ }
339
+ let decoded = match decoder.push(content, is_last) {
340
+ Ok(decoded) => decoded,
341
+ Err(()) => return Ok(PassResult::invalid()),
342
+ };
343
+ if options.enforce_legacy_hard_checks
344
+ && decoded.chars().any(is_disallowed_legacy_character)
345
+ {
346
+ return Ok(PassResult::invalid());
347
+ }
348
+ if let Some(verifier) = roundtrip.as_mut()
349
+ && !verifier.push(decoded, false)
350
+ {
351
+ return Ok(PassResult::invalid());
352
+ }
353
+ stats.observe(decoded);
354
+
355
+ if !segments_inconsistent && let Some(scanner) = segments.as_mut() {
356
+ match scanner.push(content, self.operation)? {
357
+ SegmentScan::Continue => {}
358
+ SegmentScan::Inconsistent => segments_inconsistent = true,
359
+ SegmentScan::CandidateInvalid => return Ok(PassResult::invalid()),
360
+ }
361
+ }
362
+ if is_last {
363
+ break;
364
+ }
365
+ }
366
+
367
+ if let Some(verifier) = roundtrip.as_mut()
368
+ && !verifier.finish()
369
+ {
370
+ return Ok(PassResult::invalid());
371
+ }
372
+ if !segments_inconsistent && let Some(scanner) = segments.as_mut() {
373
+ match scanner.finish(self.operation)? {
374
+ SegmentScan::Continue => {}
375
+ SegmentScan::Inconsistent => segments_inconsistent = true,
376
+ SegmentScan::CandidateInvalid => return Ok(PassResult::invalid()),
377
+ }
378
+ }
379
+ checkpoint(self.operation)?;
380
+ Ok(PassResult {
381
+ stats: Some(stats),
382
+ raw_is_multibyte_utf8: raw_utf8.is_some_and(|probe| probe.is_multibyte_utf8()),
383
+ segments_inconsistent,
384
+ #[cfg(test)]
385
+ roundtrip_allocations: roundtrip
386
+ .as_ref()
387
+ .map_or(0, RoundTripVerifier::allocation_count),
388
+ #[cfg(test)]
389
+ detector_constructions: segments
390
+ .as_ref()
391
+ .map_or(0, LegacySegmentScannerV2::detector_constructions),
392
+ })
393
+ }
394
+ }
395
+
396
+ fn automatic_detected(encoding: &'static Encoding) -> DetectedEncoding {
397
+ DetectedEncoding {
398
+ kind: EncodingKind::EncodingRs(encoding),
399
+ bom_len: 0,
400
+ source_encoding: Some(encoding.name()),
401
+ origin: EncodingOrigin::Automatic,
402
+ }
403
+ }
404
+
405
+ #[derive(Clone, Copy)]
406
+ struct PassOptions {
407
+ enforce_legacy_hard_checks: bool,
408
+ scan_legacy_segments: Option<&'static Encoding>,
409
+ probe_raw_utf8: bool,
410
+ }
411
+
412
+ impl PassOptions {
413
+ fn plain() -> Self {
414
+ Self {
415
+ enforce_legacy_hard_checks: false,
416
+ scan_legacy_segments: None,
417
+ probe_raw_utf8: false,
418
+ }
419
+ }
420
+ }
421
+
422
+ struct PassResult {
423
+ stats: Option<ValidationStats>,
424
+ raw_is_multibyte_utf8: bool,
425
+ segments_inconsistent: bool,
426
+ #[cfg(test)]
427
+ roundtrip_allocations: usize,
428
+ #[cfg(test)]
429
+ detector_constructions: usize,
430
+ }
431
+
432
+ impl PassResult {
433
+ fn invalid() -> Self {
434
+ Self {
435
+ stats: None,
436
+ raw_is_multibyte_utf8: false,
437
+ segments_inconsistent: false,
438
+ #[cfg(test)]
439
+ roundtrip_allocations: 0,
440
+ #[cfg(test)]
441
+ detector_constructions: 0,
442
+ }
443
+ }
444
+ }
445
+
446
+ struct Nomination {
447
+ encoding: &'static Encoding,
448
+ non_ascii_bytes: usize,
449
+ conflict_hex_offset: Option<usize>,
450
+ }
451
+
452
+ #[derive(Default)]
453
+ struct ValidationMemo {
454
+ entries: Vec<(&'static Encoding, Option<ValidationStats>)>,
455
+ }
456
+
457
+ impl ValidationMemo {
458
+ fn get(&self, encoding: &'static Encoding) -> Option<Option<ValidationStats>> {
459
+ self.entries
460
+ .iter()
461
+ .find(|(stored, _)| std::ptr::eq(*stored, encoding))
462
+ .map(|(_, result)| result.clone())
463
+ }
464
+
465
+ fn insert(&mut self, encoding: &'static Encoding, result: Option<ValidationStats>) {
466
+ if let Some((_, stored)) = self
467
+ .entries
468
+ .iter_mut()
469
+ .find(|(stored, _)| std::ptr::eq(*stored, encoding))
470
+ {
471
+ *stored = result;
472
+ } else {
473
+ self.entries.push((encoding, result));
474
+ }
475
+ }
476
+ }
477
+
478
+ fn checkpoint(operation: Option<&dyn WorkCheckpoint>) -> Result<(), EncodingPipelineFailure> {
479
+ match operation.map(WorkCheckpoint::check_work) {
480
+ Some(Err(stop)) => Err(EncodingPipelineFailure::Stopped(stop)),
481
+ Some(Ok(())) | None => Ok(()),
482
+ }
483
+ }
484
+
485
+ fn prefer_stop<T>(
486
+ operation: Option<&dyn WorkCheckpoint>,
487
+ result: Result<T, EncodingPipelineFailure>,
488
+ ) -> Result<T, EncodingPipelineFailure> {
489
+ checkpoint(operation)?;
490
+ result
491
+ }
492
+
493
+ fn encoding_chunk_checkpoint(
494
+ operation: Option<&dyn WorkCheckpoint>,
495
+ ) -> Result<(), EncodingPipelineFailure> {
496
+ checkpoint(operation)?;
497
+ #[cfg(test)]
498
+ if let Some(operation) = operation {
499
+ operation.stage(TestStage::EncodingChunk);
500
+ }
501
+ checkpoint(operation)
502
+ }
503
+
504
+ fn candidate_checkpoint(
505
+ operation: Option<&dyn WorkCheckpoint>,
506
+ ) -> Result<(), EncodingPipelineFailure> {
507
+ checkpoint(operation)?;
508
+ #[cfg(test)]
509
+ if let Some(operation) = operation {
510
+ operation.stage(TestStage::CandidateValidation);
511
+ }
512
+ checkpoint(operation)
513
+ }
514
+
515
+ fn segment_checkpoint(
516
+ operation: Option<&dyn WorkCheckpoint>,
517
+ ) -> Result<(), EncodingPipelineFailure> {
518
+ checkpoint(operation)?;
519
+ #[cfg(test)]
520
+ if let Some(operation) = operation {
521
+ operation.stage(TestStage::LegacySegment);
522
+ }
523
+ checkpoint(operation)
524
+ }
525
+
526
+ struct RawUtf8Probe {
527
+ decoder: StrictDecoderScratch,
528
+ valid: bool,
529
+ has_non_ascii: bool,
530
+ }
531
+
532
+ impl RawUtf8Probe {
533
+ fn new() -> Self {
534
+ Self {
535
+ decoder: StrictDecoderScratch::new(EncodingKind::EncodingRs(UTF_8)),
536
+ valid: true,
537
+ has_non_ascii: false,
538
+ }
539
+ }
540
+
541
+ fn push(&mut self, input: &[u8], is_last: bool) {
542
+ if !self.valid {
543
+ return;
544
+ }
545
+ match self.decoder.push(input, is_last) {
546
+ Ok(decoded) => self.has_non_ascii |= !decoded.is_ascii(),
547
+ Err(()) => self.valid = false,
548
+ }
549
+ }
550
+
551
+ fn is_multibyte_utf8(&self) -> bool {
552
+ self.valid && self.has_non_ascii
553
+ }
554
+ }
555
+
556
+ struct StrictDecoderScratch {
557
+ decoder: StrictDecoderKind,
558
+ output: String,
559
+ }
560
+
561
+ enum StrictDecoderKind {
562
+ Utf8 {
563
+ carry: Vec<u8>,
564
+ joined: Vec<u8>,
565
+ },
566
+ EncodingRs(encoding_rs::Decoder),
567
+ Utf32 {
568
+ little_endian: bool,
569
+ carry: Vec<u8>,
570
+ joined: Vec<u8>,
571
+ },
572
+ }
573
+
574
+ impl StrictDecoderScratch {
575
+ fn new(kind: EncodingKind) -> Self {
576
+ let decoder = match kind {
577
+ EncodingKind::EncodingRs(encoding) if encoding == UTF_8 => StrictDecoderKind::Utf8 {
578
+ carry: Vec::with_capacity(4),
579
+ joined: Vec::with_capacity(DECODE_CHUNK_BYTES + 4),
580
+ },
581
+ EncodingKind::EncodingRs(encoding) => {
582
+ StrictDecoderKind::EncodingRs(encoding.new_decoder_without_bom_handling())
583
+ }
584
+ EncodingKind::Utf32Le => StrictDecoderKind::Utf32 {
585
+ little_endian: true,
586
+ carry: Vec::with_capacity(4),
587
+ joined: Vec::with_capacity(DECODE_CHUNK_BYTES + 4),
588
+ },
589
+ EncodingKind::Utf32Be => StrictDecoderKind::Utf32 {
590
+ little_endian: false,
591
+ carry: Vec::with_capacity(4),
592
+ joined: Vec::with_capacity(DECODE_CHUNK_BYTES + 4),
593
+ },
594
+ };
595
+ Self {
596
+ decoder,
597
+ output: String::with_capacity(DECODE_CHUNK_BYTES),
598
+ }
599
+ }
600
+
601
+ fn push(&mut self, input: &[u8], is_last: bool) -> Result<&str, ()> {
602
+ self.output.clear();
603
+ match &mut self.decoder {
604
+ StrictDecoderKind::Utf8 { carry, joined } => {
605
+ joined.clear();
606
+ joined.extend_from_slice(carry);
607
+ carry.clear();
608
+ joined.extend_from_slice(input);
609
+ match std::str::from_utf8(joined) {
610
+ Ok(text) => self.output.push_str(text),
611
+ Err(error) if error.error_len().is_none() && !is_last => {
612
+ let valid_up_to = error.valid_up_to();
613
+ self.output
614
+ .push_str(std::str::from_utf8(&joined[..valid_up_to]).map_err(|_| ())?);
615
+ carry.extend_from_slice(&joined[valid_up_to..]);
616
+ }
617
+ Err(_) => return Err(()),
618
+ }
619
+ }
620
+ StrictDecoderKind::EncodingRs(decoder) => {
621
+ let mut consumed = 0_usize;
622
+ loop {
623
+ let remaining = &input[consumed..];
624
+ let capacity = decoder
625
+ .max_utf8_buffer_length_without_replacement(remaining.len())
626
+ .ok_or(())?
627
+ .max(4);
628
+ self.output.reserve(capacity);
629
+ let (result, read) = decoder.decode_to_string_without_replacement(
630
+ remaining,
631
+ &mut self.output,
632
+ is_last,
633
+ );
634
+ consumed = consumed.saturating_add(read);
635
+ match result {
636
+ DecoderResult::InputEmpty => break,
637
+ DecoderResult::OutputFull => continue,
638
+ DecoderResult::Malformed(_, _) => return Err(()),
639
+ }
640
+ }
641
+ }
642
+ StrictDecoderKind::Utf32 {
643
+ little_endian,
644
+ carry,
645
+ joined,
646
+ } => {
647
+ joined.clear();
648
+ joined.extend_from_slice(carry);
649
+ carry.clear();
650
+ joined.extend_from_slice(input);
651
+ if is_last && !joined.len().is_multiple_of(4) {
652
+ return Err(());
653
+ }
654
+ let complete_len = joined.len() / 4 * 4;
655
+ if !is_last {
656
+ carry.extend_from_slice(&joined[complete_len..]);
657
+ }
658
+ self.output.reserve(complete_len);
659
+ for raw in joined[..complete_len].as_chunks::<4>().0 {
660
+ let unit = if *little_endian {
661
+ u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]])
662
+ } else {
663
+ u32::from_be_bytes([raw[0], raw[1], raw[2], raw[3]])
664
+ };
665
+ self.output.push(char::from_u32(unit).ok_or(())?);
666
+ }
667
+ }
668
+ }
669
+ Ok(&self.output)
670
+ }
671
+ }
672
+
673
+ /// Re-encodes every decoded byte using reusable buffers and compares the entire source.
674
+ struct RoundTripVerifier {
675
+ encoder: encoding_rs::Encoder,
676
+ encoded: Vec<u8>,
677
+ expected: Vec<u8>,
678
+ encoded_start: usize,
679
+ expected_start: usize,
680
+ source_cursor: u64,
681
+ compared_cursor: u64,
682
+ #[cfg(test)]
683
+ allocations: usize,
684
+ }
685
+
686
+ impl RoundTripVerifier {
687
+ fn new(encoding: &'static Encoding) -> Self {
688
+ Self {
689
+ encoder: encoding.new_encoder(),
690
+ encoded: Vec::new(),
691
+ expected: Vec::new(),
692
+ encoded_start: 0,
693
+ expected_start: 0,
694
+ source_cursor: 0,
695
+ compared_cursor: 0,
696
+ #[cfg(test)]
697
+ allocations: 0,
698
+ }
699
+ }
700
+
701
+ fn observe_source(&mut self, bytes: &[u8]) {
702
+ self.compact_expected();
703
+ self.reserve_expected(bytes.len());
704
+ self.expected.extend_from_slice(bytes);
705
+ self.source_cursor = self.source_cursor.saturating_add(bytes.len() as u64);
706
+ }
707
+
708
+ fn push(&mut self, mut text: &str, last: bool) -> bool {
709
+ self.compact_encoded();
710
+ let mut first = true;
711
+ while first || !text.is_empty() {
712
+ first = false;
713
+ let capacity = self
714
+ .encoder
715
+ .max_buffer_length_from_utf8_without_replacement(text.len())
716
+ .unwrap_or(text.len().saturating_mul(4).saturating_add(16))
717
+ .saturating_add(16)
718
+ .max(16);
719
+ self.reserve_encoded(capacity);
720
+ let (result, read) = self.encoder.encode_from_utf8_to_vec_without_replacement(
721
+ text,
722
+ &mut self.encoded,
723
+ last,
724
+ );
725
+ if !self.compare_available() {
726
+ return false;
727
+ }
728
+ text = &text[read..];
729
+ match result {
730
+ EncoderResult::InputEmpty => return text.is_empty(),
731
+ EncoderResult::OutputFull => continue,
732
+ EncoderResult::Unmappable(_) => return false,
733
+ }
734
+ }
735
+ true
736
+ }
737
+
738
+ fn finish(&mut self) -> bool {
739
+ self.push("", true)
740
+ && self.compare_available()
741
+ && self.encoded_start == self.encoded.len()
742
+ && self.expected_start == self.expected.len()
743
+ && self.compared_cursor == self.source_cursor
744
+ }
745
+
746
+ fn compare_available(&mut self) -> bool {
747
+ let encoded = &self.encoded[self.encoded_start..];
748
+ let expected = &self.expected[self.expected_start..];
749
+ let count = encoded.len().min(expected.len());
750
+ if encoded[..count] != expected[..count] {
751
+ return false;
752
+ }
753
+ self.encoded_start += count;
754
+ self.expected_start += count;
755
+ self.compared_cursor = self.compared_cursor.saturating_add(count as u64);
756
+ true
757
+ }
758
+
759
+ fn compact_encoded(&mut self) {
760
+ compact_buffer(&mut self.encoded, &mut self.encoded_start);
761
+ }
762
+
763
+ fn compact_expected(&mut self) {
764
+ compact_buffer(&mut self.expected, &mut self.expected_start);
765
+ }
766
+
767
+ fn reserve_encoded(&mut self, additional: usize) {
768
+ #[cfg(test)]
769
+ let before = self.encoded.capacity();
770
+ self.encoded.reserve(additional);
771
+ #[cfg(test)]
772
+ if self.encoded.capacity() != before {
773
+ self.allocations = self.allocations.saturating_add(1);
774
+ }
775
+ }
776
+
777
+ fn reserve_expected(&mut self, additional: usize) {
778
+ #[cfg(test)]
779
+ let before = self.expected.capacity();
780
+ self.expected.reserve(additional);
781
+ #[cfg(test)]
782
+ if self.expected.capacity() != before {
783
+ self.allocations = self.allocations.saturating_add(1);
784
+ }
785
+ }
786
+
787
+ #[cfg(test)]
788
+ fn allocation_count(&self) -> usize {
789
+ self.allocations
790
+ }
791
+ }
792
+
793
+ fn compact_buffer(buffer: &mut Vec<u8>, start: &mut usize) {
794
+ if *start == buffer.len() {
795
+ buffer.clear();
796
+ *start = 0;
797
+ } else if *start >= DECODE_CHUNK_BYTES && *start >= buffer.len() / 2 {
798
+ buffer.copy_within(*start.., 0);
799
+ buffer.truncate(buffer.len() - *start);
800
+ *start = 0;
801
+ }
802
+ }
803
+
804
+ struct LegacySegmentScannerV2 {
805
+ whole_encoding: &'static Encoding,
806
+ pending: Vec<u8>,
807
+ start: usize,
808
+ evidence: SegmentEvidence,
809
+ cache: SegmentCache,
810
+ }
811
+
812
+ #[derive(Clone, Copy, Debug, Eq, PartialEq)]
813
+ enum SegmentScan {
814
+ Continue,
815
+ Inconsistent,
816
+ CandidateInvalid,
817
+ }
818
+
819
+ impl LegacySegmentScannerV2 {
820
+ fn new(whole_encoding: &'static Encoding) -> Self {
821
+ Self {
822
+ whole_encoding,
823
+ pending: Vec::with_capacity(LEGACY_SEGMENT_MAX_BYTES * 2),
824
+ start: 0,
825
+ evidence: SegmentEvidence::default(),
826
+ cache: SegmentCache::new(),
827
+ }
828
+ }
829
+
830
+ fn push(
831
+ &mut self,
832
+ input: &[u8],
833
+ operation: Option<&dyn WorkCheckpoint>,
834
+ ) -> Result<SegmentScan, EncodingPipelineFailure> {
835
+ for byte in input {
836
+ self.pending.push(*byte);
837
+ self.evidence.push(*byte);
838
+ if *byte == b'\n' && self.evidence.has_evidence() {
839
+ segment_checkpoint(operation)?;
840
+ let result = self.inspect_pending();
841
+ if result != SegmentScan::Continue {
842
+ return Ok(result);
843
+ }
844
+ self.clear_pending();
845
+ checkpoint(operation)?;
846
+ } else if self.pending_len() >= LEGACY_SEGMENT_MAX_BYTES {
847
+ segment_checkpoint(operation)?;
848
+ let Some(end) = self.aligned_prefix_len() else {
849
+ return Ok(SegmentScan::CandidateInvalid);
850
+ };
851
+ let absolute_end = self.start + end;
852
+ let segment = &self.pending[self.start..absolute_end];
853
+ let strong = SegmentEvidence::from_bytes(segment).is_strong_utf8();
854
+ if self
855
+ .cache
856
+ .segment_disagrees(self.whole_encoding, segment, strong)
857
+ {
858
+ return Ok(SegmentScan::Inconsistent);
859
+ }
860
+ self.start = absolute_end;
861
+ self.rebuild_evidence();
862
+ self.compact_pending();
863
+ checkpoint(operation)?;
864
+ }
865
+ }
866
+ Ok(SegmentScan::Continue)
867
+ }
868
+
869
+ fn finish(
870
+ &mut self,
871
+ operation: Option<&dyn WorkCheckpoint>,
872
+ ) -> Result<SegmentScan, EncodingPipelineFailure> {
873
+ segment_checkpoint(operation)?;
874
+ let result = self.inspect_pending();
875
+ checkpoint(operation)?;
876
+ Ok(result)
877
+ }
878
+
879
+ fn inspect_pending(&mut self) -> SegmentScan {
880
+ if !self.evidence.has_evidence() {
881
+ return SegmentScan::Continue;
882
+ }
883
+ let segment = &self.pending[self.start..];
884
+ if !self.cache.hard_validate(self.whole_encoding, segment) {
885
+ return SegmentScan::CandidateInvalid;
886
+ }
887
+ if self.cache.segment_disagrees(
888
+ self.whole_encoding,
889
+ segment,
890
+ self.evidence.is_strong_utf8(),
891
+ ) {
892
+ SegmentScan::Inconsistent
893
+ } else {
894
+ SegmentScan::Continue
895
+ }
896
+ }
897
+
898
+ fn aligned_prefix_len(&mut self) -> Option<usize> {
899
+ let pending_len = self.pending_len();
900
+ for trim in 0..=4 {
901
+ let Some(end) = pending_len.checked_sub(trim) else {
902
+ continue;
903
+ };
904
+ if self.cache.hard_validate(
905
+ self.whole_encoding,
906
+ &self.pending[self.start..self.start + end],
907
+ ) {
908
+ return Some(end);
909
+ }
910
+ }
911
+ None
912
+ }
913
+
914
+ fn pending_len(&self) -> usize {
915
+ self.pending.len() - self.start
916
+ }
917
+
918
+ fn clear_pending(&mut self) {
919
+ self.pending.clear();
920
+ self.start = 0;
921
+ self.evidence = SegmentEvidence::default();
922
+ }
923
+
924
+ fn rebuild_evidence(&mut self) {
925
+ self.evidence = SegmentEvidence::from_bytes(&self.pending[self.start..]);
926
+ }
927
+
928
+ fn compact_pending(&mut self) {
929
+ if self.start >= SEGMENT_COMPACT_THRESHOLD
930
+ || (self.start > 0 && self.start >= self.pending.len() / 2)
931
+ {
932
+ self.pending.copy_within(self.start.., 0);
933
+ self.pending.truncate(self.pending.len() - self.start);
934
+ self.start = 0;
935
+ }
936
+ }
937
+
938
+ #[cfg(test)]
939
+ fn detector_constructions(&self) -> usize {
940
+ self.cache.detector_constructions
941
+ }
942
+ }
943
+
944
+ #[derive(Default)]
945
+ struct SegmentEvidence {
946
+ len: usize,
947
+ non_ascii: usize,
948
+ utf8: Utf8Evidence,
949
+ }
950
+
951
+ impl SegmentEvidence {
952
+ fn from_bytes(bytes: &[u8]) -> Self {
953
+ let mut evidence = Self::default();
954
+ for byte in bytes {
955
+ evidence.push(*byte);
956
+ }
957
+ evidence
958
+ }
959
+
960
+ fn push(&mut self, byte: u8) {
961
+ self.len = self.len.saturating_add(1);
962
+ self.non_ascii = self.non_ascii.saturating_add(usize::from(!byte.is_ascii()));
963
+ self.utf8.push(byte);
964
+ }
965
+
966
+ fn has_evidence(&self) -> bool {
967
+ self.non_ascii >= LEGACY_EVIDENCE_BYTES || self.is_strong_utf8()
968
+ }
969
+
970
+ fn is_strong_utf8(&self) -> bool {
971
+ self.len >= UTF8_SEGMENT_MIN_BYTES
972
+ && self.non_ascii >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
973
+ && self.utf8.is_complete_and_valid()
974
+ }
975
+ }
976
+
977
+ #[derive(Default)]
978
+ struct Utf8Evidence {
979
+ valid: bool,
980
+ initialized: bool,
981
+ remaining: u8,
982
+ next_min: u8,
983
+ next_max: u8,
984
+ }
985
+
986
+ impl Utf8Evidence {
987
+ fn push(&mut self, byte: u8) {
988
+ if !self.initialized {
989
+ self.initialized = true;
990
+ self.valid = true;
991
+ }
992
+ if !self.valid {
993
+ return;
994
+ }
995
+ if self.remaining > 0 {
996
+ if !(self.next_min..=self.next_max).contains(&byte) {
997
+ self.valid = false;
998
+ return;
999
+ }
1000
+ self.remaining -= 1;
1001
+ self.next_min = 0x80;
1002
+ self.next_max = 0xBF;
1003
+ return;
1004
+ }
1005
+ match byte {
1006
+ 0x00..=0x7F => {}
1007
+ 0xC2..=0xDF => self.begin(1, 0x80, 0xBF),
1008
+ 0xE0 => self.begin(2, 0xA0, 0xBF),
1009
+ 0xE1..=0xEC | 0xEE..=0xEF => self.begin(2, 0x80, 0xBF),
1010
+ 0xED => self.begin(2, 0x80, 0x9F),
1011
+ 0xF0 => self.begin(3, 0x90, 0xBF),
1012
+ 0xF1..=0xF3 => self.begin(3, 0x80, 0xBF),
1013
+ 0xF4 => self.begin(3, 0x80, 0x8F),
1014
+ _ => self.valid = false,
1015
+ }
1016
+ }
1017
+
1018
+ fn begin(&mut self, remaining: u8, next_min: u8, next_max: u8) {
1019
+ self.remaining = remaining;
1020
+ self.next_min = next_min;
1021
+ self.next_max = next_max;
1022
+ }
1023
+
1024
+ fn is_complete_and_valid(&self) -> bool {
1025
+ (!self.initialized || self.valid) && self.remaining == 0
1026
+ }
1027
+ }
1028
+
1029
+ #[derive(Clone, Copy, Eq, PartialEq)]
1030
+ struct SegmentKey {
1031
+ high: u64,
1032
+ low: u64,
1033
+ len: usize,
1034
+ }
1035
+
1036
+ struct SegmentCacheEntry {
1037
+ key: SegmentKey,
1038
+ bytes: Box<[u8]>,
1039
+ detector_guess: Option<&'static Encoding>,
1040
+ validations: Vec<(&'static Encoding, bool)>,
1041
+ }
1042
+
1043
+ struct SegmentCache {
1044
+ entries: VecDeque<SegmentCacheEntry>,
1045
+ scratch: SegmentValidationScratch,
1046
+ #[cfg(test)]
1047
+ detector_constructions: usize,
1048
+ }
1049
+
1050
+ impl SegmentCache {
1051
+ fn new() -> Self {
1052
+ Self {
1053
+ entries: VecDeque::with_capacity(SEGMENT_CACHE_CAPACITY),
1054
+ scratch: SegmentValidationScratch::default(),
1055
+ #[cfg(test)]
1056
+ detector_constructions: 0,
1057
+ }
1058
+ }
1059
+
1060
+ fn hard_validate(&mut self, encoding: &'static Encoding, bytes: &[u8]) -> bool {
1061
+ let mut entry = self.take_entry(bytes);
1062
+ let result = if let Some((_, result)) = entry
1063
+ .validations
1064
+ .iter()
1065
+ .find(|(stored, _)| std::ptr::eq(*stored, encoding))
1066
+ {
1067
+ *result
1068
+ } else {
1069
+ let result = self.scratch.hard_validate(encoding, bytes);
1070
+ entry.validations.push((encoding, result));
1071
+ result
1072
+ };
1073
+ self.restore_entry(entry);
1074
+ result
1075
+ }
1076
+
1077
+ fn segment_disagrees(
1078
+ &mut self,
1079
+ whole_encoding: &'static Encoding,
1080
+ bytes: &[u8],
1081
+ strong_utf8: bool,
1082
+ ) -> bool {
1083
+ if strong_utf8 {
1084
+ return true;
1085
+ }
1086
+ let mut entry = self.take_entry(bytes);
1087
+ let candidate = match entry.detector_guess {
1088
+ Some(candidate) => candidate,
1089
+ None => {
1090
+ let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
1091
+ detector.feed(bytes, true);
1092
+ let candidate = detector.guess(None, Utf8Detection::Deny);
1093
+ entry.detector_guess = Some(candidate);
1094
+ #[cfg(test)]
1095
+ {
1096
+ self.detector_constructions = self.detector_constructions.saturating_add(1);
1097
+ }
1098
+ candidate
1099
+ }
1100
+ };
1101
+ let result = if std::ptr::eq(candidate, whole_encoding) || candidate.is_single_byte() {
1102
+ false
1103
+ } else if let Some((_, result)) = entry
1104
+ .validations
1105
+ .iter()
1106
+ .find(|(stored, _)| std::ptr::eq(*stored, candidate))
1107
+ {
1108
+ *result
1109
+ } else {
1110
+ let result = self.scratch.hard_validate(candidate, bytes);
1111
+ entry.validations.push((candidate, result));
1112
+ result
1113
+ };
1114
+ self.restore_entry(entry);
1115
+ result
1116
+ }
1117
+
1118
+ fn take_entry(&mut self, bytes: &[u8]) -> SegmentCacheEntry {
1119
+ let key = segment_key(bytes);
1120
+ if let Some(position) = self
1121
+ .entries
1122
+ .iter()
1123
+ .position(|entry| entry.key == key && entry.bytes.as_ref() == bytes)
1124
+ {
1125
+ return self
1126
+ .entries
1127
+ .remove(position)
1128
+ .expect("located segment cache entry must remain present");
1129
+ }
1130
+ SegmentCacheEntry {
1131
+ key,
1132
+ bytes: bytes.into(),
1133
+ detector_guess: None,
1134
+ validations: Vec::new(),
1135
+ }
1136
+ }
1137
+
1138
+ fn restore_entry(&mut self, entry: SegmentCacheEntry) {
1139
+ self.entries.push_back(entry);
1140
+ if self.entries.len() > SEGMENT_CACHE_CAPACITY {
1141
+ self.entries.pop_front();
1142
+ }
1143
+ }
1144
+ }
1145
+
1146
+ fn segment_key(bytes: &[u8]) -> SegmentKey {
1147
+ let mut high = 0xCBF2_9CE4_8422_2325_u64;
1148
+ let mut low = 0x9E37_79B9_7F4A_7C15_u64;
1149
+ for byte in bytes {
1150
+ high ^= u64::from(*byte);
1151
+ high = high.wrapping_mul(0x0000_0100_0000_01B3);
1152
+ low ^= u64::from(*byte).wrapping_add(high.rotate_left(17));
1153
+ low = low.rotate_left(13).wrapping_mul(0xC2B2_AE3D_27D4_EB4F);
1154
+ }
1155
+ SegmentKey {
1156
+ high,
1157
+ low,
1158
+ len: bytes.len(),
1159
+ }
1160
+ }
1161
+
1162
+ #[derive(Default)]
1163
+ struct SegmentValidationScratch {
1164
+ decoded: String,
1165
+ encoded: Vec<u8>,
1166
+ }
1167
+
1168
+ impl SegmentValidationScratch {
1169
+ fn hard_validate(&mut self, encoding: &'static Encoding, bytes: &[u8]) -> bool {
1170
+ self.decoded.clear();
1171
+ let mut decoder = encoding.new_decoder_without_bom_handling();
1172
+ let mut consumed = 0_usize;
1173
+ loop {
1174
+ let remaining = &bytes[consumed..];
1175
+ let Some(capacity) =
1176
+ decoder.max_utf8_buffer_length_without_replacement(remaining.len())
1177
+ else {
1178
+ return false;
1179
+ };
1180
+ self.decoded.reserve(capacity.max(4));
1181
+ let (result, read) =
1182
+ decoder.decode_to_string_without_replacement(remaining, &mut self.decoded, true);
1183
+ consumed = consumed.saturating_add(read);
1184
+ match result {
1185
+ DecoderResult::InputEmpty => break,
1186
+ DecoderResult::OutputFull => continue,
1187
+ DecoderResult::Malformed(_, _) => return false,
1188
+ }
1189
+ }
1190
+ if self.decoded.chars().any(is_disallowed_legacy_character) {
1191
+ return false;
1192
+ }
1193
+
1194
+ self.encoded.clear();
1195
+ let mut encoder = encoding.new_encoder();
1196
+ let mut remaining = self.decoded.as_str();
1197
+ let mut first = true;
1198
+ while first || !remaining.is_empty() {
1199
+ first = false;
1200
+ let capacity = encoder
1201
+ .max_buffer_length_from_utf8_without_replacement(remaining.len())
1202
+ .unwrap_or(remaining.len().saturating_mul(4).saturating_add(16))
1203
+ .saturating_add(16)
1204
+ .max(16);
1205
+ self.encoded.reserve(capacity);
1206
+ let (result, read) = encoder.encode_from_utf8_to_vec_without_replacement(
1207
+ remaining,
1208
+ &mut self.encoded,
1209
+ true,
1210
+ );
1211
+ remaining = &remaining[read..];
1212
+ match result {
1213
+ EncoderResult::InputEmpty => break,
1214
+ EncoderResult::OutputFull => continue,
1215
+ EncoderResult::Unmappable(_) => return false,
1216
+ }
1217
+ }
1218
+ remaining.is_empty() && self.encoded == bytes
1219
+ }
1220
+ }
1221
+
1222
+ #[cfg(test)]
1223
+ pub(super) fn validate_bytes_for_test(
1224
+ bytes: &[u8],
1225
+ explicit_encoding: Option<&str>,
1226
+ ) -> EncodingDecision {
1227
+ validate_source(ByteSource::Bytes(bytes), explicit_encoding, None)
1228
+ .unwrap_or_else(|_| panic!("in-memory validation cannot fail with I/O"))
1229
+ }
1230
+
1231
+ #[cfg(test)]
1232
+ pub(super) fn validate_bytes_with_chunk_limit_for_test(
1233
+ bytes: &[u8],
1234
+ explicit_encoding: Option<&str>,
1235
+ chunk_bytes: usize,
1236
+ ) -> EncodingDecision {
1237
+ assert!((1..=DECODE_CHUNK_BYTES).contains(&chunk_bytes));
1238
+ let pipeline = EncodingPipeline {
1239
+ source: ByteSource::Bytes(bytes),
1240
+ operation: None,
1241
+ chunk_bytes,
1242
+ };
1243
+ pipeline
1244
+ .validate(
1245
+ &bytes[..bytes.len().min(BINARY_PROBE_BYTES)],
1246
+ explicit_encoding,
1247
+ )
1248
+ .unwrap_or_else(|_| panic!("in-memory validation cannot fail with I/O"))
1249
+ }
1250
+
1251
+ #[cfg(test)]
1252
+ pub(super) fn legacy_metrics_for_test(
1253
+ bytes: &[u8],
1254
+ encoding: &'static Encoding,
1255
+ ) -> (Option<ValidationStats>, bool, usize, usize) {
1256
+ let pipeline = EncodingPipeline {
1257
+ source: ByteSource::Bytes(bytes),
1258
+ operation: None,
1259
+ chunk_bytes: DECODE_CHUNK_BYTES,
1260
+ };
1261
+ let detected = automatic_detected(encoding);
1262
+ let result = pipeline
1263
+ .validate_selected(
1264
+ &detected,
1265
+ PassOptions {
1266
+ enforce_legacy_hard_checks: true,
1267
+ scan_legacy_segments: Some(encoding),
1268
+ probe_raw_utf8: false,
1269
+ },
1270
+ )
1271
+ .unwrap_or_else(|_| panic!("in-memory validation cannot fail with I/O"));
1272
+ (
1273
+ result.stats,
1274
+ result.segments_inconsistent,
1275
+ result.roundtrip_allocations,
1276
+ result.detector_constructions,
1277
+ )
1278
+ }
1279
+
1280
+ #[cfg(test)]
1281
+ mod tests {
1282
+ use super::{
1283
+ ByteSource, EncodingDecision, EncodingKind, EncodingOrigin, EncodingPipelineFailure,
1284
+ EncodingRejection, LegacySegmentScannerV2, RoundTripVerifier, SegmentCache,
1285
+ SegmentCacheEntry, SegmentEvidence, SegmentScan, legacy_metrics_for_test, segment_key,
1286
+ validate_bytes_for_test, validate_bytes_with_chunk_limit_for_test, validate_source,
1287
+ };
1288
+ use crate::encoding::reference_v011::{
1289
+ ReferenceDecision, ReferenceOrigin, ReferenceRejection, classify_v011,
1290
+ };
1291
+ use crate::operation::{EpochGuard, RequestWorkGuard, TestStage, WorkCtx, WorkStop};
1292
+ use encoding_rs::GBK;
1293
+ use rmcp::model::RequestId;
1294
+ use std::sync::Arc;
1295
+ use std::sync::atomic::{AtomicU64, Ordering};
1296
+ use tokio_util::sync::CancellationToken;
1297
+
1298
+ fn normalize_production(decision: &EncodingDecision) -> ReferenceDecision {
1299
+ match decision {
1300
+ EncodingDecision::Text(validated) => ReferenceDecision::Text {
1301
+ kind: match validated.detected.kind {
1302
+ EncodingKind::EncodingRs(encoding) => encoding.name().to_string(),
1303
+ EncodingKind::Utf32Le => "UTF-32LE".to_string(),
1304
+ EncodingKind::Utf32Be => "UTF-32BE".to_string(),
1305
+ },
1306
+ bom_len: validated.detected.bom_len as usize,
1307
+ source_encoding: validated.detected.source_encoding.map(str::to_string),
1308
+ origin: match &validated.detected.origin {
1309
+ EncodingOrigin::Explicit(value) => ReferenceOrigin::Explicit(value.clone()),
1310
+ EncodingOrigin::Bom(value) => ReferenceOrigin::Bom((*value).to_string()),
1311
+ EncodingOrigin::Automatic => ReferenceOrigin::Automatic,
1312
+ },
1313
+ total_lines: validated.total_lines,
1314
+ has_trailing_newline: validated.has_trailing_newline,
1315
+ explicit_utf8_warning: validated.explicit_utf8_warning,
1316
+ note: validated.transcoding_note(),
1317
+ },
1318
+ EncodingDecision::Binary => ReferenceDecision::Binary,
1319
+ EncodingDecision::Rejected(rejection) => {
1320
+ ReferenceDecision::Rejected(normalize_rejection(rejection))
1321
+ }
1322
+ }
1323
+ }
1324
+
1325
+ fn normalize_rejection(rejection: &EncodingRejection) -> ReferenceRejection {
1326
+ match rejection {
1327
+ EncodingRejection::Ambiguous { candidates } => ReferenceRejection::Ambiguous(
1328
+ candidates
1329
+ .iter()
1330
+ .map(|candidate| (*candidate).to_string())
1331
+ .collect(),
1332
+ ),
1333
+ EncodingRejection::MixedOrInconsistent {
1334
+ conflict_hex_offset,
1335
+ } => ReferenceRejection::MixedOrInconsistent(*conflict_hex_offset),
1336
+ EncodingRejection::Iso2022JpSignature => ReferenceRejection::Iso2022JpSignature,
1337
+ EncodingRejection::Undecodable => ReferenceRejection::Undecodable,
1338
+ EncodingRejection::BomMismatch { encoding } => {
1339
+ ReferenceRejection::BomMismatch((*encoding).to_string())
1340
+ }
1341
+ EncodingRejection::ExplicitMalformed { encoding } => {
1342
+ ReferenceRejection::ExplicitMalformed(encoding.clone())
1343
+ }
1344
+ EncodingRejection::InvalidLabel { value } => {
1345
+ ReferenceRejection::InvalidLabel(value.clone())
1346
+ }
1347
+ }
1348
+ }
1349
+
1350
+ #[test]
1351
+ fn utf8_evidence_matches_the_standard_library_for_every_prefix() {
1352
+ let mut state = 0xC001_D00D_A5A5_5A5A_u64;
1353
+ for length in 0..=512 {
1354
+ let mut bytes = Vec::with_capacity(length);
1355
+ for _ in 0..length {
1356
+ state = state
1357
+ .wrapping_mul(6_364_136_223_846_793_005)
1358
+ .wrapping_add(1_442_695_040_888_963_407);
1359
+ bytes.push((state >> 32) as u8);
1360
+ let evidence = SegmentEvidence::from_bytes(&bytes);
1361
+ assert_eq!(
1362
+ evidence.utf8.is_complete_and_valid(),
1363
+ std::str::from_utf8(&bytes).is_ok(),
1364
+ "length {}",
1365
+ bytes.len()
1366
+ );
1367
+ }
1368
+ }
1369
+ }
1370
+
1371
+ #[test]
1372
+ fn repeated_segments_reuse_detector_and_roundtrip_scratch() {
1373
+ let mut line = Vec::new();
1374
+ for _ in 0..40 {
1375
+ line.extend_from_slice(&[0xD6, 0xD0]);
1376
+ }
1377
+ line.push(b'\n');
1378
+ let bytes = line.repeat(4_096);
1379
+ let (stats, inconsistent, allocations, detector_constructions) =
1380
+ legacy_metrics_for_test(&bytes, GBK);
1381
+ assert!(stats.is_some());
1382
+ assert!(!inconsistent);
1383
+ assert!(
1384
+ allocations <= 8,
1385
+ "round-trip buffers grew {allocations} times for {} chunks",
1386
+ bytes.len().div_ceil(super::DECODE_CHUNK_BYTES)
1387
+ );
1388
+ assert_eq!(detector_constructions, 1);
1389
+ }
1390
+
1391
+ #[test]
1392
+ fn unique_segments_construct_at_most_one_detector_each_and_reuse_scratch() {
1393
+ const UNIQUE_SEGMENTS: usize = 64;
1394
+ let mut segments = Vec::with_capacity(UNIQUE_SEGMENTS);
1395
+ for index in 0..UNIQUE_SEGMENTS {
1396
+ let mut segment = Vec::with_capacity(96);
1397
+ for _ in 0..40 {
1398
+ segment.extend_from_slice(&[0xD6, 0xD0]);
1399
+ }
1400
+ segment.extend_from_slice(format!("-{index:02x}").as_bytes());
1401
+ segment.push(b'\n');
1402
+ segments.push(segment);
1403
+ }
1404
+ let mut bytes = Vec::new();
1405
+ for _ in 0..2 {
1406
+ for segment in &segments {
1407
+ bytes.extend_from_slice(segment);
1408
+ }
1409
+ }
1410
+
1411
+ let (stats, inconsistent, allocations, detector_constructions) =
1412
+ legacy_metrics_for_test(&bytes, GBK);
1413
+ assert!(stats.is_some());
1414
+ assert!(!inconsistent);
1415
+ assert!(
1416
+ allocations <= 8,
1417
+ "round-trip buffers grew {allocations} times"
1418
+ );
1419
+ assert_eq!(detector_constructions, UNIQUE_SEGMENTS);
1420
+ }
1421
+
1422
+ #[test]
1423
+ fn invalid_unalignable_segment_cannot_grow_pending_past_the_contract_boundary() {
1424
+ let mut scanner = LegacySegmentScannerV2::new(GBK);
1425
+ let invalid = vec![0xFF; super::LEGACY_SEGMENT_MAX_BYTES * 8];
1426
+ let result = scanner.push(&invalid, None).unwrap();
1427
+ assert_eq!(result, SegmentScan::CandidateInvalid);
1428
+ assert!(scanner.pending_len() <= super::LEGACY_SEGMENT_MAX_BYTES);
1429
+ }
1430
+
1431
+ #[test]
1432
+ fn segment_cache_compares_full_bytes_after_a_hash_key_match() {
1433
+ let requested = b"requested-segment";
1434
+ let colliding_bytes = b"different-segment";
1435
+ let mut cache = SegmentCache::new();
1436
+ cache.entries.push_back(SegmentCacheEntry {
1437
+ key: segment_key(requested),
1438
+ bytes: colliding_bytes.as_slice().into(),
1439
+ detector_guess: Some(GBK),
1440
+ validations: vec![(GBK, true)],
1441
+ });
1442
+
1443
+ let entry = cache.take_entry(requested);
1444
+ assert_eq!(entry.bytes.as_ref(), requested);
1445
+ assert!(entry.detector_guess.is_none());
1446
+ assert!(entry.validations.is_empty());
1447
+ assert_eq!(cache.entries.len(), 1);
1448
+ assert_eq!(
1449
+ cache.entries.front().unwrap().bytes.as_ref(),
1450
+ colliding_bytes
1451
+ );
1452
+ }
1453
+
1454
+ #[test]
1455
+ fn roundtrip_verifier_compares_every_source_byte() {
1456
+ let mut exact = RoundTripVerifier::new(GBK);
1457
+ exact.observe_source(&[0xD6, 0xD0]);
1458
+ assert!(exact.push("中", false));
1459
+ assert!(exact.finish());
1460
+
1461
+ let mut mutated = RoundTripVerifier::new(GBK);
1462
+ mutated.observe_source(&[0xD6, 0xD1]);
1463
+ assert!(!mutated.push("中", false) || !mutated.finish());
1464
+ }
1465
+
1466
+ #[test]
1467
+ fn encoding_chunk_checkpoint_observes_request_cancellation() {
1468
+ let token = CancellationToken::new();
1469
+ let cancel_from_hook = token.clone();
1470
+ let chunks = Arc::new(AtomicU64::new(0));
1471
+ let chunks_from_hook = Arc::clone(&chunks);
1472
+ let hook = Arc::new(move |stage| {
1473
+ if stage == TestStage::EncodingChunk {
1474
+ let chunk = chunks_from_hook.fetch_add(1, Ordering::AcqRel) + 1;
1475
+ if chunk == 2 {
1476
+ cancel_from_hook.cancel();
1477
+ }
1478
+ }
1479
+ });
1480
+ let (_guard, operation) =
1481
+ RequestWorkGuard::new_with_hook(RequestId::Number(71), token, hook);
1482
+ let bytes = vec![b'a'; super::DECODE_CHUNK_BYTES * 2];
1483
+ let result = validate_source(ByteSource::Bytes(&bytes), None, Some(&operation));
1484
+ assert!(matches!(
1485
+ result,
1486
+ Err(EncodingPipelineFailure::Stopped(WorkStop::RequestCancelled))
1487
+ ));
1488
+ assert_eq!(chunks.load(Ordering::Acquire), 2);
1489
+ }
1490
+
1491
+ #[test]
1492
+ fn legacy_segment_checkpoint_stops_without_inspecting_another_segment() {
1493
+ let token = CancellationToken::new();
1494
+ let cancel_from_hook = token.clone();
1495
+ let segments = Arc::new(AtomicU64::new(0));
1496
+ let segments_from_hook = Arc::clone(&segments);
1497
+ let hook = Arc::new(move |stage| {
1498
+ if stage == TestStage::LegacySegment {
1499
+ segments_from_hook.fetch_add(1, Ordering::AcqRel);
1500
+ cancel_from_hook.cancel();
1501
+ }
1502
+ });
1503
+ let (_guard, operation) =
1504
+ RequestWorkGuard::new_with_hook(RequestId::Number(73), token, hook);
1505
+ let mut bytes = Vec::new();
1506
+ for _ in 0..64 {
1507
+ bytes.extend_from_slice(&[0xD6, 0xD0]);
1508
+ }
1509
+ bytes.push(b'\n');
1510
+ let result = validate_source(ByteSource::Bytes(&bytes), None, Some(&operation));
1511
+ assert!(matches!(
1512
+ result,
1513
+ Err(EncodingPipelineFailure::Stopped(WorkStop::RequestCancelled))
1514
+ ));
1515
+ assert_eq!(segments.load(Ordering::Acquire), 1);
1516
+ }
1517
+
1518
+ #[test]
1519
+ fn candidate_checkpoint_observes_epoch_retirement() {
1520
+ let generation = Arc::new(AtomicU64::new(0));
1521
+ let generation_from_hook = Arc::clone(&generation);
1522
+ let hook = Arc::new(move |stage| {
1523
+ if stage == TestStage::CandidateValidation {
1524
+ generation_from_hook.store(1, Ordering::Release);
1525
+ }
1526
+ });
1527
+ let (mut guard, operation) =
1528
+ RequestWorkGuard::new_with_hook(RequestId::Number(72), CancellationToken::new(), hook);
1529
+ let work = WorkCtx::speculative(operation, EpochGuard::new(0, generation));
1530
+ let result = validate_source(ByteSource::Bytes(b"plain text"), None, Some(&work));
1531
+ guard.disarm();
1532
+ assert!(matches!(
1533
+ result,
1534
+ Err(EncodingPipelineFailure::Stopped(WorkStop::EpochRetired))
1535
+ ));
1536
+ }
1537
+
1538
+ fn reference_corpus() -> Vec<(Vec<u8>, Option<&'static str>)> {
1539
+ let mut corpus: Vec<(Vec<u8>, Option<&str>)> = vec![
1540
+ (Vec::new(), None),
1541
+ (b"plain ASCII\r\nsecond\n".to_vec(), None),
1542
+ (b"\xEF\xBB\xBFutf8 bom\n".to_vec(), None),
1543
+ (b"\xEF\xBB\xBFbad\0tail".to_vec(), None),
1544
+ (b"\xEF\xBB\xBFbad\xFFtail".to_vec(), None),
1545
+ (b"\xFF\xFEA\0B\0".to_vec(), None),
1546
+ (b"\xFE\xFF\0A\0B".to_vec(), None),
1547
+ (vec![0xFF, 0xFE, b'A'], None),
1548
+ (vec![0xFF, 0xFE, 0, 0, b'A', 0, 0, 0], None),
1549
+ (vec![0, 0, 0xFE, 0xFF, 0, 0, 0, b'A'], None),
1550
+ (vec![0, 0, 0xFE, 0xFF, 0], None),
1551
+ (b"\x1B$Bstateful-ascii".to_vec(), None),
1552
+ (b"invalid\xFFtail".to_vec(), Some("utf-8")),
1553
+ (b"plain UTF-8 \xE7\x95\x8C".to_vec(), Some("gbk")),
1554
+ (b"A\0B\0".to_vec(), Some("utf-16le")),
1555
+ (b"\0A\0B".to_vec(), Some("utf-16be")),
1556
+ (vec![b'A', 0, 0, 0, b'B', 0, 0, 0], Some("utf-32le")),
1557
+ (vec![0, 0, 0, b'A', 0, 0, 0, b'B'], Some("utf-32be")),
1558
+ (b"\xFF\xFEA\0".to_vec(), Some("utf-16le")),
1559
+ (b"\xFE\xFF\0A".to_vec(), Some("utf-16be")),
1560
+ (vec![0xFF, 0xFE, 0, 0, b'A', 0, 0, 0], Some("utf-32le")),
1561
+ (vec![0, 0, 0xFE, 0xFF, 0, 0, 0, b'A'], Some("utf-32be")),
1562
+ (b"\x1B$B$\"\x1B(B".to_vec(), Some("iso-2022-jp")),
1563
+ (b"plain".to_vec(), Some("not-a-label")),
1564
+ ];
1565
+
1566
+ let mut gbk = Vec::new();
1567
+ let mut shift_jis = Vec::new();
1568
+ let mut big5 = Vec::new();
1569
+ let mut euc_kr = Vec::new();
1570
+ for _ in 0..96 {
1571
+ gbk.extend_from_slice(&[0xD6, 0xD0]);
1572
+ shift_jis.extend_from_slice(&[0x93, 0xFA]);
1573
+ big5.extend_from_slice(&[0xA4, 0xA4]);
1574
+ euc_kr.extend_from_slice(&[0xC7, 0xD1]);
1575
+ }
1576
+ for bytes in [&gbk, &shift_jis, &big5, &euc_kr] {
1577
+ corpus.push((bytes.clone(), None));
1578
+ }
1579
+ corpus.extend([
1580
+ (gbk.clone(), Some("gbk")),
1581
+ (shift_jis.clone(), Some("shift_jis")),
1582
+ (big5.clone(), Some("big5")),
1583
+ (euc_kr.clone(), Some("euc-kr")),
1584
+ (vec![0xE9; 40], None),
1585
+ (vec![0xE9; 40], Some("windows-1252")),
1586
+ ]);
1587
+
1588
+ let strong_utf8 = "界".repeat(11).into_bytes();
1589
+ for ascii_padding in [0, 31, 4_095, 65_535] {
1590
+ let mut mixed = strong_utf8.clone();
1591
+ mixed.extend(std::iter::repeat_n(b'a', ascii_padding));
1592
+ mixed.push(0xFF);
1593
+ corpus.push((mixed.clone(), None));
1594
+ mixed.push(b'\n');
1595
+ mixed.extend_from_slice(&gbk);
1596
+ corpus.push((mixed, None));
1597
+ }
1598
+ let mut cross_line = gbk.clone();
1599
+ cross_line.push(b'\n');
1600
+ cross_line.extend_from_slice(&shift_jis);
1601
+ corpus.push((cross_line, None));
1602
+ let mut same_line = gbk.clone();
1603
+ same_line.extend_from_slice(&shift_jis);
1604
+ corpus.push((same_line, None));
1605
+
1606
+ for length in [31, 32, 33, 4_095, 4_096, 4_097] {
1607
+ let mut bytes = Vec::with_capacity(length + 64);
1608
+ while bytes.len() + 2 <= length {
1609
+ bytes.extend_from_slice(&[0xD6, 0xD0]);
1610
+ }
1611
+ bytes.resize(length, b'a');
1612
+ bytes.push(b'\n');
1613
+ corpus.push((bytes, None));
1614
+ }
1615
+ for non_ascii in [7, 8, 9, 31, 32, 33] {
1616
+ let mut bytes = b"ascii-prefix-which-is-long-enough-".to_vec();
1617
+ bytes.extend(std::iter::repeat_n(0xE9, non_ascii));
1618
+ bytes.push(b'\n');
1619
+ corpus.push((bytes, None));
1620
+ }
1621
+
1622
+ let mut utf8_boundary = vec![b'a'; super::DECODE_CHUNK_BYTES - 1];
1623
+ utf8_boundary.extend_from_slice("界\r\n".as_bytes());
1624
+ corpus.push((utf8_boundary, None));
1625
+ let mut gbk_boundary = vec![b'a'; super::DECODE_CHUNK_BYTES - 1];
1626
+ gbk_boundary.extend_from_slice(&[0xD6, 0xD0, b'\r', b'\n']);
1627
+ gbk_boundary.extend_from_slice(&gbk);
1628
+ corpus.push((gbk_boundary, None));
1629
+ let mut escape_boundary = vec![b'a'; super::DECODE_CHUNK_BYTES - 1];
1630
+ escape_boundary.extend_from_slice(b"\x1B$Btail");
1631
+ corpus.push((escape_boundary, None));
1632
+
1633
+ let mut state = 0xA5A5_5A5A_1234_5678_u64;
1634
+ for case in 0..384 {
1635
+ let length = case * 17 % 2_049;
1636
+ let mut bytes = Vec::with_capacity(length);
1637
+ for _ in 0..length {
1638
+ state = state
1639
+ .wrapping_mul(6_364_136_223_846_793_005)
1640
+ .wrapping_add(1_442_695_040_888_963_407);
1641
+ bytes.push((state >> 32) as u8);
1642
+ }
1643
+ corpus.push((bytes, None));
1644
+ }
1645
+
1646
+ corpus
1647
+ }
1648
+
1649
+ #[test]
1650
+ fn v011_reference_differential_covers_trust_tree_and_boundaries() {
1651
+ for (case, (bytes, explicit)) in reference_corpus().into_iter().enumerate() {
1652
+ let expected = classify_v011(&bytes, explicit);
1653
+ let actual = normalize_production(&validate_bytes_for_test(&bytes, explicit));
1654
+ assert_eq!(
1655
+ actual, expected,
1656
+ "differential case {case}, explicit={explicit:?}"
1657
+ );
1658
+ }
1659
+ }
1660
+
1661
+ #[test]
1662
+ fn v011_reference_differential_is_invariant_to_reader_chunking() {
1663
+ for (case, (fixture, explicit)) in reference_corpus().into_iter().enumerate() {
1664
+ let expected = classify_v011(&fixture, explicit);
1665
+ for chunk_bytes in [1, 2, 3, 7, 31, 4_095, 65_535, 65_536] {
1666
+ let actual = normalize_production(&validate_bytes_with_chunk_limit_for_test(
1667
+ &fixture,
1668
+ explicit,
1669
+ chunk_bytes,
1670
+ ));
1671
+ assert_eq!(
1672
+ actual, expected,
1673
+ "case={case}, explicit={explicit:?}, chunk_bytes={chunk_bytes}"
1674
+ );
1675
+ }
1676
+ }
1677
+ }
1678
+ }