dsh-ops 0.0.0-stage → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +202 -0
  2. package/LICENSE +30 -0
  3. package/NOTICE +106 -0
  4. package/PROVENANCE.md +435 -0
  5. package/README.en.md +126 -0
  6. package/README.md +115 -2
  7. package/README.zh.md +116 -0
  8. package/bin/dsh-ops.mjs +1216 -0
  9. package/cordis.patch.yml +160 -0
  10. package/docs/manual-validation.md +53 -0
  11. package/docs/release-0.2.1.md +72 -0
  12. package/docs/schema-baseline.json +64 -0
  13. package/docs/schema-current.json +84 -0
  14. package/docs/schema-measurement.md +17 -0
  15. package/dsh-plugin.json +88 -0
  16. package/icon.svg +12 -0
  17. package/lib/binary.js +409 -0
  18. package/lib/config.js +198 -0
  19. package/lib/handshake.js +252 -0
  20. package/lib/index.js +108 -0
  21. package/lib/jobs.js +42 -0
  22. package/lib/policy.js +64 -0
  23. package/lib/presentation.js +63 -0
  24. package/lib/profile-install.js +61 -0
  25. package/lib/rust.js +194 -0
  26. package/lib/session-shells.js +78 -0
  27. package/lib/shells.js +998 -0
  28. package/lib/tools.js +657 -0
  29. package/locale/en.json +6 -0
  30. package/locale/zh.json +6 -0
  31. package/package.json +114 -4
  32. package/vendor/fastctx/Cargo.lock +3210 -0
  33. package/vendor/fastctx/Cargo.toml +94 -0
  34. package/vendor/fastctx/FORK.md +119 -0
  35. package/vendor/fastctx/LICENSE-APACHE +201 -0
  36. package/vendor/fastctx/NOTICE +40 -0
  37. package/vendor/fastctx/README.md +439 -0
  38. package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
  39. package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
  40. package/vendor/fastctx/UPSTREAM.md +49 -0
  41. package/vendor/fastctx/build.rs +413 -0
  42. package/vendor/fastctx/src/background_status.rs +403 -0
  43. package/vendor/fastctx/src/binary.rs +75 -0
  44. package/vendor/fastctx/src/bounded_sort.rs +500 -0
  45. package/vendor/fastctx/src/budget.rs +781 -0
  46. package/vendor/fastctx/src/cli/mod.rs +110 -0
  47. package/vendor/fastctx/src/context_guard.rs +289 -0
  48. package/vendor/fastctx/src/control/mod.rs +6 -0
  49. package/vendor/fastctx/src/control/paths.rs +49 -0
  50. package/vendor/fastctx/src/control/settings.rs +753 -0
  51. package/vendor/fastctx/src/control/transaction.rs +531 -0
  52. package/vendor/fastctx/src/edit/document.rs +535 -0
  53. package/vendor/fastctx/src/edit/locks.rs +371 -0
  54. package/vendor/fastctx/src/edit/mod.rs +213 -0
  55. package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
  56. package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
  57. package/vendor/fastctx/src/edit/private_storage.rs +234 -0
  58. package/vendor/fastctx/src/edit/replace.rs +1030 -0
  59. package/vendor/fastctx/src/edit_server.rs +53 -0
  60. package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
  61. package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
  62. package/vendor/fastctx/src/encoding.rs +1118 -0
  63. package/vendor/fastctx/src/file_executor.rs +1151 -0
  64. package/vendor/fastctx/src/file_snapshot.rs +1491 -0
  65. package/vendor/fastctx/src/glob_filter.rs +98 -0
  66. package/vendor/fastctx/src/glob_tool.rs +653 -0
  67. package/vendor/fastctx/src/grep_sink.rs +1162 -0
  68. package/vendor/fastctx/src/grep_tool.rs +2449 -0
  69. package/vendor/fastctx/src/lib.rs +45 -0
  70. package/vendor/fastctx/src/main.rs +15 -0
  71. package/vendor/fastctx/src/model.rs +51 -0
  72. package/vendor/fastctx/src/model_guidance.rs +62 -0
  73. package/vendor/fastctx/src/operation.rs +356 -0
  74. package/vendor/fastctx/src/ordered_window.rs +1235 -0
  75. package/vendor/fastctx/src/os_environment.rs +414 -0
  76. package/vendor/fastctx/src/path_codec.rs +850 -0
  77. package/vendor/fastctx/src/paths.rs +244 -0
  78. package/vendor/fastctx/src/process_identity.rs +763 -0
  79. package/vendor/fastctx/src/process_policy.rs +74 -0
  80. package/vendor/fastctx/src/read_tool/batch.rs +496 -0
  81. package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
  82. package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
  83. package/vendor/fastctx/src/read_tool/mod.rs +245 -0
  84. package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
  85. package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
  86. package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
  87. package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
  88. package/vendor/fastctx/src/render_plan.rs +468 -0
  89. package/vendor/fastctx/src/runtime/activity.rs +159 -0
  90. package/vendor/fastctx/src/runtime/hosts.rs +99 -0
  91. package/vendor/fastctx/src/runtime/journal.rs +556 -0
  92. package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
  93. package/vendor/fastctx/src/runtime/mod.rs +746 -0
  94. package/vendor/fastctx/src/runtime/protocol.rs +296 -0
  95. package/vendor/fastctx/src/runtime/session.rs +536 -0
  96. package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
  97. package/vendor/fastctx/src/search_parallelism.rs +106 -0
  98. package/vendor/fastctx/src/search_text.rs +227 -0
  99. package/vendor/fastctx/src/server.rs +359 -0
  100. package/vendor/fastctx/src/server_manifest.rs +468 -0
  101. package/vendor/fastctx/src/server_support.rs +826 -0
  102. package/vendor/fastctx/src/session.rs +629 -0
  103. package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
  104. package/vendor/fastctx/src/shell/bash.rs +263 -0
  105. package/vendor/fastctx/src/shell/buffer.rs +108 -0
  106. package/vendor/fastctx/src/shell/encoding.rs +403 -0
  107. package/vendor/fastctx/src/shell/foreground.rs +115 -0
  108. package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
  109. package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
  110. package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
  111. package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
  112. package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
  113. package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
  114. package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
  115. package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
  116. package/vendor/fastctx/src/shell/mod.rs +345 -0
  117. package/vendor/fastctx/src/shell/normalize.rs +389 -0
  118. package/vendor/fastctx/src/shell/output.rs +406 -0
  119. package/vendor/fastctx/src/shell/process.rs +493 -0
  120. package/vendor/fastctx/src/shell_server.rs +156 -0
  121. package/vendor/fastctx/src/skip_report.rs +83 -0
  122. package/vendor/fastctx/src/stdio_transport.rs +177 -0
  123. package/vendor/fastctx/src/tool_schema.rs +204 -0
  124. package/vendor/fastctx/src/traversal.rs +846 -0
  125. package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
  126. package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
  127. package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
  128. package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
  129. package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
  130. package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
  131. package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
  132. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
  133. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
  134. package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
  135. package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
  136. package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
  137. package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
  138. package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
  139. package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
  140. package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
@@ -0,0 +1,53 @@
1
+ //! The default byte-preserving replacement route in the single `fastctx` server.
2
+
3
+ use crate::budget::GLOBAL_TOKEN_BUDGET_ENV;
4
+ use crate::control::settings;
5
+ use crate::edit::ReplaceRequest;
6
+ use crate::model::ToolResponse;
7
+ use crate::server::FastCtxServer;
8
+ use crate::server_support::{BudgetRetry, run_blocking};
9
+ use rmcp::handler::server::wrapper::Parameters;
10
+ use rmcp::model::CallToolResult;
11
+ use rmcp::{tool, tool_router};
12
+ use std::sync::Arc;
13
+
14
+ #[tool_router(router = edit_tool_router, vis = "pub(crate)")]
15
+ impl FastCtxServer {
16
+ #[tool(
17
+ name = "replace",
18
+ description = "Batch find-and-replace across a file or directory (Rust regex, same engine\nas grep; no lookaround). A reference to an undefined capture group is\nrejected before any write. To delete whole lines, include \\n in the\npattern. Matching is leftmost-first and non-overlapping; unlike grep,\n`^`/`$` anchor the whole file by default — use (?m) for per-line anchors.\nRespects .gitignore; skips .git and binaries; files whose encoding cannot\nbe determined are skipped and listed. Each file is written atomically with\na concurrent-modification check, preserving its original encoding, BOM, and\nline endings. The last line states Complete or Partial.",
19
+ annotations(
20
+ title = "Batch replace file contents",
21
+ read_only_hint = false,
22
+ destructive_hint = false,
23
+ open_world_hint = false
24
+ )
25
+ )]
26
+ async fn replace(&self, Parameters(request): Parameters<ReplaceRequest>) -> CallToolResult {
27
+ let _activity = self.activity.request();
28
+ let replace = self.replace.clone();
29
+ let control_paths = self.session.control_paths.clone();
30
+ let status_shell = self.shell.clone();
31
+ run_blocking(
32
+ Arc::clone(&self.session),
33
+ Arc::clone(&self.replace_permits),
34
+ GLOBAL_TOKEN_BUDGET_ENV,
35
+ move || status_shell.background_status(None),
36
+ BudgetRetry::Never,
37
+ move || {
38
+ let max_file_size_mib = match settings::load(&control_paths)
39
+ .and_then(|settings| settings.replace_file_limit_mib())
40
+ {
41
+ Ok(limit) => limit,
42
+ Err(error) => {
43
+ return ToolResponse::error(format!(
44
+ "Cannot use replace settings: {error}. Repair the FastCtx configuration and retry."
45
+ ));
46
+ }
47
+ };
48
+ replace.replace_with_limit(request.clone(), max_file_size_mib)
49
+ },
50
+ )
51
+ .await
52
+ }
53
+ }
@@ -0,0 +1,587 @@
1
+ //! Test-only copy of the v0.1.1 encoding trust classifier.
2
+ //!
3
+ //! This module is intentionally independent from the production pipeline. Its
4
+ //! byte-oriented decision tree was extracted from commit
5
+ //! `64a6a45f88e65a2c0305e36673fa5e3f99d95384` and remains the differential oracle.
6
+
7
+ use chardetng::{EncodingDetector, Iso2022JpDetection, Utf8Detection};
8
+ use encoding_rs::{
9
+ BIG5, EUC_KR, EncoderResult, Encoding, GBK, SHIFT_JIS, UTF_8, UTF_16BE, UTF_16LE, WINDOWS_1252,
10
+ };
11
+
12
+ const BINARY_PROBE_BYTES: usize = 8 * 1024;
13
+ const LEGACY_EVIDENCE_BYTES: usize = 32;
14
+ const LEGACY_SEGMENT_MAX_BYTES: usize = 4 * 1024;
15
+ const UTF8_SEGMENT_MIN_BYTES: usize = 32;
16
+ const UTF8_SEGMENT_MIN_NON_ASCII_BYTES: usize = 8;
17
+ const FIXED_LEGACY_ENCODINGS: [(&str, &Encoding); 5] = [
18
+ ("windows-1252", WINDOWS_1252),
19
+ ("gbk", GBK),
20
+ ("shift_jis", SHIFT_JIS),
21
+ ("big5", BIG5),
22
+ ("euc-kr", EUC_KR),
23
+ ];
24
+
25
+ #[derive(Debug, Eq, PartialEq)]
26
+ pub(super) enum ReferenceDecision {
27
+ Text {
28
+ kind: String,
29
+ bom_len: usize,
30
+ source_encoding: Option<String>,
31
+ origin: ReferenceOrigin,
32
+ total_lines: usize,
33
+ has_trailing_newline: bool,
34
+ explicit_utf8_warning: bool,
35
+ note: Option<String>,
36
+ },
37
+ Binary,
38
+ Rejected(ReferenceRejection),
39
+ }
40
+
41
+ #[derive(Debug, Eq, PartialEq)]
42
+ pub(super) enum ReferenceOrigin {
43
+ Explicit(String),
44
+ Bom(String),
45
+ Automatic,
46
+ }
47
+
48
+ #[derive(Debug, Eq, PartialEq)]
49
+ pub(super) enum ReferenceRejection {
50
+ Ambiguous(Vec<String>),
51
+ MixedOrInconsistent(Option<usize>),
52
+ Iso2022JpSignature,
53
+ Undecodable,
54
+ BomMismatch(String),
55
+ ExplicitMalformed(String),
56
+ InvalidLabel(String),
57
+ }
58
+
59
+ pub(super) fn classify_v011(bytes: &[u8], explicit_encoding: Option<&str>) -> ReferenceDecision {
60
+ let prefix = &bytes[..bytes.len().min(BINARY_PROBE_BYTES)];
61
+ if let Some(value) = explicit_encoding {
62
+ let detected = match explicit_detected_encoding(value, prefix) {
63
+ Ok(detected) => detected,
64
+ Err(rejection) => return ReferenceDecision::Rejected(rejection),
65
+ };
66
+ return match validate_selected(bytes, &detected, false) {
67
+ Some(stats) => {
68
+ let explicit_utf8_warning = is_legacy_encoding(detected.kind)
69
+ && std::str::from_utf8(bytes).is_ok_and(|text| !text.is_ascii());
70
+ text_decision(detected, stats, explicit_utf8_warning)
71
+ }
72
+ None => ReferenceDecision::Rejected(ReferenceRejection::ExplicitMalformed(
73
+ value.to_string(),
74
+ )),
75
+ };
76
+ }
77
+
78
+ if let Some(detected) = bom_detected_encoding(prefix) {
79
+ if detected.kind == ReferenceKind::EncodingRs(UTF_8) && has_binary_nul(prefix) {
80
+ return ReferenceDecision::Binary;
81
+ }
82
+ return match validate_selected(bytes, &detected, false) {
83
+ Some(stats) => text_decision(detected, stats, false),
84
+ None => ReferenceDecision::Rejected(ReferenceRejection::BomMismatch(
85
+ detected
86
+ .source_encoding
87
+ .map(str::to_string)
88
+ .or_else(|| match &detected.origin {
89
+ ReferenceOrigin::Bom(encoding) => Some(encoding.clone()),
90
+ _ => None,
91
+ })
92
+ .unwrap_or_else(|| "UTF-8".to_string()),
93
+ )),
94
+ };
95
+ }
96
+
97
+ if has_binary_nul(prefix) {
98
+ return ReferenceDecision::Binary;
99
+ }
100
+ let utf8 = ReferenceDetected {
101
+ kind: ReferenceKind::EncodingRs(UTF_8),
102
+ bom_len: 0,
103
+ source_encoding: None,
104
+ origin: ReferenceOrigin::Automatic,
105
+ };
106
+ if let Some(stats) = validate_selected(bytes, &utf8, false) {
107
+ if stats.has_iso_2022_escape {
108
+ return ReferenceDecision::Rejected(ReferenceRejection::Iso2022JpSignature);
109
+ }
110
+ return text_decision(utf8, stats, false);
111
+ }
112
+ if has_reference_binary_magic(prefix) {
113
+ return ReferenceDecision::Binary;
114
+ }
115
+
116
+ let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
117
+ detector.feed(bytes, true);
118
+ let candidate = detector.guess(None, Utf8Detection::Deny);
119
+ let non_ascii_bytes = bytes.iter().filter(|byte| !byte.is_ascii()).count();
120
+ let candidate_detected = automatic_detected(candidate);
121
+ if let Some(conflict_hex_offset) = first_conflicting_utf8_hex_offset(bytes) {
122
+ return ReferenceDecision::Rejected(ReferenceRejection::MixedOrInconsistent(Some(
123
+ conflict_hex_offset,
124
+ )));
125
+ }
126
+ if non_ascii_bytes >= LEGACY_EVIDENCE_BYTES
127
+ && let Some(stats) = validate_selected(bytes, &candidate_detected, true)
128
+ {
129
+ if legacy_segments_are_inconsistent(bytes, candidate) {
130
+ return ReferenceDecision::Rejected(ReferenceRejection::MixedOrInconsistent(None));
131
+ }
132
+ if !candidate.is_single_byte() {
133
+ return text_decision(candidate_detected, stats, false);
134
+ }
135
+ }
136
+
137
+ let mut candidates = Vec::new();
138
+ for (label, encoding) in FIXED_LEGACY_ENCODINGS {
139
+ if validate_selected(bytes, &automatic_detected(encoding), true).is_some() {
140
+ candidates.push(label.to_string());
141
+ }
142
+ }
143
+ if candidates.is_empty() {
144
+ ReferenceDecision::Rejected(ReferenceRejection::Undecodable)
145
+ } else {
146
+ ReferenceDecision::Rejected(ReferenceRejection::Ambiguous(candidates))
147
+ }
148
+ }
149
+
150
+ #[derive(Clone, Copy, Eq, PartialEq)]
151
+ enum ReferenceKind {
152
+ EncodingRs(&'static Encoding),
153
+ Utf32Le,
154
+ Utf32Be,
155
+ }
156
+
157
+ struct ReferenceDetected {
158
+ kind: ReferenceKind,
159
+ bom_len: usize,
160
+ source_encoding: Option<&'static str>,
161
+ origin: ReferenceOrigin,
162
+ }
163
+
164
+ fn automatic_detected(encoding: &'static Encoding) -> ReferenceDetected {
165
+ ReferenceDetected {
166
+ kind: ReferenceKind::EncodingRs(encoding),
167
+ bom_len: 0,
168
+ source_encoding: Some(encoding.name()),
169
+ origin: ReferenceOrigin::Automatic,
170
+ }
171
+ }
172
+
173
+ fn text_decision(
174
+ detected: ReferenceDetected,
175
+ stats: ReferenceStats,
176
+ explicit_utf8_warning: bool,
177
+ ) -> ReferenceDecision {
178
+ let kind = match detected.kind {
179
+ ReferenceKind::EncodingRs(encoding) => encoding.name().to_string(),
180
+ ReferenceKind::Utf32Le => "UTF-32LE".to_string(),
181
+ ReferenceKind::Utf32Be => "UTF-32BE".to_string(),
182
+ };
183
+ let note = transcoding_note(
184
+ detected.source_encoding,
185
+ &detected.origin,
186
+ explicit_utf8_warning,
187
+ );
188
+ ReferenceDecision::Text {
189
+ kind,
190
+ bom_len: detected.bom_len,
191
+ source_encoding: detected.source_encoding.map(str::to_string),
192
+ origin: detected.origin,
193
+ total_lines: stats.total_lines,
194
+ has_trailing_newline: stats.has_trailing_newline,
195
+ explicit_utf8_warning,
196
+ note,
197
+ }
198
+ }
199
+
200
+ fn transcoding_note(
201
+ source_encoding: Option<&'static str>,
202
+ origin: &ReferenceOrigin,
203
+ explicit_utf8_warning: bool,
204
+ ) -> Option<String> {
205
+ let encoding = source_encoding?;
206
+ match origin {
207
+ ReferenceOrigin::Explicit(_) if explicit_utf8_warning => Some(format!(
208
+ "(Note: decoded from {encoding} as requested; output is UTF-8. Warning: the raw bytes are also valid UTF-8 — if this looks garbled, retry with encoding=\"utf-8\" or omit encoding.)"
209
+ )),
210
+ ReferenceOrigin::Explicit(_) => Some(format!(
211
+ "(Note: decoded from {encoding} as requested; output is UTF-8.)"
212
+ )),
213
+ ReferenceOrigin::Bom(_) | ReferenceOrigin::Automatic => {
214
+ Some(format!("(Note: decoded from {encoding}; output is UTF-8.)"))
215
+ }
216
+ }
217
+ }
218
+
219
+ fn has_binary_nul(bytes: &[u8]) -> bool {
220
+ bytes.iter().take(BINARY_PROBE_BYTES).any(|byte| *byte == 0)
221
+ }
222
+
223
+ fn has_reference_binary_magic(bytes: &[u8]) -> bool {
224
+ bytes.starts_with(b"PK\x03\x04")
225
+ || bytes.starts_with(b"\x1F\x8B")
226
+ || bytes.starts_with(b"\x37\x7A\xBC\xAF\x27\x1C")
227
+ || bytes.starts_with(b"\x28\xB5\x2F\xFD")
228
+ || bytes.starts_with(b"\x7FELF")
229
+ || bytes.starts_with(b"MZ")
230
+ || [
231
+ b"\xFE\xED\xFA\xCE".as_slice(),
232
+ b"\xFE\xED\xFA\xCF".as_slice(),
233
+ b"\xCE\xFA\xED\xFE".as_slice(),
234
+ b"\xCF\xFA\xED\xFE".as_slice(),
235
+ ]
236
+ .iter()
237
+ .any(|magic| bytes.starts_with(magic))
238
+ || bytes.starts_with(b"SQLite format 3\0")
239
+ || bytes.starts_with(b"\0asm")
240
+ || bytes.get(257..262) == Some(b"ustar")
241
+ }
242
+
243
+ fn bom_detected_encoding(bytes: &[u8]) -> Option<ReferenceDetected> {
244
+ if bytes.starts_with(b"\x00\x00\xFE\xFF") {
245
+ return Some(ReferenceDetected {
246
+ kind: ReferenceKind::Utf32Be,
247
+ bom_len: 4,
248
+ source_encoding: Some("UTF-32BE"),
249
+ origin: ReferenceOrigin::Bom("UTF-32BE".to_string()),
250
+ });
251
+ }
252
+ if bytes.starts_with(b"\xFF\xFE\x00\x00") {
253
+ return Some(ReferenceDetected {
254
+ kind: ReferenceKind::Utf32Le,
255
+ bom_len: 4,
256
+ source_encoding: Some("UTF-32LE"),
257
+ origin: ReferenceOrigin::Bom("UTF-32LE".to_string()),
258
+ });
259
+ }
260
+ if bytes.starts_with(b"\xEF\xBB\xBF") {
261
+ return Some(ReferenceDetected {
262
+ kind: ReferenceKind::EncodingRs(UTF_8),
263
+ bom_len: 3,
264
+ source_encoding: None,
265
+ origin: ReferenceOrigin::Bom("UTF-8".to_string()),
266
+ });
267
+ }
268
+ if bytes.starts_with(b"\xFF\xFE") {
269
+ return Some(ReferenceDetected {
270
+ kind: ReferenceKind::EncodingRs(UTF_16LE),
271
+ bom_len: 2,
272
+ source_encoding: Some("UTF-16LE"),
273
+ origin: ReferenceOrigin::Bom("UTF-16LE".to_string()),
274
+ });
275
+ }
276
+ if bytes.starts_with(b"\xFE\xFF") {
277
+ return Some(ReferenceDetected {
278
+ kind: ReferenceKind::EncodingRs(UTF_16BE),
279
+ bom_len: 2,
280
+ source_encoding: Some("UTF-16BE"),
281
+ origin: ReferenceOrigin::Bom("UTF-16BE".to_string()),
282
+ });
283
+ }
284
+ None
285
+ }
286
+
287
+ fn explicit_detected_encoding(
288
+ value: &str,
289
+ prefix: &[u8],
290
+ ) -> Result<ReferenceDetected, ReferenceRejection> {
291
+ let label = value.trim_matches(|character: char| character.is_ascii_whitespace());
292
+ let (kind, source_encoding) = if label.eq_ignore_ascii_case("utf-32le") {
293
+ (ReferenceKind::Utf32Le, Some("UTF-32LE"))
294
+ } else if label.eq_ignore_ascii_case("utf-32be") {
295
+ (ReferenceKind::Utf32Be, Some("UTF-32BE"))
296
+ } else {
297
+ let Some(encoding) = Encoding::for_label_no_replacement(label.as_bytes()) else {
298
+ return Err(ReferenceRejection::InvalidLabel(value.to_string()));
299
+ };
300
+ (
301
+ ReferenceKind::EncodingRs(encoding),
302
+ (encoding != UTF_8).then_some(encoding.name()),
303
+ )
304
+ };
305
+ Ok(ReferenceDetected {
306
+ kind,
307
+ bom_len: matching_bom_len(kind, prefix),
308
+ source_encoding,
309
+ origin: ReferenceOrigin::Explicit(value.to_string()),
310
+ })
311
+ }
312
+
313
+ fn matching_bom_len(kind: ReferenceKind, bytes: &[u8]) -> usize {
314
+ match kind {
315
+ ReferenceKind::Utf32Le if bytes.starts_with(b"\xFF\xFE\x00\x00") => 4,
316
+ ReferenceKind::Utf32Be if bytes.starts_with(b"\x00\x00\xFE\xFF") => 4,
317
+ ReferenceKind::EncodingRs(encoding)
318
+ if encoding == UTF_8 && bytes.starts_with(b"\xEF\xBB\xBF") =>
319
+ {
320
+ 3
321
+ }
322
+ ReferenceKind::EncodingRs(encoding)
323
+ if encoding == UTF_16LE && bytes.starts_with(b"\xFF\xFE") =>
324
+ {
325
+ 2
326
+ }
327
+ ReferenceKind::EncodingRs(encoding)
328
+ if encoding == UTF_16BE && bytes.starts_with(b"\xFE\xFF") =>
329
+ {
330
+ 2
331
+ }
332
+ _ => 0,
333
+ }
334
+ }
335
+
336
+ fn is_legacy_encoding(kind: ReferenceKind) -> bool {
337
+ matches!(
338
+ kind,
339
+ ReferenceKind::EncodingRs(encoding)
340
+ if encoding != UTF_8 && encoding != UTF_16LE && encoding != UTF_16BE
341
+ )
342
+ }
343
+
344
+ struct ReferenceStats {
345
+ total_lines: usize,
346
+ has_trailing_newline: bool,
347
+ has_iso_2022_escape: bool,
348
+ }
349
+
350
+ fn validate_selected(
351
+ bytes: &[u8],
352
+ detected: &ReferenceDetected,
353
+ enforce_legacy_hard_checks: bool,
354
+ ) -> Option<ReferenceStats> {
355
+ let content = bytes.get(detected.bom_len..)?;
356
+ let decoded = decode_bytes(content, detected.kind)?;
357
+ if enforce_legacy_hard_checks {
358
+ let ReferenceKind::EncodingRs(encoding) = detected.kind else {
359
+ return None;
360
+ };
361
+ if !hard_validate_bytes(encoding, content) {
362
+ return None;
363
+ }
364
+ }
365
+ let decoded_any = !decoded.is_empty();
366
+ let newline_count = decoded.bytes().filter(|byte| *byte == b'\n').count();
367
+ Some(ReferenceStats {
368
+ total_lines: if decoded_any { newline_count + 1 } else { 0 },
369
+ has_trailing_newline: decoded.ends_with('\n'),
370
+ has_iso_2022_escape: contains_iso_2022_escape(decoded.as_bytes()),
371
+ })
372
+ }
373
+
374
+ fn decode_bytes(bytes: &[u8], kind: ReferenceKind) -> Option<String> {
375
+ match kind {
376
+ ReferenceKind::EncodingRs(encoding) => encoding
377
+ .decode_without_bom_handling_and_without_replacement(bytes)
378
+ .map(|text| text.into_owned()),
379
+ ReferenceKind::Utf32Le => decode_utf32(bytes, true),
380
+ ReferenceKind::Utf32Be => decode_utf32(bytes, false),
381
+ }
382
+ }
383
+
384
+ fn decode_utf32(bytes: &[u8], little_endian: bool) -> Option<String> {
385
+ if !bytes.len().is_multiple_of(4) {
386
+ return None;
387
+ }
388
+ let mut output = String::with_capacity(bytes.len());
389
+ for raw in bytes.as_chunks::<4>().0 {
390
+ let unit = if little_endian {
391
+ u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]])
392
+ } else {
393
+ u32::from_be_bytes([raw[0], raw[1], raw[2], raw[3]])
394
+ };
395
+ output.push(char::from_u32(unit)?);
396
+ }
397
+ Some(output)
398
+ }
399
+
400
+ fn hard_validate_bytes(encoding: &'static Encoding, bytes: &[u8]) -> bool {
401
+ let Some(decoded) = encoding.decode_without_bom_handling_and_without_replacement(bytes) else {
402
+ return false;
403
+ };
404
+ if decoded.chars().any(is_disallowed_legacy_character) {
405
+ return false;
406
+ }
407
+ let mut encoder = encoding.new_encoder();
408
+ let capacity = encoder
409
+ .max_buffer_length_from_utf8_without_replacement(decoded.len())
410
+ .unwrap_or(decoded.len().saturating_mul(4).saturating_add(16))
411
+ .saturating_add(16)
412
+ .max(16);
413
+ let mut encoded = Vec::with_capacity(capacity);
414
+ let (result, read) =
415
+ encoder.encode_from_utf8_to_vec_without_replacement(&decoded, &mut encoded, true);
416
+ result == EncoderResult::InputEmpty && read == decoded.len() && encoded == bytes
417
+ }
418
+
419
+ fn is_disallowed_legacy_character(character: char) -> bool {
420
+ let value = character as u32;
421
+ (value <= 0x1F && !matches!(character, '\t' | '\n' | '\r'))
422
+ || (0x80..=0x9F).contains(&value)
423
+ || (0xFDD0..=0xFDEF).contains(&value)
424
+ || value & 0xFFFF == 0xFFFE
425
+ || value & 0xFFFF == 0xFFFF
426
+ }
427
+
428
+ fn first_conflicting_utf8_hex_offset(bytes: &[u8]) -> Option<usize> {
429
+ let mut scanner = ReferenceUtf8PrefixScanner::default();
430
+ scanner.push(bytes).or_else(|| scanner.finish())
431
+ }
432
+
433
+ #[derive(Default)]
434
+ struct ReferenceUtf8PrefixScanner {
435
+ absolute_base: usize,
436
+ valid_bytes: usize,
437
+ non_ascii_bytes: usize,
438
+ carry: Vec<u8>,
439
+ abandoned: bool,
440
+ }
441
+
442
+ impl ReferenceUtf8PrefixScanner {
443
+ fn push(&mut self, input: &[u8]) -> Option<usize> {
444
+ if self.abandoned {
445
+ return None;
446
+ }
447
+ let mut combined = std::mem::take(&mut self.carry);
448
+ combined.extend_from_slice(input);
449
+ match std::str::from_utf8(&combined) {
450
+ Ok(_) => {
451
+ self.observe_valid(&combined);
452
+ self.absolute_base = self.absolute_base.saturating_add(combined.len());
453
+ None
454
+ }
455
+ Err(error) => {
456
+ let valid = &combined[..error.valid_up_to()];
457
+ self.observe_valid(valid);
458
+ if error.error_len().is_some() {
459
+ let conflict = self.absolute_base.saturating_add(error.valid_up_to());
460
+ if self.clear_prefix() {
461
+ return Some(conflict / 16 + 1);
462
+ }
463
+ self.abandoned = true;
464
+ return None;
465
+ }
466
+ self.absolute_base = self.absolute_base.saturating_add(error.valid_up_to());
467
+ self.carry
468
+ .extend_from_slice(&combined[error.valid_up_to()..]);
469
+ None
470
+ }
471
+ }
472
+ }
473
+
474
+ fn finish(mut self) -> Option<usize> {
475
+ if self.abandoned || self.carry.is_empty() {
476
+ return None;
477
+ }
478
+ let carry = std::mem::take(&mut self.carry);
479
+ match std::str::from_utf8(&carry) {
480
+ Ok(_) => None,
481
+ Err(error) => {
482
+ self.observe_valid(&carry[..error.valid_up_to()]);
483
+ self.clear_prefix()
484
+ .then_some((self.absolute_base + error.valid_up_to()) / 16 + 1)
485
+ }
486
+ }
487
+ }
488
+
489
+ fn observe_valid(&mut self, bytes: &[u8]) {
490
+ self.valid_bytes = self.valid_bytes.saturating_add(bytes.len());
491
+ self.non_ascii_bytes = self
492
+ .non_ascii_bytes
493
+ .saturating_add(bytes.iter().filter(|byte| !byte.is_ascii()).count());
494
+ }
495
+
496
+ fn clear_prefix(&self) -> bool {
497
+ self.valid_bytes >= UTF8_SEGMENT_MIN_BYTES
498
+ && self.non_ascii_bytes >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
499
+ }
500
+ }
501
+
502
+ fn legacy_segments_are_inconsistent(bytes: &[u8], whole_encoding: &'static Encoding) -> bool {
503
+ let mut scanner = ReferenceLegacySegmentScanner::new(whole_encoding);
504
+ scanner.push(bytes) || scanner.finish()
505
+ }
506
+
507
+ struct ReferenceLegacySegmentScanner {
508
+ whole_encoding: &'static Encoding,
509
+ pending: Vec<u8>,
510
+ }
511
+
512
+ impl ReferenceLegacySegmentScanner {
513
+ fn new(whole_encoding: &'static Encoding) -> Self {
514
+ Self {
515
+ whole_encoding,
516
+ pending: Vec::with_capacity(LEGACY_SEGMENT_MAX_BYTES),
517
+ }
518
+ }
519
+
520
+ fn push(&mut self, input: &[u8]) -> bool {
521
+ for byte in input {
522
+ self.pending.push(*byte);
523
+ if *byte == b'\n' && has_segment_evidence(&self.pending) {
524
+ if self.inspect_pending() {
525
+ return true;
526
+ }
527
+ self.pending.clear();
528
+ } else if self.pending.len() >= LEGACY_SEGMENT_MAX_BYTES {
529
+ let end = aligned_prefix_len(self.whole_encoding, &self.pending);
530
+ if end == 0 {
531
+ continue;
532
+ }
533
+ if segment_disagrees(self.whole_encoding, &self.pending[..end]) {
534
+ return true;
535
+ }
536
+ self.pending.drain(..end);
537
+ }
538
+ }
539
+ false
540
+ }
541
+
542
+ fn finish(&mut self) -> bool {
543
+ self.inspect_pending()
544
+ }
545
+
546
+ fn inspect_pending(&self) -> bool {
547
+ has_segment_evidence(&self.pending)
548
+ && hard_validate_bytes(self.whole_encoding, &self.pending)
549
+ && segment_disagrees(self.whole_encoding, &self.pending)
550
+ }
551
+ }
552
+
553
+ fn aligned_prefix_len(encoding: &'static Encoding, bytes: &[u8]) -> usize {
554
+ (0..=4)
555
+ .filter_map(|trim| bytes.len().checked_sub(trim))
556
+ .find(|end| hard_validate_bytes(encoding, &bytes[..*end]))
557
+ .unwrap_or(0)
558
+ }
559
+
560
+ fn has_segment_evidence(bytes: &[u8]) -> bool {
561
+ bytes.iter().filter(|byte| !byte.is_ascii()).count() >= LEGACY_EVIDENCE_BYTES
562
+ || is_strong_utf8_segment(bytes)
563
+ }
564
+
565
+ fn segment_disagrees(whole_encoding: &'static Encoding, bytes: &[u8]) -> bool {
566
+ if is_strong_utf8_segment(bytes) {
567
+ return true;
568
+ }
569
+ let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
570
+ detector.feed(bytes, true);
571
+ let candidate = detector.guess(None, Utf8Detection::Deny);
572
+ candidate != whole_encoding
573
+ && !candidate.is_single_byte()
574
+ && hard_validate_bytes(candidate, bytes)
575
+ }
576
+
577
+ fn is_strong_utf8_segment(bytes: &[u8]) -> bool {
578
+ bytes.len() >= UTF8_SEGMENT_MIN_BYTES
579
+ && bytes.iter().filter(|byte| !byte.is_ascii()).count() >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
580
+ && std::str::from_utf8(bytes).is_ok()
581
+ }
582
+
583
+ fn contains_iso_2022_escape(bytes: &[u8]) -> bool {
584
+ bytes
585
+ .windows(2)
586
+ .any(|pair| pair[0] == 0x1B && matches!(pair[1], b'(' | b'$'))
587
+ }