dsh-ops 0.0.0-stage → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +202 -0
- package/LICENSE +30 -0
- package/NOTICE +106 -0
- package/PROVENANCE.md +435 -0
- package/README.en.md +126 -0
- package/README.md +115 -2
- package/README.zh.md +116 -0
- package/bin/dsh-ops.mjs +1216 -0
- package/cordis.patch.yml +160 -0
- package/docs/manual-validation.md +53 -0
- package/docs/release-0.2.1.md +72 -0
- package/docs/schema-baseline.json +64 -0
- package/docs/schema-current.json +84 -0
- package/docs/schema-measurement.md +17 -0
- package/dsh-plugin.json +88 -0
- package/icon.svg +12 -0
- package/lib/binary.js +409 -0
- package/lib/config.js +198 -0
- package/lib/handshake.js +252 -0
- package/lib/index.js +108 -0
- package/lib/jobs.js +42 -0
- package/lib/policy.js +64 -0
- package/lib/presentation.js +63 -0
- package/lib/profile-install.js +61 -0
- package/lib/rust.js +194 -0
- package/lib/session-shells.js +78 -0
- package/lib/shells.js +998 -0
- package/lib/tools.js +657 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +114 -4
- package/vendor/fastctx/Cargo.lock +3210 -0
- package/vendor/fastctx/Cargo.toml +94 -0
- package/vendor/fastctx/FORK.md +119 -0
- package/vendor/fastctx/LICENSE-APACHE +201 -0
- package/vendor/fastctx/NOTICE +40 -0
- package/vendor/fastctx/README.md +439 -0
- package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
- package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
- package/vendor/fastctx/UPSTREAM.md +49 -0
- package/vendor/fastctx/build.rs +413 -0
- package/vendor/fastctx/src/background_status.rs +403 -0
- package/vendor/fastctx/src/binary.rs +75 -0
- package/vendor/fastctx/src/bounded_sort.rs +500 -0
- package/vendor/fastctx/src/budget.rs +781 -0
- package/vendor/fastctx/src/cli/mod.rs +110 -0
- package/vendor/fastctx/src/context_guard.rs +289 -0
- package/vendor/fastctx/src/control/mod.rs +6 -0
- package/vendor/fastctx/src/control/paths.rs +49 -0
- package/vendor/fastctx/src/control/settings.rs +753 -0
- package/vendor/fastctx/src/control/transaction.rs +531 -0
- package/vendor/fastctx/src/edit/document.rs +535 -0
- package/vendor/fastctx/src/edit/locks.rs +371 -0
- package/vendor/fastctx/src/edit/mod.rs +213 -0
- package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
- package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
- package/vendor/fastctx/src/edit/private_storage.rs +234 -0
- package/vendor/fastctx/src/edit/replace.rs +1030 -0
- package/vendor/fastctx/src/edit_server.rs +53 -0
- package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
- package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
- package/vendor/fastctx/src/encoding.rs +1118 -0
- package/vendor/fastctx/src/file_executor.rs +1151 -0
- package/vendor/fastctx/src/file_snapshot.rs +1491 -0
- package/vendor/fastctx/src/glob_filter.rs +98 -0
- package/vendor/fastctx/src/glob_tool.rs +653 -0
- package/vendor/fastctx/src/grep_sink.rs +1162 -0
- package/vendor/fastctx/src/grep_tool.rs +2449 -0
- package/vendor/fastctx/src/lib.rs +45 -0
- package/vendor/fastctx/src/main.rs +15 -0
- package/vendor/fastctx/src/model.rs +51 -0
- package/vendor/fastctx/src/model_guidance.rs +62 -0
- package/vendor/fastctx/src/operation.rs +356 -0
- package/vendor/fastctx/src/ordered_window.rs +1235 -0
- package/vendor/fastctx/src/os_environment.rs +414 -0
- package/vendor/fastctx/src/path_codec.rs +850 -0
- package/vendor/fastctx/src/paths.rs +244 -0
- package/vendor/fastctx/src/process_identity.rs +763 -0
- package/vendor/fastctx/src/process_policy.rs +74 -0
- package/vendor/fastctx/src/read_tool/batch.rs +496 -0
- package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
- package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
- package/vendor/fastctx/src/read_tool/mod.rs +245 -0
- package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
- package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
- package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
- package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
- package/vendor/fastctx/src/render_plan.rs +468 -0
- package/vendor/fastctx/src/runtime/activity.rs +159 -0
- package/vendor/fastctx/src/runtime/hosts.rs +99 -0
- package/vendor/fastctx/src/runtime/journal.rs +556 -0
- package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
- package/vendor/fastctx/src/runtime/mod.rs +746 -0
- package/vendor/fastctx/src/runtime/protocol.rs +296 -0
- package/vendor/fastctx/src/runtime/session.rs +536 -0
- package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
- package/vendor/fastctx/src/search_parallelism.rs +106 -0
- package/vendor/fastctx/src/search_text.rs +227 -0
- package/vendor/fastctx/src/server.rs +359 -0
- package/vendor/fastctx/src/server_manifest.rs +468 -0
- package/vendor/fastctx/src/server_support.rs +826 -0
- package/vendor/fastctx/src/session.rs +629 -0
- package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
- package/vendor/fastctx/src/shell/bash.rs +263 -0
- package/vendor/fastctx/src/shell/buffer.rs +108 -0
- package/vendor/fastctx/src/shell/encoding.rs +403 -0
- package/vendor/fastctx/src/shell/foreground.rs +115 -0
- package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
- package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
- package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
- package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
- package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
- package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
- package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
- package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
- package/vendor/fastctx/src/shell/mod.rs +345 -0
- package/vendor/fastctx/src/shell/normalize.rs +389 -0
- package/vendor/fastctx/src/shell/output.rs +406 -0
- package/vendor/fastctx/src/shell/process.rs +493 -0
- package/vendor/fastctx/src/shell_server.rs +156 -0
- package/vendor/fastctx/src/skip_report.rs +83 -0
- package/vendor/fastctx/src/stdio_transport.rs +177 -0
- package/vendor/fastctx/src/tool_schema.rs +204 -0
- package/vendor/fastctx/src/traversal.rs +846 -0
- package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
//! The default byte-preserving replacement route in the single `fastctx` server.
|
|
2
|
+
|
|
3
|
+
use crate::budget::GLOBAL_TOKEN_BUDGET_ENV;
|
|
4
|
+
use crate::control::settings;
|
|
5
|
+
use crate::edit::ReplaceRequest;
|
|
6
|
+
use crate::model::ToolResponse;
|
|
7
|
+
use crate::server::FastCtxServer;
|
|
8
|
+
use crate::server_support::{BudgetRetry, run_blocking};
|
|
9
|
+
use rmcp::handler::server::wrapper::Parameters;
|
|
10
|
+
use rmcp::model::CallToolResult;
|
|
11
|
+
use rmcp::{tool, tool_router};
|
|
12
|
+
use std::sync::Arc;
|
|
13
|
+
|
|
14
|
+
#[tool_router(router = edit_tool_router, vis = "pub(crate)")]
|
|
15
|
+
impl FastCtxServer {
|
|
16
|
+
#[tool(
|
|
17
|
+
name = "replace",
|
|
18
|
+
description = "Batch find-and-replace across a file or directory (Rust regex, same engine\nas grep; no lookaround). A reference to an undefined capture group is\nrejected before any write. To delete whole lines, include \\n in the\npattern. Matching is leftmost-first and non-overlapping; unlike grep,\n`^`/`$` anchor the whole file by default — use (?m) for per-line anchors.\nRespects .gitignore; skips .git and binaries; files whose encoding cannot\nbe determined are skipped and listed. Each file is written atomically with\na concurrent-modification check, preserving its original encoding, BOM, and\nline endings. The last line states Complete or Partial.",
|
|
19
|
+
annotations(
|
|
20
|
+
title = "Batch replace file contents",
|
|
21
|
+
read_only_hint = false,
|
|
22
|
+
destructive_hint = false,
|
|
23
|
+
open_world_hint = false
|
|
24
|
+
)
|
|
25
|
+
)]
|
|
26
|
+
async fn replace(&self, Parameters(request): Parameters<ReplaceRequest>) -> CallToolResult {
|
|
27
|
+
let _activity = self.activity.request();
|
|
28
|
+
let replace = self.replace.clone();
|
|
29
|
+
let control_paths = self.session.control_paths.clone();
|
|
30
|
+
let status_shell = self.shell.clone();
|
|
31
|
+
run_blocking(
|
|
32
|
+
Arc::clone(&self.session),
|
|
33
|
+
Arc::clone(&self.replace_permits),
|
|
34
|
+
GLOBAL_TOKEN_BUDGET_ENV,
|
|
35
|
+
move || status_shell.background_status(None),
|
|
36
|
+
BudgetRetry::Never,
|
|
37
|
+
move || {
|
|
38
|
+
let max_file_size_mib = match settings::load(&control_paths)
|
|
39
|
+
.and_then(|settings| settings.replace_file_limit_mib())
|
|
40
|
+
{
|
|
41
|
+
Ok(limit) => limit,
|
|
42
|
+
Err(error) => {
|
|
43
|
+
return ToolResponse::error(format!(
|
|
44
|
+
"Cannot use replace settings: {error}. Repair the FastCtx configuration and retry."
|
|
45
|
+
));
|
|
46
|
+
}
|
|
47
|
+
};
|
|
48
|
+
replace.replace_with_limit(request.clone(), max_file_size_mib)
|
|
49
|
+
},
|
|
50
|
+
)
|
|
51
|
+
.await
|
|
52
|
+
}
|
|
53
|
+
}
|
|
@@ -0,0 +1,587 @@
|
|
|
1
|
+
//! Test-only copy of the v0.1.1 encoding trust classifier.
|
|
2
|
+
//!
|
|
3
|
+
//! This module is intentionally independent from the production pipeline. Its
|
|
4
|
+
//! byte-oriented decision tree was extracted from commit
|
|
5
|
+
//! `64a6a45f88e65a2c0305e36673fa5e3f99d95384` and remains the differential oracle.
|
|
6
|
+
|
|
7
|
+
use chardetng::{EncodingDetector, Iso2022JpDetection, Utf8Detection};
|
|
8
|
+
use encoding_rs::{
|
|
9
|
+
BIG5, EUC_KR, EncoderResult, Encoding, GBK, SHIFT_JIS, UTF_8, UTF_16BE, UTF_16LE, WINDOWS_1252,
|
|
10
|
+
};
|
|
11
|
+
|
|
12
|
+
const BINARY_PROBE_BYTES: usize = 8 * 1024;
|
|
13
|
+
const LEGACY_EVIDENCE_BYTES: usize = 32;
|
|
14
|
+
const LEGACY_SEGMENT_MAX_BYTES: usize = 4 * 1024;
|
|
15
|
+
const UTF8_SEGMENT_MIN_BYTES: usize = 32;
|
|
16
|
+
const UTF8_SEGMENT_MIN_NON_ASCII_BYTES: usize = 8;
|
|
17
|
+
const FIXED_LEGACY_ENCODINGS: [(&str, &Encoding); 5] = [
|
|
18
|
+
("windows-1252", WINDOWS_1252),
|
|
19
|
+
("gbk", GBK),
|
|
20
|
+
("shift_jis", SHIFT_JIS),
|
|
21
|
+
("big5", BIG5),
|
|
22
|
+
("euc-kr", EUC_KR),
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
#[derive(Debug, Eq, PartialEq)]
|
|
26
|
+
pub(super) enum ReferenceDecision {
|
|
27
|
+
Text {
|
|
28
|
+
kind: String,
|
|
29
|
+
bom_len: usize,
|
|
30
|
+
source_encoding: Option<String>,
|
|
31
|
+
origin: ReferenceOrigin,
|
|
32
|
+
total_lines: usize,
|
|
33
|
+
has_trailing_newline: bool,
|
|
34
|
+
explicit_utf8_warning: bool,
|
|
35
|
+
note: Option<String>,
|
|
36
|
+
},
|
|
37
|
+
Binary,
|
|
38
|
+
Rejected(ReferenceRejection),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
#[derive(Debug, Eq, PartialEq)]
|
|
42
|
+
pub(super) enum ReferenceOrigin {
|
|
43
|
+
Explicit(String),
|
|
44
|
+
Bom(String),
|
|
45
|
+
Automatic,
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
#[derive(Debug, Eq, PartialEq)]
|
|
49
|
+
pub(super) enum ReferenceRejection {
|
|
50
|
+
Ambiguous(Vec<String>),
|
|
51
|
+
MixedOrInconsistent(Option<usize>),
|
|
52
|
+
Iso2022JpSignature,
|
|
53
|
+
Undecodable,
|
|
54
|
+
BomMismatch(String),
|
|
55
|
+
ExplicitMalformed(String),
|
|
56
|
+
InvalidLabel(String),
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
pub(super) fn classify_v011(bytes: &[u8], explicit_encoding: Option<&str>) -> ReferenceDecision {
|
|
60
|
+
let prefix = &bytes[..bytes.len().min(BINARY_PROBE_BYTES)];
|
|
61
|
+
if let Some(value) = explicit_encoding {
|
|
62
|
+
let detected = match explicit_detected_encoding(value, prefix) {
|
|
63
|
+
Ok(detected) => detected,
|
|
64
|
+
Err(rejection) => return ReferenceDecision::Rejected(rejection),
|
|
65
|
+
};
|
|
66
|
+
return match validate_selected(bytes, &detected, false) {
|
|
67
|
+
Some(stats) => {
|
|
68
|
+
let explicit_utf8_warning = is_legacy_encoding(detected.kind)
|
|
69
|
+
&& std::str::from_utf8(bytes).is_ok_and(|text| !text.is_ascii());
|
|
70
|
+
text_decision(detected, stats, explicit_utf8_warning)
|
|
71
|
+
}
|
|
72
|
+
None => ReferenceDecision::Rejected(ReferenceRejection::ExplicitMalformed(
|
|
73
|
+
value.to_string(),
|
|
74
|
+
)),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
if let Some(detected) = bom_detected_encoding(prefix) {
|
|
79
|
+
if detected.kind == ReferenceKind::EncodingRs(UTF_8) && has_binary_nul(prefix) {
|
|
80
|
+
return ReferenceDecision::Binary;
|
|
81
|
+
}
|
|
82
|
+
return match validate_selected(bytes, &detected, false) {
|
|
83
|
+
Some(stats) => text_decision(detected, stats, false),
|
|
84
|
+
None => ReferenceDecision::Rejected(ReferenceRejection::BomMismatch(
|
|
85
|
+
detected
|
|
86
|
+
.source_encoding
|
|
87
|
+
.map(str::to_string)
|
|
88
|
+
.or_else(|| match &detected.origin {
|
|
89
|
+
ReferenceOrigin::Bom(encoding) => Some(encoding.clone()),
|
|
90
|
+
_ => None,
|
|
91
|
+
})
|
|
92
|
+
.unwrap_or_else(|| "UTF-8".to_string()),
|
|
93
|
+
)),
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
if has_binary_nul(prefix) {
|
|
98
|
+
return ReferenceDecision::Binary;
|
|
99
|
+
}
|
|
100
|
+
let utf8 = ReferenceDetected {
|
|
101
|
+
kind: ReferenceKind::EncodingRs(UTF_8),
|
|
102
|
+
bom_len: 0,
|
|
103
|
+
source_encoding: None,
|
|
104
|
+
origin: ReferenceOrigin::Automatic,
|
|
105
|
+
};
|
|
106
|
+
if let Some(stats) = validate_selected(bytes, &utf8, false) {
|
|
107
|
+
if stats.has_iso_2022_escape {
|
|
108
|
+
return ReferenceDecision::Rejected(ReferenceRejection::Iso2022JpSignature);
|
|
109
|
+
}
|
|
110
|
+
return text_decision(utf8, stats, false);
|
|
111
|
+
}
|
|
112
|
+
if has_reference_binary_magic(prefix) {
|
|
113
|
+
return ReferenceDecision::Binary;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
|
|
117
|
+
detector.feed(bytes, true);
|
|
118
|
+
let candidate = detector.guess(None, Utf8Detection::Deny);
|
|
119
|
+
let non_ascii_bytes = bytes.iter().filter(|byte| !byte.is_ascii()).count();
|
|
120
|
+
let candidate_detected = automatic_detected(candidate);
|
|
121
|
+
if let Some(conflict_hex_offset) = first_conflicting_utf8_hex_offset(bytes) {
|
|
122
|
+
return ReferenceDecision::Rejected(ReferenceRejection::MixedOrInconsistent(Some(
|
|
123
|
+
conflict_hex_offset,
|
|
124
|
+
)));
|
|
125
|
+
}
|
|
126
|
+
if non_ascii_bytes >= LEGACY_EVIDENCE_BYTES
|
|
127
|
+
&& let Some(stats) = validate_selected(bytes, &candidate_detected, true)
|
|
128
|
+
{
|
|
129
|
+
if legacy_segments_are_inconsistent(bytes, candidate) {
|
|
130
|
+
return ReferenceDecision::Rejected(ReferenceRejection::MixedOrInconsistent(None));
|
|
131
|
+
}
|
|
132
|
+
if !candidate.is_single_byte() {
|
|
133
|
+
return text_decision(candidate_detected, stats, false);
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
let mut candidates = Vec::new();
|
|
138
|
+
for (label, encoding) in FIXED_LEGACY_ENCODINGS {
|
|
139
|
+
if validate_selected(bytes, &automatic_detected(encoding), true).is_some() {
|
|
140
|
+
candidates.push(label.to_string());
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
if candidates.is_empty() {
|
|
144
|
+
ReferenceDecision::Rejected(ReferenceRejection::Undecodable)
|
|
145
|
+
} else {
|
|
146
|
+
ReferenceDecision::Rejected(ReferenceRejection::Ambiguous(candidates))
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
#[derive(Clone, Copy, Eq, PartialEq)]
|
|
151
|
+
enum ReferenceKind {
|
|
152
|
+
EncodingRs(&'static Encoding),
|
|
153
|
+
Utf32Le,
|
|
154
|
+
Utf32Be,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
struct ReferenceDetected {
|
|
158
|
+
kind: ReferenceKind,
|
|
159
|
+
bom_len: usize,
|
|
160
|
+
source_encoding: Option<&'static str>,
|
|
161
|
+
origin: ReferenceOrigin,
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
fn automatic_detected(encoding: &'static Encoding) -> ReferenceDetected {
|
|
165
|
+
ReferenceDetected {
|
|
166
|
+
kind: ReferenceKind::EncodingRs(encoding),
|
|
167
|
+
bom_len: 0,
|
|
168
|
+
source_encoding: Some(encoding.name()),
|
|
169
|
+
origin: ReferenceOrigin::Automatic,
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
fn text_decision(
|
|
174
|
+
detected: ReferenceDetected,
|
|
175
|
+
stats: ReferenceStats,
|
|
176
|
+
explicit_utf8_warning: bool,
|
|
177
|
+
) -> ReferenceDecision {
|
|
178
|
+
let kind = match detected.kind {
|
|
179
|
+
ReferenceKind::EncodingRs(encoding) => encoding.name().to_string(),
|
|
180
|
+
ReferenceKind::Utf32Le => "UTF-32LE".to_string(),
|
|
181
|
+
ReferenceKind::Utf32Be => "UTF-32BE".to_string(),
|
|
182
|
+
};
|
|
183
|
+
let note = transcoding_note(
|
|
184
|
+
detected.source_encoding,
|
|
185
|
+
&detected.origin,
|
|
186
|
+
explicit_utf8_warning,
|
|
187
|
+
);
|
|
188
|
+
ReferenceDecision::Text {
|
|
189
|
+
kind,
|
|
190
|
+
bom_len: detected.bom_len,
|
|
191
|
+
source_encoding: detected.source_encoding.map(str::to_string),
|
|
192
|
+
origin: detected.origin,
|
|
193
|
+
total_lines: stats.total_lines,
|
|
194
|
+
has_trailing_newline: stats.has_trailing_newline,
|
|
195
|
+
explicit_utf8_warning,
|
|
196
|
+
note,
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
fn transcoding_note(
|
|
201
|
+
source_encoding: Option<&'static str>,
|
|
202
|
+
origin: &ReferenceOrigin,
|
|
203
|
+
explicit_utf8_warning: bool,
|
|
204
|
+
) -> Option<String> {
|
|
205
|
+
let encoding = source_encoding?;
|
|
206
|
+
match origin {
|
|
207
|
+
ReferenceOrigin::Explicit(_) if explicit_utf8_warning => Some(format!(
|
|
208
|
+
"(Note: decoded from {encoding} as requested; output is UTF-8. Warning: the raw bytes are also valid UTF-8 — if this looks garbled, retry with encoding=\"utf-8\" or omit encoding.)"
|
|
209
|
+
)),
|
|
210
|
+
ReferenceOrigin::Explicit(_) => Some(format!(
|
|
211
|
+
"(Note: decoded from {encoding} as requested; output is UTF-8.)"
|
|
212
|
+
)),
|
|
213
|
+
ReferenceOrigin::Bom(_) | ReferenceOrigin::Automatic => {
|
|
214
|
+
Some(format!("(Note: decoded from {encoding}; output is UTF-8.)"))
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
fn has_binary_nul(bytes: &[u8]) -> bool {
|
|
220
|
+
bytes.iter().take(BINARY_PROBE_BYTES).any(|byte| *byte == 0)
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
fn has_reference_binary_magic(bytes: &[u8]) -> bool {
|
|
224
|
+
bytes.starts_with(b"PK\x03\x04")
|
|
225
|
+
|| bytes.starts_with(b"\x1F\x8B")
|
|
226
|
+
|| bytes.starts_with(b"\x37\x7A\xBC\xAF\x27\x1C")
|
|
227
|
+
|| bytes.starts_with(b"\x28\xB5\x2F\xFD")
|
|
228
|
+
|| bytes.starts_with(b"\x7FELF")
|
|
229
|
+
|| bytes.starts_with(b"MZ")
|
|
230
|
+
|| [
|
|
231
|
+
b"\xFE\xED\xFA\xCE".as_slice(),
|
|
232
|
+
b"\xFE\xED\xFA\xCF".as_slice(),
|
|
233
|
+
b"\xCE\xFA\xED\xFE".as_slice(),
|
|
234
|
+
b"\xCF\xFA\xED\xFE".as_slice(),
|
|
235
|
+
]
|
|
236
|
+
.iter()
|
|
237
|
+
.any(|magic| bytes.starts_with(magic))
|
|
238
|
+
|| bytes.starts_with(b"SQLite format 3\0")
|
|
239
|
+
|| bytes.starts_with(b"\0asm")
|
|
240
|
+
|| bytes.get(257..262) == Some(b"ustar")
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
fn bom_detected_encoding(bytes: &[u8]) -> Option<ReferenceDetected> {
|
|
244
|
+
if bytes.starts_with(b"\x00\x00\xFE\xFF") {
|
|
245
|
+
return Some(ReferenceDetected {
|
|
246
|
+
kind: ReferenceKind::Utf32Be,
|
|
247
|
+
bom_len: 4,
|
|
248
|
+
source_encoding: Some("UTF-32BE"),
|
|
249
|
+
origin: ReferenceOrigin::Bom("UTF-32BE".to_string()),
|
|
250
|
+
});
|
|
251
|
+
}
|
|
252
|
+
if bytes.starts_with(b"\xFF\xFE\x00\x00") {
|
|
253
|
+
return Some(ReferenceDetected {
|
|
254
|
+
kind: ReferenceKind::Utf32Le,
|
|
255
|
+
bom_len: 4,
|
|
256
|
+
source_encoding: Some("UTF-32LE"),
|
|
257
|
+
origin: ReferenceOrigin::Bom("UTF-32LE".to_string()),
|
|
258
|
+
});
|
|
259
|
+
}
|
|
260
|
+
if bytes.starts_with(b"\xEF\xBB\xBF") {
|
|
261
|
+
return Some(ReferenceDetected {
|
|
262
|
+
kind: ReferenceKind::EncodingRs(UTF_8),
|
|
263
|
+
bom_len: 3,
|
|
264
|
+
source_encoding: None,
|
|
265
|
+
origin: ReferenceOrigin::Bom("UTF-8".to_string()),
|
|
266
|
+
});
|
|
267
|
+
}
|
|
268
|
+
if bytes.starts_with(b"\xFF\xFE") {
|
|
269
|
+
return Some(ReferenceDetected {
|
|
270
|
+
kind: ReferenceKind::EncodingRs(UTF_16LE),
|
|
271
|
+
bom_len: 2,
|
|
272
|
+
source_encoding: Some("UTF-16LE"),
|
|
273
|
+
origin: ReferenceOrigin::Bom("UTF-16LE".to_string()),
|
|
274
|
+
});
|
|
275
|
+
}
|
|
276
|
+
if bytes.starts_with(b"\xFE\xFF") {
|
|
277
|
+
return Some(ReferenceDetected {
|
|
278
|
+
kind: ReferenceKind::EncodingRs(UTF_16BE),
|
|
279
|
+
bom_len: 2,
|
|
280
|
+
source_encoding: Some("UTF-16BE"),
|
|
281
|
+
origin: ReferenceOrigin::Bom("UTF-16BE".to_string()),
|
|
282
|
+
});
|
|
283
|
+
}
|
|
284
|
+
None
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
fn explicit_detected_encoding(
|
|
288
|
+
value: &str,
|
|
289
|
+
prefix: &[u8],
|
|
290
|
+
) -> Result<ReferenceDetected, ReferenceRejection> {
|
|
291
|
+
let label = value.trim_matches(|character: char| character.is_ascii_whitespace());
|
|
292
|
+
let (kind, source_encoding) = if label.eq_ignore_ascii_case("utf-32le") {
|
|
293
|
+
(ReferenceKind::Utf32Le, Some("UTF-32LE"))
|
|
294
|
+
} else if label.eq_ignore_ascii_case("utf-32be") {
|
|
295
|
+
(ReferenceKind::Utf32Be, Some("UTF-32BE"))
|
|
296
|
+
} else {
|
|
297
|
+
let Some(encoding) = Encoding::for_label_no_replacement(label.as_bytes()) else {
|
|
298
|
+
return Err(ReferenceRejection::InvalidLabel(value.to_string()));
|
|
299
|
+
};
|
|
300
|
+
(
|
|
301
|
+
ReferenceKind::EncodingRs(encoding),
|
|
302
|
+
(encoding != UTF_8).then_some(encoding.name()),
|
|
303
|
+
)
|
|
304
|
+
};
|
|
305
|
+
Ok(ReferenceDetected {
|
|
306
|
+
kind,
|
|
307
|
+
bom_len: matching_bom_len(kind, prefix),
|
|
308
|
+
source_encoding,
|
|
309
|
+
origin: ReferenceOrigin::Explicit(value.to_string()),
|
|
310
|
+
})
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
fn matching_bom_len(kind: ReferenceKind, bytes: &[u8]) -> usize {
|
|
314
|
+
match kind {
|
|
315
|
+
ReferenceKind::Utf32Le if bytes.starts_with(b"\xFF\xFE\x00\x00") => 4,
|
|
316
|
+
ReferenceKind::Utf32Be if bytes.starts_with(b"\x00\x00\xFE\xFF") => 4,
|
|
317
|
+
ReferenceKind::EncodingRs(encoding)
|
|
318
|
+
if encoding == UTF_8 && bytes.starts_with(b"\xEF\xBB\xBF") =>
|
|
319
|
+
{
|
|
320
|
+
3
|
|
321
|
+
}
|
|
322
|
+
ReferenceKind::EncodingRs(encoding)
|
|
323
|
+
if encoding == UTF_16LE && bytes.starts_with(b"\xFF\xFE") =>
|
|
324
|
+
{
|
|
325
|
+
2
|
|
326
|
+
}
|
|
327
|
+
ReferenceKind::EncodingRs(encoding)
|
|
328
|
+
if encoding == UTF_16BE && bytes.starts_with(b"\xFE\xFF") =>
|
|
329
|
+
{
|
|
330
|
+
2
|
|
331
|
+
}
|
|
332
|
+
_ => 0,
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
fn is_legacy_encoding(kind: ReferenceKind) -> bool {
|
|
337
|
+
matches!(
|
|
338
|
+
kind,
|
|
339
|
+
ReferenceKind::EncodingRs(encoding)
|
|
340
|
+
if encoding != UTF_8 && encoding != UTF_16LE && encoding != UTF_16BE
|
|
341
|
+
)
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
struct ReferenceStats {
|
|
345
|
+
total_lines: usize,
|
|
346
|
+
has_trailing_newline: bool,
|
|
347
|
+
has_iso_2022_escape: bool,
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
fn validate_selected(
|
|
351
|
+
bytes: &[u8],
|
|
352
|
+
detected: &ReferenceDetected,
|
|
353
|
+
enforce_legacy_hard_checks: bool,
|
|
354
|
+
) -> Option<ReferenceStats> {
|
|
355
|
+
let content = bytes.get(detected.bom_len..)?;
|
|
356
|
+
let decoded = decode_bytes(content, detected.kind)?;
|
|
357
|
+
if enforce_legacy_hard_checks {
|
|
358
|
+
let ReferenceKind::EncodingRs(encoding) = detected.kind else {
|
|
359
|
+
return None;
|
|
360
|
+
};
|
|
361
|
+
if !hard_validate_bytes(encoding, content) {
|
|
362
|
+
return None;
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
let decoded_any = !decoded.is_empty();
|
|
366
|
+
let newline_count = decoded.bytes().filter(|byte| *byte == b'\n').count();
|
|
367
|
+
Some(ReferenceStats {
|
|
368
|
+
total_lines: if decoded_any { newline_count + 1 } else { 0 },
|
|
369
|
+
has_trailing_newline: decoded.ends_with('\n'),
|
|
370
|
+
has_iso_2022_escape: contains_iso_2022_escape(decoded.as_bytes()),
|
|
371
|
+
})
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
fn decode_bytes(bytes: &[u8], kind: ReferenceKind) -> Option<String> {
|
|
375
|
+
match kind {
|
|
376
|
+
ReferenceKind::EncodingRs(encoding) => encoding
|
|
377
|
+
.decode_without_bom_handling_and_without_replacement(bytes)
|
|
378
|
+
.map(|text| text.into_owned()),
|
|
379
|
+
ReferenceKind::Utf32Le => decode_utf32(bytes, true),
|
|
380
|
+
ReferenceKind::Utf32Be => decode_utf32(bytes, false),
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
fn decode_utf32(bytes: &[u8], little_endian: bool) -> Option<String> {
|
|
385
|
+
if !bytes.len().is_multiple_of(4) {
|
|
386
|
+
return None;
|
|
387
|
+
}
|
|
388
|
+
let mut output = String::with_capacity(bytes.len());
|
|
389
|
+
for raw in bytes.as_chunks::<4>().0 {
|
|
390
|
+
let unit = if little_endian {
|
|
391
|
+
u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]])
|
|
392
|
+
} else {
|
|
393
|
+
u32::from_be_bytes([raw[0], raw[1], raw[2], raw[3]])
|
|
394
|
+
};
|
|
395
|
+
output.push(char::from_u32(unit)?);
|
|
396
|
+
}
|
|
397
|
+
Some(output)
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
fn hard_validate_bytes(encoding: &'static Encoding, bytes: &[u8]) -> bool {
|
|
401
|
+
let Some(decoded) = encoding.decode_without_bom_handling_and_without_replacement(bytes) else {
|
|
402
|
+
return false;
|
|
403
|
+
};
|
|
404
|
+
if decoded.chars().any(is_disallowed_legacy_character) {
|
|
405
|
+
return false;
|
|
406
|
+
}
|
|
407
|
+
let mut encoder = encoding.new_encoder();
|
|
408
|
+
let capacity = encoder
|
|
409
|
+
.max_buffer_length_from_utf8_without_replacement(decoded.len())
|
|
410
|
+
.unwrap_or(decoded.len().saturating_mul(4).saturating_add(16))
|
|
411
|
+
.saturating_add(16)
|
|
412
|
+
.max(16);
|
|
413
|
+
let mut encoded = Vec::with_capacity(capacity);
|
|
414
|
+
let (result, read) =
|
|
415
|
+
encoder.encode_from_utf8_to_vec_without_replacement(&decoded, &mut encoded, true);
|
|
416
|
+
result == EncoderResult::InputEmpty && read == decoded.len() && encoded == bytes
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
fn is_disallowed_legacy_character(character: char) -> bool {
|
|
420
|
+
let value = character as u32;
|
|
421
|
+
(value <= 0x1F && !matches!(character, '\t' | '\n' | '\r'))
|
|
422
|
+
|| (0x80..=0x9F).contains(&value)
|
|
423
|
+
|| (0xFDD0..=0xFDEF).contains(&value)
|
|
424
|
+
|| value & 0xFFFF == 0xFFFE
|
|
425
|
+
|| value & 0xFFFF == 0xFFFF
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
fn first_conflicting_utf8_hex_offset(bytes: &[u8]) -> Option<usize> {
|
|
429
|
+
let mut scanner = ReferenceUtf8PrefixScanner::default();
|
|
430
|
+
scanner.push(bytes).or_else(|| scanner.finish())
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
#[derive(Default)]
|
|
434
|
+
struct ReferenceUtf8PrefixScanner {
|
|
435
|
+
absolute_base: usize,
|
|
436
|
+
valid_bytes: usize,
|
|
437
|
+
non_ascii_bytes: usize,
|
|
438
|
+
carry: Vec<u8>,
|
|
439
|
+
abandoned: bool,
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
impl ReferenceUtf8PrefixScanner {
|
|
443
|
+
fn push(&mut self, input: &[u8]) -> Option<usize> {
|
|
444
|
+
if self.abandoned {
|
|
445
|
+
return None;
|
|
446
|
+
}
|
|
447
|
+
let mut combined = std::mem::take(&mut self.carry);
|
|
448
|
+
combined.extend_from_slice(input);
|
|
449
|
+
match std::str::from_utf8(&combined) {
|
|
450
|
+
Ok(_) => {
|
|
451
|
+
self.observe_valid(&combined);
|
|
452
|
+
self.absolute_base = self.absolute_base.saturating_add(combined.len());
|
|
453
|
+
None
|
|
454
|
+
}
|
|
455
|
+
Err(error) => {
|
|
456
|
+
let valid = &combined[..error.valid_up_to()];
|
|
457
|
+
self.observe_valid(valid);
|
|
458
|
+
if error.error_len().is_some() {
|
|
459
|
+
let conflict = self.absolute_base.saturating_add(error.valid_up_to());
|
|
460
|
+
if self.clear_prefix() {
|
|
461
|
+
return Some(conflict / 16 + 1);
|
|
462
|
+
}
|
|
463
|
+
self.abandoned = true;
|
|
464
|
+
return None;
|
|
465
|
+
}
|
|
466
|
+
self.absolute_base = self.absolute_base.saturating_add(error.valid_up_to());
|
|
467
|
+
self.carry
|
|
468
|
+
.extend_from_slice(&combined[error.valid_up_to()..]);
|
|
469
|
+
None
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
fn finish(mut self) -> Option<usize> {
|
|
475
|
+
if self.abandoned || self.carry.is_empty() {
|
|
476
|
+
return None;
|
|
477
|
+
}
|
|
478
|
+
let carry = std::mem::take(&mut self.carry);
|
|
479
|
+
match std::str::from_utf8(&carry) {
|
|
480
|
+
Ok(_) => None,
|
|
481
|
+
Err(error) => {
|
|
482
|
+
self.observe_valid(&carry[..error.valid_up_to()]);
|
|
483
|
+
self.clear_prefix()
|
|
484
|
+
.then_some((self.absolute_base + error.valid_up_to()) / 16 + 1)
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
fn observe_valid(&mut self, bytes: &[u8]) {
|
|
490
|
+
self.valid_bytes = self.valid_bytes.saturating_add(bytes.len());
|
|
491
|
+
self.non_ascii_bytes = self
|
|
492
|
+
.non_ascii_bytes
|
|
493
|
+
.saturating_add(bytes.iter().filter(|byte| !byte.is_ascii()).count());
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
fn clear_prefix(&self) -> bool {
|
|
497
|
+
self.valid_bytes >= UTF8_SEGMENT_MIN_BYTES
|
|
498
|
+
&& self.non_ascii_bytes >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
|
|
499
|
+
}
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
fn legacy_segments_are_inconsistent(bytes: &[u8], whole_encoding: &'static Encoding) -> bool {
|
|
503
|
+
let mut scanner = ReferenceLegacySegmentScanner::new(whole_encoding);
|
|
504
|
+
scanner.push(bytes) || scanner.finish()
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
struct ReferenceLegacySegmentScanner {
|
|
508
|
+
whole_encoding: &'static Encoding,
|
|
509
|
+
pending: Vec<u8>,
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
impl ReferenceLegacySegmentScanner {
|
|
513
|
+
fn new(whole_encoding: &'static Encoding) -> Self {
|
|
514
|
+
Self {
|
|
515
|
+
whole_encoding,
|
|
516
|
+
pending: Vec::with_capacity(LEGACY_SEGMENT_MAX_BYTES),
|
|
517
|
+
}
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
fn push(&mut self, input: &[u8]) -> bool {
|
|
521
|
+
for byte in input {
|
|
522
|
+
self.pending.push(*byte);
|
|
523
|
+
if *byte == b'\n' && has_segment_evidence(&self.pending) {
|
|
524
|
+
if self.inspect_pending() {
|
|
525
|
+
return true;
|
|
526
|
+
}
|
|
527
|
+
self.pending.clear();
|
|
528
|
+
} else if self.pending.len() >= LEGACY_SEGMENT_MAX_BYTES {
|
|
529
|
+
let end = aligned_prefix_len(self.whole_encoding, &self.pending);
|
|
530
|
+
if end == 0 {
|
|
531
|
+
continue;
|
|
532
|
+
}
|
|
533
|
+
if segment_disagrees(self.whole_encoding, &self.pending[..end]) {
|
|
534
|
+
return true;
|
|
535
|
+
}
|
|
536
|
+
self.pending.drain(..end);
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
false
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
fn finish(&mut self) -> bool {
|
|
543
|
+
self.inspect_pending()
|
|
544
|
+
}
|
|
545
|
+
|
|
546
|
+
fn inspect_pending(&self) -> bool {
|
|
547
|
+
has_segment_evidence(&self.pending)
|
|
548
|
+
&& hard_validate_bytes(self.whole_encoding, &self.pending)
|
|
549
|
+
&& segment_disagrees(self.whole_encoding, &self.pending)
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
fn aligned_prefix_len(encoding: &'static Encoding, bytes: &[u8]) -> usize {
|
|
554
|
+
(0..=4)
|
|
555
|
+
.filter_map(|trim| bytes.len().checked_sub(trim))
|
|
556
|
+
.find(|end| hard_validate_bytes(encoding, &bytes[..*end]))
|
|
557
|
+
.unwrap_or(0)
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
fn has_segment_evidence(bytes: &[u8]) -> bool {
|
|
561
|
+
bytes.iter().filter(|byte| !byte.is_ascii()).count() >= LEGACY_EVIDENCE_BYTES
|
|
562
|
+
|| is_strong_utf8_segment(bytes)
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
fn segment_disagrees(whole_encoding: &'static Encoding, bytes: &[u8]) -> bool {
|
|
566
|
+
if is_strong_utf8_segment(bytes) {
|
|
567
|
+
return true;
|
|
568
|
+
}
|
|
569
|
+
let mut detector = EncodingDetector::new(Iso2022JpDetection::Allow);
|
|
570
|
+
detector.feed(bytes, true);
|
|
571
|
+
let candidate = detector.guess(None, Utf8Detection::Deny);
|
|
572
|
+
candidate != whole_encoding
|
|
573
|
+
&& !candidate.is_single_byte()
|
|
574
|
+
&& hard_validate_bytes(candidate, bytes)
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
fn is_strong_utf8_segment(bytes: &[u8]) -> bool {
|
|
578
|
+
bytes.len() >= UTF8_SEGMENT_MIN_BYTES
|
|
579
|
+
&& bytes.iter().filter(|byte| !byte.is_ascii()).count() >= UTF8_SEGMENT_MIN_NON_ASCII_BYTES
|
|
580
|
+
&& std::str::from_utf8(bytes).is_ok()
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
fn contains_iso_2022_escape(bytes: &[u8]) -> bool {
|
|
584
|
+
bytes
|
|
585
|
+
.windows(2)
|
|
586
|
+
.any(|pair| pair[0] == 0x1B && matches!(pair[1], b'(' | b'$'))
|
|
587
|
+
}
|