dsh-ops 0.0.0-stage → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +202 -0
  2. package/LICENSE +30 -0
  3. package/NOTICE +106 -0
  4. package/PROVENANCE.md +435 -0
  5. package/README.en.md +126 -0
  6. package/README.md +115 -2
  7. package/README.zh.md +116 -0
  8. package/bin/dsh-ops.mjs +1216 -0
  9. package/cordis.patch.yml +160 -0
  10. package/docs/manual-validation.md +53 -0
  11. package/docs/release-0.2.1.md +72 -0
  12. package/docs/schema-baseline.json +64 -0
  13. package/docs/schema-current.json +84 -0
  14. package/docs/schema-measurement.md +17 -0
  15. package/dsh-plugin.json +88 -0
  16. package/icon.svg +12 -0
  17. package/lib/binary.js +409 -0
  18. package/lib/config.js +198 -0
  19. package/lib/handshake.js +252 -0
  20. package/lib/index.js +108 -0
  21. package/lib/jobs.js +42 -0
  22. package/lib/policy.js +64 -0
  23. package/lib/presentation.js +63 -0
  24. package/lib/profile-install.js +61 -0
  25. package/lib/rust.js +194 -0
  26. package/lib/session-shells.js +78 -0
  27. package/lib/shells.js +998 -0
  28. package/lib/tools.js +657 -0
  29. package/locale/en.json +6 -0
  30. package/locale/zh.json +6 -0
  31. package/package.json +114 -4
  32. package/vendor/fastctx/Cargo.lock +3210 -0
  33. package/vendor/fastctx/Cargo.toml +94 -0
  34. package/vendor/fastctx/FORK.md +119 -0
  35. package/vendor/fastctx/LICENSE-APACHE +201 -0
  36. package/vendor/fastctx/NOTICE +40 -0
  37. package/vendor/fastctx/README.md +439 -0
  38. package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
  39. package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
  40. package/vendor/fastctx/UPSTREAM.md +49 -0
  41. package/vendor/fastctx/build.rs +413 -0
  42. package/vendor/fastctx/src/background_status.rs +403 -0
  43. package/vendor/fastctx/src/binary.rs +75 -0
  44. package/vendor/fastctx/src/bounded_sort.rs +500 -0
  45. package/vendor/fastctx/src/budget.rs +781 -0
  46. package/vendor/fastctx/src/cli/mod.rs +110 -0
  47. package/vendor/fastctx/src/context_guard.rs +289 -0
  48. package/vendor/fastctx/src/control/mod.rs +6 -0
  49. package/vendor/fastctx/src/control/paths.rs +49 -0
  50. package/vendor/fastctx/src/control/settings.rs +753 -0
  51. package/vendor/fastctx/src/control/transaction.rs +531 -0
  52. package/vendor/fastctx/src/edit/document.rs +535 -0
  53. package/vendor/fastctx/src/edit/locks.rs +371 -0
  54. package/vendor/fastctx/src/edit/mod.rs +213 -0
  55. package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
  56. package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
  57. package/vendor/fastctx/src/edit/private_storage.rs +234 -0
  58. package/vendor/fastctx/src/edit/replace.rs +1030 -0
  59. package/vendor/fastctx/src/edit_server.rs +53 -0
  60. package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
  61. package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
  62. package/vendor/fastctx/src/encoding.rs +1118 -0
  63. package/vendor/fastctx/src/file_executor.rs +1151 -0
  64. package/vendor/fastctx/src/file_snapshot.rs +1491 -0
  65. package/vendor/fastctx/src/glob_filter.rs +98 -0
  66. package/vendor/fastctx/src/glob_tool.rs +653 -0
  67. package/vendor/fastctx/src/grep_sink.rs +1162 -0
  68. package/vendor/fastctx/src/grep_tool.rs +2449 -0
  69. package/vendor/fastctx/src/lib.rs +45 -0
  70. package/vendor/fastctx/src/main.rs +15 -0
  71. package/vendor/fastctx/src/model.rs +51 -0
  72. package/vendor/fastctx/src/model_guidance.rs +62 -0
  73. package/vendor/fastctx/src/operation.rs +356 -0
  74. package/vendor/fastctx/src/ordered_window.rs +1235 -0
  75. package/vendor/fastctx/src/os_environment.rs +414 -0
  76. package/vendor/fastctx/src/path_codec.rs +850 -0
  77. package/vendor/fastctx/src/paths.rs +244 -0
  78. package/vendor/fastctx/src/process_identity.rs +763 -0
  79. package/vendor/fastctx/src/process_policy.rs +74 -0
  80. package/vendor/fastctx/src/read_tool/batch.rs +496 -0
  81. package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
  82. package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
  83. package/vendor/fastctx/src/read_tool/mod.rs +245 -0
  84. package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
  85. package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
  86. package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
  87. package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
  88. package/vendor/fastctx/src/render_plan.rs +468 -0
  89. package/vendor/fastctx/src/runtime/activity.rs +159 -0
  90. package/vendor/fastctx/src/runtime/hosts.rs +99 -0
  91. package/vendor/fastctx/src/runtime/journal.rs +556 -0
  92. package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
  93. package/vendor/fastctx/src/runtime/mod.rs +746 -0
  94. package/vendor/fastctx/src/runtime/protocol.rs +296 -0
  95. package/vendor/fastctx/src/runtime/session.rs +536 -0
  96. package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
  97. package/vendor/fastctx/src/search_parallelism.rs +106 -0
  98. package/vendor/fastctx/src/search_text.rs +227 -0
  99. package/vendor/fastctx/src/server.rs +359 -0
  100. package/vendor/fastctx/src/server_manifest.rs +468 -0
  101. package/vendor/fastctx/src/server_support.rs +826 -0
  102. package/vendor/fastctx/src/session.rs +629 -0
  103. package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
  104. package/vendor/fastctx/src/shell/bash.rs +263 -0
  105. package/vendor/fastctx/src/shell/buffer.rs +108 -0
  106. package/vendor/fastctx/src/shell/encoding.rs +403 -0
  107. package/vendor/fastctx/src/shell/foreground.rs +115 -0
  108. package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
  109. package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
  110. package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
  111. package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
  112. package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
  113. package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
  114. package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
  115. package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
  116. package/vendor/fastctx/src/shell/mod.rs +345 -0
  117. package/vendor/fastctx/src/shell/normalize.rs +389 -0
  118. package/vendor/fastctx/src/shell/output.rs +406 -0
  119. package/vendor/fastctx/src/shell/process.rs +493 -0
  120. package/vendor/fastctx/src/shell_server.rs +156 -0
  121. package/vendor/fastctx/src/skip_report.rs +83 -0
  122. package/vendor/fastctx/src/stdio_transport.rs +177 -0
  123. package/vendor/fastctx/src/tool_schema.rs +204 -0
  124. package/vendor/fastctx/src/traversal.rs +846 -0
  125. package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
  126. package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
  127. package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
  128. package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
  129. package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
  130. package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
  131. package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
  132. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
  133. package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
  134. package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
  135. package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
  136. package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
  137. package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
  138. package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
  139. package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
  140. package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
@@ -0,0 +1,74 @@
1
+ //! Shared process-launch policy for FastCtx-owned non-interactive children.
2
+
3
+ use std::ffi::OsStr;
4
+ use std::process::Command;
5
+
6
+ #[cfg(windows)]
7
+ const CREATE_NO_WINDOW: u32 = 0x0800_0000;
8
+
9
+ /// Creates a command without allocating or inheriting a console window on Windows.
10
+ pub(crate) fn noninteractive_command(program: impl AsRef<OsStr>) -> Command {
11
+ let mut command = Command::new(program);
12
+ apply_noninteractive_policy(&mut command);
13
+ command
14
+ }
15
+
16
+ /// Applies the platform policy to an existing non-interactive command.
17
+ pub(crate) fn apply_noninteractive_policy(command: &mut Command) {
18
+ #[cfg(windows)]
19
+ {
20
+ use std::os::windows::process::CommandExt;
21
+
22
+ command.creation_flags(CREATE_NO_WINDOW);
23
+ }
24
+ #[cfg(not(windows))]
25
+ {
26
+ let _ = command;
27
+ }
28
+ }
29
+
30
+ /// Composes extra Windows creation flags without dropping the no-window invariant.
31
+ #[cfg(windows)]
32
+ pub(crate) const fn noninteractive_creation_flags(additional: u32) -> u32 {
33
+ CREATE_NO_WINDOW | additional
34
+ }
35
+
36
+ #[cfg(test)]
37
+ mod tests {
38
+ #[cfg(windows)]
39
+ #[test]
40
+ fn creation_flag_composition_preserves_the_no_window_contract() {
41
+ const INDEPENDENT_CREATE_NO_WINDOW: u32 = 0x0800_0000;
42
+ const CREATE_SUSPENDED: u32 = 0x0000_0004;
43
+
44
+ let flags = super::noninteractive_creation_flags(CREATE_SUSPENDED);
45
+ assert_eq!(
46
+ flags & INDEPENDENT_CREATE_NO_WINDOW,
47
+ INDEPENDENT_CREATE_NO_WINDOW
48
+ );
49
+ assert_eq!(flags & CREATE_SUSPENDED, CREATE_SUSPENDED);
50
+ }
51
+
52
+ #[cfg(windows)]
53
+ #[test]
54
+ fn noninteractive_child_has_no_console() {
55
+ const PROBE: &str = "FASTCTX_TEST_NO_WINDOW_PROBE";
56
+ if std::env::var_os(PROBE).is_some() {
57
+ use windows_sys::Win32::System::Console::GetConsoleWindow;
58
+
59
+ // SAFETY: GetConsoleWindow takes no arguments and returns a borrowed HWND.
60
+ assert!(unsafe { GetConsoleWindow() }.is_null());
61
+ return;
62
+ }
63
+
64
+ let status = super::noninteractive_command(std::env::current_exe().unwrap())
65
+ .args([
66
+ "--exact",
67
+ "process_policy::tests::noninteractive_child_has_no_console",
68
+ ])
69
+ .env(PROBE, "1")
70
+ .status()
71
+ .unwrap();
72
+ assert!(status.success());
73
+ }
74
+ }
@@ -0,0 +1,496 @@
1
+ //! Request-ordered text batching with one shared token budget and exact continuations.
2
+
3
+ use super::{BatchReadEntry, ReadRequest, image_file, pdf, text_file};
4
+ use crate::binary::detect_binary_type;
5
+ use crate::budget::{READ_TOKEN_BUDGET_ENV, TokenBudget, estimate_tokens, tool_token_budget};
6
+ use crate::encoding::canonical_encoding_label;
7
+ use crate::model::ToolResponse;
8
+ use crate::paths::{
9
+ canonical_existing, display_path, io_error_message, is_local_file_uri_input,
10
+ missing_read_file_message, parse_input_path, parse_local_path_input,
11
+ };
12
+ use serde::Serialize;
13
+ use std::collections::HashSet;
14
+ use std::fs;
15
+ use std::io::Read;
16
+
17
+ const MAX_BATCH_ENTRIES: usize = 32;
18
+
19
+ #[derive(Clone, Debug, Serialize)]
20
+ struct ContinuationEntry {
21
+ path: String,
22
+ #[serde(skip_serializing_if = "Option::is_none")]
23
+ offset: Option<usize>,
24
+ #[serde(skip_serializing_if = "Option::is_none")]
25
+ limit: Option<usize>,
26
+ #[serde(skip_serializing_if = "Option::is_none")]
27
+ encoding: Option<String>,
28
+ }
29
+
30
+ struct PreparedEntry {
31
+ path: String,
32
+ outcome: PreparedOutcome,
33
+ }
34
+
35
+ enum PreparedOutcome {
36
+ Content(text_file::BatchTextContent),
37
+ Message(String),
38
+ }
39
+
40
+ pub(super) fn read_text_files(mut request: ReadRequest) -> ToolResponse {
41
+ let mut entries = request
42
+ .files
43
+ .take()
44
+ .expect("batch shape was validated by read_file");
45
+ if !(1..=MAX_BATCH_ENTRIES).contains(&entries.len()) {
46
+ return ToolResponse::error(format!(
47
+ "Invalid files value: expected 1 to 32 entries, got {}.",
48
+ entries.len()
49
+ ));
50
+ }
51
+ for (parameter, present) in [
52
+ ("offset", request.offset.is_some()),
53
+ ("encoding", request.encoding.is_some()),
54
+ ] {
55
+ if present {
56
+ return ToolResponse::error(format!(
57
+ "The top-level {parameter} parameter cannot be combined with files; set it inside the files entries instead."
58
+ ));
59
+ }
60
+ }
61
+ if request.limit == Some(0) {
62
+ return ToolResponse::error("Invalid limit value: 0. Expected an integer >= 1.");
63
+ }
64
+ if let Some(default_limit) = request.limit {
65
+ for entry in &mut entries {
66
+ entry.limit.get_or_insert(default_limit);
67
+ }
68
+ }
69
+ for (parameter, present) in [
70
+ ("pages", request.pages.is_some()),
71
+ ("pdf_mode", request.pdf_mode.is_some()),
72
+ ("view", request.view.is_some()),
73
+ ] {
74
+ if present {
75
+ return ToolResponse::error(format!(
76
+ "The {parameter} parameter cannot be combined with files; PDFs, images, and hex view are single-file reads."
77
+ ));
78
+ }
79
+ }
80
+ let entries = match validate_entries(entries) {
81
+ Ok(entries) => entries,
82
+ Err(error) => return ToolResponse::error(error),
83
+ };
84
+ let budget = match tool_token_budget(READ_TOKEN_BUDGET_ENV) {
85
+ Ok(budget) => budget,
86
+ Err(error) => return ToolResponse::error(error),
87
+ };
88
+ pack_entries(entries, budget)
89
+ }
90
+
91
+ fn validate_entries(mut entries: Vec<BatchReadEntry>) -> Result<Vec<BatchReadEntry>, String> {
92
+ let mut seen = HashSet::with_capacity(entries.len());
93
+ for entry in &mut entries {
94
+ if entry.offset == Some(0) {
95
+ return Err("Invalid offset value: 0. Expected an integer >= 1.".to_string());
96
+ }
97
+ if entry.limit == Some(0) {
98
+ return Err("Invalid limit value: 0. Expected an integer >= 1.".to_string());
99
+ }
100
+ let from_uri = is_local_file_uri_input(&entry.path);
101
+ let parsed = parse_local_path_input(&entry.path)?;
102
+ if let Some(encoding) = entry.encoding.as_deref()
103
+ && let Err(rejection) = canonical_encoding_label(encoding)
104
+ {
105
+ return Err(rejection.message(""));
106
+ }
107
+ // A relative entry is an existence problem, not a request-shape one, so it is
108
+ // reported in its own segment and never discards its neighbors. Keeping it out
109
+ // of canonicalization also stops it from resolving into a false duplicate.
110
+ // The URL crate may spell a Windows file URI with an 8.3 component even when the
111
+ // equivalent native input used its long name. Canonicalization expands that platform
112
+ // alias; Unix keeps the URI's lexical symlink spelling so it matches the plain input.
113
+ let normalized_input_path = if cfg!(windows) && parsed.is_absolute() {
114
+ canonical_existing(&parsed).unwrap_or_else(|_| parsed.clone())
115
+ } else {
116
+ parsed.clone()
117
+ };
118
+ let normalized_input = display_path(&normalized_input_path);
119
+ let key_path = if parsed.is_absolute() {
120
+ canonical_existing(&parsed).unwrap_or_else(|_| parsed.clone())
121
+ } else {
122
+ parsed.clone()
123
+ };
124
+ let key_path = display_path(&key_path);
125
+ entry.path = if from_uri {
126
+ normalized_input
127
+ } else {
128
+ continuation_path(&entry.path)
129
+ };
130
+ #[cfg(windows)]
131
+ let key_path = key_path.to_ascii_lowercase();
132
+ let key = (key_path, entry.offset, entry.limit, entry.encoding.clone());
133
+ if !seen.insert(key) {
134
+ return Err(format!(
135
+ "Duplicate files entry: two entries request the exact same text interval for {} (including offset, limit, and encoding).",
136
+ entry.path
137
+ ));
138
+ }
139
+ }
140
+ Ok(entries)
141
+ }
142
+
143
+ fn pack_entries(entries: Vec<BatchReadEntry>, budget: TokenBudget) -> ToolResponse {
144
+ let total = entries.len();
145
+ let mut progress = entries
146
+ .iter()
147
+ .map(ContinuationEntry::from_request)
148
+ .map(Some)
149
+ .collect::<Vec<_>>();
150
+ let mut segments = Vec::new();
151
+ // How many leading entries this response has fully settled (content shown in
152
+ // full or an inline problem reported). A budget break stops the loop, so every
153
+ // entry after the break was never attempted and must not be counted against
154
+ // the ones already delivered (#32).
155
+ let mut delivered = 0_usize;
156
+ // The index whose content is only partially shown, if any; it stays in the
157
+ // continuation array but counts as processed for the tally.
158
+ let mut partially_shown = false;
159
+
160
+ for (index, entry) in entries.iter().enumerate() {
161
+ let prepared = prepare_entry(entry, budget.value);
162
+ match prepared.outcome {
163
+ PreparedOutcome::Message(message) => {
164
+ let segment = format!("=== {} ===\n{message}", prepared.path);
165
+ let mut proposed = progress.clone();
166
+ proposed[index] = None;
167
+ if !candidate_fits(
168
+ &segments,
169
+ &segment,
170
+ &proposed,
171
+ total,
172
+ delivered,
173
+ partially_shown,
174
+ budget.value,
175
+ ) {
176
+ if segments.is_empty() {
177
+ return budget_too_small(budget);
178
+ }
179
+ break;
180
+ }
181
+ segments.push(segment);
182
+ progress = proposed;
183
+ delivered += 1;
184
+ }
185
+ PreparedOutcome::Content(content) => {
186
+ let shown = largest_fitting_prefix(
187
+ &segments,
188
+ &prepared.path,
189
+ entry,
190
+ &content,
191
+ &progress,
192
+ index,
193
+ total,
194
+ budget.value,
195
+ );
196
+ if shown == 0 {
197
+ if segments.is_empty() {
198
+ return budget_too_small(budget);
199
+ }
200
+ break;
201
+ }
202
+ let proposed = progress_after(entry, &content, shown);
203
+ let segment = content_segment(&prepared.path, &content, shown);
204
+ progress[index] = proposed;
205
+ segments.push(segment);
206
+ if shown < content.lines.len() || !content.slice_complete {
207
+ partially_shown = true;
208
+ break;
209
+ }
210
+ delivered += 1;
211
+ }
212
+ }
213
+ }
214
+
215
+ ToolResponse::text(render_response(
216
+ &segments,
217
+ &progress,
218
+ total,
219
+ delivered,
220
+ partially_shown,
221
+ ))
222
+ }
223
+
224
+ fn prepare_entry(entry: &BatchReadEntry, collection_budget: usize) -> PreparedEntry {
225
+ let parsed = parse_input_path(&entry.path);
226
+ let input_display = display_path(&parsed);
227
+ if !parsed.is_absolute() {
228
+ return PreparedEntry {
229
+ path: input_display,
230
+ outcome: PreparedOutcome::Message(missing_read_file_message(&entry.path)),
231
+ };
232
+ }
233
+ let metadata = match fs::metadata(&parsed) {
234
+ Ok(metadata) => metadata,
235
+ Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
236
+ return PreparedEntry {
237
+ path: input_display,
238
+ outcome: PreparedOutcome::Message(missing_read_file_message(&entry.path)),
239
+ };
240
+ }
241
+ Err(error) => {
242
+ return PreparedEntry {
243
+ path: input_display,
244
+ outcome: PreparedOutcome::Message(io_error_message(&parsed, &error)),
245
+ };
246
+ }
247
+ };
248
+ let path = canonical_existing(&parsed).unwrap_or(parsed);
249
+ let path_display = display_path(&path);
250
+ if metadata.is_dir() {
251
+ return PreparedEntry {
252
+ path: path_display.clone(),
253
+ outcome: PreparedOutcome::Message(format!(
254
+ "{path_display} is a directory, not a file. Use the glob tool to list its contents."
255
+ )),
256
+ };
257
+ }
258
+ if !metadata.is_file() {
259
+ return PreparedEntry {
260
+ path: path_display.clone(),
261
+ outcome: PreparedOutcome::Message(format!(
262
+ "Cannot read non-regular file: {path_display}. Only regular files are supported."
263
+ )),
264
+ };
265
+ }
266
+ let mut prefix = Vec::new();
267
+ if let Err(error) =
268
+ fs::File::open(&path).and_then(|file| file.take(8 * 1024).read_to_end(&mut prefix))
269
+ {
270
+ return PreparedEntry {
271
+ path: path_display,
272
+ outcome: PreparedOutcome::Message(io_error_message(&path, &error)),
273
+ };
274
+ }
275
+ if pdf::is_pdf(&path, &prefix) {
276
+ return PreparedEntry {
277
+ path: path_display,
278
+ outcome: PreparedOutcome::Message(
279
+ "PDF files cannot be included in files. Read this file separately with file_path and optional pages/pdf_mode."
280
+ .to_string(),
281
+ ),
282
+ };
283
+ }
284
+ if image_file::detect_image_mime(&path, &prefix).is_some() {
285
+ return PreparedEntry {
286
+ path: path_display,
287
+ outcome: PreparedOutcome::Message(
288
+ "Image files cannot be included in files. Read this file separately with file_path."
289
+ .to_string(),
290
+ ),
291
+ };
292
+ }
293
+ let outcome = match text_file::read_batch_text_file(
294
+ &path,
295
+ &path_display,
296
+ entry.offset,
297
+ entry.limit,
298
+ entry.encoding.as_deref(),
299
+ detect_binary_type(&prefix),
300
+ collection_budget,
301
+ ) {
302
+ Ok(content) => PreparedOutcome::Content(content),
303
+ Err(message) => PreparedOutcome::Message(message),
304
+ };
305
+ PreparedEntry {
306
+ path: path_display,
307
+ outcome,
308
+ }
309
+ }
310
+
311
+ #[allow(clippy::too_many_arguments)]
312
+ fn largest_fitting_prefix(
313
+ segments: &[String],
314
+ path: &str,
315
+ entry: &BatchReadEntry,
316
+ content: &text_file::BatchTextContent,
317
+ progress: &[Option<ContinuationEntry>],
318
+ index: usize,
319
+ total: usize,
320
+ budget: usize,
321
+ ) -> usize {
322
+ let maximum = content.lines.len();
323
+ let fits = |shown: usize| {
324
+ let mut proposed = progress.to_vec();
325
+ proposed[index] = progress_after(entry, content, shown);
326
+ let segment = content_segment(path, content, shown);
327
+ // During the binary search this entry is by definition the partially
328
+ // shown one; the delivered count comes from the caller via `segments`.
329
+ candidate_fits(
330
+ segments,
331
+ &segment,
332
+ &proposed,
333
+ total,
334
+ segments.len(),
335
+ true,
336
+ budget,
337
+ )
338
+ };
339
+
340
+ // Probing the whole slice first is what makes the search sound: dropping the last line
341
+ // can also drop this file from the continuation array, so `fits` is only monotonic
342
+ // below `maximum`. Returning here also keeps the common all-fits entry at one probe
343
+ // instead of a full binary search that re-tokenizes the whole response each step.
344
+ if fits(maximum) {
345
+ return maximum;
346
+ }
347
+ if maximum <= 1 {
348
+ return 0;
349
+ }
350
+ let mut best = 0;
351
+ let mut low = 1;
352
+ let mut high = maximum - 1;
353
+ while low <= high {
354
+ let shown = low + (high - low) / 2;
355
+ if fits(shown) {
356
+ best = best.max(shown);
357
+ low = shown.saturating_add(1);
358
+ } else if shown == 1 {
359
+ break;
360
+ } else {
361
+ high = shown - 1;
362
+ }
363
+ }
364
+ best
365
+ }
366
+
367
+ fn progress_after(
368
+ entry: &BatchReadEntry,
369
+ content: &text_file::BatchTextContent,
370
+ shown: usize,
371
+ ) -> Option<ContinuationEntry> {
372
+ let last = content.first.saturating_add(shown.saturating_sub(1));
373
+ if last >= content.total_lines {
374
+ return None;
375
+ }
376
+ // The continuation's limit counts the requested window, not the file: the next
377
+ // call should read exactly the lines this request promised but did not show.
378
+ // A limit that already ran past EOF carries no remainder, so it is dropped —
379
+ // an explicit cap of "whatever is left" is what an omitted limit already means.
380
+ let remaining_lines = content.total_lines - last;
381
+ let limit = entry.limit.and_then(|limit| {
382
+ let requested_end = content.first.saturating_add(limit.saturating_sub(1));
383
+ let value = requested_end.min(content.total_lines) - last;
384
+ (value > 0 && value < remaining_lines).then_some(value)
385
+ });
386
+ Some(ContinuationEntry {
387
+ path: entry.path.clone(),
388
+ offset: Some(last.saturating_add(1)),
389
+ limit,
390
+ encoding: entry.encoding.clone(),
391
+ })
392
+ }
393
+
394
+ fn content_segment(path: &str, content: &text_file::BatchTextContent, shown: usize) -> String {
395
+ let last = content.first.saturating_add(shown.saturating_sub(1));
396
+ let header = if content.total_is_known {
397
+ format!(
398
+ "=== {path} (lines {}-{last} of {}) ===",
399
+ content.first, content.total_lines
400
+ )
401
+ } else {
402
+ format!("=== {path} (lines {}-{last}) ===", content.first)
403
+ };
404
+ let mut lines = Vec::with_capacity(shown + 2);
405
+ lines.push(header);
406
+ if let Some(note) = &content.transcoding_note {
407
+ lines.push(note.clone());
408
+ }
409
+ lines.extend(content.lines[..shown].iter().cloned());
410
+ lines.join("\n")
411
+ }
412
+
413
+ fn candidate_fits(
414
+ segments: &[String],
415
+ candidate: &str,
416
+ progress: &[Option<ContinuationEntry>],
417
+ total: usize,
418
+ delivered: usize,
419
+ partially_shown: bool,
420
+ budget: usize,
421
+ ) -> bool {
422
+ let mut proposed = segments.to_vec();
423
+ proposed.push(candidate.to_string());
424
+ estimate_tokens(&render_response(
425
+ &proposed,
426
+ progress,
427
+ total,
428
+ delivered,
429
+ partially_shown,
430
+ )) <= budget
431
+ }
432
+
433
+ fn render_response(
434
+ segments: &[String],
435
+ progress: &[Option<ContinuationEntry>],
436
+ total: usize,
437
+ delivered: usize,
438
+ partially_shown: bool,
439
+ ) -> String {
440
+ let terminal = batch_terminal(progress, total, delivered, partially_shown);
441
+ if segments.is_empty() {
442
+ terminal
443
+ } else {
444
+ format!("{}\n\n{terminal}", segments.join("\n\n"))
445
+ }
446
+ }
447
+
448
+ fn batch_terminal(
449
+ progress: &[Option<ContinuationEntry>],
450
+ total: usize,
451
+ delivered: usize,
452
+ partially_shown: bool,
453
+ ) -> String {
454
+ let pending = progress.iter().flatten().collect::<Vec<_>>();
455
+ if pending.is_empty() {
456
+ let noun = if total == 1 { "entry" } else { "entries" };
457
+ return format!("(Complete: {total} {noun} processed.)");
458
+ }
459
+ // `delivered` counts fully settled entries; a partially shown entry counts as
460
+ // processed too, because its continuation carries the exact resume point. The
461
+ // remainder of `total` was never attempted — the budget broke before it — so
462
+ // saying "0 of N" for them would tell the model its delivered content did not
463
+ // happen (#32).
464
+ let processed = if partially_shown {
465
+ delivered + 1
466
+ } else {
467
+ delivered
468
+ };
469
+ let noun = if processed == 1 { "entry" } else { "entries" };
470
+ let json = serde_json::to_string(&pending).expect("continuation entries serialize");
471
+ format!(
472
+ "(Partial: {processed} {noun} in progress, {total} requested. Continue with files={json}.)"
473
+ )
474
+ }
475
+
476
+ fn budget_too_small(budget: TokenBudget) -> ToolResponse {
477
+ ToolResponse::error(format!(
478
+ "{}={} is too small to return the required continuation note. Increase it and retry.",
479
+ budget.variable, budget.value
480
+ ))
481
+ }
482
+
483
+ impl ContinuationEntry {
484
+ fn from_request(entry: &BatchReadEntry) -> Self {
485
+ Self {
486
+ path: entry.path.clone(),
487
+ offset: entry.offset,
488
+ limit: entry.limit,
489
+ encoding: entry.encoding.clone(),
490
+ }
491
+ }
492
+ }
493
+
494
+ fn continuation_path(input: &str) -> String {
495
+ display_path(&parse_input_path(input))
496
+ }
@@ -0,0 +1,141 @@
1
+ //! Sixteen-byte paged hexadecimal view for any regular file.
2
+
3
+ use super::DEFAULT_HEX_LINE_LIMIT;
4
+ use crate::budget::{TokenBudget, assemble_text, estimate_tokens};
5
+ use crate::model::ToolResponse;
6
+ use crate::paths::io_error_message;
7
+ use std::fmt::Write as _;
8
+ use std::fs::File;
9
+ use std::io::{Read, Seek, SeekFrom};
10
+ use std::path::Path;
11
+
12
+ const BYTES_PER_LINE: u64 = 16;
13
+ const HEX_COLUMN_WIDTH: usize = 48;
14
+
15
+ pub(super) fn read_hex_file(
16
+ path: &Path,
17
+ offset: Option<usize>,
18
+ limit: Option<usize>,
19
+ budget: TokenBudget,
20
+ ) -> ToolResponse {
21
+ let offset = offset.unwrap_or(1);
22
+ let limit = limit.unwrap_or(DEFAULT_HEX_LINE_LIMIT);
23
+ if offset == 0 {
24
+ return ToolResponse::error("Invalid offset value: 0. Expected an integer >= 1.");
25
+ }
26
+ if limit == 0 {
27
+ return ToolResponse::error("Invalid limit value: 0. Expected an integer >= 1.");
28
+ }
29
+
30
+ let mut file = match File::open(path) {
31
+ Ok(file) => file,
32
+ Err(error) => return ToolResponse::error(io_error_message(path, &error)),
33
+ };
34
+ let file_size = match file.metadata() {
35
+ Ok(metadata) => metadata.len(),
36
+ Err(error) => return ToolResponse::error(io_error_message(path, &error)),
37
+ };
38
+ if file_size == 0 {
39
+ return ToolResponse::text("Warning: the file exists but is empty.");
40
+ }
41
+ let total_lines = file_size / BYTES_PER_LINE + u64::from(file_size % BYTES_PER_LINE != 0);
42
+ let offset_line = offset as u64;
43
+ if offset_line > total_lines {
44
+ let noun = if total_lines == 1 { "line" } else { "lines" };
45
+ return ToolResponse::text(format!(
46
+ "Warning: the file has only {total_lines} {noun}, but offset={offset} was requested."
47
+ ));
48
+ }
49
+
50
+ let byte_offset = (offset_line - 1) * BYTES_PER_LINE;
51
+ if let Err(error) = file.seek(SeekFrom::Start(byte_offset)) {
52
+ return ToolResponse::error(io_error_message(path, &error));
53
+ }
54
+ let remaining = total_lines - offset_line + 1;
55
+ let budget_probe = budget.value.saturating_mul(4).saturating_add(1) as u64;
56
+ let candidate_lines = remaining.min(limit as u64).min(budget_probe.max(1));
57
+ let mut rendered = Vec::with_capacity(candidate_lines.min(usize::MAX as u64) as usize);
58
+ for line_index in 0..candidate_lines {
59
+ let mut bytes = [0_u8; BYTES_PER_LINE as usize];
60
+ let mut read = 0_usize;
61
+ while read < bytes.len() {
62
+ match file.read(&mut bytes[read..]) {
63
+ Ok(0) => break,
64
+ Ok(count) => read += count,
65
+ Err(error) => return ToolResponse::error(io_error_message(path, &error)),
66
+ }
67
+ }
68
+ if read == 0 {
69
+ break;
70
+ }
71
+ rendered.push(format_hex_line(
72
+ byte_offset + line_index * BYTES_PER_LINE,
73
+ &bytes[..read],
74
+ ));
75
+ }
76
+
77
+ loop {
78
+ if rendered.is_empty() {
79
+ return ToolResponse::error(format!(
80
+ "{}={} is too small to return the required continuation note. Increase it and retry.",
81
+ budget.variable, budget.value
82
+ ));
83
+ }
84
+ let shown = rendered.len() as u64;
85
+ let last = offset_line + shown - 1;
86
+ let terminal = if last < total_lines {
87
+ format!(
88
+ "(Partial: {} of {total_lines} shown. Continue with offset={}.)",
89
+ line_span(offset_line, last),
90
+ last + 1
91
+ )
92
+ } else {
93
+ format!(
94
+ "(Complete: reached end of file; {} of {total_lines} shown.)",
95
+ line_span(offset_line, last)
96
+ )
97
+ };
98
+ let output = assemble_text(&rendered, &[terminal]);
99
+ if estimate_tokens(&output) <= budget.value {
100
+ return ToolResponse::text(output);
101
+ }
102
+ rendered.pop();
103
+ }
104
+ }
105
+
106
+ fn format_hex_line(offset: u64, bytes: &[u8]) -> String {
107
+ let mut hex_column = String::with_capacity(HEX_COLUMN_WIDTH);
108
+ for index in 0..BYTES_PER_LINE as usize {
109
+ if index > 0 {
110
+ hex_column.push(' ');
111
+ }
112
+ if index == 8 {
113
+ hex_column.push(' ');
114
+ }
115
+ if let Some(byte) = bytes.get(index) {
116
+ let _ = write!(hex_column, "{byte:02x}");
117
+ } else {
118
+ hex_column.push_str(" ");
119
+ }
120
+ }
121
+ debug_assert_eq!(hex_column.len(), HEX_COLUMN_WIDTH);
122
+ let ascii = bytes
123
+ .iter()
124
+ .map(|byte| {
125
+ if (0x20..=0x7E).contains(byte) {
126
+ char::from(*byte)
127
+ } else {
128
+ '.'
129
+ }
130
+ })
131
+ .collect::<String>();
132
+ format!("{offset:08x} {hex_column} |{ascii}|")
133
+ }
134
+
135
+ fn line_span(first: u64, last: u64) -> String {
136
+ if first == last {
137
+ format!("line {first}")
138
+ } else {
139
+ format!("lines {first}-{last}")
140
+ }
141
+ }