dsh-ops 0.0.0-stage → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +202 -0
- package/LICENSE +30 -0
- package/NOTICE +106 -0
- package/PROVENANCE.md +435 -0
- package/README.en.md +126 -0
- package/README.md +115 -2
- package/README.zh.md +116 -0
- package/bin/dsh-ops.mjs +1216 -0
- package/cordis.patch.yml +160 -0
- package/docs/manual-validation.md +53 -0
- package/docs/release-0.2.1.md +72 -0
- package/docs/schema-baseline.json +64 -0
- package/docs/schema-current.json +84 -0
- package/docs/schema-measurement.md +17 -0
- package/dsh-plugin.json +88 -0
- package/icon.svg +12 -0
- package/lib/binary.js +409 -0
- package/lib/config.js +198 -0
- package/lib/handshake.js +252 -0
- package/lib/index.js +108 -0
- package/lib/jobs.js +42 -0
- package/lib/policy.js +64 -0
- package/lib/presentation.js +63 -0
- package/lib/profile-install.js +61 -0
- package/lib/rust.js +194 -0
- package/lib/session-shells.js +78 -0
- package/lib/shells.js +998 -0
- package/lib/tools.js +657 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +114 -4
- package/vendor/fastctx/Cargo.lock +3210 -0
- package/vendor/fastctx/Cargo.toml +94 -0
- package/vendor/fastctx/FORK.md +119 -0
- package/vendor/fastctx/LICENSE-APACHE +201 -0
- package/vendor/fastctx/NOTICE +40 -0
- package/vendor/fastctx/README.md +439 -0
- package/vendor/fastctx/THIRD_PARTY_LICENSES.md +17 -0
- package/vendor/fastctx/THIRD_PARTY_LICENSES_RUST.md +7914 -0
- package/vendor/fastctx/UPSTREAM.md +49 -0
- package/vendor/fastctx/build.rs +413 -0
- package/vendor/fastctx/src/background_status.rs +403 -0
- package/vendor/fastctx/src/binary.rs +75 -0
- package/vendor/fastctx/src/bounded_sort.rs +500 -0
- package/vendor/fastctx/src/budget.rs +781 -0
- package/vendor/fastctx/src/cli/mod.rs +110 -0
- package/vendor/fastctx/src/context_guard.rs +289 -0
- package/vendor/fastctx/src/control/mod.rs +6 -0
- package/vendor/fastctx/src/control/paths.rs +49 -0
- package/vendor/fastctx/src/control/settings.rs +753 -0
- package/vendor/fastctx/src/control/transaction.rs +531 -0
- package/vendor/fastctx/src/edit/document.rs +535 -0
- package/vendor/fastctx/src/edit/locks.rs +371 -0
- package/vendor/fastctx/src/edit/mod.rs +213 -0
- package/vendor/fastctx/src/edit/private_storage/unix.rs +315 -0
- package/vendor/fastctx/src/edit/private_storage/windows.rs +793 -0
- package/vendor/fastctx/src/edit/private_storage.rs +234 -0
- package/vendor/fastctx/src/edit/replace.rs +1030 -0
- package/vendor/fastctx/src/edit_server.rs +53 -0
- package/vendor/fastctx/src/encoding/reference_v011.rs +587 -0
- package/vendor/fastctx/src/encoding/snapshot_pipeline.rs +1678 -0
- package/vendor/fastctx/src/encoding.rs +1118 -0
- package/vendor/fastctx/src/file_executor.rs +1151 -0
- package/vendor/fastctx/src/file_snapshot.rs +1491 -0
- package/vendor/fastctx/src/glob_filter.rs +98 -0
- package/vendor/fastctx/src/glob_tool.rs +653 -0
- package/vendor/fastctx/src/grep_sink.rs +1162 -0
- package/vendor/fastctx/src/grep_tool.rs +2449 -0
- package/vendor/fastctx/src/lib.rs +45 -0
- package/vendor/fastctx/src/main.rs +15 -0
- package/vendor/fastctx/src/model.rs +51 -0
- package/vendor/fastctx/src/model_guidance.rs +62 -0
- package/vendor/fastctx/src/operation.rs +356 -0
- package/vendor/fastctx/src/ordered_window.rs +1235 -0
- package/vendor/fastctx/src/os_environment.rs +414 -0
- package/vendor/fastctx/src/path_codec.rs +850 -0
- package/vendor/fastctx/src/paths.rs +244 -0
- package/vendor/fastctx/src/process_identity.rs +763 -0
- package/vendor/fastctx/src/process_policy.rs +74 -0
- package/vendor/fastctx/src/read_tool/batch.rs +496 -0
- package/vendor/fastctx/src/read_tool/hex_file.rs +141 -0
- package/vendor/fastctx/src/read_tool/image_file.rs +88 -0
- package/vendor/fastctx/src/read_tool/mod.rs +245 -0
- package/vendor/fastctx/src/read_tool/pdf.rs +470 -0
- package/vendor/fastctx/src/read_tool/pdf_disabled.rs +47 -0
- package/vendor/fastctx/src/read_tool/pdf_engine.rs +664 -0
- package/vendor/fastctx/src/read_tool/text_file.rs +351 -0
- package/vendor/fastctx/src/render_plan.rs +468 -0
- package/vendor/fastctx/src/runtime/activity.rs +159 -0
- package/vendor/fastctx/src/runtime/hosts.rs +99 -0
- package/vendor/fastctx/src/runtime/journal.rs +556 -0
- package/vendor/fastctx/src/runtime/local_ipc.rs +186 -0
- package/vendor/fastctx/src/runtime/mod.rs +746 -0
- package/vendor/fastctx/src/runtime/protocol.rs +296 -0
- package/vendor/fastctx/src/runtime/session.rs +536 -0
- package/vendor/fastctx/src/runtime/windows_process.rs +66 -0
- package/vendor/fastctx/src/search_parallelism.rs +106 -0
- package/vendor/fastctx/src/search_text.rs +227 -0
- package/vendor/fastctx/src/server.rs +359 -0
- package/vendor/fastctx/src/server_manifest.rs +468 -0
- package/vendor/fastctx/src/server_support.rs +826 -0
- package/vendor/fastctx/src/session.rs +629 -0
- package/vendor/fastctx/src/shell/apply_patch_hint.rs +41 -0
- package/vendor/fastctx/src/shell/bash.rs +263 -0
- package/vendor/fastctx/src/shell/buffer.rs +108 -0
- package/vendor/fastctx/src/shell/encoding.rs +403 -0
- package/vendor/fastctx/src/shell/foreground.rs +115 -0
- package/vendor/fastctx/src/shell/jobs/admission.rs +91 -0
- package/vendor/fastctx/src/shell/jobs/background.rs +146 -0
- package/vendor/fastctx/src/shell/jobs/host.rs +830 -0
- package/vendor/fastctx/src/shell/jobs/identity.rs +29 -0
- package/vendor/fastctx/src/shell/jobs/mod.rs +1513 -0
- package/vendor/fastctx/src/shell/jobs/model.rs +244 -0
- package/vendor/fastctx/src/shell/jobs/output_log.rs +1148 -0
- package/vendor/fastctx/src/shell/jobs/store.rs +1300 -0
- package/vendor/fastctx/src/shell/mod.rs +345 -0
- package/vendor/fastctx/src/shell/normalize.rs +389 -0
- package/vendor/fastctx/src/shell/output.rs +406 -0
- package/vendor/fastctx/src/shell/process.rs +493 -0
- package/vendor/fastctx/src/shell_server.rs +156 -0
- package/vendor/fastctx/src/skip_report.rs +83 -0
- package/vendor/fastctx/src/stdio_transport.rs +177 -0
- package/vendor/fastctx/src/tool_schema.rs +204 -0
- package/vendor/fastctx/src/traversal.rs +846 -0
- package/vendor/fastctx/third-party/pdfium-7763/LICENSE +9 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/abseil.txt +202 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/agg23.txt +14 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/fast_float.txt +27 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/freetype.txt +169 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/icu.txt +542 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/lcms.txt +27 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.ijg +260 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libjpeg_turbo.md +135 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libopenjpeg.txt +32 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libpng.txt +134 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/libtiff.txt +21 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/llvm-libc.txt +278 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/pdfium.txt +230 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/simdutf.txt +18 -0
- package/vendor/fastctx/third-party/pdfium-7763/licenses/zlib.txt +29 -0
|
@@ -0,0 +1,1148 @@
|
|
|
1
|
+
// Modified by dsh-ops (https://github.com/T-Auto/dsh-ops): removed read_range/record_end/log_len, the incremental readers that only the deleted TUI job-tail viewer called.
|
|
2
|
+
//! Append-only plain-text background logs and their derived line-offset index.
|
|
3
|
+
|
|
4
|
+
use super::model::{OUTPUT_INDEX_FILE, OUTPUT_LOG_FILE, StoredLine};
|
|
5
|
+
use crate::paths::display_path;
|
|
6
|
+
use crate::shell::normalize::{NormalizedEvent, StreamEncoding};
|
|
7
|
+
use std::fs::{File, OpenOptions};
|
|
8
|
+
use std::io::{BufWriter, Read, Seek, SeekFrom, Write};
|
|
9
|
+
use std::path::{Path, PathBuf};
|
|
10
|
+
use std::time::{Duration, Instant};
|
|
11
|
+
|
|
12
|
+
pub(super) const INDEX_HEADER: &[u8; 8] = b"FCTXIDX1";
|
|
13
|
+
pub(super) const INDEX_ENTRY_BYTES: u64 = 24;
|
|
14
|
+
const FLUSH_BYTES: u64 = 64 * 1024;
|
|
15
|
+
const FLUSH_IDLE: Duration = Duration::from_millis(250);
|
|
16
|
+
const MAX_STORED_LINE_BYTES: u64 = 64 * 1024;
|
|
17
|
+
const MAX_RECOVERY_SCAN_BYTES: u64 = 256 * 1024;
|
|
18
|
+
|
|
19
|
+
/// One fixed-width derived index entry. The log remains the source of truth;
|
|
20
|
+
/// this sidecar only makes line-number lookup independent of log length.
|
|
21
|
+
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
|
22
|
+
pub(super) struct LineIndexEntry {
|
|
23
|
+
pub(super) start: u64,
|
|
24
|
+
pub(super) content_end: u64,
|
|
25
|
+
pub(super) record_end: u64,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
impl LineIndexEntry {
|
|
29
|
+
fn encode(self, output: &mut Vec<u8>) {
|
|
30
|
+
output.extend_from_slice(&self.start.to_le_bytes());
|
|
31
|
+
output.extend_from_slice(&self.content_end.to_le_bytes());
|
|
32
|
+
output.extend_from_slice(&self.record_end.to_le_bytes());
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
pub(super) fn decode(bytes: &[u8]) -> Option<Self> {
|
|
36
|
+
let bytes: &[u8; 24] = bytes.try_into().ok()?;
|
|
37
|
+
let start = u64::from_le_bytes(bytes[0..8].try_into().ok()?);
|
|
38
|
+
let content_end = u64::from_le_bytes(bytes[8..16].try_into().ok()?);
|
|
39
|
+
let record_end = u64::from_le_bytes(bytes[16..24].try_into().ok()?);
|
|
40
|
+
(start <= content_end && content_end <= record_end).then_some(Self {
|
|
41
|
+
start,
|
|
42
|
+
content_end,
|
|
43
|
+
record_end,
|
|
44
|
+
})
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/// The detached supervisor's sole writer for a schema-v3 job. Both files are
|
|
49
|
+
/// append-only; the index is flushed only after the corresponding log bytes.
|
|
50
|
+
#[derive(Debug)]
|
|
51
|
+
pub(crate) struct OutputLogWriter {
|
|
52
|
+
log_path: PathBuf,
|
|
53
|
+
index_path: PathBuf,
|
|
54
|
+
log: BufWriter<File>,
|
|
55
|
+
index: File,
|
|
56
|
+
pending_index: Vec<u8>,
|
|
57
|
+
position: u64,
|
|
58
|
+
current_start: u64,
|
|
59
|
+
current_has_content: bool,
|
|
60
|
+
total_lines: u64,
|
|
61
|
+
durable_lines: u64,
|
|
62
|
+
committed_position: u64,
|
|
63
|
+
indexed_position: u64,
|
|
64
|
+
stream_encoding: Option<StreamEncoding>,
|
|
65
|
+
started: bool,
|
|
66
|
+
bytes_since_flush: u64,
|
|
67
|
+
last_flush: Instant,
|
|
68
|
+
byte_limit: u64,
|
|
69
|
+
quota_exceeded: bool,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
impl OutputLogWriter {
|
|
73
|
+
#[cfg(test)]
|
|
74
|
+
pub(crate) fn new(directory: &Path) -> Result<Self, String> {
|
|
75
|
+
Self::with_limit(directory, u64::MAX)
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
pub(crate) fn with_limit(directory: &Path, byte_limit: u64) -> Result<Self, String> {
|
|
79
|
+
if byte_limit < INDEX_HEADER.len() as u64 + INDEX_ENTRY_BYTES {
|
|
80
|
+
return Err(format!(
|
|
81
|
+
"background output limit {byte_limit} bytes is too small for the log index"
|
|
82
|
+
));
|
|
83
|
+
}
|
|
84
|
+
let log_path = directory.join(OUTPUT_LOG_FILE);
|
|
85
|
+
let index_path = directory.join(OUTPUT_INDEX_FILE);
|
|
86
|
+
let log = create_private_file(&log_path, "output log")?;
|
|
87
|
+
let mut index = create_private_file(&index_path, "output index")?;
|
|
88
|
+
if let Err(error) = index.write_all(INDEX_HEADER).and_then(|()| index.flush()) {
|
|
89
|
+
let _ = std::fs::remove_file(&log_path);
|
|
90
|
+
let _ = std::fs::remove_file(&index_path);
|
|
91
|
+
return Err(format!(
|
|
92
|
+
"cannot initialize background output index {}: {error}",
|
|
93
|
+
display_path(&index_path)
|
|
94
|
+
));
|
|
95
|
+
}
|
|
96
|
+
Ok(Self {
|
|
97
|
+
log_path,
|
|
98
|
+
index_path,
|
|
99
|
+
log: BufWriter::with_capacity(FLUSH_BYTES as usize, log),
|
|
100
|
+
index,
|
|
101
|
+
pending_index: Vec::with_capacity(FLUSH_BYTES as usize),
|
|
102
|
+
position: 0,
|
|
103
|
+
current_start: 0,
|
|
104
|
+
current_has_content: false,
|
|
105
|
+
total_lines: 0,
|
|
106
|
+
durable_lines: 0,
|
|
107
|
+
committed_position: 0,
|
|
108
|
+
indexed_position: 0,
|
|
109
|
+
stream_encoding: None,
|
|
110
|
+
started: false,
|
|
111
|
+
bytes_since_flush: 0,
|
|
112
|
+
last_flush: Instant::now(),
|
|
113
|
+
byte_limit,
|
|
114
|
+
quota_exceeded: false,
|
|
115
|
+
})
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/// Appends one normalized stream event and returns the committed line number when a line ends
|
|
119
|
+
/// or the quota seals a final readable prefix.
|
|
120
|
+
pub(crate) fn append(&mut self, event: NormalizedEvent) -> Result<Option<u64>, String> {
|
|
121
|
+
let quota_was_exceeded = self.quota_exceeded;
|
|
122
|
+
let committed = match event {
|
|
123
|
+
NormalizedEvent::Start(encoding) => {
|
|
124
|
+
if self.started {
|
|
125
|
+
return Err(
|
|
126
|
+
"cannot restart an initialized background output stream".to_string()
|
|
127
|
+
);
|
|
128
|
+
}
|
|
129
|
+
self.started = true;
|
|
130
|
+
self.stream_encoding = encoding;
|
|
131
|
+
if let Some(encoding) = encoding
|
|
132
|
+
&& !self.write_fixed(stream_bom(encoding))?
|
|
133
|
+
{
|
|
134
|
+
self.quota_exceeded = true;
|
|
135
|
+
}
|
|
136
|
+
self.current_start = self.position;
|
|
137
|
+
None
|
|
138
|
+
}
|
|
139
|
+
NormalizedEvent::Bytes(bytes) => {
|
|
140
|
+
self.require_started()?;
|
|
141
|
+
if !bytes.is_empty() && !self.quota_exceeded {
|
|
142
|
+
let written = self.write_content(&bytes)?;
|
|
143
|
+
if written < bytes.len() {
|
|
144
|
+
self.quota_exceeded = true;
|
|
145
|
+
}
|
|
146
|
+
if written > 0 {
|
|
147
|
+
self.current_has_content = true;
|
|
148
|
+
}
|
|
149
|
+
if self.quota_exceeded && self.current_has_content {
|
|
150
|
+
Some(self.commit_line(false)?)
|
|
151
|
+
} else {
|
|
152
|
+
None
|
|
153
|
+
}
|
|
154
|
+
} else if !bytes.is_empty() {
|
|
155
|
+
self.quota_exceeded = true;
|
|
156
|
+
None
|
|
157
|
+
} else {
|
|
158
|
+
None
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
NormalizedEvent::LineEnd { terminated } => {
|
|
162
|
+
self.require_started()?;
|
|
163
|
+
if self.quota_exceeded {
|
|
164
|
+
if self.current_has_content {
|
|
165
|
+
Some(self.commit_line(false)?)
|
|
166
|
+
} else {
|
|
167
|
+
None
|
|
168
|
+
}
|
|
169
|
+
} else {
|
|
170
|
+
match self.try_commit_line(terminated)? {
|
|
171
|
+
Some(line) => Some(line),
|
|
172
|
+
None => {
|
|
173
|
+
self.quota_exceeded = true;
|
|
174
|
+
self.current_has_content
|
|
175
|
+
.then(|| self.commit_line(false))
|
|
176
|
+
.transpose()?
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
};
|
|
182
|
+
if (!quota_was_exceeded && self.quota_exceeded)
|
|
183
|
+
|| self.bytes_since_flush >= FLUSH_BYTES
|
|
184
|
+
|| self
|
|
185
|
+
.committed_position
|
|
186
|
+
.saturating_sub(self.indexed_position)
|
|
187
|
+
>= MAX_RECOVERY_SCAN_BYTES
|
|
188
|
+
{
|
|
189
|
+
self.flush()?;
|
|
190
|
+
}
|
|
191
|
+
Ok(committed)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
pub(crate) fn flush_if_idle(&mut self) -> Result<(), String> {
|
|
195
|
+
if self.has_buffered_data() && self.last_flush.elapsed() >= FLUSH_IDLE {
|
|
196
|
+
self.flush()?;
|
|
197
|
+
}
|
|
198
|
+
Ok(())
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/// Seals bytes already received as an unterminated final line, then flushes.
|
|
202
|
+
/// This is also used when capture fails after some bytes have reached disk.
|
|
203
|
+
pub(crate) fn finish(&mut self) -> Result<(), String> {
|
|
204
|
+
if self.current_has_content {
|
|
205
|
+
self.commit_line(false)?;
|
|
206
|
+
}
|
|
207
|
+
self.flush()
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
pub(crate) const fn total_lines(&self) -> u64 {
|
|
211
|
+
self.total_lines
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
pub(crate) const fn quota_exceeded(&self) -> bool {
|
|
215
|
+
self.quota_exceeded
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
pub(crate) const fn persisted_log_bytes(&self) -> u64 {
|
|
219
|
+
self.position
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
pub(crate) const fn persisted_index_bytes(&self) -> u64 {
|
|
223
|
+
INDEX_HEADER.len() as u64 + self.total_lines.saturating_mul(INDEX_ENTRY_BYTES)
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
pub(crate) const fn byte_limit(&self) -> u64 {
|
|
227
|
+
self.byte_limit
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
pub(crate) fn preserved_lines(&self) -> u64 {
|
|
231
|
+
let log_len = self
|
|
232
|
+
.log
|
|
233
|
+
.get_ref()
|
|
234
|
+
.metadata()
|
|
235
|
+
.map(|metadata| metadata.len())
|
|
236
|
+
.unwrap_or(self.indexed_position);
|
|
237
|
+
self.durable_lines.saturating_add(
|
|
238
|
+
self.pending_index
|
|
239
|
+
.as_chunks::<{ INDEX_ENTRY_BYTES as usize }>()
|
|
240
|
+
.0
|
|
241
|
+
.iter()
|
|
242
|
+
.filter_map(|bytes| LineIndexEntry::decode(bytes))
|
|
243
|
+
.take_while(|entry| entry.record_end <= log_len)
|
|
244
|
+
.count() as u64,
|
|
245
|
+
)
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
fn require_started(&self) -> Result<(), String> {
|
|
249
|
+
if self.started {
|
|
250
|
+
Ok(())
|
|
251
|
+
} else {
|
|
252
|
+
Err("background output arrived before its stream encoding was established".to_string())
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
fn try_commit_line(&mut self, terminated: bool) -> Result<Option<u64>, String> {
|
|
257
|
+
let required = INDEX_ENTRY_BYTES.saturating_add(if terminated {
|
|
258
|
+
line_ending(self.stream_encoding).len() as u64
|
|
259
|
+
} else {
|
|
260
|
+
0
|
|
261
|
+
});
|
|
262
|
+
if self.combined_bytes().saturating_add(required) > self.byte_limit {
|
|
263
|
+
return Ok(None);
|
|
264
|
+
}
|
|
265
|
+
self.commit_line(terminated).map(Some)
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
fn commit_line(&mut self, terminated: bool) -> Result<u64, String> {
|
|
269
|
+
let content_end = self.position;
|
|
270
|
+
if terminated {
|
|
271
|
+
self.write_fixed(line_ending(self.stream_encoding))?;
|
|
272
|
+
}
|
|
273
|
+
LineIndexEntry {
|
|
274
|
+
start: self.current_start,
|
|
275
|
+
content_end,
|
|
276
|
+
record_end: self.position,
|
|
277
|
+
}
|
|
278
|
+
.encode(&mut self.pending_index);
|
|
279
|
+
self.total_lines = self.total_lines.saturating_add(1);
|
|
280
|
+
self.committed_position = self.position;
|
|
281
|
+
self.current_start = self.position;
|
|
282
|
+
self.current_has_content = false;
|
|
283
|
+
Ok(self.total_lines)
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
fn write_content(&mut self, bytes: &[u8]) -> Result<usize, String> {
|
|
287
|
+
let reserve =
|
|
288
|
+
INDEX_ENTRY_BYTES.saturating_add(line_ending(self.stream_encoding).len() as u64);
|
|
289
|
+
let available = self
|
|
290
|
+
.byte_limit
|
|
291
|
+
.saturating_sub(self.combined_bytes())
|
|
292
|
+
.saturating_sub(reserve);
|
|
293
|
+
let width = stream_width(self.stream_encoding);
|
|
294
|
+
let accepted = usize::try_from(available.min(bytes.len() as u64)).unwrap_or(bytes.len());
|
|
295
|
+
let accepted = accepted - accepted % width;
|
|
296
|
+
self.write_log_unchecked(&bytes[..accepted])?;
|
|
297
|
+
Ok(accepted)
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
fn write_fixed(&mut self, bytes: &[u8]) -> Result<bool, String> {
|
|
301
|
+
if self.combined_bytes().saturating_add(bytes.len() as u64) > self.byte_limit {
|
|
302
|
+
return Ok(false);
|
|
303
|
+
}
|
|
304
|
+
self.write_log_unchecked(bytes)?;
|
|
305
|
+
Ok(true)
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
fn write_log_unchecked(&mut self, bytes: &[u8]) -> Result<(), String> {
|
|
309
|
+
self.log.write_all(bytes).map_err(|error| {
|
|
310
|
+
format!(
|
|
311
|
+
"cannot append background output log {}: {error}",
|
|
312
|
+
display_path(&self.log_path)
|
|
313
|
+
)
|
|
314
|
+
})?;
|
|
315
|
+
self.position = self.position.saturating_add(bytes.len() as u64);
|
|
316
|
+
self.bytes_since_flush = self.bytes_since_flush.saturating_add(bytes.len() as u64);
|
|
317
|
+
Ok(())
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
fn combined_bytes(&self) -> u64 {
|
|
321
|
+
self.position
|
|
322
|
+
.saturating_add(INDEX_HEADER.len() as u64)
|
|
323
|
+
.saturating_add(self.total_lines.saturating_mul(INDEX_ENTRY_BYTES))
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
fn has_buffered_data(&self) -> bool {
|
|
327
|
+
!self.log.buffer().is_empty() || !self.pending_index.is_empty()
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
fn flush(&mut self) -> Result<(), String> {
|
|
331
|
+
self.log.flush().map_err(|error| {
|
|
332
|
+
format!(
|
|
333
|
+
"cannot flush background output log {}: {error}",
|
|
334
|
+
display_path(&self.log_path)
|
|
335
|
+
)
|
|
336
|
+
})?;
|
|
337
|
+
if !self.pending_index.is_empty() {
|
|
338
|
+
self.index.write_all(&self.pending_index).map_err(|error| {
|
|
339
|
+
format!(
|
|
340
|
+
"cannot append background output index {}: {error}",
|
|
341
|
+
display_path(&self.index_path)
|
|
342
|
+
)
|
|
343
|
+
})?;
|
|
344
|
+
self.index.flush().map_err(|error| {
|
|
345
|
+
format!(
|
|
346
|
+
"cannot flush background output index {}: {error}",
|
|
347
|
+
display_path(&self.index_path)
|
|
348
|
+
)
|
|
349
|
+
})?;
|
|
350
|
+
self.pending_index.clear();
|
|
351
|
+
self.indexed_position = self.committed_position;
|
|
352
|
+
self.durable_lines = self.total_lines;
|
|
353
|
+
}
|
|
354
|
+
self.bytes_since_flush = 0;
|
|
355
|
+
self.last_flush = Instant::now();
|
|
356
|
+
Ok(())
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
/// A point-in-time reader over a schema-v3 log. The fixed-width sidecar makes
|
|
361
|
+
/// line lookup O(requested lines), independent of the full log length.
|
|
362
|
+
#[derive(Debug)]
|
|
363
|
+
pub(super) struct OutputLogReader {
|
|
364
|
+
log_path: PathBuf,
|
|
365
|
+
log: File,
|
|
366
|
+
index: File,
|
|
367
|
+
log_len: u64,
|
|
368
|
+
bom_len: u64,
|
|
369
|
+
indexed_lines: u64,
|
|
370
|
+
tail_entries: Vec<LineIndexEntry>,
|
|
371
|
+
stream_encoding: Option<StreamEncoding>,
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
#[derive(Debug)]
|
|
375
|
+
pub(super) struct BoundedLines {
|
|
376
|
+
pub(super) lines: Vec<StoredLine>,
|
|
377
|
+
pub(super) complete: bool,
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
impl OutputLogReader {
|
|
381
|
+
/// Opens one stable snapshot. Unindexed bytes are ignored for a live stream,
|
|
382
|
+
/// but become recoverable final lines once the job/capture is terminal.
|
|
383
|
+
pub(super) fn open(directory: &Path, include_unindexed_tail: bool) -> Result<Self, String> {
|
|
384
|
+
let log_path = directory.join(OUTPUT_LOG_FILE);
|
|
385
|
+
let index_path = directory.join(OUTPUT_INDEX_FILE);
|
|
386
|
+
let mut log = File::open(&log_path).map_err(|error| {
|
|
387
|
+
format!(
|
|
388
|
+
"Cannot read background output log {}: {error}",
|
|
389
|
+
display_path(&log_path)
|
|
390
|
+
)
|
|
391
|
+
})?;
|
|
392
|
+
let mut index = File::open(&index_path).map_err(|error| {
|
|
393
|
+
format!(
|
|
394
|
+
"Cannot read background output index {}: {error}. The full log remains readable at {}.",
|
|
395
|
+
display_path(&index_path),
|
|
396
|
+
display_path(&log_path)
|
|
397
|
+
)
|
|
398
|
+
})?;
|
|
399
|
+
// Sample the index length BEFORE the log length (2026-07-23). The live
|
|
400
|
+
// writer flushes log bytes before their index entries, so this order
|
|
401
|
+
// guarantees record_end <= log_len for every sampled entry. Sampling
|
|
402
|
+
// the log first races with an in-between writer flush and misreports a
|
|
403
|
+
// healthy index as damaged.
|
|
404
|
+
let index_len = index
|
|
405
|
+
.metadata()
|
|
406
|
+
.map_err(|error| {
|
|
407
|
+
format!(
|
|
408
|
+
"Cannot inspect background output index {}: {error}",
|
|
409
|
+
display_path(&index_path)
|
|
410
|
+
)
|
|
411
|
+
})?
|
|
412
|
+
.len();
|
|
413
|
+
let log_len = log
|
|
414
|
+
.metadata()
|
|
415
|
+
.map_err(|error| {
|
|
416
|
+
format!(
|
|
417
|
+
"Cannot inspect background output log {}: {error}",
|
|
418
|
+
display_path(&log_path)
|
|
419
|
+
)
|
|
420
|
+
})?
|
|
421
|
+
.len();
|
|
422
|
+
let (stream_encoding, bom_len) = detect_log_encoding(&mut log, log_len)?;
|
|
423
|
+
if index_len < INDEX_HEADER.len() as u64 {
|
|
424
|
+
return Err(damaged_index(&index_path, &log_path));
|
|
425
|
+
}
|
|
426
|
+
let mut header = [0_u8; INDEX_HEADER.len()];
|
|
427
|
+
index
|
|
428
|
+
.read_exact(&mut header)
|
|
429
|
+
.map_err(|_| damaged_index(&index_path, &log_path))?;
|
|
430
|
+
if &header != INDEX_HEADER {
|
|
431
|
+
return Err(damaged_index(&index_path, &log_path));
|
|
432
|
+
}
|
|
433
|
+
let indexed_lines = index_len.saturating_sub(INDEX_HEADER.len() as u64) / INDEX_ENTRY_BYTES;
|
|
434
|
+
let indexed_end = if indexed_lines == 0 {
|
|
435
|
+
bom_len
|
|
436
|
+
} else {
|
|
437
|
+
read_validated_index_entry(
|
|
438
|
+
&mut index,
|
|
439
|
+
indexed_lines,
|
|
440
|
+
indexed_lines,
|
|
441
|
+
bom_len,
|
|
442
|
+
log_len,
|
|
443
|
+
&index_path,
|
|
444
|
+
&log_path,
|
|
445
|
+
)?
|
|
446
|
+
.record_end
|
|
447
|
+
};
|
|
448
|
+
if indexed_end < bom_len || indexed_end > log_len {
|
|
449
|
+
return Err(damaged_index(&index_path, &log_path));
|
|
450
|
+
}
|
|
451
|
+
let tail_entries = if include_unindexed_tail && indexed_end < log_len {
|
|
452
|
+
recover_tail_entries(&mut log, indexed_end, log_len, stream_encoding, &log_path)?
|
|
453
|
+
} else {
|
|
454
|
+
Vec::new()
|
|
455
|
+
};
|
|
456
|
+
Ok(Self {
|
|
457
|
+
log_path,
|
|
458
|
+
log,
|
|
459
|
+
index,
|
|
460
|
+
log_len,
|
|
461
|
+
bom_len,
|
|
462
|
+
indexed_lines,
|
|
463
|
+
tail_entries,
|
|
464
|
+
stream_encoding,
|
|
465
|
+
})
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
pub(super) fn path(&self) -> &Path {
|
|
469
|
+
&self.log_path
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
pub(super) fn total_lines(&self) -> u64 {
|
|
473
|
+
self.indexed_lines
|
|
474
|
+
.saturating_add(self.tail_entries.len() as u64)
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
/// Reads an inclusive line range from the index.
|
|
478
|
+
///
|
|
479
|
+
/// The serve path reads job output through `read_prefix_bounded`/`read_suffix_bounded`, so this
|
|
480
|
+
/// reference path exists only for the reader tests below, which pin index decoding to it.
|
|
481
|
+
#[cfg(test)]
|
|
482
|
+
pub(super) fn read_range(&mut self, first: u64, last: u64) -> Result<Vec<StoredLine>, String> {
|
|
483
|
+
if first == 0 || first > last {
|
|
484
|
+
return Ok(Vec::new());
|
|
485
|
+
}
|
|
486
|
+
let last = last.min(self.total_lines());
|
|
487
|
+
let mut lines = Vec::with_capacity(
|
|
488
|
+
usize::try_from(last.saturating_sub(first).saturating_add(1)).unwrap_or(0),
|
|
489
|
+
);
|
|
490
|
+
for seq in first..=last {
|
|
491
|
+
let entry = self.entry(seq)?;
|
|
492
|
+
lines.push(read_stored_line(
|
|
493
|
+
&mut self.log,
|
|
494
|
+
seq,
|
|
495
|
+
entry,
|
|
496
|
+
self.stream_encoding,
|
|
497
|
+
&self.log_path,
|
|
498
|
+
)?);
|
|
499
|
+
}
|
|
500
|
+
Ok(lines)
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
pub(super) fn read_prefix_bounded(
|
|
504
|
+
&mut self,
|
|
505
|
+
first: u64,
|
|
506
|
+
last: u64,
|
|
507
|
+
max_lines: usize,
|
|
508
|
+
max_bytes: usize,
|
|
509
|
+
) -> Result<BoundedLines, String> {
|
|
510
|
+
if first == 0 || first > last || max_lines == 0 {
|
|
511
|
+
return Ok(BoundedLines {
|
|
512
|
+
lines: Vec::new(),
|
|
513
|
+
complete: first > last,
|
|
514
|
+
});
|
|
515
|
+
}
|
|
516
|
+
let last = last.min(self.total_lines());
|
|
517
|
+
let mut lines = Vec::new();
|
|
518
|
+
let mut stored_bytes = 0_usize;
|
|
519
|
+
let mut next = first;
|
|
520
|
+
while next <= last && lines.len() < max_lines {
|
|
521
|
+
let line = self.read_one(next)?;
|
|
522
|
+
let would_exceed =
|
|
523
|
+
!lines.is_empty() && stored_bytes.saturating_add(line.bytes.len()) > max_bytes;
|
|
524
|
+
if would_exceed {
|
|
525
|
+
break;
|
|
526
|
+
}
|
|
527
|
+
stored_bytes = stored_bytes.saturating_add(line.bytes.len());
|
|
528
|
+
lines.push(line);
|
|
529
|
+
next = next.saturating_add(1);
|
|
530
|
+
}
|
|
531
|
+
Ok(BoundedLines {
|
|
532
|
+
lines,
|
|
533
|
+
complete: next > last,
|
|
534
|
+
})
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
pub(super) fn read_suffix_bounded(
|
|
538
|
+
&mut self,
|
|
539
|
+
first: u64,
|
|
540
|
+
last: u64,
|
|
541
|
+
max_lines: usize,
|
|
542
|
+
max_bytes: usize,
|
|
543
|
+
) -> Result<BoundedLines, String> {
|
|
544
|
+
if first == 0 || first > last || max_lines == 0 {
|
|
545
|
+
return Ok(BoundedLines {
|
|
546
|
+
lines: Vec::new(),
|
|
547
|
+
complete: first > last,
|
|
548
|
+
});
|
|
549
|
+
}
|
|
550
|
+
let last = last.min(self.total_lines());
|
|
551
|
+
let mut lines = Vec::new();
|
|
552
|
+
let mut stored_bytes = 0_usize;
|
|
553
|
+
let mut next = last;
|
|
554
|
+
loop {
|
|
555
|
+
let line = self.read_one(next)?;
|
|
556
|
+
let would_exceed =
|
|
557
|
+
!lines.is_empty() && stored_bytes.saturating_add(line.bytes.len()) > max_bytes;
|
|
558
|
+
if would_exceed {
|
|
559
|
+
break;
|
|
560
|
+
}
|
|
561
|
+
stored_bytes = stored_bytes.saturating_add(line.bytes.len());
|
|
562
|
+
lines.push(line);
|
|
563
|
+
if next == first || lines.len() >= max_lines {
|
|
564
|
+
break;
|
|
565
|
+
}
|
|
566
|
+
next -= 1;
|
|
567
|
+
}
|
|
568
|
+
let complete = lines.last().is_some_and(|line| line.seq == first);
|
|
569
|
+
lines.reverse();
|
|
570
|
+
Ok(BoundedLines { lines, complete })
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
fn entry(&mut self, seq: u64) -> Result<LineIndexEntry, String> {
|
|
574
|
+
if seq <= self.indexed_lines {
|
|
575
|
+
let index_path = self.log_path.with_file_name(OUTPUT_INDEX_FILE);
|
|
576
|
+
return read_validated_index_entry(
|
|
577
|
+
&mut self.index,
|
|
578
|
+
seq,
|
|
579
|
+
self.indexed_lines,
|
|
580
|
+
self.bom_len,
|
|
581
|
+
self.log_len,
|
|
582
|
+
&index_path,
|
|
583
|
+
&self.log_path,
|
|
584
|
+
);
|
|
585
|
+
}
|
|
586
|
+
let tail_index = usize::try_from(seq.saturating_sub(self.indexed_lines + 1))
|
|
587
|
+
.map_err(|_| "background output line number is too large".to_string())?;
|
|
588
|
+
self.tail_entries.get(tail_index).copied().ok_or_else(|| {
|
|
589
|
+
format!(
|
|
590
|
+
"Cannot read line {seq} from background output log {}: only {} lines are available.",
|
|
591
|
+
display_path(&self.log_path),
|
|
592
|
+
self.total_lines()
|
|
593
|
+
)
|
|
594
|
+
})
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
fn read_one(&mut self, seq: u64) -> Result<StoredLine, String> {
|
|
598
|
+
let entry = self.entry(seq)?;
|
|
599
|
+
read_stored_line(
|
|
600
|
+
&mut self.log,
|
|
601
|
+
seq,
|
|
602
|
+
entry,
|
|
603
|
+
self.stream_encoding,
|
|
604
|
+
&self.log_path,
|
|
605
|
+
)
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
fn read_index_entry(
|
|
610
|
+
index: &mut File,
|
|
611
|
+
seq: u64,
|
|
612
|
+
index_path: &Path,
|
|
613
|
+
log_path: &Path,
|
|
614
|
+
) -> Result<LineIndexEntry, String> {
|
|
615
|
+
let offset = (INDEX_HEADER.len() as u64)
|
|
616
|
+
.checked_add(seq.saturating_sub(1).saturating_mul(INDEX_ENTRY_BYTES))
|
|
617
|
+
.ok_or_else(|| damaged_index(index_path, log_path))?;
|
|
618
|
+
index
|
|
619
|
+
.seek(SeekFrom::Start(offset))
|
|
620
|
+
.and_then(|_| {
|
|
621
|
+
let mut bytes = [0_u8; INDEX_ENTRY_BYTES as usize];
|
|
622
|
+
index.read_exact(&mut bytes)?;
|
|
623
|
+
Ok(bytes)
|
|
624
|
+
})
|
|
625
|
+
.map_err(|_| damaged_index(index_path, log_path))
|
|
626
|
+
.and_then(|bytes| {
|
|
627
|
+
LineIndexEntry::decode(&bytes).ok_or_else(|| damaged_index(index_path, log_path))
|
|
628
|
+
})
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
fn read_validated_index_entry(
|
|
632
|
+
index: &mut File,
|
|
633
|
+
seq: u64,
|
|
634
|
+
indexed_lines: u64,
|
|
635
|
+
bom_len: u64,
|
|
636
|
+
log_len: u64,
|
|
637
|
+
index_path: &Path,
|
|
638
|
+
log_path: &Path,
|
|
639
|
+
) -> Result<LineIndexEntry, String> {
|
|
640
|
+
let entry = read_index_entry(index, seq, index_path, log_path)?;
|
|
641
|
+
let begins_at_expected_offset = if seq == 1 {
|
|
642
|
+
entry.start == bom_len
|
|
643
|
+
} else {
|
|
644
|
+
read_index_entry(index, seq - 1, index_path, log_path)?.record_end == entry.start
|
|
645
|
+
};
|
|
646
|
+
let ends_at_expected_offset = if seq == indexed_lines {
|
|
647
|
+
entry.record_end <= log_len
|
|
648
|
+
} else {
|
|
649
|
+
entry.record_end == read_index_entry(index, seq + 1, index_path, log_path)?.start
|
|
650
|
+
};
|
|
651
|
+
if !begins_at_expected_offset || !ends_at_expected_offset || entry.record_end > log_len {
|
|
652
|
+
return Err(damaged_index(index_path, log_path));
|
|
653
|
+
}
|
|
654
|
+
Ok(entry)
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
fn read_stored_line(
|
|
658
|
+
log: &mut File,
|
|
659
|
+
seq: u64,
|
|
660
|
+
entry: LineIndexEntry,
|
|
661
|
+
stream_encoding: Option<StreamEncoding>,
|
|
662
|
+
path: &Path,
|
|
663
|
+
) -> Result<StoredLine, String> {
|
|
664
|
+
let total_bytes = entry.content_end.saturating_sub(entry.start);
|
|
665
|
+
let shown_bytes = total_bytes.min(MAX_STORED_LINE_BYTES);
|
|
666
|
+
let length = usize::try_from(shown_bytes).map_err(|_| {
|
|
667
|
+
format!(
|
|
668
|
+
"Cannot read line {seq} from background output log {}: the line is too large to address.",
|
|
669
|
+
display_path(path)
|
|
670
|
+
)
|
|
671
|
+
})?;
|
|
672
|
+
let mut bytes = vec![0_u8; length];
|
|
673
|
+
log.seek(SeekFrom::Start(entry.start))
|
|
674
|
+
.and_then(|_| log.read_exact(&mut bytes))
|
|
675
|
+
.map_err(|error| {
|
|
676
|
+
format!(
|
|
677
|
+
"Cannot read line {seq} from background output log {}: {error}",
|
|
678
|
+
display_path(path)
|
|
679
|
+
)
|
|
680
|
+
})?;
|
|
681
|
+
Ok(StoredLine {
|
|
682
|
+
seq,
|
|
683
|
+
bytes,
|
|
684
|
+
total_bytes,
|
|
685
|
+
stream_encoding,
|
|
686
|
+
legacy_text: None,
|
|
687
|
+
known_truncated: shown_bytes < total_bytes,
|
|
688
|
+
})
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
fn detect_log_encoding(
|
|
692
|
+
log: &mut File,
|
|
693
|
+
log_len: u64,
|
|
694
|
+
) -> Result<(Option<StreamEncoding>, u64), String> {
|
|
695
|
+
let mut prefix = [0_u8; 4];
|
|
696
|
+
let length = usize::try_from(log_len.min(4)).unwrap_or(4);
|
|
697
|
+
log.seek(SeekFrom::Start(0))
|
|
698
|
+
.and_then(|_| log.read_exact(&mut prefix[..length]))
|
|
699
|
+
.map_err(|error| format!("Cannot inspect the background output log encoding: {error}"))?;
|
|
700
|
+
let bytes = &prefix[..length];
|
|
701
|
+
let detected = if bytes.starts_with(stream_bom(StreamEncoding::Utf32Be)) {
|
|
702
|
+
(Some(StreamEncoding::Utf32Be), 4)
|
|
703
|
+
} else if bytes.starts_with(stream_bom(StreamEncoding::Utf32Le)) {
|
|
704
|
+
(Some(StreamEncoding::Utf32Le), 4)
|
|
705
|
+
} else if bytes.starts_with(stream_bom(StreamEncoding::Utf16Be)) {
|
|
706
|
+
(Some(StreamEncoding::Utf16Be), 2)
|
|
707
|
+
} else if bytes.starts_with(stream_bom(StreamEncoding::Utf16Le)) {
|
|
708
|
+
(Some(StreamEncoding::Utf16Le), 2)
|
|
709
|
+
} else {
|
|
710
|
+
(None, 0)
|
|
711
|
+
};
|
|
712
|
+
Ok(detected)
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
fn recover_tail_entries(
|
|
716
|
+
log: &mut File,
|
|
717
|
+
start: u64,
|
|
718
|
+
end: u64,
|
|
719
|
+
stream_encoding: Option<StreamEncoding>,
|
|
720
|
+
path: &Path,
|
|
721
|
+
) -> Result<Vec<LineIndexEntry>, String> {
|
|
722
|
+
let length = end.saturating_sub(start);
|
|
723
|
+
if length > MAX_RECOVERY_SCAN_BYTES {
|
|
724
|
+
let ending = line_ending(stream_encoding);
|
|
725
|
+
let terminated = file_ends_with(log, ending, end, path)?;
|
|
726
|
+
return Ok(vec![LineIndexEntry {
|
|
727
|
+
start,
|
|
728
|
+
content_end: end.saturating_sub(if terminated { ending.len() as u64 } else { 0 }),
|
|
729
|
+
record_end: end,
|
|
730
|
+
}]);
|
|
731
|
+
}
|
|
732
|
+
let mut bytes = vec![0_u8; usize::try_from(length).unwrap_or(0)];
|
|
733
|
+
log.seek(SeekFrom::Start(start))
|
|
734
|
+
.and_then(|_| log.read_exact(&mut bytes))
|
|
735
|
+
.map_err(|error| {
|
|
736
|
+
format!(
|
|
737
|
+
"Cannot recover the final background output bytes from {}: {error}",
|
|
738
|
+
display_path(path)
|
|
739
|
+
)
|
|
740
|
+
})?;
|
|
741
|
+
let ending = line_ending(stream_encoding);
|
|
742
|
+
let width = ending.len();
|
|
743
|
+
let mut entries = Vec::new();
|
|
744
|
+
let mut line_start = 0_usize;
|
|
745
|
+
let mut cursor = 0_usize;
|
|
746
|
+
while cursor.saturating_add(width) <= bytes.len() {
|
|
747
|
+
if &bytes[cursor..cursor + width] == ending {
|
|
748
|
+
entries.push(LineIndexEntry {
|
|
749
|
+
start: start + line_start as u64,
|
|
750
|
+
content_end: start + cursor as u64,
|
|
751
|
+
record_end: start + (cursor + width) as u64,
|
|
752
|
+
});
|
|
753
|
+
cursor += width;
|
|
754
|
+
line_start = cursor;
|
|
755
|
+
} else {
|
|
756
|
+
cursor += width;
|
|
757
|
+
}
|
|
758
|
+
}
|
|
759
|
+
if line_start < bytes.len() {
|
|
760
|
+
entries.push(LineIndexEntry {
|
|
761
|
+
start: start + line_start as u64,
|
|
762
|
+
content_end: end,
|
|
763
|
+
record_end: end,
|
|
764
|
+
});
|
|
765
|
+
}
|
|
766
|
+
Ok(entries)
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
fn file_ends_with(log: &mut File, suffix: &[u8], end: u64, path: &Path) -> Result<bool, String> {
|
|
770
|
+
if end < suffix.len() as u64 {
|
|
771
|
+
return Ok(false);
|
|
772
|
+
}
|
|
773
|
+
let mut actual = vec![0_u8; suffix.len()];
|
|
774
|
+
log.seek(SeekFrom::Start(end - suffix.len() as u64))
|
|
775
|
+
.and_then(|_| log.read_exact(&mut actual))
|
|
776
|
+
.map_err(|error| {
|
|
777
|
+
format!(
|
|
778
|
+
"Cannot inspect the final background output bytes in {}: {error}",
|
|
779
|
+
display_path(path)
|
|
780
|
+
)
|
|
781
|
+
})?;
|
|
782
|
+
Ok(actual == suffix)
|
|
783
|
+
}
|
|
784
|
+
|
|
785
|
+
fn damaged_index(index_path: &Path, log_path: &Path) -> String {
|
|
786
|
+
format!(
|
|
787
|
+
"Cannot read background output index {}: it is damaged. The full log remains readable at {}.",
|
|
788
|
+
display_path(index_path),
|
|
789
|
+
display_path(log_path)
|
|
790
|
+
)
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
fn create_private_file(path: &Path, label: &str) -> Result<File, String> {
|
|
794
|
+
let mut options = OpenOptions::new();
|
|
795
|
+
options.create_new(true).write(true);
|
|
796
|
+
#[cfg(unix)]
|
|
797
|
+
{
|
|
798
|
+
use std::os::unix::fs::OpenOptionsExt;
|
|
799
|
+
options.mode(0o600);
|
|
800
|
+
}
|
|
801
|
+
options.open(path).map_err(|error| {
|
|
802
|
+
format!(
|
|
803
|
+
"cannot create background {label} {}: {error}",
|
|
804
|
+
display_path(path)
|
|
805
|
+
)
|
|
806
|
+
})
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
pub(super) const fn stream_bom(encoding: StreamEncoding) -> &'static [u8] {
|
|
810
|
+
match encoding {
|
|
811
|
+
StreamEncoding::Utf16Le => &[0xff, 0xfe],
|
|
812
|
+
StreamEncoding::Utf16Be => &[0xfe, 0xff],
|
|
813
|
+
StreamEncoding::Utf32Le => &[0xff, 0xfe, 0x00, 0x00],
|
|
814
|
+
StreamEncoding::Utf32Be => &[0x00, 0x00, 0xfe, 0xff],
|
|
815
|
+
}
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
pub(super) const fn line_ending(encoding: Option<StreamEncoding>) -> &'static [u8] {
|
|
819
|
+
match encoding {
|
|
820
|
+
None => b"\n",
|
|
821
|
+
Some(StreamEncoding::Utf16Le) => &[b'\n', 0],
|
|
822
|
+
Some(StreamEncoding::Utf16Be) => &[0, b'\n'],
|
|
823
|
+
Some(StreamEncoding::Utf32Le) => &[b'\n', 0, 0, 0],
|
|
824
|
+
Some(StreamEncoding::Utf32Be) => &[0, 0, 0, b'\n'],
|
|
825
|
+
}
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
const fn stream_width(encoding: Option<StreamEncoding>) -> usize {
|
|
829
|
+
match encoding {
|
|
830
|
+
None => 1,
|
|
831
|
+
Some(StreamEncoding::Utf16Le | StreamEncoding::Utf16Be) => 2,
|
|
832
|
+
Some(StreamEncoding::Utf32Le | StreamEncoding::Utf32Be) => 4,
|
|
833
|
+
}
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
#[cfg(test)]
|
|
837
|
+
mod tests {
|
|
838
|
+
use super::{
|
|
839
|
+
INDEX_ENTRY_BYTES, INDEX_HEADER, LineIndexEntry, OutputLogReader, OutputLogWriter,
|
|
840
|
+
};
|
|
841
|
+
use crate::shell::normalize::{NormalizedEvent, StreamEncoding};
|
|
842
|
+
use std::io::Write as _;
|
|
843
|
+
|
|
844
|
+
#[cfg(windows)]
|
|
845
|
+
fn mark_sparse(file: &std::fs::File) {
|
|
846
|
+
use std::os::windows::io::AsRawHandle;
|
|
847
|
+
use windows_sys::Win32::System::IO::DeviceIoControl;
|
|
848
|
+
|
|
849
|
+
const FSCTL_SET_SPARSE: u32 = 590_020;
|
|
850
|
+
let mut returned = 0_u32;
|
|
851
|
+
// SAFETY: the file handle is valid for the duration of the synchronous
|
|
852
|
+
// call; this control code has no input or output buffers.
|
|
853
|
+
let marked = unsafe {
|
|
854
|
+
DeviceIoControl(
|
|
855
|
+
file.as_raw_handle(),
|
|
856
|
+
FSCTL_SET_SPARSE,
|
|
857
|
+
std::ptr::null(),
|
|
858
|
+
0,
|
|
859
|
+
std::ptr::null_mut(),
|
|
860
|
+
0,
|
|
861
|
+
&mut returned,
|
|
862
|
+
std::ptr::null_mut(),
|
|
863
|
+
)
|
|
864
|
+
};
|
|
865
|
+
assert_ne!(
|
|
866
|
+
marked,
|
|
867
|
+
0,
|
|
868
|
+
"failed to mark the query-complexity fixture sparse: {}",
|
|
869
|
+
std::io::Error::last_os_error()
|
|
870
|
+
);
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
#[cfg(not(windows))]
|
|
874
|
+
fn mark_sparse(_file: &std::fs::File) {}
|
|
875
|
+
|
|
876
|
+
#[test]
|
|
877
|
+
fn writer_keeps_complete_long_lines_and_a_fixed_line_index() {
|
|
878
|
+
let temp = tempfile::tempdir().unwrap();
|
|
879
|
+
let mut writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
880
|
+
let payload = vec![b'x'; 400_000];
|
|
881
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
882
|
+
for chunk in payload.chunks(16 * 1024) {
|
|
883
|
+
writer
|
|
884
|
+
.append(NormalizedEvent::Bytes(chunk.to_vec()))
|
|
885
|
+
.unwrap();
|
|
886
|
+
}
|
|
887
|
+
assert_eq!(
|
|
888
|
+
writer
|
|
889
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
890
|
+
.unwrap(),
|
|
891
|
+
Some(1)
|
|
892
|
+
);
|
|
893
|
+
writer.finish().unwrap();
|
|
894
|
+
|
|
895
|
+
let log = std::fs::read(temp.path().join("output.log")).unwrap();
|
|
896
|
+
assert_eq!(&log[..payload.len()], payload.as_slice());
|
|
897
|
+
assert_eq!(&log[payload.len()..], b"\n");
|
|
898
|
+
let index = std::fs::read(temp.path().join("output.idx")).unwrap();
|
|
899
|
+
assert_eq!(&index[..INDEX_HEADER.len()], INDEX_HEADER);
|
|
900
|
+
assert_eq!(
|
|
901
|
+
index.len() as u64,
|
|
902
|
+
INDEX_HEADER.len() as u64 + INDEX_ENTRY_BYTES
|
|
903
|
+
);
|
|
904
|
+
assert_eq!(
|
|
905
|
+
LineIndexEntry::decode(&index[INDEX_HEADER.len()..]),
|
|
906
|
+
Some(LineIndexEntry {
|
|
907
|
+
start: 0,
|
|
908
|
+
content_end: payload.len() as u64,
|
|
909
|
+
record_end: payload.len() as u64 + 1,
|
|
910
|
+
})
|
|
911
|
+
);
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
#[test]
|
|
915
|
+
fn wide_stream_log_is_a_plain_bom_marked_text_file() {
|
|
916
|
+
let temp = tempfile::tempdir().unwrap();
|
|
917
|
+
let mut writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
918
|
+
writer
|
|
919
|
+
.append(NormalizedEvent::Start(Some(StreamEncoding::Utf16Le)))
|
|
920
|
+
.unwrap();
|
|
921
|
+
writer
|
|
922
|
+
.append(NormalizedEvent::Bytes(vec![b'a', 0]))
|
|
923
|
+
.unwrap();
|
|
924
|
+
writer
|
|
925
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
926
|
+
.unwrap();
|
|
927
|
+
writer.finish().unwrap();
|
|
928
|
+
assert_eq!(
|
|
929
|
+
std::fs::read(temp.path().join("output.log")).unwrap(),
|
|
930
|
+
[0xff, 0xfe, b'a', 0, b'\n', 0]
|
|
931
|
+
);
|
|
932
|
+
}
|
|
933
|
+
|
|
934
|
+
#[test]
|
|
935
|
+
fn terminal_reader_recovers_the_log_tail_after_a_partial_index_write() {
|
|
936
|
+
let temp = tempfile::tempdir().unwrap();
|
|
937
|
+
let mut writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
938
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
939
|
+
writer
|
|
940
|
+
.append(NormalizedEvent::Bytes(b"first".to_vec()))
|
|
941
|
+
.unwrap();
|
|
942
|
+
writer
|
|
943
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
944
|
+
.unwrap();
|
|
945
|
+
writer.finish().unwrap();
|
|
946
|
+
drop(writer);
|
|
947
|
+
|
|
948
|
+
std::fs::OpenOptions::new()
|
|
949
|
+
.append(true)
|
|
950
|
+
.open(temp.path().join("output.log"))
|
|
951
|
+
.unwrap()
|
|
952
|
+
.write_all(b"second\nthird")
|
|
953
|
+
.unwrap();
|
|
954
|
+
let mut partial_entry = Vec::new();
|
|
955
|
+
LineIndexEntry {
|
|
956
|
+
start: 6,
|
|
957
|
+
content_end: 12,
|
|
958
|
+
record_end: 13,
|
|
959
|
+
}
|
|
960
|
+
.encode(&mut partial_entry);
|
|
961
|
+
std::fs::OpenOptions::new()
|
|
962
|
+
.append(true)
|
|
963
|
+
.open(temp.path().join("output.idx"))
|
|
964
|
+
.unwrap()
|
|
965
|
+
.write_all(&partial_entry[..7])
|
|
966
|
+
.unwrap();
|
|
967
|
+
|
|
968
|
+
let live = OutputLogReader::open(temp.path(), false).unwrap();
|
|
969
|
+
assert_eq!(live.total_lines(), 1);
|
|
970
|
+
let mut terminal = OutputLogReader::open(temp.path(), true).unwrap();
|
|
971
|
+
assert_eq!(terminal.total_lines(), 3);
|
|
972
|
+
assert_eq!(
|
|
973
|
+
terminal
|
|
974
|
+
.read_range(1, 3)
|
|
975
|
+
.unwrap()
|
|
976
|
+
.into_iter()
|
|
977
|
+
.map(|line| (line.seq, line.bytes))
|
|
978
|
+
.collect::<Vec<_>>(),
|
|
979
|
+
[
|
|
980
|
+
(1, b"first".to_vec()),
|
|
981
|
+
(2, b"second".to_vec()),
|
|
982
|
+
(3, b"third".to_vec()),
|
|
983
|
+
]
|
|
984
|
+
);
|
|
985
|
+
}
|
|
986
|
+
|
|
987
|
+
#[test]
|
|
988
|
+
fn reader_bounds_response_memory_without_truncating_the_plain_log() {
|
|
989
|
+
let temp = tempfile::tempdir().unwrap();
|
|
990
|
+
let mut writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
991
|
+
let payload = vec![b'z'; 400_000];
|
|
992
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
993
|
+
writer
|
|
994
|
+
.append(NormalizedEvent::Bytes(payload.clone()))
|
|
995
|
+
.unwrap();
|
|
996
|
+
writer.finish().unwrap();
|
|
997
|
+
drop(writer);
|
|
998
|
+
|
|
999
|
+
let mut reader = OutputLogReader::open(temp.path(), true).unwrap();
|
|
1000
|
+
let line = reader.read_range(1, 1).unwrap().remove(0);
|
|
1001
|
+
assert_eq!(line.total_bytes, payload.len() as u64);
|
|
1002
|
+
assert_eq!(line.bytes.len(), 64 * 1024);
|
|
1003
|
+
assert!(line.known_truncated);
|
|
1004
|
+
assert_eq!(
|
|
1005
|
+
std::fs::read(temp.path().join("output.log")).unwrap(),
|
|
1006
|
+
payload
|
|
1007
|
+
);
|
|
1008
|
+
}
|
|
1009
|
+
|
|
1010
|
+
#[test]
|
|
1011
|
+
fn reader_answers_from_the_index_without_scanning_a_sparse_huge_log() {
|
|
1012
|
+
const SPARSE_LOG_BYTES: u64 = 64 * 1024 * 1024 * 1024;
|
|
1013
|
+
|
|
1014
|
+
let temp = tempfile::tempdir().unwrap();
|
|
1015
|
+
let writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
1016
|
+
drop(writer);
|
|
1017
|
+
let log_path = temp.path().join("output.log");
|
|
1018
|
+
let log = std::fs::OpenOptions::new()
|
|
1019
|
+
.write(true)
|
|
1020
|
+
.open(&log_path)
|
|
1021
|
+
.unwrap();
|
|
1022
|
+
mark_sparse(&log);
|
|
1023
|
+
log.set_len(SPARSE_LOG_BYTES).unwrap();
|
|
1024
|
+
let mut entry = Vec::new();
|
|
1025
|
+
LineIndexEntry {
|
|
1026
|
+
start: 0,
|
|
1027
|
+
content_end: SPARSE_LOG_BYTES,
|
|
1028
|
+
record_end: SPARSE_LOG_BYTES,
|
|
1029
|
+
}
|
|
1030
|
+
.encode(&mut entry);
|
|
1031
|
+
std::fs::OpenOptions::new()
|
|
1032
|
+
.append(true)
|
|
1033
|
+
.open(temp.path().join("output.idx"))
|
|
1034
|
+
.unwrap()
|
|
1035
|
+
.write_all(&entry)
|
|
1036
|
+
.unwrap();
|
|
1037
|
+
|
|
1038
|
+
let mut reader = OutputLogReader::open(temp.path(), true).unwrap();
|
|
1039
|
+
let line = reader.read_range(1, 1).unwrap().remove(0);
|
|
1040
|
+
assert_eq!(line.total_bytes, SPARSE_LOG_BYTES);
|
|
1041
|
+
assert_eq!(line.bytes.len(), 64 * 1024);
|
|
1042
|
+
assert!(line.known_truncated);
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
#[test]
|
|
1046
|
+
fn reader_rejects_a_locally_valid_but_discontinuous_middle_index_entry() {
|
|
1047
|
+
let temp = tempfile::tempdir().unwrap();
|
|
1048
|
+
let mut writer = OutputLogWriter::new(temp.path()).unwrap();
|
|
1049
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
1050
|
+
for payload in [b"one".as_slice(), b"two", b"three", b"four"] {
|
|
1051
|
+
writer
|
|
1052
|
+
.append(NormalizedEvent::Bytes(payload.to_vec()))
|
|
1053
|
+
.unwrap();
|
|
1054
|
+
writer
|
|
1055
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
1056
|
+
.unwrap();
|
|
1057
|
+
}
|
|
1058
|
+
writer.finish().unwrap();
|
|
1059
|
+
drop(writer);
|
|
1060
|
+
|
|
1061
|
+
let index_path = temp.path().join("output.idx");
|
|
1062
|
+
let mut index = std::fs::read(&index_path).unwrap();
|
|
1063
|
+
let second_record_end = INDEX_HEADER.len() + INDEX_ENTRY_BYTES as usize + 16;
|
|
1064
|
+
index[second_record_end..second_record_end + 8].copy_from_slice(&7_u64.to_le_bytes());
|
|
1065
|
+
std::fs::write(index_path, index).unwrap();
|
|
1066
|
+
|
|
1067
|
+
let mut reader = OutputLogReader::open(temp.path(), true).unwrap();
|
|
1068
|
+
let error = reader.read_range(2, 2).unwrap_err();
|
|
1069
|
+
assert!(error.contains("index"));
|
|
1070
|
+
assert!(error.contains("damaged"));
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
#[test]
|
|
1074
|
+
fn combined_quota_bounds_index_heavy_empty_lines() {
|
|
1075
|
+
const LIMIT: u64 = 1_024;
|
|
1076
|
+
let temp = tempfile::tempdir().unwrap();
|
|
1077
|
+
let mut writer = OutputLogWriter::with_limit(temp.path(), LIMIT).unwrap();
|
|
1078
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
1079
|
+
for _ in 0..10_000 {
|
|
1080
|
+
writer
|
|
1081
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
1082
|
+
.unwrap();
|
|
1083
|
+
}
|
|
1084
|
+
writer.finish().unwrap();
|
|
1085
|
+
assert!(writer.quota_exceeded());
|
|
1086
|
+
let log_len = std::fs::metadata(temp.path().join("output.log"))
|
|
1087
|
+
.unwrap()
|
|
1088
|
+
.len();
|
|
1089
|
+
let index_len = std::fs::metadata(temp.path().join("output.idx"))
|
|
1090
|
+
.unwrap()
|
|
1091
|
+
.len();
|
|
1092
|
+
assert!(log_len + index_len <= LIMIT, "{log_len}+{index_len}");
|
|
1093
|
+
assert_eq!(
|
|
1094
|
+
index_len,
|
|
1095
|
+
INDEX_HEADER.len() as u64 + writer.total_lines() * 24
|
|
1096
|
+
);
|
|
1097
|
+
assert!(writer.total_lines() > 0);
|
|
1098
|
+
assert!(writer.total_lines() < 10_000);
|
|
1099
|
+
}
|
|
1100
|
+
|
|
1101
|
+
#[test]
|
|
1102
|
+
fn combined_quota_keeps_a_readable_prefix_and_stops_all_future_growth() {
|
|
1103
|
+
const LIMIT: u64 = 4_096;
|
|
1104
|
+
let temp = tempfile::tempdir().unwrap();
|
|
1105
|
+
let mut writer = OutputLogWriter::with_limit(temp.path(), LIMIT).unwrap();
|
|
1106
|
+
writer.append(NormalizedEvent::Start(None)).unwrap();
|
|
1107
|
+
writer
|
|
1108
|
+
.append(NormalizedEvent::Bytes(vec![b'x'; 32 * 1024]))
|
|
1109
|
+
.unwrap();
|
|
1110
|
+
assert!(writer.quota_exceeded());
|
|
1111
|
+
assert_eq!(writer.total_lines(), 1);
|
|
1112
|
+
writer
|
|
1113
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
1114
|
+
.unwrap();
|
|
1115
|
+
writer.finish().unwrap();
|
|
1116
|
+
let before = (
|
|
1117
|
+
std::fs::metadata(temp.path().join("output.log"))
|
|
1118
|
+
.unwrap()
|
|
1119
|
+
.len(),
|
|
1120
|
+
std::fs::metadata(temp.path().join("output.idx"))
|
|
1121
|
+
.unwrap()
|
|
1122
|
+
.len(),
|
|
1123
|
+
);
|
|
1124
|
+
for _ in 0..100 {
|
|
1125
|
+
writer
|
|
1126
|
+
.append(NormalizedEvent::Bytes(vec![b'y'; 4_096]))
|
|
1127
|
+
.unwrap();
|
|
1128
|
+
writer
|
|
1129
|
+
.append(NormalizedEvent::LineEnd { terminated: true })
|
|
1130
|
+
.unwrap();
|
|
1131
|
+
}
|
|
1132
|
+
writer.finish().unwrap();
|
|
1133
|
+
let after = (
|
|
1134
|
+
std::fs::metadata(temp.path().join("output.log"))
|
|
1135
|
+
.unwrap()
|
|
1136
|
+
.len(),
|
|
1137
|
+
std::fs::metadata(temp.path().join("output.idx"))
|
|
1138
|
+
.unwrap()
|
|
1139
|
+
.len(),
|
|
1140
|
+
);
|
|
1141
|
+
assert_eq!(after, before);
|
|
1142
|
+
assert!(after.0 + after.1 <= LIMIT);
|
|
1143
|
+
let mut reader = OutputLogReader::open(temp.path(), true).unwrap();
|
|
1144
|
+
let line = reader.read_range(1, 1).unwrap().remove(0);
|
|
1145
|
+
assert!(!line.bytes.is_empty());
|
|
1146
|
+
assert!(line.bytes.iter().all(|byte| *byte == b'x'));
|
|
1147
|
+
}
|
|
1148
|
+
}
|