@hadden-industries/markdown-quality 1.0.1 → 1.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/quality.js CHANGED
@@ -4,6 +4,7 @@ import { selectDocuments, readDocument } from "./documents.js";
4
4
  import { resolveTool, manifest } from "./native-tool.js";
5
5
  import { createDocumentAnalyzer } from "./document-analysis.js";
6
6
  import { replaceDocument } from "./replacement.js";
7
+ import { createNativeStaging } from "./native-staging.js";
7
8
  import {
8
9
  decode,
9
10
  fail,
@@ -44,7 +45,21 @@ export function createResult(mode = "check", explicit = false) {
44
45
  export async function runQuality(options = {}) {
45
46
  const mode = options.mode ?? "check";
46
47
  const result = createResult(mode, options.files !== undefined);
47
- let analyzer;
48
+ let analyzer,
49
+ staging,
50
+ stagingAttempted = false;
51
+ async function closeAnalysis() {
52
+ if (analyzer) {
53
+ const owned = analyzer;
54
+ analyzer = undefined;
55
+ await owned.close();
56
+ }
57
+ if (staging) {
58
+ const owned = staging;
59
+ staging = undefined;
60
+ owned.close();
61
+ }
62
+ }
48
63
  try {
49
64
  if (!["check", "format", "inspect"].includes(mode))
50
65
  fail("INVALID_OPERATION", "Unknown operation.");
@@ -71,48 +86,157 @@ export async function runQuality(options = {}) {
71
86
  let total = 0,
72
87
  outputTotal = 0,
73
88
  diagnosticBytes = 0;
74
- for (const file of result.selection.files) {
75
- const item = readDocument(context.root, file);
76
- if (mode === "format" && item.stat.nlink !== 1n)
77
- fail("HARD_LINK", "Hard-linked documents cannot be formatted.");
78
- total += item.bytes.length;
79
- if (total > limits.totalBytes)
80
- fail("BATCH_LIMIT", "Total document bytes exceed the limit.");
81
- const original = decode(item.bytes);
82
- const formatted = await analyzer.analyze({ text: original, file, mode });
83
- const outputBytes = Buffer.byteLength(formatted.output);
84
- if (outputBytes > limits.fileBytes)
85
- fail("DOCUMENT_LIMIT", "Formatted document exceeds the size limit.");
86
- outputTotal += outputBytes;
87
- if (outputTotal > limits.totalBytes)
88
- fail("BATCH_LIMIT", "Total formatted document bytes exceed the limit.");
89
- const diagnostics = formatted.diagnostics;
90
- if (mode === "check" && !item.bytes.equals(Buffer.from(formatted.output)))
91
- diagnostics.push({
92
- source: "formatter",
93
- rule: "layout",
94
- line: 1,
95
- column: 1,
96
- severity: "error",
97
- message: "Document requires formatting.",
98
- });
99
- diagnosticBudget(diagnostics);
100
- for (const d of diagnostics) {
101
- const diagnostic = { path: file, ...d };
102
- diagnosticBytes += Buffer.byteLength(JSON.stringify(diagnostic)) + 1;
89
+ let prepared = [],
90
+ preparedBytes = 0;
91
+ let inputs = [],
92
+ inputBytes = 0;
93
+ async function flushPrepared() {
94
+ if (!prepared.length) return;
95
+ const group = prepared;
96
+ prepared = [];
97
+ preparedBytes = 0;
98
+ if (group.length > 1 && !stagingAttempted) {
99
+ stagingAttempted = true;
100
+ const started = performance.now();
101
+ staging = createNativeStaging(context.root);
102
+ const elapsed = performance.now() - started;
103
+ for (const record of group) record.elapsedMs += elapsed;
104
+ }
105
+ const verified = await analyzer.verify(
106
+ group.map(({ formatted, elapsedMs, precheck }) => ({
107
+ output: formatted.output,
108
+ elapsedMs,
109
+ precheck: formatted.prechecked ? precheck : undefined,
110
+ })),
111
+ staging?.token,
112
+ );
113
+ for (const [index, { file, item, formatted }] of group.entries()) {
114
+ if (formatted.deferredError)
115
+ fail(formatted.deferredError.code, formatted.deferredError.message);
116
+ const outputBytes = Buffer.byteLength(formatted.output);
117
+ if (outputBytes > limits.fileBytes)
118
+ fail("DOCUMENT_LIMIT", "Formatted document exceeds the size limit.");
119
+ outputTotal += outputBytes;
120
+ if (outputTotal > limits.totalBytes)
121
+ fail(
122
+ "BATCH_LIMIT",
123
+ "Total formatted document bytes exceed the limit.",
124
+ );
125
+ const diagnostics = [
126
+ ...formatted.diagnostics.slice(0, formatted.whitespaceCount),
127
+ ...verified.diagnostics[index],
128
+ ...formatted.diagnostics.slice(formatted.whitespaceCount),
129
+ ];
103
130
  if (
104
- result.diagnostics.length >= limits.diagnostics ||
105
- diagnosticBytes > limits.diagnosticBytes
131
+ mode === "check" &&
132
+ !item.bytes.equals(Buffer.from(formatted.output))
106
133
  )
107
- fail(
108
- "DIAGNOSTIC_LIMIT",
109
- "Diagnostic count or output bytes exceed the limit.",
134
+ diagnostics.push({
135
+ source: "formatter",
136
+ rule: "layout",
137
+ line: 1,
138
+ column: 1,
139
+ severity: "error",
140
+ message: "Document requires formatting.",
141
+ });
142
+ diagnosticBudget(diagnostics);
143
+ for (const d of diagnostics) {
144
+ const diagnostic = { path: file, ...d };
145
+ diagnosticBytes += Buffer.byteLength(JSON.stringify(diagnostic)) + 1;
146
+ if (
147
+ result.diagnostics.length >= limits.diagnostics ||
148
+ diagnosticBytes > limits.diagnosticBytes
149
+ )
150
+ fail(
151
+ "DIAGNOSTIC_LIMIT",
152
+ "Diagnostic count or output bytes exceed the limit.",
153
+ );
154
+ result.diagnostics.push(diagnostic);
155
+ }
156
+ if (mode === "format")
157
+ candidates.push({ file, item, output: formatted.output });
158
+ }
159
+ }
160
+ async function flushInputs() {
161
+ if (!inputs.length) return;
162
+ const group = inputs;
163
+ inputs = [];
164
+ inputBytes = 0;
165
+ let setupMs = 0;
166
+ if (group.length > 1 && !stagingAttempted) {
167
+ stagingAttempted = true;
168
+ const started = performance.now();
169
+ staging = createNativeStaging(context.root);
170
+ setupMs = performance.now() - started;
171
+ }
172
+ let checks;
173
+ let overheadMs = setupMs;
174
+ if (staging) {
175
+ const started = performance.now();
176
+ checks = await analyzer.precheck(
177
+ group.map((record) => record.original),
178
+ staging.token,
179
+ );
180
+ overheadMs += Math.max(
181
+ 0,
182
+ performance.now() - started - checks.workerMs,
183
+ );
184
+ }
185
+ for (const [index, { file, item, original }] of group.entries()) {
186
+ const precheck = checks?.checks[index];
187
+ let elapsedMs = overheadMs + (checks?.elapsedMs[index] ?? 0);
188
+ let formatted;
189
+ try {
190
+ const started = performance.now();
191
+ formatted = await analyzer.prepare(
192
+ { text: original, file, mode, precheck },
193
+ 30_000 - elapsedMs,
110
194
  );
111
- result.diagnostics.push(diagnostic);
195
+ elapsedMs += performance.now() - started;
196
+ } catch (error) {
197
+ await flushPrepared();
198
+ throw error;
199
+ }
200
+ const retainedBytes =
201
+ item.bytes.length + Buffer.byteLength(formatted.output);
202
+ if (
203
+ prepared.length >= 32 ||
204
+ preparedBytes + retainedBytes > 4 * 1024 * 1024 ||
205
+ formatted.deferredError
206
+ )
207
+ await flushPrepared();
208
+ prepared.push({ file, item, formatted, elapsedMs, precheck });
209
+ preparedBytes += retainedBytes;
210
+ if (formatted.deferredError) await flushPrepared();
112
211
  }
113
- if (mode === "format")
114
- candidates.push({ file, item, output: formatted.output });
212
+ await flushPrepared();
115
213
  }
214
+ for (const file of result.selection.files) {
215
+ let item, original;
216
+ try {
217
+ item = readDocument(context.root, file);
218
+ if (mode === "format" && item.stat.nlink !== 1n)
219
+ fail("HARD_LINK", "Hard-linked documents cannot be formatted.");
220
+ total += item.bytes.length;
221
+ if (total > limits.totalBytes)
222
+ fail("BATCH_LIMIT", "Total document bytes exceed the limit.");
223
+ original = decode(item.bytes);
224
+ } catch (error) {
225
+ await flushInputs();
226
+ throw error;
227
+ }
228
+ if (
229
+ inputs.length >= 32 ||
230
+ inputBytes + item.bytes.length > 4 * 1024 * 1024
231
+ )
232
+ await flushInputs();
233
+ inputs.push({ file, item, original });
234
+ inputBytes += item.bytes.length;
235
+ }
236
+ await flushInputs();
237
+ await flushPrepared();
238
+ // Quiesce the worker and remove private payloads before admitting any write.
239
+ await closeAnalysis();
116
240
  result.diagnostics.sort((a, b) =>
117
241
  a.path < b.path
118
242
  ? -1
@@ -150,7 +274,17 @@ export async function runQuality(options = {}) {
150
274
  error instanceof OperationError ? error.message : "Operation failed.",
151
275
  });
152
276
  } finally {
153
- if (analyzer) await analyzer.close();
277
+ try {
278
+ await closeAnalysis();
279
+ } catch (error) {
280
+ result.outcome = "error";
281
+ result.exitCode = 2;
282
+ result.errors.push({
283
+ code: error instanceof OperationError ? error.code : "OPERATION_FAILED",
284
+ message:
285
+ error instanceof OperationError ? error.message : "Operation failed.",
286
+ });
287
+ }
154
288
  }
155
289
  return result;
156
290
  }
@@ -0,0 +1,57 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-only
2
+ // Authored Markdown whitespace policy; literal code bodies are the only exemption.
3
+ import { parse } from "./analysis.js";
4
+ import { codeBodyRows } from "./literal-layout.js";
5
+ import { fail, limits } from "./contracts.js";
6
+
7
+ /** Find every non-code trailing space/tab, or fail at the document diagnostic bound. */
8
+ export function checkTrailingWhitespace(text, memo) {
9
+ if (!/[ \t](?:\r\n|\r|\n|$)/u.test(text)) return [];
10
+ const rows = text.split(/(\r\n|\r|\n)/u);
11
+ const codeRows = codeBodyRows(text, memo ? memo.parse(text) : parse(text));
12
+ const diagnostics = [];
13
+ for (let row = 0; row < rows.length; row += 2) {
14
+ if (codeRows.has(row)) continue;
15
+ const trailing = /[ \t]+$/u.exec(rows[row]);
16
+ if (!trailing) continue;
17
+ if (diagnostics.length >= limits.documentDiagnostics)
18
+ fail(
19
+ "DIAGNOSTIC_LIMIT",
20
+ "Diagnostic count or output bytes exceed the limit.",
21
+ );
22
+ diagnostics.push({
23
+ source: "formatter",
24
+ rule: "trailing-whitespace",
25
+ line: row / 2 + 1,
26
+ column: trailing.index + 1,
27
+ severity: "error",
28
+ message:
29
+ "Trailing spaces and tabs are forbidden outside code-block contents; literal-sensitive fixes require manual editing.",
30
+ });
31
+ }
32
+ return diagnostics;
33
+ }
34
+
35
+ /** Propose trimming with parsed hard breaks made explicit; caller must verify semantics. */
36
+ export function normalizeTrailingWhitespace(text, memo) {
37
+ // Most documents need no policy repair. Avoid another Markdown parse there.
38
+ if (!/[ \t](?:\r\n|\r|\n|$)/u.test(text)) return text;
39
+ const tree = memo ? memo.parse(text) : parse(text);
40
+ const rows = text.split(/(\r\n|\r|\n)/u);
41
+ const codeRows = codeBodyRows(text, tree);
42
+ const hardBreakRows = new Set();
43
+ function visit(node) {
44
+ // Only the parser can distinguish an actual hard break from spaces at a
45
+ // paragraph end, in a literal, or beside escaped punctuation.
46
+ if (node.type === "break" && text[node.position.start.offset] === " ")
47
+ hardBreakRows.add((node.position.start.line - 1) * 2);
48
+ for (const child of node.children ?? []) visit(child);
49
+ }
50
+ visit(tree);
51
+ for (let row = 0; row < rows.length; row += 2) {
52
+ if (codeRows.has(row) || !/[ \t]+$/u.test(rows[row])) continue;
53
+ rows[row] = rows[row].replace(/[ \t]+$/u, "");
54
+ if (hardBreakRows.has(row)) rows[row] += "\\";
55
+ }
56
+ return rows.join("");
57
+ }