omitly-mcp 0.1.6 → 0.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -24,12 +24,14 @@
24
24
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
25
25
  import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
26
26
  import { spawn } from "node:child_process";
27
- import { existsSync, readFileSync } from "node:fs";
27
+ import { existsSync, readFileSync, realpathSync } from "node:fs";
28
28
  import { createRequire } from "node:module";
29
29
  import * as path from "node:path";
30
+ import { pathToFileURL } from "node:url";
30
31
  import { z } from "zod";
31
32
  import { ENGINE_OUTPUT_CAP_BYTES, resolveEngineTimeoutMs } from "./engine-config.js";
32
33
  import { allowedRoot, confineInput, confineOutput } from "./paths.js";
34
+ import { isSealErrorVerdict, normalizeSealResult, SEAL_VERDICTS, } from "./seal.js";
33
35
  /**
34
36
  * WASM fallback for the four detection/verify tools — bundled in the npm
35
37
  * package (see package.json "files"/"build:wasm"), so `find_sensitive_regions`,
@@ -126,6 +128,289 @@ const regionSchema = z.object({
126
128
  height: z.number().positive(),
127
129
  reason: z.string().optional().describe("audit-log reason, e.g. 'PII: SSN'"),
128
130
  });
131
+ /**
132
+ * Output schemas (#540) — the enforcement point for "no raw text ever leaves
133
+ * this server". Every object is `.strict()` (unknown keys are a validation
134
+ * ERROR, not silently dropped) and preview-carrying fields are always the
135
+ * engine's MASKED preview, never a raw-text field — so an engine or handler
136
+ * regression that starts emitting a raw value fails the tool call instead of
137
+ * quietly shipping it. Handlers below build `structuredContent` by explicit
138
+ * field allowlist (never by spreading the raw engine response), which is the
139
+ * second half of that guarantee: even a schema gap can't leak a field the
140
+ * handler never copied over in the first place.
141
+ *
142
+ * Field shapes here are drawn directly from what `crates/omitly-cli/src/main.rs`
143
+ * actually serializes (see the module doc's JSON contracts) — not guessed from
144
+ * the handler's existing text formatting.
145
+ *
146
+ * IMPORTANT — why every handler calls `checked()` rather than leaving
147
+ * validation to the SDK: `@modelcontextprotocol/sdk`'s `validateToolOutput`
148
+ * (server/mcp.js) opens with `if (result.isError) { return; }`, i.e. it skips
149
+ * `outputSchema` validation entirely whenever a result is flagged as an error.
150
+ * Per #338's policy, "the document is NOT clean" is exactly what sets
151
+ * `isError: true` here — so the SDK would skip validation on precisely the
152
+ * three branches that carry `findings`/`survivors`/`regions`, the ones where a
153
+ * leaked field matters most. Flipping that `isError` policy is not an option
154
+ * (#338 settled it deliberately), so the schemas are enforced here instead of
155
+ * being relied on to fire downstream.
156
+ */
157
+ /**
158
+ * Validate a `structuredContent` payload against its declared output schema
159
+ * before it leaves this process — see the note above on the SDK's `isError`
160
+ * bypass. Returns the PARSED value (so any non-strict nested object also has
161
+ * unknown keys stripped), and throws otherwise, which fails the tool call.
162
+ *
163
+ * The thrown message names issue PATHS and zod issue CODES only — never the
164
+ * offending value. Some zod messages (`invalid_enum_value`, `invalid_literal`)
165
+ * embed what they received, and echoing that into an error string would leak
166
+ * the very field the schema just rejected.
167
+ *
168
+ * Exported for `schema-enforcement.test.ts`.
169
+ */
170
+ export function checked(schema, value) {
171
+ const parsed = schema.safeParse(value);
172
+ if (!parsed.success) {
173
+ const where = parsed.error.issues
174
+ .map((i) => `${i.path.join(".") || "<root>"}: ${i.code}`)
175
+ .join("; ") || "unknown";
176
+ throw new Error(`omitly-mcp internal error: tool output failed its declared outputSchema ` +
177
+ `(${where}). Refusing to return it. This is a bug — please report it; ` +
178
+ `no document content is included in this message by design.`);
179
+ }
180
+ return parsed.data;
181
+ }
182
+ /** A candidate/found region as `find`/`locate_text` return it — coordinates
183
+ * plus a MASKED preview (e.g. '•••-••-6789'), never the raw matched text. */
184
+ const outputRegionSchema = z
185
+ .object({
186
+ page: z.number().int(),
187
+ x: z.number(),
188
+ y: z.number(),
189
+ width: z.number(),
190
+ height: z.number(),
191
+ kind: z.string(),
192
+ preview: z.string(),
193
+ })
194
+ .strict();
195
+ /** Copy only the known-safe fields off an engine-returned region — defense in
196
+ * depth alongside the `.strict()` schema above: a raw-text field the engine
197
+ * started emitting is dropped here even before validation would catch it. */
198
+ function toOutputRegion(r) {
199
+ return {
200
+ page: r.page,
201
+ x: r.x,
202
+ y: r.y,
203
+ width: r.width,
204
+ height: r.height,
205
+ kind: r.kind,
206
+ preview: r.preview,
207
+ };
208
+ }
209
+ export const findSensitiveRegionsOutputSchema = z
210
+ .object({
211
+ count: z.number().int().min(0),
212
+ regions: z.array(outputRegionSchema),
213
+ note: z.string().optional(),
214
+ })
215
+ .strict();
216
+ export const locateTextOutputSchema = z
217
+ .object({
218
+ count: z.number().int().min(0),
219
+ regions: z.array(outputRegionSchema),
220
+ })
221
+ .strict();
222
+ const bboxOutputSchema = z
223
+ .object({ x: z.number(), y: z.number(), width: z.number(), height: z.number() })
224
+ .strict();
225
+ /** Mirrors `redaction_core::model::Verification` — `{result:"pass"}` or
226
+ * `{result:"fail",detail}`. `detail` is a structural diagnostic ("page 3
227
+ * could not be re-read") produced by the verify pass itself, never text
228
+ * extracted from the document. */
229
+ const verificationOutputSchema = z.union([
230
+ z.object({ result: z.literal("pass") }).strict(),
231
+ z.object({ result: z.literal("fail"), detail: z.string() }).strict(),
232
+ ]);
233
+ const auditEntryOutputSchema = z
234
+ .object({
235
+ page: z.number().int(),
236
+ bbox: bboxOutputSchema,
237
+ timestamp: z.string(),
238
+ reason_code: z.string(),
239
+ verification: verificationOutputSchema,
240
+ })
241
+ .strict();
242
+ const auditOutputSchema = z
243
+ .object({
244
+ verdict: z.enum(["pass", "fail"]),
245
+ regions: z.array(auditEntryOutputSchema),
246
+ warnings: z.array(z.string()),
247
+ metadataScrubbed: z.boolean(),
248
+ license: z.string(),
249
+ licensedTo: z.string().nullable(),
250
+ disclosures: z.array(z.string()),
251
+ })
252
+ .strict();
253
+ /** Copy only the known-safe fields off the engine's `audit` object — same
254
+ * allowlist defense as `toOutputRegion`. */
255
+ function toOutputAudit(a) {
256
+ return {
257
+ verdict: a.verdict,
258
+ regions: (a.regions ?? []).map((e) => ({
259
+ page: e.page,
260
+ bbox: { x: e.bbox.x, y: e.bbox.y, width: e.bbox.width, height: e.bbox.height },
261
+ timestamp: e.timestamp,
262
+ reason_code: e.reason_code,
263
+ verification: e.verification?.result === "fail"
264
+ ? { result: "fail", detail: e.verification.detail }
265
+ : { result: "pass" },
266
+ })),
267
+ warnings: a.warnings ?? [],
268
+ metadataScrubbed: a.metadataScrubbed,
269
+ license: a.license,
270
+ licensedTo: a.licensedTo ?? null,
271
+ disclosures: a.disclosures ?? [],
272
+ };
273
+ }
274
+ export const redactByEntityOutputSchema = z
275
+ .object({
276
+ output: z.string().nullable(),
277
+ redactedCount: z.number().int().min(0),
278
+ redacted: z.array(z.object({ page: z.number().int(), kind: z.string(), preview: z.string() }).strict()),
279
+ verdict: z.enum(["pass", "fail"]).nullable(),
280
+ audit: auditOutputSchema.nullable(),
281
+ })
282
+ .strict();
283
+ export const redactPdfOutputSchema = z
284
+ .object({
285
+ output: z.string(),
286
+ regionCount: z.number().int().min(1),
287
+ verdict: z.enum(["pass", "fail"]),
288
+ audit: auditOutputSchema,
289
+ })
290
+ .strict();
291
+ /** A single leak/finding surfaced by `verify_redaction` — never the raw
292
+ * matched text. `reason`/`detail`/`class` are structural (which check
293
+ * failed, why), `preview` is the engine's masked preview. */
294
+ const verifyFindingOutputSchema = z
295
+ .object({
296
+ page: z.number().int().optional(),
297
+ kind: z.string().optional(),
298
+ class: z.string().optional(),
299
+ reason: z.string().optional(),
300
+ detail: z.string().optional(),
301
+ preview: z.string().optional(),
302
+ })
303
+ .strict();
304
+ export const verifyRedactionOutputSchema = z
305
+ .object({
306
+ clean: z.boolean(),
307
+ mode: z.enum(["sidecar", "rescan"]),
308
+ verdict: z.enum(["pass", "fail"]).optional(),
309
+ totalFindings: z.number().int().min(0),
310
+ findings: z.array(verifyFindingOutputSchema),
311
+ })
312
+ .strict();
313
+ export const createPdfOutputSchema = z.object({ output: z.string() }).strict();
314
+ const offPageOutputSchema = z
315
+ .object({ source: z.string(), kind: z.string(), preview: z.string() })
316
+ .strict();
317
+ const coverageOutputSchema = z
318
+ .object({
319
+ pagesScanned: z.number().int().nullable(),
320
+ pagesTotal: z.number().int().nullable(),
321
+ notScanned: z.array(z.string()),
322
+ priorRevisionsScanned: z.boolean(),
323
+ metadataScanned: z.boolean(),
324
+ acroformScanned: z.boolean(),
325
+ attachmentsScanned: z.boolean(),
326
+ formXobjectsScanned: z.number().int(),
327
+ annotationAppearancesScanned: z.number().int(),
328
+ pagesFailedCount: z.number().int(),
329
+ })
330
+ .strict();
331
+ /** Copy only the known-safe fields off the engine's `coverage` object — same
332
+ * allowlist defense as the other `toOutput*` helpers, and it doubles as the
333
+ * camelCase reshape `check_redaction`'s handler already needed. */
334
+ function toOutputCoverage(cov, notScanned) {
335
+ return {
336
+ pagesScanned: typeof cov.pages_scanned === "number" ? cov.pages_scanned : null,
337
+ pagesTotal: typeof cov.pages_total === "number" ? cov.pages_total : null,
338
+ notScanned,
339
+ priorRevisionsScanned: !!cov.prior_revisions_scanned,
340
+ metadataScanned: !!cov.metadata_scanned,
341
+ acroformScanned: !!cov.acroform_scanned,
342
+ attachmentsScanned: !!cov.attachments_scanned,
343
+ formXobjectsScanned: cov.form_xobjects_scanned ?? 0,
344
+ annotationAppearancesScanned: cov.annotation_appearances_scanned ?? 0,
345
+ pagesFailedCount: (cov.pages_failed ?? []).length,
346
+ };
347
+ }
348
+ const licenseProvenanceOutputSchema = z
349
+ .object({
350
+ valid: z.boolean(),
351
+ licenseId: z.string(),
352
+ licensedTo: z.string(),
353
+ product: z.string(),
354
+ reason: z.string().nullable(),
355
+ })
356
+ .strict();
357
+ /** Mirrors `NormalizedSealResult` (seal.ts) minus its internal `ok` — every
358
+ * field here is already hardened by `normalizeSealResult` (unrecognized
359
+ * verdicts collapse to `seal_invalid`/`sealValid:false`), and this is the
360
+ * same allowlist discipline as `toOutputRegion`/`toOutputAudit`: fingerprints,
361
+ * hashes, filenames and structural verdicts only — never document text. */
362
+ export const verifySealOutputSchema = z
363
+ .object({
364
+ verdict: z.enum(SEAL_VERDICTS),
365
+ sealValid: z.boolean().nullable(),
366
+ sealVersion: z.string().nullable(),
367
+ carriesAuditReport: z.boolean().nullable(),
368
+ sealFingerprint: z.string().nullable(),
369
+ allPassed: z.boolean().nullable(),
370
+ metadataScrubbed: z.boolean().nullable(),
371
+ regionCount: z.number().int().nullable(),
372
+ pageCount: z.number().int().nullable(),
373
+ warnings: z.array(z.string()).nullable(),
374
+ licenseProvenance: licenseProvenanceOutputSchema.nullable(),
375
+ inputSha256: z.string().nullable(),
376
+ outputSha256: z.string().nullable(),
377
+ sourceFilename: z.string().nullable(),
378
+ outputFilename: z.string().nullable(),
379
+ })
380
+ .strict();
381
+ /** Copy only the known-safe fields off a `normalizeSealResult()` result —
382
+ * same allowlist defense as the other `toOutput*` helpers, even though
383
+ * `normalizeSealResult` itself is already a hardening layer. */
384
+ function toOutputSeal(res) {
385
+ return {
386
+ verdict: res.verdict,
387
+ sealValid: res.sealValid,
388
+ sealVersion: res.sealVersion,
389
+ carriesAuditReport: res.carriesAuditReport,
390
+ sealFingerprint: res.sealFingerprint,
391
+ allPassed: res.allPassed,
392
+ metadataScrubbed: res.metadataScrubbed,
393
+ regionCount: res.regionCount,
394
+ pageCount: res.pageCount,
395
+ warnings: res.warnings,
396
+ licenseProvenance: res.licenseProvenance,
397
+ inputSha256: res.inputSha256,
398
+ outputSha256: res.outputSha256,
399
+ sourceFilename: res.sourceFilename,
400
+ outputFilename: res.outputFilename,
401
+ };
402
+ }
403
+ export const checkRedactionOutputSchema = z
404
+ .object({
405
+ clean: z.boolean(),
406
+ totalFindings: z.number().int().min(0),
407
+ byKind: z.record(z.string(), z.number().int()),
408
+ regions: z.array(outputRegionSchema),
409
+ survivors: z.array(outputRegionSchema),
410
+ offPage: z.array(offPageOutputSchema),
411
+ coverage: coverageOutputSchema,
412
+ })
413
+ .strict();
129
414
  /**
130
415
  * Invoke the local Omitly engine CLI with a JSON request on stdin and parse the
131
416
  * JSON response from stdout. Contract (proposed for the redaction-core CLI):
@@ -214,7 +499,6 @@ function runPdfEngine(request) {
214
499
  }
215
500
  return spawnEngine(PDF_BIN, request);
216
501
  }
217
- const server = new McpServer({ name: "omitly-mcp", version: VERSION });
218
502
  const REGIONS = z.enum(["generic", "us", "au"]);
219
503
  /** Run "find" via the native engine if configured, else the bundled wasm
220
504
  * fallback (see the wasm block near the top of this file). `regions`
@@ -230,383 +514,648 @@ function findViaEngineOrWasm(confinedPath, regions) {
230
514
  }
231
515
  return res;
232
516
  }
233
- server.tool("find_sensitive_regions", "Scan a PDF on-device and return candidate regions that look like PII — " +
234
- "emails, US SSNs, phone numbers, card numbers, and Australian identifiers " +
235
- "(TFN, ABN, ACN, Medicare, Centrelink CRN, IHI, BSB; kinds 'tfn'/'abn'/" +
236
- "'acn'/'medicare'/'crn'/'ihi'/'bsb') — each with the page and exact " +
237
- "coordinates (in PDF " +
238
- "points) the redaction engine needs. Use this FIRST so you select regions " +
239
- "by entity ('redact every TFN') and pass the returned coordinates straight " +
240
- "to redact_pdf, instead of guessing geometry from a rendered page. Numeric " +
241
- "kinds are check-digit validated where a published algorithm exists (CRN " +
242
- "has none — its matches are format-only). Candidates are best-effort " +
243
- "pattern matches for review — not a completeness guarantee and not a " +
244
- "compliance assessment; the file is never uploaded — detection runs " +
245
- "locally. Each candidate carries a MASKED preview (e.g. '•••-••-6789'), " +
246
- "never the raw value: the secret stays on the machine. You don't need the " +
247
- "plaintext to redact — drive it by page + coordinates. (A human reviewer " +
248
- "has the file open locally for full context.)", {
249
- pdfPath: z.string().describe("absolute path to the PDF to scan"),
250
- regions: z
251
- .array(REGIONS)
252
- .optional()
253
- .describe("narrow LISTED pattern kinds to these regional packs (generic kinds " +
254
- "always listed; confirmed under-mark survivors always report); omit " +
255
- "to scan everything — the safe default"),
256
- }, async ({ pdfPath, regions }) => {
257
- try {
258
- const res = await findViaEngineOrWasm(confineInput(pdfPath, ROOT), regions);
259
- if (!res?.ok) {
517
+ /**
518
+ * Registers all 8 tools on a given `McpServer` instance. Extracted from
519
+ * module-level top code (#540) so tests can build an in-memory server +
520
+ * client pair and drive the tools directly, without a child-process stdio
521
+ * client. `index.ts`'s own module body stays a thin bin entry — see the
522
+ * `isMainModule` guard at the bottom of this file.
523
+ */
524
+ export function registerTools(server) {
525
+ server.registerTool("find_sensitive_regions", {
526
+ description: "Scan a PDF on-device and return candidate regions that look like PII — " +
527
+ "emails, US SSNs, phone numbers, card numbers, and Australian identifiers " +
528
+ "(TFN, ABN, ACN, Medicare, Centrelink CRN, IHI, BSB; kinds 'tfn'/'abn'/" +
529
+ "'acn'/'medicare'/'crn'/'ihi'/'bsb') — each with the page and exact " +
530
+ "coordinates (in PDF " +
531
+ "points) the redaction engine needs. Use this FIRST so you select regions " +
532
+ "by entity ('redact every TFN') and pass the returned coordinates straight " +
533
+ "to redact_pdf, instead of guessing geometry from a rendered page. Numeric " +
534
+ "kinds are check-digit validated where a published algorithm exists (CRN " +
535
+ "has none — its matches are format-only). Candidates are best-effort " +
536
+ "pattern matches for review — not a completeness guarantee and not a " +
537
+ "compliance assessment; the file is never uploaded — detection runs " +
538
+ "locally. Each candidate carries a MASKED preview (e.g. '•••-••-6789'), " +
539
+ "never the raw value: the secret stays on the machine. You don't need the " +
540
+ "plaintext to redact — drive it by page + coordinates. (A human reviewer " +
541
+ "has the file open locally for full context.)",
542
+ inputSchema: {
543
+ pdfPath: z.string().describe("absolute path to the PDF to scan"),
544
+ regions: z
545
+ .array(REGIONS)
546
+ .optional()
547
+ .describe("narrow LISTED pattern kinds to these regional packs (generic kinds " +
548
+ "always listed; confirmed under-mark survivors always report); omit " +
549
+ "to scan everything — the safe default"),
550
+ },
551
+ outputSchema: findSensitiveRegionsOutputSchema,
552
+ }, async ({ pdfPath, regions }) => {
553
+ try {
554
+ const res = await findViaEngineOrWasm(confineInput(pdfPath, ROOT), regions);
555
+ if (!res?.ok) {
556
+ return {
557
+ content: [{ type: "text", text: `Scan failed: ${res?.error ?? "unknown error"}` }],
558
+ isError: true,
559
+ };
560
+ }
561
+ const found = res.regions ?? [];
562
+ const summary = `Found ${found.length} candidate region(s).\n` +
563
+ `Pass any subset to redact_pdf as its "regions" argument (page + x/y/width/height carry over).\n\n` +
564
+ (res.note ? `Note: ${res.note}\n\n` : "") +
565
+ JSON.stringify(found, null, 2);
566
+ const structuredContent = checked(findSensitiveRegionsOutputSchema, {
567
+ count: found.length,
568
+ regions: found.map(toOutputRegion),
569
+ ...(res.note ? { note: res.note } : {}),
570
+ });
571
+ return { content: [{ type: "text", text: summary }], structuredContent };
572
+ }
573
+ catch (e) {
260
574
  return {
261
- content: [{ type: "text", text: `Scan failed: ${res?.error ?? "unknown error"}` }],
575
+ content: [{ type: "text", text: `Could not scan: ${e.message}` }],
262
576
  isError: true,
263
577
  };
264
578
  }
265
- const found = res.regions ?? [];
266
- const summary = `Found ${found.length} candidate region(s).\n` +
267
- `Pass any subset to redact_pdf as its "regions" argument (page + x/y/width/height carry over).\n\n` +
268
- (res.note ? `Note: ${res.note}\n\n` : "") +
269
- JSON.stringify(found, null, 2);
270
- return { content: [{ type: "text", text: summary }] };
271
- }
272
- catch (e) {
273
- return {
274
- content: [{ type: "text", text: `Could not scan: ${e.message}` }],
275
- isError: true,
276
- };
277
- }
278
- });
279
- server.tool("locate_text", "Locate exact text strings in a PDF and return each occurrence's page and " +
280
- "coordinates (in PDF points). Use this for what pattern-matching can't catch " +
281
- "— names, addresses, account references — by doing the entity recognition " +
282
- "YOURSELF and passing the literal strings here; the engine resolves where " +
283
- "they sit so you never guess geometry from a rendered page. Feed the returned " +
284
- "regions straight to redact_pdf. Case-insensitive; a string the PDF splits " +
285
- "across text operators may not match as one run. Each hit returns a masked " +
286
- "preview, not the raw text. Nothing is uploaded.", {
287
- pdfPath: z.string().describe("absolute path to the PDF to search"),
288
- texts: z.array(z.string()).min(1).describe("literal strings to locate"),
289
- }, async ({ pdfPath, texts }) => {
290
- try {
291
- const confined = confineInput(pdfPath, ROOT);
292
- const res = ENGINE_BIN
293
- ? await runEngine({ command: "locate_text", pdfPath: confined, texts })
294
- : runWasmLocate(readFileSync(confined), texts);
295
- if (!res?.ok) {
579
+ });
580
+ server.registerTool("locate_text", {
581
+ description: "Locate exact text strings in a PDF and return each occurrence's page and " +
582
+ "coordinates (in PDF points). Use this for what pattern-matching can't catch " +
583
+ "— names, addresses, account references — by doing the entity recognition " +
584
+ "YOURSELF and passing the literal strings here; the engine resolves where " +
585
+ "they sit so you never guess geometry from a rendered page. Feed the returned " +
586
+ "regions straight to redact_pdf. Case-insensitive; a string the PDF splits " +
587
+ "across text operators may not match as one run. Each hit returns a masked " +
588
+ "preview, not the raw text. Nothing is uploaded.",
589
+ inputSchema: {
590
+ pdfPath: z.string().describe("absolute path to the PDF to search"),
591
+ texts: z.array(z.string()).min(1).describe("literal strings to locate"),
592
+ },
593
+ outputSchema: locateTextOutputSchema,
594
+ }, async ({ pdfPath, texts }) => {
595
+ try {
596
+ const confined = confineInput(pdfPath, ROOT);
597
+ const res = ENGINE_BIN
598
+ ? await runEngine({ command: "locate_text", pdfPath: confined, texts })
599
+ : runWasmLocate(readFileSync(confined), texts);
600
+ if (!res?.ok) {
601
+ return {
602
+ content: [{ type: "text", text: `Search failed: ${res?.error ?? "unknown error"}` }],
603
+ isError: true,
604
+ };
605
+ }
606
+ const regions = res.regions ?? [];
607
+ const summary = `Located ${regions.length} occurrence(s). Pass any subset to redact_pdf as "regions".\n\n` +
608
+ JSON.stringify(regions, null, 2);
609
+ const structuredContent = checked(locateTextOutputSchema, {
610
+ count: regions.length,
611
+ regions: regions.map(toOutputRegion),
612
+ });
613
+ return { content: [{ type: "text", text: summary }], structuredContent };
614
+ }
615
+ catch (e) {
296
616
  return {
297
- content: [{ type: "text", text: `Search failed: ${res?.error ?? "unknown error"}` }],
617
+ content: [{ type: "text", text: `Could not search: ${e.message}` }],
298
618
  isError: true,
299
619
  };
300
620
  }
301
- const regions = res.regions ?? [];
302
- const summary = `Located ${regions.length} occurrence(s). Pass any subset to redact_pdf as "regions".\n\n` +
303
- JSON.stringify(regions, null, 2);
304
- return { content: [{ type: "text", text: summary }] };
305
- }
306
- catch (e) {
307
- return {
308
- content: [{ type: "text", text: `Could not search: ${e.message}` }],
309
- isError: true,
310
- };
311
- }
312
- });
313
- server.tool("redact_by_entity", "Find and redact PII in a PDF in ONE on-device step: scan, keep only the " +
314
- "requested entity kinds (email/ssn/phone/card plus the Australian " +
315
- "tfn/abn/acn/medicare/crn/ihi/bsb — omit `kinds` to redact every kind " +
316
- "detected), remove them, verify, and return what was redacted plus the " +
317
- "audit log. This is the 'just scrub the obvious PII' shortcut; when you need " +
318
- "to review before removing, call find_sensitive_regions first. Same caveat as " +
319
- "the detector: matches are best-effort pattern matching, not a completeness " +
320
- "guarantee and not a compliance assessment. Nothing is uploaded.", {
321
- pdfPath: z.string().describe("absolute path to the source PDF"),
322
- outputPath: z.string().describe("absolute path to write the redacted PDF"),
323
- kinds: z
324
- .array(z.enum([
325
- "email",
326
- "ssn",
327
- "phone",
328
- "card",
329
- "tfn",
330
- "abn",
331
- "acn",
332
- "medicare",
333
- "crn",
334
- "ihi",
335
- "bsb",
336
- ]))
337
- .optional()
338
- .describe("entity kinds to redact; omit to redact all detected"),
339
- regions: z
340
- .array(REGIONS)
341
- .optional()
342
- .describe("narrow to these regional packs (intersects with `kinds`); omit for all"),
343
- drawBox: z.boolean().optional().describe("also paint a black bar (default: opaque fill only)"),
344
- }, async ({ pdfPath, outputPath, kinds, regions, drawBox }) => {
345
- try {
346
- const res = await runEngine({
347
- command: "redact_entities",
348
- pdfPath: confineInput(pdfPath, ROOT),
349
- outputPath: confineOutput(outputPath, ROOT),
350
- kinds,
351
- regions,
352
- drawBox,
353
- });
354
- if (!res?.ok) {
621
+ });
622
+ server.registerTool("redact_by_entity", {
623
+ description: "Find and redact PII in a PDF in ONE on-device step: scan, keep only the " +
624
+ "requested entity kinds (email/ssn/phone/card plus the Australian " +
625
+ "tfn/abn/acn/medicare/crn/ihi/bsb — omit `kinds` to redact every kind " +
626
+ "detected), remove them, verify, and return what was redacted plus the " +
627
+ "audit log. This is the 'just scrub the obvious PII' shortcut; when you need " +
628
+ "to review before removing, call find_sensitive_regions first. Same caveat as " +
629
+ "the detector: matches are best-effort pattern matching, not a completeness " +
630
+ "guarantee and not a compliance assessment. Nothing is uploaded.",
631
+ inputSchema: {
632
+ pdfPath: z.string().describe("absolute path to the source PDF"),
633
+ outputPath: z.string().describe("absolute path to write the redacted PDF"),
634
+ kinds: z
635
+ .array(z.enum([
636
+ "email",
637
+ "ssn",
638
+ "phone",
639
+ "card",
640
+ "tfn",
641
+ "abn",
642
+ "acn",
643
+ "medicare",
644
+ "crn",
645
+ "ihi",
646
+ "bsb",
647
+ ]))
648
+ .optional()
649
+ .describe("entity kinds to redact; omit to redact all detected"),
650
+ regions: z
651
+ .array(REGIONS)
652
+ .optional()
653
+ .describe("narrow to these regional packs (intersects with `kinds`); omit for all"),
654
+ drawBox: z.boolean().optional().describe("also paint a black bar (default: opaque fill only)"),
655
+ },
656
+ outputSchema: redactByEntityOutputSchema,
657
+ }, async ({ pdfPath, outputPath, kinds, regions, drawBox }) => {
658
+ try {
659
+ const res = await runEngine({
660
+ command: "redact_entities",
661
+ pdfPath: confineInput(pdfPath, ROOT),
662
+ outputPath: confineOutput(outputPath, ROOT),
663
+ kinds,
664
+ regions,
665
+ drawBox,
666
+ });
667
+ if (!res?.ok) {
668
+ return {
669
+ content: [{ type: "text", text: `Redaction failed: ${res?.error ?? "unknown error"}` }],
670
+ isError: true,
671
+ };
672
+ }
673
+ const redacted = res.redacted ?? [];
674
+ if (redacted.length === 0) {
675
+ const structuredContent = checked(redactByEntityOutputSchema, {
676
+ output: null,
677
+ redactedCount: 0,
678
+ redacted: [],
679
+ verdict: null,
680
+ audit: null,
681
+ });
682
+ return {
683
+ content: [{ type: "text", text: `No matching entities found — nothing was written.` }],
684
+ structuredContent,
685
+ };
686
+ }
687
+ const audit = toOutputAudit(res.audit);
688
+ const redactedOut = redacted.map((r) => ({ page: r.page, kind: r.kind, preview: r.preview }));
689
+ const structuredContent = checked(redactByEntityOutputSchema, {
690
+ output: res.output,
691
+ redactedCount: redactedOut.length,
692
+ redacted: redactedOut,
693
+ verdict: audit.verdict,
694
+ audit,
695
+ });
696
+ return {
697
+ content: [
698
+ {
699
+ type: "text",
700
+ text: `Redacted ${redacted.length} entit${redacted.length === 1 ? "y" : "ies"} → ${res.output}\n` +
701
+ `Verification: ${res.audit?.verdict ?? "see audit log"}\n\n` +
702
+ `Removed:\n${JSON.stringify(redacted, null, 2)}\n\n` +
703
+ `Audit log:\n${JSON.stringify(res.audit, null, 2)}`,
704
+ },
705
+ ],
706
+ structuredContent,
707
+ };
708
+ }
709
+ catch (e) {
355
710
  return {
356
- content: [{ type: "text", text: `Redaction failed: ${res?.error ?? "unknown error"}` }],
711
+ content: [{ type: "text", text: `Could not run redaction: ${e.message}` }],
357
712
  isError: true,
358
713
  };
359
714
  }
360
- const redacted = res.redacted ?? [];
361
- if (redacted.length === 0) {
715
+ });
716
+ server.registerTool("redact_pdf", {
717
+ description: "Permanently redact regions of a PDF on-device using Omitly. Removes the " +
718
+ "underlying text and image data (not a black box over it), verifies nothing " +
719
+ "survives in each region, and returns a signed audit log. The file is never " +
720
+ "uploaded — redaction happens locally.",
721
+ inputSchema: {
722
+ pdfPath: z.string().describe("absolute path to the source PDF"),
723
+ outputPath: z.string().describe("absolute path to write the redacted PDF"),
724
+ regions: z.array(regionSchema).min(1).describe("regions to remove"),
725
+ },
726
+ outputSchema: redactPdfOutputSchema,
727
+ }, async ({ pdfPath, outputPath, regions }) => {
728
+ try {
729
+ const res = await runEngine({
730
+ command: "redact",
731
+ pdfPath: confineInput(pdfPath, ROOT),
732
+ outputPath: confineOutput(outputPath, ROOT),
733
+ regions,
734
+ });
735
+ if (!res?.ok) {
736
+ return {
737
+ content: [{ type: "text", text: `Redaction failed: ${res?.error ?? "unknown error"}` }],
738
+ isError: true,
739
+ };
740
+ }
741
+ const audit = toOutputAudit(res.audit);
742
+ const structuredContent = checked(redactPdfOutputSchema, {
743
+ output: res.output,
744
+ regionCount: regions.length,
745
+ verdict: audit.verdict,
746
+ audit,
747
+ });
362
748
  return {
363
- content: [{ type: "text", text: `No matching entities found — nothing was written.` }],
749
+ structuredContent,
750
+ content: [
751
+ {
752
+ type: "text",
753
+ text: `Redacted ${regions.length} region(s) → ${res.output}\n` +
754
+ `Verification: ${res.audit?.verdict ?? "see audit log"}\n\n` +
755
+ `Audit log:\n${JSON.stringify(res.audit, null, 2)}`,
756
+ },
757
+ ],
364
758
  };
365
759
  }
366
- return {
367
- content: [
368
- {
369
- type: "text",
370
- text: `Redacted ${redacted.length} entit${redacted.length === 1 ? "y" : "ies"} → ${res.output}\n` +
371
- `Verification: ${res.audit?.verdict ?? "see audit log"}\n\n` +
372
- `Removed:\n${JSON.stringify(redacted, null, 2)}\n\n` +
373
- `Audit log:\n${JSON.stringify(res.audit, null, 2)}`,
374
- },
375
- ],
376
- };
377
- }
378
- catch (e) {
379
- return {
380
- content: [{ type: "text", text: `Could not run redaction: ${e.message}` }],
381
- isError: true,
382
- };
383
- }
384
- });
385
- server.tool("redact_pdf", "Permanently redact regions of a PDF on-device using Omitly. Removes the " +
386
- "underlying text and image data (not a black box over it), verifies nothing " +
387
- "survives in each region, and returns a signed audit log. The file is never " +
388
- "uploaded — redaction happens locally.", {
389
- pdfPath: z.string().describe("absolute path to the source PDF"),
390
- outputPath: z.string().describe("absolute path to write the redacted PDF"),
391
- regions: z.array(regionSchema).min(1).describe("regions to remove"),
392
- }, async ({ pdfPath, outputPath, regions }) => {
393
- try {
394
- const res = await runEngine({
395
- command: "redact",
396
- pdfPath: confineInput(pdfPath, ROOT),
397
- outputPath: confineOutput(outputPath, ROOT),
398
- regions,
399
- });
400
- if (!res?.ok) {
760
+ catch (e) {
401
761
  return {
402
- content: [{ type: "text", text: `Redaction failed: ${res?.error ?? "unknown error"}` }],
762
+ content: [{ type: "text", text: `Could not run redaction: ${e.message}` }],
403
763
  isError: true,
404
764
  };
405
765
  }
406
- return {
407
- content: [
408
- {
409
- type: "text",
410
- text: `Redacted ${regions.length} region(s) → ${res.output}\n` +
411
- `Verification: ${res.audit?.verdict ?? "see audit log"}\n\n` +
412
- `Audit log:\n${JSON.stringify(res.audit, null, 2)}`,
413
- },
414
- ],
415
- };
416
- }
417
- catch (e) {
418
- return {
419
- content: [{ type: "text", text: `Could not run redaction: ${e.message}` }],
420
- isError: true,
421
- };
422
- }
423
- });
424
- server.tool("verify_redaction", "Re-scan an already-redacted PDF on-device and confirm nothing recoverable " +
425
- "remains. With a configured native engine and this file's own " +
426
- "`<path>.audit.json` sidecar (written by redact_pdf/redact_by_entity), this " +
427
- "re-checks exactly the regions that were redacted — the strongest form of " +
428
- "this check. Without a native engine (or without that sidecar — e.g. the " +
429
- "file wasn't redacted by this tool), it falls back to a general on-device " +
430
- "re-scan of the whole file and reports whether anything is still " +
431
- "detectable — a good-faith re-check, not a claim of the same rigor as the " +
432
- "sidecar-based path.", {
433
- pdfPath: z.string().describe("absolute path to the redacted PDF to verify"),
434
- }, async ({ pdfPath }) => {
435
- try {
436
- const confined = confineInput(pdfPath, ROOT);
437
- if (ENGINE_BIN) {
438
- const hasSidecar = existsSync(`${confined}.audit.json`);
439
- if (hasSidecar) {
440
- const res = await runEngine({ command: "verify", pdfPath: confined });
766
+ });
767
+ server.registerTool("verify_redaction", {
768
+ description: "Re-scan an already-redacted PDF on-device and confirm nothing recoverable " +
769
+ "remains. With a configured native engine and this file's own " +
770
+ "`<path>.audit.json` sidecar (written by redact_pdf/redact_by_entity), this " +
771
+ "re-checks exactly the regions that were redacted — the strongest form of " +
772
+ "this check. Without a native engine (or without that sidecar — e.g. the " +
773
+ "file wasn't redacted by this tool), it falls back to a general on-device " +
774
+ "re-scan of the whole file and reports whether anything is still " +
775
+ "detectable — a good-faith re-check, not a claim of the same rigor as the " +
776
+ "sidecar-based path.",
777
+ inputSchema: {
778
+ pdfPath: z.string().describe("absolute path to the redacted PDF to verify"),
779
+ },
780
+ outputSchema: verifyRedactionOutputSchema,
781
+ }, async ({ pdfPath }) => {
782
+ try {
783
+ const confined = confineInput(pdfPath, ROOT);
784
+ if (ENGINE_BIN) {
785
+ const hasSidecar = existsSync(`${confined}.audit.json`);
786
+ if (hasSidecar) {
787
+ const res = await runEngine({ command: "verify", pdfPath: confined });
788
+ if (!res?.ok) {
789
+ return {
790
+ content: [
791
+ { type: "text", text: `Could not verify: ${res?.error ?? "unknown error"}` },
792
+ ],
793
+ isError: true,
794
+ };
795
+ }
796
+ // Strict allowlist — this is the one branch that used to forward the
797
+ // raw engine response verbatim (JSON.stringify(res)) with no field
798
+ // filter at all. `findings` only ever carries structural
799
+ // page/reason/class/detail strings, never document text.
800
+ const verdict = res.verdict === "pass" ? "pass" : "fail";
801
+ const findings = [
802
+ ...(res.regions ?? [])
803
+ .filter((r) => r.verification?.result === "fail")
804
+ .map((r) => ({
805
+ page: r.page,
806
+ reason: r.reason,
807
+ detail: r.verification.detail,
808
+ })),
809
+ ...(res.hiddenContent ?? [])
810
+ .filter((c) => c.verification?.result === "fail")
811
+ .map((c) => ({
812
+ class: c.class,
813
+ detail: c.verification.detail,
814
+ })),
815
+ ];
816
+ const structuredContent = checked(verifyRedactionOutputSchema, {
817
+ clean: verdict === "pass",
818
+ mode: "sidecar",
819
+ verdict,
820
+ totalFindings: findings.length,
821
+ findings,
822
+ });
823
+ return {
824
+ content: [{ type: "text", text: JSON.stringify(res, null, 2) }],
825
+ structuredContent,
826
+ isError: res?.ok === false || verdict === "fail",
827
+ };
828
+ }
829
+ }
830
+ // Fallback: general re-scan (wasm if no native engine; the native `find`
831
+ // command otherwise) — no sidecar, so nothing to check regions against,
832
+ // but a non-empty result still means recoverable PII survived.
833
+ const res = ENGINE_BIN
834
+ ? await runEngine({ command: "find", pdfPath: confined })
835
+ : runWasmScan(readFileSync(confined));
836
+ if (!res?.ok) {
441
837
  return {
442
- content: [{ type: "text", text: JSON.stringify(res, null, 2) }],
443
- isError: res?.ok === false || res?.verdict === "fail",
838
+ content: [{ type: "text", text: `Could not verify: ${res?.error ?? "unknown error"}` }],
839
+ isError: true,
444
840
  };
445
841
  }
842
+ const foundRegions = res.regions ?? [];
843
+ // `res.clean`/`res.total_findings` only exist on the WASM scan shape —
844
+ // the native engine's plain "find" command returns just {ok,count,
845
+ // regions} (see the module doc's JSON contract), so falling back to
846
+ // `0` here always read as "clean" regardless of `regions.length` when
847
+ // a native engine was configured. Fall back to the regions actually
848
+ // returned, not a bare 0.
849
+ const clean = res.clean ?? (res.total_findings ?? foundRegions.length) === 0;
850
+ const summary = clean
851
+ ? `✅ No recoverable PII found on the surfaces scanned (general re-scan — no redaction sidecar to check specific regions against).`
852
+ : `⚠️ Found ${res.total_findings ?? foundRegions.length ?? 0} recoverable item(s) — this file is not clean.\n\n${JSON.stringify(res, null, 2)}`;
853
+ const structuredContent = checked(verifyRedactionOutputSchema, {
854
+ clean,
855
+ mode: "rescan",
856
+ totalFindings: res.total_findings ?? foundRegions.length,
857
+ findings: foundRegions.map((r) => ({
858
+ page: r.page,
859
+ kind: r.kind,
860
+ preview: r.preview,
861
+ })),
862
+ });
863
+ return { content: [{ type: "text", text: summary }], structuredContent, isError: !clean };
446
864
  }
447
- // Fallback: general re-scan (wasm if no native engine; the native `find`
448
- // command otherwise) — no sidecar, so nothing to check regions against,
449
- // but a non-empty result still means recoverable PII survived.
450
- const res = ENGINE_BIN
451
- ? await runEngine({ command: "find", pdfPath: confined })
452
- : runWasmScan(readFileSync(confined));
453
- if (!res?.ok) {
865
+ catch (e) {
454
866
  return {
455
- content: [{ type: "text", text: `Could not verify: ${res?.error ?? "unknown error"}` }],
867
+ content: [{ type: "text", text: `Could not verify: ${e.message}` }],
456
868
  isError: true,
457
869
  };
458
870
  }
459
- const clean = res.clean ?? (res.total_findings ?? 0) === 0;
460
- const summary = clean
461
- ? `✅ No recoverable PII found on the surfaces scanned (general re-scan — no redaction sidecar to check specific regions against).`
462
- : `⚠️ Found ${res.total_findings ?? res.regions?.length ?? 0} recoverable item(s) — this file is not clean.\n\n${JSON.stringify(res, null, 2)}`;
463
- return { content: [{ type: "text", text: summary }], isError: !clean };
464
- }
465
- catch (e) {
466
- return {
467
- content: [{ type: "text", text: `Could not verify: ${e.message}` }],
468
- isError: true,
469
- };
470
- }
471
- });
472
- server.tool("create_pdf", "Generate a clean PDF from Markdown (or raw HTML) on-device, rendered through " +
473
- "a real browser engine so it looks printed — not like a script's best guess. " +
474
- "Give it Markdown inline via `source` (or a file via `sourcePath`) and an " +
475
- "`outputPath`; it writes the PDF and returns the path. Use this instead of " +
476
- "writing a one-off reportlab/LaTeX/pandoc script. Nothing is uploaded.", {
477
- outputPath: z.string().describe("absolute path to write the PDF"),
478
- source: z.string().optional().describe("inline Markdown/HTML (omit if using sourcePath)"),
479
- sourcePath: z.string().optional().describe("absolute path to a Markdown/HTML file"),
480
- format: z.enum(["markdown", "html"]).optional().describe("input format (default: markdown)"),
481
- title: z.string().optional().describe("document <title> / metadata"),
482
- css: z.string().optional().describe("extra CSS appended after the default print styles"),
483
- }, async ({ outputPath, source, sourcePath, format, title, css }) => {
484
- try {
485
- const res = await runPdfEngine({
486
- command: "create",
487
- outputPath: confineOutput(outputPath, ROOT),
488
- source,
489
- sourcePath: sourcePath === undefined ? undefined : confineInput(sourcePath, ROOT),
490
- format,
491
- title,
492
- css,
493
- });
494
- if (!res?.ok) {
871
+ });
872
+ server.registerTool("verify_seal", {
873
+ description: "Verify a PDF's embedded Omitly audit report and trailing Ed25519 " +
874
+ "tamper-evidence seal — on-device, nothing uploaded. Distinct from " +
875
+ "`verify_redaction`: that tool re-checks whether redacted regions are " +
876
+ "still empty; this tool cryptographically checks whether the delivered " +
877
+ "bytes have changed since they were sealed. The seal proves INTEGRITY, " +
878
+ "NOT IDENTITY: the signing key is per-install and travels with the file, " +
879
+ "so a valid seal means 'unchanged since sealed by the holder of this " +
880
+ "key', never 'produced by Omitly'. Compare `sealFingerprint` " +
881
+ "out-of-band against the fingerprint the sender published if origin " +
882
+ "matters. " +
883
+ "Requires a configured native engine — there is no wasm seal-verification " +
884
+ "path, so this always needs OMITLY_ENGINE_DIR/OMITLY_REDACT_BIN. A " +
885
+ "`seal_unsupported_version` verdict means this verifier is too old to " +
886
+ "check the seal at all — that is neither a pass nor a fail; update the " +
887
+ "verifier rather than trusting or rejecting the file on that basis.",
888
+ inputSchema: {
889
+ pdfPath: z.string().describe("absolute path to the PDF to check for an Omitly audit report and seal"),
890
+ },
891
+ outputSchema: verifySealOutputSchema,
892
+ }, async ({ pdfPath }) => {
893
+ try {
894
+ const confined = confineInput(pdfPath, ROOT);
895
+ const raw = await runEngine({ command: "verify_seal", pdfPath: confined });
896
+ if (!raw?.ok) {
897
+ return {
898
+ content: [{ type: "text", text: `Could not verify seal: ${raw?.error ?? "unknown error"}` }],
899
+ isError: true,
900
+ };
901
+ }
902
+ const res = normalizeSealResult(raw);
903
+ const structuredContent = checked(verifySealOutputSchema, toOutputSeal(res));
904
+ if (res.verdict === "seal_unsupported_version") {
905
+ return {
906
+ content: [
907
+ {
908
+ type: "text",
909
+ text: `⚠️ This file carries a seal version (${res.sealVersion ?? "unknown"}) this verifier ` +
910
+ "does not implement. Nothing was cryptographically checked — this is " +
911
+ "neither a pass nor a fail. Update the verifier to get a real verdict." +
912
+ (res.carriesAuditReport
913
+ ? " The file DOES carry an Omitly audit report, which raises the stakes: " +
914
+ "an altered-and-relabelled Omitly output can look exactly like this " +
915
+ "to a verifier that's too old to check the seal — treat this as needing " +
916
+ "escalation, not a benign version mismatch."
917
+ : " No Omitly audit report was found alongside it.") +
918
+ `\n\n${JSON.stringify(res, null, 2)}`,
919
+ },
920
+ ],
921
+ structuredContent,
922
+ };
923
+ }
924
+ // Invariant #2 (CLAUDE.md): the seal key is PER-INSTALL and rides inside
925
+ // the artifact, so a valid seal proves the bytes are unchanged since
926
+ // sealing — it does NOT prove Omitly produced them. Saying "produced by
927
+ // Omitly" here would be the exact overclaim `forged_key_seal_verifies_
928
+ // but_fingerprint_differs` exists to prevent, so the pass line states
929
+ // integrity only and points at the fingerprint for out-of-band origin.
930
+ const summary = res.verdict === "verified"
931
+ ? `✅ Seal valid — these bytes are unchanged since they were sealed. That is an ` +
932
+ `INTEGRITY check, not proof of origin: the signing key is per-install and ships ` +
933
+ `with the file, so compare the fingerprint (${res.sealFingerprint ?? "none reported"}) ` +
934
+ `out-of-band against what the sender published if you need to know who sealed it.`
935
+ : `⚠️ Seal verdict: ${res.verdict} — do not treat this file's audit trail as trustworthy.`;
936
+ return {
937
+ content: [{ type: "text", text: `${summary}\n\n${JSON.stringify(res, null, 2)}` }],
938
+ structuredContent,
939
+ isError: isSealErrorVerdict(res.verdict),
940
+ };
941
+ }
942
+ catch (e) {
495
943
  return {
496
- content: [{ type: "text", text: `PDF generation failed: ${res?.error ?? "unknown error"}` }],
944
+ content: [{ type: "text", text: `Could not verify seal: ${e.message}` }],
497
945
  isError: true,
498
946
  };
499
947
  }
500
- return { content: [{ type: "text", text: `Created PDF → ${res.output}` }] };
501
- }
502
- catch (e) {
503
- return {
504
- content: [{ type: "text", text: `Could not generate PDF: ${e.message}` }],
505
- isError: true,
506
- };
507
- }
508
- });
509
- server.tool("check_redaction", "Audit an ALREADY-redacted PDF and report whether sensitive text still survives " +
510
- "underneath the redaction — the 'did my black boxes actually remove the data?' " +
511
- "check. Most tools redact by drawing a rectangle over text while leaving the " +
512
- "characters in the file, where they stay selectable and extractable. This " +
513
- "re-extracts the text on-device and flags any emails, SSNs, phone or card " +
514
- "numbers that are still present, each with a MASKED preview — the raw value " +
515
- "never leaves the machine. It checks the page text layer, text surviving UNDER " +
516
- "redaction marks, incremental-update prior revisions (the classic 'redacted then " +
517
- "saved, original still in the file' failure), document metadata, AcroForm field " +
518
- "values and embedded attachments, and returns a coverage report so a clean result " +
519
- "is scoped to what was inspected. A non-empty result means the redaction leaked. " +
520
- "Nothing is uploaded. (Pattern-based: names/addresses, image-only text, and the " +
521
- "surfaces listed as not-inspected aren't covered; absence of hits isn't proof of " +
522
- "completeness.)", {
523
- pdfPath: z.string().describe("absolute path to the supposedly-redacted PDF to audit"),
524
- }, async ({ pdfPath }) => {
525
- try {
526
- const res = await findViaEngineOrWasm(confineInput(pdfPath, ROOT), undefined);
527
- if (!res?.ok) {
948
+ });
949
+ server.registerTool("create_pdf", {
950
+ description: "Generate a clean PDF from Markdown (or raw HTML) on-device, rendered through " +
951
+ "a real browser engine so it looks printed — not like a script's best guess. " +
952
+ "Give it Markdown inline via `source` (or a file via `sourcePath`) and an " +
953
+ "`outputPath`; it writes the PDF and returns the path. Use this instead of " +
954
+ "writing a one-off reportlab/LaTeX/pandoc script. Nothing is uploaded.",
955
+ inputSchema: {
956
+ outputPath: z.string().describe("absolute path to write the PDF"),
957
+ source: z.string().optional().describe("inline Markdown/HTML (omit if using sourcePath)"),
958
+ sourcePath: z.string().optional().describe("absolute path to a Markdown/HTML file"),
959
+ format: z.enum(["markdown", "html"]).optional().describe("input format (default: markdown)"),
960
+ title: z.string().optional().describe("document <title> / metadata"),
961
+ css: z.string().optional().describe("extra CSS appended after the default print styles"),
962
+ },
963
+ outputSchema: createPdfOutputSchema,
964
+ }, async ({ outputPath, source, sourcePath, format, title, css }) => {
965
+ try {
966
+ const res = await runPdfEngine({
967
+ command: "create",
968
+ outputPath: confineOutput(outputPath, ROOT),
969
+ source,
970
+ sourcePath: sourcePath === undefined ? undefined : confineInput(sourcePath, ROOT),
971
+ format,
972
+ title,
973
+ css,
974
+ });
975
+ if (!res?.ok) {
976
+ return {
977
+ content: [{ type: "text", text: `PDF generation failed: ${res?.error ?? "unknown error"}` }],
978
+ isError: true,
979
+ };
980
+ }
981
+ const structuredContent = checked(createPdfOutputSchema, { output: res.output });
982
+ return { content: [{ type: "text", text: `Created PDF → ${res.output}` }], structuredContent };
983
+ }
984
+ catch (e) {
528
985
  return {
529
- content: [{ type: "text", text: `Audit failed: ${res?.error ?? "unknown error"}` }],
986
+ content: [{ type: "text", text: `Could not generate PDF: ${e.message}` }],
530
987
  isError: true,
531
988
  };
532
989
  }
533
- const regions = res.regions ?? [];
534
- const survivors = res.survivors ?? [];
535
- const offPage = res.off_page ?? [];
536
- const cov = res.coverage ?? {};
537
- const total = res.total_findings ?? regions.length;
538
- const clean = res.clean ?? total === 0;
539
- // Honest coverage disclosure — what was inspected, and what was NOT, so a
540
- // "clean" result is scoped and never reads as "no PII anywhere".
541
- const scanned = [
542
- `${cov.pages_scanned ?? "?"}/${cov.pages_total ?? "?"} page text layer`,
543
- cov.prior_revisions_scanned && "prior (superseded) revisions",
544
- cov.metadata_scanned && "metadata",
545
- cov.acroform_scanned && "form fields",
546
- cov.attachments_scanned && "attachments",
547
- (cov.form_xobjects_scanned ?? 0) > 0 && `${cov.form_xobjects_scanned} Form XObject(s)`,
548
- (cov.annotation_appearances_scanned ?? 0) > 0 &&
549
- `${cov.annotation_appearances_scanned} annotation appearance(s)`,
550
- ].filter(Boolean);
551
- const notScanned = [
552
- ...(cov.not_scanned ?? []),
553
- ...((cov.pages_failed ?? []).length
554
- ? [`${cov.pages_failed.length} page(s) that could not be parsed`]
555
- : []),
556
- ];
557
- const scope = `Scanned: ${scanned.join(", ")}.` +
558
- (notScanned.length ? `\nNOT inspected: ${notScanned.join("; ")}.` : "");
559
- if (clean) {
990
+ });
991
+ server.registerTool("check_redaction", {
992
+ description: "Audit an ALREADY-redacted PDF and report whether sensitive text still survives " +
993
+ "underneath the redaction — the 'did my black boxes actually remove the data?' " +
994
+ "check. Most tools redact by drawing a rectangle over text while leaving the " +
995
+ "characters in the file, where they stay selectable and extractable. This " +
996
+ "re-extracts the text on-device and flags any emails, SSNs, phone or card " +
997
+ "numbers that are still present, each with a MASKED preview — the raw value " +
998
+ "never leaves the machine. It checks the page text layer, text surviving UNDER " +
999
+ "redaction marks, incremental-update prior revisions (the classic 'redacted then " +
1000
+ "saved, original still in the file' failure), document metadata, AcroForm field " +
1001
+ "values and embedded attachments, and returns a coverage report so a clean result " +
1002
+ "is scoped to what was inspected. A non-empty result means the redaction leaked. " +
1003
+ "Nothing is uploaded. (Pattern-based: names/addresses, image-only text, and the " +
1004
+ "surfaces listed as not-inspected aren't covered; absence of hits isn't proof of " +
1005
+ "completeness.)",
1006
+ inputSchema: {
1007
+ pdfPath: z.string().describe("absolute path to the supposedly-redacted PDF to audit"),
1008
+ },
1009
+ outputSchema: checkRedactionOutputSchema,
1010
+ }, async ({ pdfPath }) => {
1011
+ try {
1012
+ const res = await findViaEngineOrWasm(confineInput(pdfPath, ROOT), undefined);
1013
+ if (!res?.ok) {
1014
+ return {
1015
+ content: [{ type: "text", text: `Audit failed: ${res?.error ?? "unknown error"}` }],
1016
+ isError: true,
1017
+ };
1018
+ }
1019
+ const regions = res.regions ?? [];
1020
+ const survivors = res.survivors ?? [];
1021
+ const offPage = res.off_page ?? [];
1022
+ const cov = res.coverage ?? {};
1023
+ const total = res.total_findings ?? regions.length;
1024
+ const clean = res.clean ?? total === 0;
1025
+ // Honest coverage disclosure — what was inspected, and what was NOT, so a
1026
+ // "clean" result is scoped and never reads as "no PII anywhere".
1027
+ const scanned = [
1028
+ `${cov.pages_scanned ?? "?"}/${cov.pages_total ?? "?"} page text layer`,
1029
+ cov.prior_revisions_scanned && "prior (superseded) revisions",
1030
+ cov.metadata_scanned && "metadata",
1031
+ cov.acroform_scanned && "form fields",
1032
+ cov.attachments_scanned && "attachments",
1033
+ (cov.form_xobjects_scanned ?? 0) > 0 && `${cov.form_xobjects_scanned} Form XObject(s)`,
1034
+ (cov.annotation_appearances_scanned ?? 0) > 0 &&
1035
+ `${cov.annotation_appearances_scanned} annotation appearance(s)`,
1036
+ ].filter(Boolean);
1037
+ const notScanned = [
1038
+ ...(cov.not_scanned ?? []),
1039
+ ...((cov.pages_failed ?? []).length
1040
+ ? [`${cov.pages_failed.length} page(s) that could not be parsed`]
1041
+ : []),
1042
+ ];
1043
+ const scope = `Scanned: ${scanned.join(", ")}.` +
1044
+ (notScanned.length ? `\nNOT inspected: ${notScanned.join("; ")}.` : "");
1045
+ const coverage = toOutputCoverage(cov, notScanned);
1046
+ if (clean) {
1047
+ const structuredContent = checked(checkRedactionOutputSchema, {
1048
+ clean: true,
1049
+ totalFindings: 0,
1050
+ byKind: {},
1051
+ regions: [],
1052
+ survivors: [],
1053
+ offPage: [],
1054
+ coverage,
1055
+ });
1056
+ return {
1057
+ content: [
1058
+ {
1059
+ type: "text",
1060
+ text: `✅ No recoverable PII found on the surfaces scanned in ${pdfPath}.\n\n` +
1061
+ `${scope}\n\n` +
1062
+ `This is a scoped result, not a completeness guarantee — the surfaces listed as ` +
1063
+ `NOT inspected, plus names/addresses (which need human or model review), are out of ` +
1064
+ `scope. For a removed-and-verified result with a signed audit log, use the Omitly app.`,
1065
+ },
1066
+ ],
1067
+ structuredContent,
1068
+ };
1069
+ }
1070
+ const byKind = {};
1071
+ for (const r of [...regions, ...survivors, ...offPage])
1072
+ byKind[r.kind] = (byKind[r.kind] ?? 0) + 1;
1073
+ const tally = Object.entries(byKind)
1074
+ .map(([k, n]) => `${n} ${k}${n > 1 ? "s" : ""}`)
1075
+ .join(", ");
1076
+ const parts = [];
1077
+ if (regions.length)
1078
+ parts.push(`Text layer (${regions.length}):\n${JSON.stringify(regions, null, 2)}`);
1079
+ if (survivors.length)
1080
+ parts.push(`Surviving UNDER a redaction mark (${survivors.length}):\n${JSON.stringify(survivors, null, 2)}`);
1081
+ if (offPage.length)
1082
+ parts.push(`Off-page — prior revisions / metadata / form fields / attachments (${offPage.length}):\n` +
1083
+ `${JSON.stringify(offPage, null, 2)}`);
1084
+ const structuredContent = checked(checkRedactionOutputSchema, {
1085
+ clean: false,
1086
+ totalFindings: total,
1087
+ byKind,
1088
+ regions: regions.map(toOutputRegion),
1089
+ survivors: survivors.map(toOutputRegion),
1090
+ offPage: offPage.map((f) => ({ source: f.source, kind: f.kind, preview: f.preview })),
1091
+ coverage,
1092
+ });
560
1093
  return {
561
1094
  content: [
562
1095
  {
563
1096
  type: "text",
564
- text: `✅ No recoverable PII found on the surfaces scanned in ${pdfPath}.\n\n` +
1097
+ text: `⚠️ LEAK: this "redacted" PDF still contains ${total} sensitive item(s) (${tally}) — ` +
1098
+ `the redaction did not actually remove the data.\n\n` +
1099
+ `${parts.join("\n\n")}\n\n` +
565
1100
  `${scope}\n\n` +
566
- `This is a scoped result, not a completeness guarantee — the surfaces listed as ` +
567
- `NOT inspected, plus names/addresses (which need human or model review), are out of ` +
568
- `scope. For a removed-and-verified result with a signed audit log, use the Omitly app.`,
1101
+ `Previews are masked; the raw values stayed on-device. To actually remove this data ` +
1102
+ `(not just cover it) and get an independent verification + signed audit log, redact it ` +
1103
+ `with Omitly — https://omitly.app`,
569
1104
  },
570
1105
  ],
1106
+ structuredContent,
1107
+ isError: true,
571
1108
  };
572
1109
  }
573
- const byKind = {};
574
- for (const r of [...regions, ...survivors, ...offPage])
575
- byKind[r.kind] = (byKind[r.kind] ?? 0) + 1;
576
- const tally = Object.entries(byKind)
577
- .map(([k, n]) => `${n} ${k}${n > 1 ? "s" : ""}`)
578
- .join(", ");
579
- const parts = [];
580
- if (regions.length)
581
- parts.push(`Text layer (${regions.length}):\n${JSON.stringify(regions, null, 2)}`);
582
- if (survivors.length)
583
- parts.push(`Surviving UNDER a redaction mark (${survivors.length}):\n${JSON.stringify(survivors, null, 2)}`);
584
- if (offPage.length)
585
- parts.push(`Off-page — prior revisions / metadata / form fields / attachments (${offPage.length}):\n` +
586
- `${JSON.stringify(offPage, null, 2)}`);
587
- return {
588
- content: [
589
- {
590
- type: "text",
591
- text: `⚠️ LEAK: this "redacted" PDF still contains ${total} sensitive item(s) (${tally}) — ` +
592
- `the redaction did not actually remove the data.\n\n` +
593
- `${parts.join("\n\n")}\n\n` +
594
- `${scope}\n\n` +
595
- `Previews are masked; the raw values stayed on-device. To actually remove this data ` +
596
- `(not just cover it) and get an independent verification + signed audit log, redact it ` +
597
- `with Omitly — https://omitly.app`,
598
- },
599
- ],
600
- isError: true,
601
- };
1110
+ catch (e) {
1111
+ return {
1112
+ content: [{ type: "text", text: `Could not audit: ${e.message}` }],
1113
+ isError: true,
1114
+ };
1115
+ }
1116
+ });
1117
+ } // end registerTools
1118
+ /** Create a fresh `McpServer` with all 8 tools registered — used by both the
1119
+ * bin entry below and tests. */
1120
+ export function createServer() {
1121
+ const server = new McpServer({ name: "omitly-mcp", version: VERSION });
1122
+ registerTools(server);
1123
+ return server;
1124
+ }
1125
+ /**
1126
+ * Thin bin entry: only connect stdio when this file is run directly (`npx
1127
+ * omitly-mcp` / the `bin` entry), never when a test imports it.
1128
+ *
1129
+ * `process.argv[1]` MUST be resolved through `realpathSync` before the
1130
+ * comparison — this is Node's own documented form of the check ("Determining
1131
+ * if a module is the entry point"). npm, npx and `npm link` all install a
1132
+ * `bin` as a SYMLINK (`node_modules/.bin/omitly-mcp` → `../omitly-mcp/dist/
1133
+ * index.js`), and Node resolves `import.meta.url` THROUGH that symlink while
1134
+ * leaving `process.argv[1]` as the unresolved link path. Comparing the two raw
1135
+ * therefore evaluates FALSE on the single most common real invocation
1136
+ * (`npx omitly-mcp`), and the server would exit silently — no stdio
1137
+ * connection, no error, no stderr. Measured on node v25.9.0: raw comparison
1138
+ * false via both an absolute and a relative `.bin`-style symlink, true with
1139
+ * `realpathSync`. `bin-entry.test.ts` runs the BUILT bin through a
1140
+ * `.bin`-style symlink and speaks real MCP stdio to it, so this cannot regress
1141
+ * silently again.
1142
+ */
1143
+ function computeIsMainModule() {
1144
+ const entry = process.argv[1];
1145
+ if (entry === undefined)
1146
+ return false;
1147
+ try {
1148
+ return import.meta.url === pathToFileURL(realpathSync(entry)).href;
602
1149
  }
603
- catch (e) {
604
- return {
605
- content: [{ type: "text", text: `Could not audit: ${e.message}` }],
606
- isError: true,
607
- };
1150
+ catch {
1151
+ // argv[1] isn't a resolvable path (`node --eval`, a deleted file) — then
1152
+ // this module was not the entry point.
1153
+ return false;
608
1154
  }
609
- });
610
- const transport = new StdioServerTransport();
611
- await server.connect(transport);
1155
+ }
1156
+ if (computeIsMainModule()) {
1157
+ const server = createServer();
1158
+ const transport = new StdioServerTransport();
1159
+ await server.connect(transport);
1160
+ }
612
1161
  //# sourceMappingURL=index.js.map