staysfixed 0.9.1 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +182 -0
  2. package/README.md +17 -5
  3. package/docs/getting-started.md +10 -0
  4. package/docs/how-v2-works.md +5 -2
  5. package/package.json +2 -2
  6. package/src/guard/api.js +107 -3
  7. package/src/guard/run.js +154 -20
  8. package/src/report/console.js +235 -17
  9. package/src/report/html.js +75 -19
  10. package/src/types.js +5 -0
  11. package/src/v2/adapters/android-driver.js +62 -12
  12. package/src/v2/adapters/contract.js +18 -4
  13. package/src/v2/adapters/electron.js +96 -14
  14. package/src/v2/adapters/http.js +264 -23
  15. package/src/v2/adapters/ios-driver.js +22 -4
  16. package/src/v2/adapters/ios.js +5 -2
  17. package/src/v2/adapters/isolate.js +78 -5
  18. package/src/v2/adapters/process.js +350 -92
  19. package/src/v2/adapters/web-driver.js +23 -1
  20. package/src/v2/adapters/web.js +42 -3
  21. package/src/v2/adapters/windows.js +32 -15
  22. package/src/v2/check.js +526 -19
  23. package/src/v2/cli.js +345 -3
  24. package/src/v2/cluster.js +112 -4
  25. package/src/v2/coverage.js +293 -8
  26. package/src/v2/detect.js +182 -9
  27. package/src/v2/doctor.js +253 -30
  28. package/src/v2/init.js +102 -10
  29. package/src/v2/mcp/server.js +4 -1
  30. package/src/v2/mcp/tools.js +291 -24
  31. package/src/v2/normalise.js +11 -0
  32. package/src/v2/observation.js +57 -5
  33. package/src/v2/reference.js +133 -14
  34. package/src/v2/refusal.js +389 -0
  35. package/src/v2/remote.js +24 -3
  36. package/src/v2/run.js +306 -16
  37. package/src/v2/sealed.js +14 -2
  38. package/src/v2/ship.js +286 -22
  39. package/src/v2/store.js +101 -2
  40. package/src/v2/types.js +5 -0
  41. package/src/v2/waiver.js +9 -2
  42. package/src/watch/panel.js +12 -1
@@ -32,6 +32,7 @@
32
32
  import fs from 'node:fs';
33
33
  import fsp from 'node:fs/promises';
34
34
  import path from 'node:path';
35
+ import { fileURLToPath } from 'node:url';
35
36
 
36
37
  import { isExpected, messageOf } from '../../core/errors.js';
37
38
  import { findConfigFile, rootForConfig } from '../../core/paths.js';
@@ -75,16 +76,114 @@ import {
75
76
  /**
76
77
  * What every tool call is handed.
77
78
  *
79
+ * `audience` is who is on the other end of this particular call. It is NOT a second set of
80
+ * answers - see THE TWO AUDIENCES below - and nothing in this file is allowed to decide a
81
+ * fact by it.
82
+ *
78
83
  * @typedef {object} ToolContext
79
84
  * @property {string} root
80
85
  * @property {string} cwd
81
86
  * @property {string} version
82
87
  * @property {string} protocolVersion
88
+ * @property {'agent'|'person'} [audience] Who asked. Defaults to an agent: over MCP is what
89
+ * this file is for, and only the command line, which
90
+ * calls these functions directly, says otherwise.
83
91
  */
84
92
 
85
93
  /** @typedef {{type: 'text', text: string}|{type: 'image', data: string, mimeType: string}} ContentItem */
86
94
  /** @typedef {{content: ContentItem[], structuredContent?: Record<string, unknown>, isError?: boolean}} ToolResult */
87
95
 
96
+ // ---------------------------------------------------------------------------
97
+ // THE TWO AUDIENCES
98
+ // ---------------------------------------------------------------------------
99
+
100
+ /**
101
+ * The same truth, in the words the reader can act on.
102
+ *
103
+ * Every tool here assembles its answer exactly once, and that must never change: a person
104
+ * and an agent told different things about the same product is the precise bug this whole
105
+ * tool exists to catch, and shipping it inside the tool itself would be a poor joke. What
106
+ * genuinely differs is the handful of sentences naming a NEXT STEP, because the two readers
107
+ * cannot take the same one. An agent calls `staysfixed_check`. A person types
108
+ * `staysfixed check` and has no MCP tool at all.
109
+ *
110
+ * Measured 2026-08-31, after `coverage`, `explain`, `prove`, `waive` and `intent` were given
111
+ * commands of their own: `staysfixed intent "..." --touches src/total.js` printed, as its
112
+ * closing line, "Now run staysfixed_check." - the name of a tool the reader does not have,
113
+ * cannot type, and will not find in `staysfixed --help`. `explain` offered `include:
114
+ * ["evidence"]`, `prove` asked for `{ "revert": ["..."] }`, and `coverage` sent a person to
115
+ * `staysfixed_capabilities`. Four commands, each ending in an instruction addressed to
116
+ * somebody else.
117
+ *
118
+ * So: facts are written once, and any sentence naming a step asks this phrasebook for the
119
+ * words. Nothing in here may change what is true - only how it is reached. The one field
120
+ * that is not wording is `by`, and it is here for the same reason: it records which of these
121
+ * two actually made the call, and it has to agree with everything else in the reply.
122
+ *
123
+ * @typedef {object} Voice
124
+ * @property {'agent'|'person'} who
125
+ * @property {boolean} isPerson
126
+ * @property {string} by What goes on the record as who did this.
127
+ * @property {string} check How this reader runs a check.
128
+ * @property {string} capabilities How this reader asks what the machine can do.
129
+ * @property {string} explainCall A whole example call to explain one finding.
130
+ * @property {string} proveCall A whole example call to prove a cause.
131
+ * @property {string} revertArg Just the "what to put back" part of that call.
132
+ * @property {string} waiveCall A whole example call to record one as intended.
133
+ * @property {string} askForEvidence
134
+ * @property {string} askForMachines
135
+ * @property {string} budgetEnds What happens when the five run out, said to this reader.
136
+ */
137
+
138
+ /** @type {Record<'agent'|'person', Voice>} */
139
+ const VOICES = {
140
+ agent: {
141
+ who: 'agent',
142
+ isPerson: false,
143
+ by: 'an agent, over MCP',
144
+ check: 'staysfixed_check',
145
+ capabilities: 'staysfixed_capabilities',
146
+ explainCall: '{ "finding": "f-a1b2c3" }',
147
+ proveCall: '{ "finding": "f-a1b2c3", "revert": ["src/checkout/total.js"] }',
148
+ revertArg: '{ "revert": ["src/checkout/total.js"] }',
149
+ waiveCall: '{ "finding": "f-a1b2c3", "because": "..." }',
150
+ askForEvidence: 'Ask with include: ["evidence"].',
151
+ askForMachines: 'A person can run `staysfixed doctor --machines` to ask',
152
+ budgetEnds: 'before a person has to look',
153
+ },
154
+ person: {
155
+ who: 'person',
156
+ isPerson: true,
157
+ by: 'a person, at the command line',
158
+ check: 'staysfixed check',
159
+ capabilities: 'staysfixed doctor',
160
+ explainCall: 'staysfixed explain f-a1b2c3',
161
+ proveCall: 'staysfixed prove f-a1b2c3 --revert src/checkout/total.js',
162
+ revertArg: '--revert src/checkout/total.js',
163
+ waiveCall: 'staysfixed waive f-a1b2c3 --because "..."',
164
+ askForEvidence: 'Add --evidence to see it.',
165
+ askForMachines: 'Run `staysfixed doctor --machines` to ask',
166
+ // "before a person has to look" is the agent's version of this, and read by the person
167
+ // themselves it says nothing at all - they ARE the person, sitting right there.
168
+ budgetEnds: 'before this refuses and the rest have to be fixed rather than recorded',
169
+ },
170
+ };
171
+
172
+ /**
173
+ * Which of the two asked.
174
+ *
175
+ * Anything that did not say is a program: over MCP is what this file is for, and the only
176
+ * caller that is not speaking MCP is `askTheToolSet` in src/v2/cli.js, which says so.
177
+ * Defaulting the other way would put "a person, at the command line" on records sealed by
178
+ * an agent, which is the same untrue record in the opposite direction.
179
+ *
180
+ * @param {ToolContext} ctx
181
+ * @returns {Voice}
182
+ */
183
+ export function voiceFor(ctx) {
184
+ return ctx?.audience === 'person' ? VOICES.person : VOICES.agent;
185
+ }
186
+
88
187
  // ---------------------------------------------------------------------------
89
188
  // Constants that are policy, not preference
90
189
  // ---------------------------------------------------------------------------
@@ -210,16 +309,21 @@ export async function loadEngine(refresh = false) {
210
309
  }
211
310
 
212
311
  /**
213
- * What an agent gets when it asks for something only the engine can do and the
312
+ * What a caller gets when it asks for something only the engine can do and the
214
313
  * engine is not there. Written as instructions to whoever is integrating,
215
314
  * because that is the only person who will ever read it.
216
315
  *
316
+ * The last line names a next step, so it is the one line that has to know who is reading:
317
+ * `staysfixed_capabilities` is a tool a person at a terminal does not have, and `staysfixed
318
+ * doctor` is the same question asked in the words they can type.
319
+ *
217
320
  * @param {Engine} engine
218
321
  * @param {string} part
219
322
  * @param {string} needs The exact signature the missing function must have.
323
+ * @param {Voice} voice
220
324
  * @returns {ToolResult}
221
325
  */
222
- function engineMissing(engine, part, needs) {
326
+ function engineMissing(engine, part, needs, voice) {
223
327
  const names = (/** @type {Record<string, string[]>} */ (ENGINE_PARTS)[part] ?? []).join(' or ');
224
328
  return problem(
225
329
  [
@@ -230,7 +334,7 @@ function engineMissing(engine, part, needs) {
230
334
  '',
231
335
  `It needs: ${needs}`,
232
336
  '',
233
- 'Everything else still works. Call staysfixed_capabilities for what this copy can do.',
337
+ `Everything else still works. ${voice.isPerson ? 'Run' : 'Call'} ${voice.capabilities} for what this copy can do.`,
234
338
  ].join('\n')
235
339
  );
236
340
  }
@@ -362,7 +466,8 @@ export function toolDefinitions() {
362
466
  touches: {
363
467
  type: 'array',
364
468
  items: { type: 'string' },
365
- description: 'Files, folders or named areas you expect this to affect, e.g. ["src/checkout/total.js", "the basket page"]. A difference outside this list cannot be waived.',
469
+ description:
470
+ 'Files, folders or named areas you expect this to affect, e.g. ["src/checkout/total.js", "the basket page"]. A difference outside this list cannot be waived. Paths may be relative to the project or absolute - an absolute path inside the project is stored relative to it, and the reply says exactly what was sealed.',
366
471
  },
367
472
  expect: { type: 'array', items: { type: 'string' }, description: 'Differences you expect this change to produce, in your own words. Optional, and it makes the check sharper.' },
368
473
  },
@@ -468,7 +573,10 @@ export function toolDefinitions() {
468
573
  'What was NOT checked. The ways in that no journey has ever opened, the surfaces this machine cannot reach at all, anything refused because doing it twice would not have been reversible, and the things this tool can never see on any machine. Read it before you tell anyone a change is safe: a clean check only covers what was walked, and this is the list of what was not.',
469
574
  inputSchema: {
470
575
  type: 'object',
471
- properties: { format: { type: 'string', enum: ['text', 'json'] } },
576
+ properties: {
577
+ format: { type: 'string', enum: ['text', 'json'] },
578
+ offline: { type: 'boolean', description: 'Do not look for any other machine at all. Faster, and then this answer cannot say anything about remote runners either way.' },
579
+ },
472
580
  additionalProperties: false,
473
581
  },
474
582
  },
@@ -684,6 +792,67 @@ const RESULT_SHAPES = [
684
792
  // intent
685
793
  // ---------------------------------------------------------------------------
686
794
 
795
+ /**
796
+ * The name of a file, said the way the rest of the tool says it.
797
+ *
798
+ * An agent holds ABSOLUTE paths. That is what its own editing tools hand it and that is
799
+ * what it types back here, so `touches` arrives as `/Users/…/tiny/cli.js` while every
800
+ * address, every source file and every intent comparison downstream is written relative to
801
+ * the project. Nothing lines those two up, and the gate that decides whether a difference
802
+ * was declared fell straight through to "one word in common is not a match".
803
+ *
804
+ * Measured on a product with ONE file in it: seal `/var/…/tiny/cli.js`, edit that same
805
+ * file, run a check, and `staysfixed_waive` answered "Refused. This is outside what you
806
+ * sealed… Only the word 'cli' lines up." The one file in the product, the file the agent
807
+ * had just edited, reported as something it never said it was touching. Naming the same
808
+ * file `cli.js` was accepted. So the two spellings meant opposite things, and the one an
809
+ * agent naturally reaches for was the broken one.
810
+ *
811
+ * Only a real path that really sits inside this project is rewritten. Everything else is
812
+ * passed through exactly as written, because `touches` also takes plain areas — "the
813
+ * basket page" — and mangling those would break the match it is there to make.
814
+ *
815
+ * @param {string} named What the agent called it.
816
+ * @param {string} root The project root.
817
+ * @returns {string}
818
+ */
819
+ function insideTheProject(named, root) {
820
+ let where = named;
821
+ if (where.startsWith('file://')) {
822
+ try {
823
+ where = fileURLToPath(where);
824
+ } catch {
825
+ return named;
826
+ }
827
+ }
828
+ if (!path.isAbsolute(where)) return named;
829
+
830
+ /**
831
+ * @param {string} from
832
+ * @param {string} to
833
+ * @returns {string|null}
834
+ */
835
+ const under = (from, to) => {
836
+ const rel = path.relative(from, to);
837
+ return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel) ? rel : null;
838
+ };
839
+
840
+ const plain = under(root, where);
841
+ if (plain) return plain;
842
+
843
+ // The same folder under two names. On a Mac `/tmp` is a link to `/private/tmp`, so the
844
+ // path an agent reports and the root this tool worked out can be the identical folder
845
+ // spelled two ways, and comparing the text alone says they are unrelated.
846
+ try {
847
+ const real = under(fs.realpathSync(root), fs.realpathSync(where));
848
+ if (real) return real;
849
+ } catch {
850
+ // A path that does not exist cannot be resolved, and that is not an error here: an
851
+ // agent may name a file it is about to create. Its own words stand.
852
+ }
853
+ return named;
854
+ }
855
+
687
856
  /**
688
857
  * @param {ToolContext} ctx
689
858
  * @param {Record<string, any>} input
@@ -691,7 +860,7 @@ const RESULT_SHAPES = [
691
860
  */
692
861
  async function toolIntent(ctx, input) {
693
862
  const summary = text(input.summary);
694
- const touches = stringList(input.touches) ?? [];
863
+ const touches = (stringList(input.touches) ?? []).map((t) => insideTheProject(t, ctx.root));
695
864
  const expect = stringList(input.expect) ?? [];
696
865
 
697
866
  if (!summary) return problem('Say what you meant to change, in one plain sentence: { "summary": "...", "touches": ["..."] }.');
@@ -703,11 +872,20 @@ async function toolIntent(ctx, input) {
703
872
 
704
873
  const store = storeFor(ctx);
705
874
  const product = await productFor(ctx.root);
875
+ const voice = voiceFor(ctx);
706
876
 
707
877
  // src/v2/intent.js does the sealing, and it does more than write a file down: it
708
878
  // fingerprints the working tree at this moment, so whether the intent was written before
709
879
  // the edits or after them stops being a promise and becomes something anybody can check.
710
- const intent = await sealIntent(store, { product, summary, touches, expect, by: 'an agent, over MCP' });
880
+ //
881
+ // WHO SEALED IT IS PART OF THE EVIDENCE. `by` used to be the constant string 'an agent,
882
+ // over MCP' whoever called, so from the day `staysfixed intent` became a command every
883
+ // intent a person sealed at their own terminal went on the record as an agent's - measured
884
+ // 2026-08-31, the first CLI run wrote exactly that into
885
+ // .staysfixed/v2/intents/<product>.json. This record is what a waiver is judged against
886
+ // months later, by somebody deciding whether a claim of "I meant that" is worth anything,
887
+ // and a record naming the wrong actor is a record nobody can use.
888
+ const intent = await sealIntent(store, { product, summary, touches, expect, by: voice.by });
711
889
 
712
890
  // Sealing a new intent does NOT hand out a fresh five. The budget is counted
713
891
  // against the reference, precisely so an agent that has spent its waivers
@@ -726,8 +904,8 @@ async function toolIntent(ctx, input) {
726
904
  expect.length ? `Expecting to see: ${expect.join('; ')}.` : '',
727
905
  intent.ordering,
728
906
  '',
729
- `You may waive at most ${WAIVER_BUDGET} differences before a person has to look, and ${spent} of those are already spent since the last time a build shipped. Sealing another intent does not give you more, and you can only waive a difference that falls inside what you just named.`,
730
- 'Now run staysfixed_check.',
907
+ `You may waive at most ${WAIVER_BUDGET} differences ${voice.budgetEnds}, and ${spent} of those are already spent since the last time a build shipped. Sealing another intent does not give you more, and you can only waive a difference that falls inside what you just named.`,
908
+ `Now run ${voice.check}.`,
731
909
  ]
732
910
  .filter(Boolean)
733
911
  .join('\n'),
@@ -749,7 +927,7 @@ async function toolCheck(ctx, input) {
749
927
  const engine = await loadEngine();
750
928
  const run = engine.parts.check;
751
929
  if (!run) {
752
- return engineMissing(engine, 'check', 'check({cwd, configFile, against, paired, journeys, only}) returning a CheckResult - the shape at the top of src/v2/cli.js.');
930
+ return engineMissing(engine, 'check', 'check({cwd, configFile, against, paired, journeys, only}) returning a CheckResult - the shape at the top of src/v2/cli.js.', voiceFor(ctx));
753
931
  }
754
932
 
755
933
  const store = storeFor(ctx);
@@ -1257,11 +1435,12 @@ function renderFinding(f) {
1257
1435
  * @returns {Promise<ToolResult>}
1258
1436
  */
1259
1437
  async function toolExplain(ctx, input) {
1438
+ const voice = voiceFor(ctx);
1260
1439
  const id = text(input.finding);
1261
- if (!id) return problem('Say which finding to explain, e.g. { "finding": "f-a1b2c3" }. The ids come from staysfixed_check.');
1440
+ if (!id) return problem(`Say which finding to explain, e.g. ${voice.explainCall}. The ids come from ${voice.check}.`);
1262
1441
 
1263
1442
  const last = await readCheckRecord(storeFor(ctx));
1264
- if (!last) return problem('No check has run in this copy yet, so there is nothing to explain. Run staysfixed_check first.');
1443
+ if (!last) return problem(`No check has run in this copy yet, so there is nothing to explain. Run ${voice.check} first.`);
1265
1444
  const f = last.findings.find((x) => x.id === id);
1266
1445
  if (!f) {
1267
1446
  const ids = last.findings.slice(0, 12).map((x) => x.id);
@@ -1331,7 +1510,7 @@ async function toolExplain(ctx, input) {
1331
1510
  }
1332
1511
  } else if (f.evidence) {
1333
1512
  out.push('');
1334
- out.push('Evidence was kept and not sent. Ask with include: ["evidence"].');
1513
+ out.push(`Evidence was kept and not shown. ${voice.askForEvidence}`);
1335
1514
  }
1336
1515
 
1337
1516
  if (deep?.error) {
@@ -1354,7 +1533,15 @@ async function toolExplain(ctx, input) {
1354
1533
  content.push({ type: 'image', data: png.toString('base64'), mimeType: 'image/png' });
1355
1534
  }
1356
1535
  } else if (pictures.length) {
1357
- content.push({ type: 'text', text: `${pictures.length} picture(s) were kept as evidence and not sent. Ask with include: ["pixels"] if a picture would settle it.` });
1536
+ // A terminal cannot be handed a PNG, so "ask with include: [\"pixels\"]" is not an
1537
+ // instruction a person can carry out - and telling them a picture exists without saying
1538
+ // WHERE leaves them worse off than saying nothing. They get the paths and open them.
1539
+ content.push({
1540
+ type: 'text',
1541
+ text: voice.isPerson
1542
+ ? `${pictures.length} picture(s) were kept as evidence: ${pictures.slice(0, MAX_IMAGES).join(', ')}${pictures.length > MAX_IMAGES ? `, and ${pictures.length - MAX_IMAGES} more` : ''}. Open them if a picture would settle it.`
1543
+ : `${pictures.length} picture(s) were kept as evidence and not sent. Ask with include: ["pixels"] if a picture would settle it.`,
1544
+ });
1358
1545
  }
1359
1546
 
1360
1547
  return { content };
@@ -1383,24 +1570,26 @@ function picturesFrom(f, deep) {
1383
1570
  * @returns {Promise<ToolResult>}
1384
1571
  */
1385
1572
  async function toolProve(ctx, input) {
1573
+ const voice = voiceFor(ctx);
1386
1574
  const engine = await loadEngine();
1387
1575
  const run = engine.parts.prove;
1388
1576
  if (!run) {
1389
1577
  return engineMissing(
1390
1578
  engine,
1391
1579
  'prove',
1392
- 'prove({cwd, configFile, finding, revert}) returning {gone: boolean, detail?: string}. src/v2/cause.js already has proveCause(), but it takes an engine-internal finding and a loaded project, which this surface does not have - a small facade in src/v2/check.js is all that is needed.'
1580
+ 'prove({cwd, configFile, finding, revert}) returning {gone: boolean, detail?: string}. src/v2/cause.js already has proveCause(), but it takes an engine-internal finding and a loaded project, which this surface does not have - a small facade in src/v2/check.js is all that is needed.',
1581
+ voice
1393
1582
  );
1394
1583
  }
1395
1584
 
1396
1585
  const id = text(input.finding);
1397
1586
  const revert = stringList(input.revert);
1398
- if (!id) return problem('Say which finding you are trying to explain, e.g. { "finding": "f-a1b2c3", "revert": ["src/total.js"] }.');
1399
- if (!revert) return problem('Name what to put back to the reference for one run, e.g. { "revert": ["src/checkout/total.js"] }. Without that there is no claim to test.');
1587
+ if (!id) return problem(`Say which finding you are trying to explain, e.g. ${voice.proveCall}.`);
1588
+ if (!revert) return problem(`Name what to put back to the reference for one run, e.g. ${voice.revertArg}. Without that there is no claim to test.`);
1400
1589
 
1401
1590
  const last = await readCheckRecord(storeFor(ctx));
1402
1591
  const f = last?.findings.find((x) => x.id === id);
1403
- if (!f) return problem(`The last check has no finding called "${id}". Run staysfixed_check first, then prove one of the ids it gives you.`);
1592
+ if (!f) return problem(`The last check has no finding called "${id}". Run ${voice.check} first, then prove one of the ids it gives you.`);
1404
1593
 
1405
1594
  /** @type {any} */
1406
1595
  const result = (await run({ cwd: ctx.root, finding: id, revert })) ?? {};
@@ -1441,14 +1630,15 @@ async function toolProve(ctx, input) {
1441
1630
  * @returns {Promise<ToolResult>}
1442
1631
  */
1443
1632
  async function toolWaive(ctx, input) {
1633
+ const voice = voiceFor(ctx);
1444
1634
  const id = text(input.finding);
1445
1635
  const because = text(input.because);
1446
- if (!id) return problem('Say which finding, e.g. { "finding": "f-a1b2c3", "because": "..." }.');
1636
+ if (!id) return problem(`Say which finding, e.g. ${voice.waiveCall}.`);
1447
1637
  if (!because) return problem('Say why this difference is what you meant, in one plain sentence. A waiver with no reason is worth nothing to whoever reads it later.');
1448
1638
 
1449
1639
  const store = storeFor(ctx);
1450
1640
  const last = await readCheckRecord(store);
1451
- if (!last) return problem('No check has run in this copy yet, so there is no difference to waive. Run staysfixed_check first.');
1641
+ if (!last) return problem(`No check has run in this copy yet, so there is no difference to waive. Run ${voice.check} first.`);
1452
1642
  const f = last.findings.find((x) => x.id === id);
1453
1643
  if (!f) return problem(`The last check has no finding called "${id}". You can only waive something this tool actually reported.`);
1454
1644
 
@@ -1464,7 +1654,13 @@ async function toolWaive(ctx, input) {
1464
1654
  finding: f,
1465
1655
  why: because,
1466
1656
  check: { at: last.at, runId: last.result?.runId },
1467
- by: 'an agent, over MCP',
1657
+ // Same reason as the intent above: from the day `staysfixed waive` became a command,
1658
+ // a constant here would have put an agent's name on every waiver a person recorded at
1659
+ // their own terminal, and a waiver is read months later by somebody weighing exactly
1660
+ // that. `audience` rides along so the sealed-class refusal is worded at whoever is
1661
+ // actually reading it - see sayRefusal in src/v2/sealed.js.
1662
+ by: voice.by,
1663
+ audience: voice.who,
1468
1664
  });
1469
1665
 
1470
1666
  // A refusal is the tool working, not the tool being difficult, and the wording is the
@@ -1496,11 +1692,22 @@ async function toolWaive(ctx, input) {
1496
1692
  * here at all. A report missing either half would let somebody read "everything
1497
1693
  * walked" and believe the product was covered when the phone was never touched.
1498
1694
  *
1695
+ * THE SURVEY IS THE ONE THE PERSON GETS. This call used to ask for it with `offline: true`
1696
+ * written into the code, and then print the answer under the words "Cannot be reached from
1697
+ * this machine at all". Offline means the ssh config is not even READ, so on a Mac with two
1698
+ * machines named in it the agent was told, flatly, "No Windows desktop is reachable from
1699
+ * here", while `staysfixed doctor` on the same Mac names both and says nothing is known
1700
+ * about them either way. Forcing the answer and then reporting the forced answer as a fact
1701
+ * about the world is the exact failure this tool exists to catch, so the question is now
1702
+ * asked the same way `staysfixed_capabilities` and `doctor` ask it, and a caller who wants
1703
+ * no network at all says so.
1704
+ *
1499
1705
  * @param {ToolContext} ctx
1500
1706
  * @param {Record<string, any>} input
1501
1707
  * @returns {Promise<ToolResult>}
1502
1708
  */
1503
1709
  async function toolCoverage(ctx, input) {
1710
+ const voice = voiceFor(ctx);
1504
1711
  const engine = await loadEngine();
1505
1712
  const last = await readCheckRecord(storeFor(ctx));
1506
1713
 
@@ -1508,7 +1715,7 @@ async function toolCoverage(ctx, input) {
1508
1715
  let caps = null;
1509
1716
  if (engine.parts.capabilities) {
1510
1717
  try {
1511
- caps = await engine.parts.capabilities({ cwd: ctx.root, offline: true });
1718
+ caps = await engine.parts.capabilities({ cwd: ctx.root, offline: input.offline === true });
1512
1719
  } catch {
1513
1720
  // The machine survey is one half of the answer, not the whole of it. Losing
1514
1721
  // it must not lose the half that came from the run.
@@ -1517,12 +1724,40 @@ async function toolCoverage(ctx, input) {
1517
1724
  }
1518
1725
 
1519
1726
  const coverage = last?.result?.coverage ?? null;
1520
- const unreachable = (caps?.surfaces ?? []).filter((/** @type {any} */ s) => s.status === 'unavailable');
1727
+ const unavailable = (caps?.surfaces ?? []).filter((/** @type {any} */ s) => s.status === 'unavailable');
1728
+ // Two completely different sentences were being printed under one heading. "There is no
1729
+ // Android app in this repository" is not "this machine cannot reach an Android device" —
1730
+ // the first is nothing to check, the second is a hole in the coverage — and `doctor`
1731
+ // already tells a person those apart in so many words. The agent was told both were the
1732
+ // machine's fault, which reads as a crippled install rather than a project that simply
1733
+ // has no phone app in it.
1734
+ const nothingOfThatKind = unavailable.filter((/** @type {any} */ s) => s.notInThisProject === true);
1735
+ const unreachable = unavailable.filter((/** @type {any} */ s) => s.notInThisProject !== true);
1521
1736
  const partial = (caps?.surfaces ?? []).filter((/** @type {any} */ s) => s.status === 'partial');
1737
+ // Machines the survey NAMED and never asked. Nothing above rests on having asked them,
1738
+ // and saying so is the difference between "there is no Windows desktop" and "nobody
1739
+ // knocked". The command that knocks belongs to a person, not to an agent: dialling
1740
+ // somebody's ssh config is what `doctor` deliberately stopped doing unasked.
1741
+ const notDialled = (caps?.hosts ?? []).filter((/** @type {any} */ h) => h.reachable !== true && /not dialled/i.test(String(h.how ?? '')));
1742
+ // And the other way the machine list can come back empty: somebody asked for no network
1743
+ // at all, here or in STAYSFIXED_OFFLINE, and then the ssh config is not even read. An
1744
+ // empty list looks identical to "there are no other machines", so which of the two it is
1745
+ // has to be said. That is the whole defect this call had: force the answer, then report
1746
+ // the forced answer as a finding about the world.
1747
+ const lookedForNoMachines = input.offline === true || process.env.STAYSFIXED_OFFLINE !== undefined;
1522
1748
 
1523
1749
  if (input.format === 'json') {
1524
1750
  const payload = {
1525
1751
  lastCheckAt: last?.at ?? null,
1752
+ // A cold start has no numbers to give, and nulls beside an empty `unopened` list read
1753
+ // like a product with no holes in it. This says which of the two it is in one field.
1754
+ anyCheckHasRun: last !== null,
1755
+ note:
1756
+ last === null
1757
+ ? 'No check has run in this copy yet, so nothing at all has been covered and every count below is empty because nothing was measured, not because nothing was missed.'
1758
+ : coverage
1759
+ ? null
1760
+ : 'The last run did not report what it covered, so how deep it went is unknown. Treat its clean result with suspicion.',
1526
1761
  covers: caps?.covers ?? null,
1527
1762
  walked: coverage?.journeys ?? null,
1528
1763
  doorsKnown: coverage?.doorsKnown ?? null,
@@ -1540,6 +1775,15 @@ async function toolCoverage(ctx, input) {
1540
1775
  .filter((/** @type {{doors?: number}} */ g) => typeof g.doors !== 'number')
1541
1776
  .map((/** @type {{what: string, why?: string, unlockedBy?: string}} */ g) => ({ what: g.what, why: g.why ?? null, unlockedBy: g.unlockedBy ?? null })),
1542
1777
  surfacesOutOfReach: unreachable.map((/** @type {any} */ s) => ({ name: s.name, why: s.summary, needs: s.needs })),
1778
+ // Kept apart from `surfacesOutOfReach` on purpose: nothing of this kind exists in the
1779
+ // project, so there is nothing here to check and it is not a limit of the machine.
1780
+ surfacesNotInThisProject: nothingOfThatKind.map((/** @type {any} */ s) => ({ name: s.name, why: s.summary })),
1781
+ // Named in the ssh config, never asked. Whatever is in `surfacesOutOfReach` above,
1782
+ // none of it rests on having knocked on these.
1783
+ machinesNotDialled: notDialled.map((/** @type {any} */ h) => ({ name: h.name, why: h.how })),
1784
+ // True when this answer was taken without looking for other machines at all, so
1785
+ // `machinesNotDialled` being empty means nobody looked, not that there are none.
1786
+ machinesNotLookedFor: lookedForNoMachines,
1543
1787
  surfacesPartial: partial.map((/** @type {any} */ s) => ({ name: s.name, why: s.summary })),
1544
1788
  neverVisible: caps?.limits ?? null,
1545
1789
  };
@@ -1604,6 +1848,12 @@ async function toolCoverage(ctx, input) {
1604
1848
  for (const d of noAdapter) out.push(`- ${d.surface}: ${d.why}`);
1605
1849
  }
1606
1850
 
1851
+ if (nothingOfThatKind.length) {
1852
+ out.push('');
1853
+ out.push('There is nothing of these kinds in this project, so there was nothing here to check and this is not a limit of the machine:');
1854
+ for (const s of nothingOfThatKind) out.push(`- ${s.name}: ${s.summary}`);
1855
+ }
1856
+
1607
1857
  if (unreachable.length) {
1608
1858
  out.push('');
1609
1859
  out.push('Cannot be reached from this machine at all, so nothing there has been checked by anything:');
@@ -1612,9 +1862,26 @@ async function toolCoverage(ctx, input) {
1612
1862
  for (const need of s.needs ?? []) out.push(` it would take: ${need.fix ?? need.what}`);
1613
1863
  }
1614
1864
  }
1865
+
1866
+ // Said even when nothing above needed a second machine. A machine quietly left out of
1867
+ // this answer is a runner somebody may be looking for, and "no Windows desktop is
1868
+ // reachable from here" printed over an ssh config naming two of them is not a finding
1869
+ // about the world — it is a question nobody asked.
1870
+ if (lookedForNoMachines) {
1871
+ out.push('');
1872
+ out.push(
1873
+ 'This answer was taken without looking for any other machine, so nothing above is a statement about what this machine can reach. Ask again without offline to find out.'
1874
+ );
1875
+ } else if (notDialled.length) {
1876
+ out.push('');
1877
+ out.push(
1878
+ `${notDialled.length} ${notDialled.length === 1 ? 'machine is named in your ssh config and was' : 'machines are named in your ssh config and were'} not dialled, so nothing above rests on having asked ${notDialled.length === 1 ? 'it' : 'them'}: ${notDialled.map((/** @type {any} */ h) => h.name).join(', ')}. ${voice.askForMachines}; this tool does not connect to anybody's machines on its own.`
1879
+ );
1880
+ }
1881
+
1615
1882
  if (partial.length) {
1616
1883
  out.push('');
1617
- out.push(`Reachable, but not everything on them can be watched: ${partial.map((/** @type {any} */ s) => s.name).join(', ')}. staysfixed_capabilities says what each limit is.`);
1884
+ out.push(`Reachable, but not everything on them can be watched: ${partial.map((/** @type {any} */ s) => s.name).join(', ')}. ${voice.capabilities} says what each limit is.`);
1618
1885
  }
1619
1886
 
1620
1887
  if (Array.isArray(caps?.limits) && caps.limits.length) {
@@ -114,6 +114,17 @@ export const DEFAULT_RULES = [
114
114
  pattern: '\\b[0-9a-fA-F]{16,31}\\b',
115
115
  with: '<hex>',
116
116
  },
117
+ {
118
+ id: 'asset.bundled',
119
+ kind: 'replace',
120
+ what: 'The hash in a bundler\'s own asset filename — /_next/static/chunks/main-9f2c1a.js, /assets/index-4b8e21.css, app.7d3f9a1c.js.',
121
+ why:
122
+ 'A bundler renames its output whenever the source changes, so editing one line renames several files. Measured on a Next.js app: one source edit produced four extra findings, all of them chunk filenames, and the change a person actually made was underneath them. The rename is not news — the code change is, and that is reported on its own.',
123
+ wouldHide:
124
+ 'A deliberate change to an asset filename, which nobody makes by hand. Deliberately narrow: the hash is only taken inside a path that a bundler owns, so a content hash anywhere else still changes and still shows — see id.hex, which leaves 32-and-longer hex alone for exactly that reason.',
125
+ pattern: '(/_next/static/[^"\'\\s]*?|/assets/[^"\'\\s]*?|\\b[\\w.-]+)[.-][0-9a-fA-F]{6,}(\\.(?:js|mjs|css|map))',
126
+ with: '$1.<asset>$2',
127
+ },
117
128
  {
118
129
  id: 'id.pid',
119
130
  kind: 'replace',
@@ -879,13 +879,43 @@ export function wobbleStorm(wobble) {
879
879
  * and wobbles now is a finding in itself: the change made something unpredictable. Nothing is
880
880
  * "wrong" at that address and it still needs fixing.
881
881
  *
882
+ * AND THE CASE WHERE THERE IS NOTHING TO SUBTRACT FROM, because there was no change. When
883
+ * nobody has edited anything, the build being checked and the build on record as working are
884
+ * the same build — the same id, the same folder of stored runs — and the record of "the old
885
+ * build" is read back as the newest run in that folder, which is the run the LAST check left
886
+ * there. So an everyday check on an untouched tree compares this run against the previous run
887
+ * of one build and reports whatever flickered between them as a change nobody asked for.
888
+ *
889
+ * Measured 2026-08-31 on a stock Next.js app with two pages and a link between them. Next's
890
+ * own link prefetch is started by the browser and cancelled when the page is torn down at the
891
+ * end of the walk, so it lands in about four runs in five; ten checks of that untouched app,
892
+ * minutes apart, gave three reports of a difference, one report that "the change made
893
+ * something non-deterministic", and six clean ones. Identical bytes, four different answers.
894
+ *
895
+ * Two runs of one build are a wobble measurement — that is this tool's own word for it, and
896
+ * `measureWobble` above refuses to be handed two different builds precisely because the
897
+ * distinction matters. So when both sides carry the same build id, nothing here may be called
898
+ * a change, and nothing may be called newly unpredictable either: the address was already
899
+ * unpredictable and all that happened is that it was watched for longer.
900
+ *
901
+ * Nothing is hidden by this and the count does not shrink. Every difference is still counted
902
+ * and still named — it moves from the change list to the wobble list, where it says the true
903
+ * thing about the product: this address does not sit still. That is more than "1 thing
904
+ * behaves differently" told anybody, not less. Tell it the reference build id and it applies;
905
+ * leave that out and every existing caller behaves exactly as it did before.
906
+ *
882
907
  * @param {Difference[]} differences
883
908
  * @param {Wobble} wobble Measured on the candidate build.
884
- * @param {{referenceWobble?: Wobble, steadyInReference?: string[]}} [opts]
909
+ * @param {{referenceWobble?: Wobble, steadyInReference?: string[], referenceBuildId?: string, candidateBuildId?: string}} [opts]
885
910
  * @returns {WobbleSubtraction}
886
911
  */
887
912
  export function subtractWobble(differences, wobble, opts = {}) {
888
913
  const unstableNow = new Set(wobble.unstable);
914
+ const candidateBuildId = opts.candidateBuildId ?? wobble.buildId;
915
+ // Both sides have to actually name a build. An empty id means "we do not know", and two
916
+ // things we do not know are not the same thing — reading them as a match would quietly
917
+ // silence every real comparison that failed to record its build.
918
+ const sameBuild = Boolean(opts.referenceBuildId) && opts.referenceBuildId === candidateBuildId;
889
919
  // NOT symmetric, and that is deliberate. Subtracting the OLD build's wobble as well was
890
920
  // tried on 2026-08-30 and taken straight back out: a path the old build answered randomly
891
921
  // and the new build now answers the same way every time is a REAL change — somebody made
@@ -899,7 +929,11 @@ export function subtractWobble(differences, wobble, opts = {}) {
899
929
  const noise = [];
900
930
 
901
931
  for (const d of differences) {
902
- const wobbling = unstableNow.has(d.path);
932
+ // Same build on both sides means the two runs being compared are two runs of ONE build,
933
+ // so the disagreement is the build arguing with itself whether or not this check's own
934
+ // two runs happened to catch it doing so. It is wobble by definition, and it is filed as
935
+ // wobble rather than dropped, so the address is still counted and still named.
936
+ const wobbling = sameBuild || unstableNow.has(d.path);
903
937
  // Copy rather than mutate: the caller's list is often the stored diff, and a flag written
904
938
  // into it becomes a fact nobody can trace back to whoever decided it.
905
939
  const flagged = { ...d, real: !wobbling, wobbling };
@@ -909,7 +943,11 @@ export function subtractWobble(differences, wobble, opts = {}) {
909
943
 
910
944
  const referenceWobble = opts.referenceWobble;
911
945
  const steadyBefore = opts.steadyInReference ? new Set(opts.steadyInReference) : null;
912
- const couldTell = Boolean((referenceWobble && referenceWobble.measured) || steadyBefore);
946
+ // "The change made something unpredictable" needs a change to blame. With one build on both
947
+ // sides there was none, and the run that used to say it was reading the previous check of
948
+ // the same build as though it were the old build. It could not tell, and saying so is the
949
+ // honest answer — never "nothing became unpredictable", which claims a measurement.
950
+ const couldTell = !sameBuild && Boolean((referenceWobble && referenceWobble.measured) || steadyBefore);
913
951
 
914
952
  /** @type {WobbleEntry[]} */
915
953
  let newlyUnstable = [];
@@ -930,8 +968,9 @@ export function subtractWobble(differences, wobble, opts = {}) {
930
968
  noise,
931
969
  newlyUnstable,
932
970
  couldTellNewlyUnstable: couldTell,
933
- note: subtractionNote(wobble, couldTell, real.length, noise.length, newlyUnstable.length),
971
+ note: subtractionNote(wobble, couldTell, real.length, noise.length, newlyUnstable.length, sameBuild),
934
972
  };
973
+ if (sameBuild) out.sameBuild = true;
935
974
  if (storm.stormy) {
936
975
  out.couldNotTell = true;
937
976
  out.couldNotTellWhy = storm.why;
@@ -949,9 +988,22 @@ export function subtractWobble(differences, wobble, opts = {}) {
949
988
  * @param {number} realCount
950
989
  * @param {number} noiseCount
951
990
  * @param {number} newlyUnstableCount
991
+ * @param {boolean} [sameBuild] Both sides are one build, so none of this is a change.
952
992
  * @returns {string}
953
993
  */
954
- function subtractionNote(wobble, couldTell, realCount, noiseCount, newlyUnstableCount) {
994
+ function subtractionNote(wobble, couldTell, realCount, noiseCount, newlyUnstableCount, sameBuild = false) {
995
+ if (sameBuild) {
996
+ // Said first and said plainly, because a quiet answer nobody can account for is worth
997
+ // nothing. This is not "we found no differences" — it is "the two things compared were
998
+ // one build, so a difference between them could only ever have been the build arguing
999
+ // with itself", and the reader needs the second sentence to trust the first.
1000
+ const n = noiseCount;
1001
+ return (
1002
+ `${n === 0 ? 'Two runs of one build, and they agreed about everything.' : `${n} address${n === 1 ? '' : 'es'} answered differently across two runs of one build, which is this build disagreeing with itself and not a change anybody made.`} ` +
1003
+ 'Nothing has been edited since it was shipped, so there was nothing here that could have been a change. ' +
1004
+ 'Change the code, or ship again to cut a fresh record, to have something to compare against.'
1005
+ );
1006
+ }
955
1007
  if (!wobble.measured) {
956
1008
  return `The new build was only run once, so none of its own noise has been subtracted. All ${realCount} difference${realCount === 1 ? '' : 's'} here may include things that change on every run. Run it twice for a clean list.`;
957
1009
  }