carrick 0.3.82 → 0.3.83

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +46 -19
  2. package/bin/carrick.mjs +45 -2
  3. package/dist/contract.d.ts +17 -0
  4. package/dist/contract.js.map +1 -1
  5. package/dist/hook/apply-patch.d.ts +13 -0
  6. package/dist/hook/apply-patch.js +100 -0
  7. package/dist/hook/apply-patch.js.map +1 -0
  8. package/dist/hook/post-edit.d.ts +30 -1
  9. package/dist/hook/post-edit.js +95 -24
  10. package/dist/hook/post-edit.js.map +1 -1
  11. package/dist/hook/reuse.d.ts +97 -0
  12. package/dist/hook/reuse.js +245 -0
  13. package/dist/hook/reuse.js.map +1 -0
  14. package/dist/hook/stop.d.ts +3 -0
  15. package/dist/hook/stop.js +76 -0
  16. package/dist/hook/stop.js.map +1 -0
  17. package/dist/hook/user-prompt.d.ts +9 -0
  18. package/dist/hook/user-prompt.js +79 -0
  19. package/dist/hook/user-prompt.js.map +1 -0
  20. package/dist/init/codex.d.ts +51 -0
  21. package/dist/init/codex.js +167 -0
  22. package/dist/init/codex.js.map +1 -0
  23. package/dist/init/connect.d.ts +13 -0
  24. package/dist/init/connect.js +21 -15
  25. package/dist/init/connect.js.map +1 -1
  26. package/dist/init/doctor.d.ts +36 -0
  27. package/dist/init/doctor.js +97 -2
  28. package/dist/init/doctor.js.map +1 -1
  29. package/dist/init/files.d.ts +15 -0
  30. package/dist/init/files.js +20 -1
  31. package/dist/init/files.js.map +1 -1
  32. package/dist/init/outdated.d.ts +54 -0
  33. package/dist/init/outdated.js +175 -0
  34. package/dist/init/outdated.js.map +1 -0
  35. package/dist/init/output.d.ts +35 -3
  36. package/dist/init/output.js +115 -23
  37. package/dist/init/output.js.map +1 -1
  38. package/dist/init/projects.d.ts +27 -13
  39. package/dist/init/projects.js +48 -50
  40. package/dist/init/projects.js.map +1 -1
  41. package/dist/init/remove.d.ts +0 -2
  42. package/dist/init/remove.js +61 -15
  43. package/dist/init/remove.js.map +1 -1
  44. package/dist/init/repos.d.ts +34 -0
  45. package/dist/init/repos.js +77 -0
  46. package/dist/init/repos.js.map +1 -1
  47. package/dist/init/run.d.ts +73 -6
  48. package/dist/init/run.js +357 -94
  49. package/dist/init/run.js.map +1 -1
  50. package/dist/init/settings.d.ts +20 -0
  51. package/dist/init/settings.js +42 -4
  52. package/dist/init/settings.js.map +1 -1
  53. package/dist/init/task-skills.d.ts +27 -0
  54. package/dist/init/task-skills.js +64 -1
  55. package/dist/init/task-skills.js.map +1 -1
  56. package/dist/init/workspace-file.d.ts +73 -0
  57. package/dist/init/workspace-file.js +173 -0
  58. package/dist/init/workspace-file.js.map +1 -0
  59. package/package.json +6 -6
  60. package/plugin/hooks/hooks.json +11 -0
  61. package/sidecar/dist/src/capture/check-classify.d.ts +10 -1
  62. package/sidecar/dist/src/capture/check-classify.js +66 -8
  63. package/sidecar/dist/src/capture/check-deep.d.ts +16 -4
  64. package/sidecar/dist/src/capture/check-deep.js +21 -17
  65. package/sidecar/dist/src/capture/check-fields.d.ts +83 -0
  66. package/sidecar/dist/src/capture/check-fields.js +259 -0
  67. package/sidecar/dist/src/capture/check-probe.d.ts +21 -1
  68. package/sidecar/dist/src/capture/check-probe.js +39 -0
  69. package/sidecar/dist/src/capture/check.js +9 -2
  70. package/templates/skills/carrick-census.md +18 -9
  71. package/templates/skills/carrick-drift.md +7 -1
  72. package/templates/skills/carrick-impact.md +3 -1
  73. package/templates/skills/carrick-reuse.md +43 -26
@@ -0,0 +1,259 @@
1
+ /**
2
+ * Field-level report for a pair the judge called incompatible
3
+ * (carrick-tools/carrick-cloud#1118).
4
+ *
5
+ * `tsc` decides. Its elaboration names ONE field and then elides the rest
6
+ * ("Type 'A' is not assignable to type 'B'. Property 'x' is missing"), which is
7
+ * not enough for a reader deciding what to change: a create endpoint that
8
+ * requires `username` while the client sends `userName` reads as one missing
9
+ * property with no hint that the sent object carries a near-neighbour.
10
+ *
11
+ * This walk enumerates the differing fields of the SAME two types the judge
12
+ * compared, in the SAME program, using the compiler's own assignability
13
+ * relation. It is not a second judge:
14
+ * - it runs only after the bucket is decided and never changes one;
15
+ * - it never contradicts: a difference is only named when the checker itself
16
+ * says the two member types do not assign, and when it finds nothing it
17
+ * adds nothing and the raw tsc text stands alone;
18
+ * - it refuses the shapes where a member list is not an account of the type
19
+ * (a union root, a receiver with an index signature), rather than guessing
20
+ * about them.
21
+ *
22
+ * It reports one thing the judge structurally cannot: an optionality gap in the
23
+ * direction that still assigns (the sending side always provides a field the
24
+ * receiving side declares optional). That is a real drift between two sources
25
+ * — the receiver carries a branch that never runs — and no assignment error can
26
+ * exist for it. It is stated as an observation beside the verdict, never as the
27
+ * verdict.
28
+ *
29
+ * Seam: node builtins + `typescript` + this bundle only.
30
+ */
31
+ import ts from 'typescript';
32
+ /**
33
+ * Cap on named fields. A mismatch with more differing members than this is
34
+ * better described as two unrelated shapes than as a list, and the text says
35
+ * how many more there are rather than pretending the list is complete.
36
+ */
37
+ export const MAX_NAMED_FIELDS = 8;
38
+ /** Printed member types are for reading, not for re-parsing. */
39
+ const MAX_TYPE_TEXT = 80;
40
+ /** Depth cap on the structural descent. Deeper differences are reported at the
41
+ * deepest ancestor the walk reached, never dropped. */
42
+ const MAX_FIELD_DEPTH = 4;
43
+ function assignabilityOf(checker) {
44
+ const fn = checker.isTypeAssignableTo;
45
+ return typeof fn === 'function' ? fn.bind(checker) : undefined;
46
+ }
47
+ /**
48
+ * Field reports for every plan whose probe the program could read, keyed by
49
+ * pair id. A plan with no entry has no report, which is not a claim that its
50
+ * types agree.
51
+ */
52
+ export function pairFieldReports(opened, plans) {
53
+ const results = new Map();
54
+ if (!opened)
55
+ return results;
56
+ const { program, checker, probesDir } = opened;
57
+ const isAssignableTo = assignabilityOf(checker);
58
+ if (!isAssignableTo)
59
+ return results;
60
+ for (const plan of plans) {
61
+ const file = program.getSourceFile(`${probesDir}/probes/${plan.fileName}`.split('\\').join('/'));
62
+ const source = file ??
63
+ program
64
+ .getSourceFiles()
65
+ .find((sf) => sf.fileName.endsWith(`/probes/${plan.fileName}`));
66
+ if (!source)
67
+ continue;
68
+ // Whatever the judge's decisive assignment actually sent: the GraphQL
69
+ // comparand where one exists, then the JSON wire form where one exists.
70
+ // Reading `sent` there instead would describe a type the judge did not
71
+ // compare, which is the one way this walk could contradict it.
72
+ const declared = declaredConstType(source, checker, 'sentComparand') ??
73
+ declaredConstType(source, checker, 'sent');
74
+ const expected = declaredConstType(source, checker, 'expected');
75
+ if (!declared || !expected)
76
+ continue;
77
+ const wire = declaredConstType(source, checker, 'sentWire');
78
+ const compared = wire ?? declared;
79
+ const report = diffReport(compared.type, expected.type, checker, isAssignableTo, expected.node);
80
+ report.wireApplied =
81
+ wire !== undefined &&
82
+ !(isAssignableTo(wire.type, declared.type) && isAssignableTo(declared.type, wire.type));
83
+ results.set(plan.pairId, report);
84
+ }
85
+ return results;
86
+ }
87
+ /** The type of one of the probe's declared consts, read where it is declared. */
88
+ function declaredConstType(file, checker, name) {
89
+ for (const statement of file.statements) {
90
+ if (!ts.isVariableStatement(statement))
91
+ continue;
92
+ for (const declaration of statement.declarationList.declarations) {
93
+ if (!ts.isIdentifier(declaration.name) || declaration.name.text !== name)
94
+ continue;
95
+ const type = checker.getTypeAtLocation(declaration.name);
96
+ if (!type)
97
+ return undefined;
98
+ return { type, node: declaration.name };
99
+ }
100
+ }
101
+ return undefined;
102
+ }
103
+ function diffReport(sent, expected, checker, isAssignableTo, at) {
104
+ const found = [];
105
+ walk(sent, expected, '', 0, { checker, isAssignableTo, at, found });
106
+ found.sort((a, b) => a.path === b.path ? compareText(a.nature, b.nature) : compareText(a.path, b.path));
107
+ return {
108
+ differences: found.slice(0, MAX_NAMED_FIELDS),
109
+ truncated: Math.max(0, found.length - MAX_NAMED_FIELDS),
110
+ wireApplied: false,
111
+ };
112
+ }
113
+ function compareText(a, b) {
114
+ return a < b ? -1 : a > b ? 1 : 0;
115
+ }
116
+ /**
117
+ * A shape whose members can be compared one by one without diverging from the
118
+ * whole-type relation.
119
+ *
120
+ * The object flag is the first and widest of these: a union or an intersection
121
+ * does not carry it, and neither does a primitive, so a root the judge compared
122
+ * as a whole is never taken apart into members one of its constituents happens
123
+ * to share.
124
+ *
125
+ * An index signature is excluded because it makes the member list an incomplete
126
+ * account of the type: a field the sender provides that the receiver's index
127
+ * signature accepts is not a field the receiver "declares no such field" for,
128
+ * and saying so would be false. Arrays and tuples carry a numeric one, so the
129
+ * same clause keeps `length` and `push` out of a field list; an element
130
+ * difference is reported at the field that holds the array.
131
+ */
132
+ function isComparableObject(type, checker) {
133
+ if (type.flags & (ts.TypeFlags.Any | ts.TypeFlags.Unknown | ts.TypeFlags.Never)) {
134
+ return false;
135
+ }
136
+ if ((type.flags & ts.TypeFlags.Object) === 0)
137
+ return false;
138
+ if (checker.getIndexInfosOfType(type).length > 0)
139
+ return false;
140
+ return true;
141
+ }
142
+ function walk(sent, expected, path, depth, ctx) {
143
+ const { checker } = ctx;
144
+ if (!isComparableObject(sent, checker) || !isComparableObject(expected, checker)) {
145
+ return;
146
+ }
147
+ const sentProps = new Map(sent.getProperties().map((p) => [p.getName(), p]));
148
+ const expectedProps = new Map(expected.getProperties().map((p) => [p.getName(), p]));
149
+ /** Members the receiver declares and the sender has no member for, optional
150
+ * ones included. Only the REQUIRED ones are a difference; the rest still
151
+ * count here, because a receiver waiting on a member it never gets is what
152
+ * makes a sender-only member worth naming beside it. */
153
+ let absent = 0;
154
+ for (const [name, expectedProp] of expectedProps) {
155
+ const at = join(path, name);
156
+ const sentProp = sentProps.get(name);
157
+ if (!sentProp) {
158
+ absent += 1;
159
+ // An optional member the sender omits is what optional MEANS. Naming it
160
+ // would state that the receiver requires it, which is false, and it is
161
+ // not what the judge rejected the pair for.
162
+ if (!isOptional(expectedProp)) {
163
+ ctx.found.push({ path: at, nature: 'missing_in_sent' });
164
+ }
165
+ continue;
166
+ }
167
+ const sentOptional = isOptional(sentProp);
168
+ const expectedOptional = isOptional(expectedProp);
169
+ if (sentOptional && !expectedOptional) {
170
+ ctx.found.push({ path: at, nature: 'optional_in_sent' });
171
+ }
172
+ else if (!sentOptional && expectedOptional) {
173
+ // No assignment error exists for this direction, which is exactly why
174
+ // the judge cannot report it and this walk must.
175
+ ctx.found.push({ path: at, nature: 'optional_in_expected' });
176
+ }
177
+ const sentType = memberType(sentProp, ctx);
178
+ const expectedType = memberType(expectedProp, ctx);
179
+ if (ctx.isAssignableTo(sentType, expectedType))
180
+ continue;
181
+ const sentInner = checker.getNonNullableType(sentType);
182
+ const expectedInner = checker.getNonNullableType(expectedType);
183
+ if (depth + 1 < MAX_FIELD_DEPTH &&
184
+ isComparableObject(sentInner, checker) &&
185
+ isComparableObject(expectedInner, checker)) {
186
+ walk(sentInner, expectedInner, at, depth + 1, ctx);
187
+ continue;
188
+ }
189
+ ctx.found.push({
190
+ path: at,
191
+ nature: 'type_differs',
192
+ sentText: printType(sentType, ctx),
193
+ expectedText: printType(expectedType, ctx),
194
+ });
195
+ }
196
+ // A field the sender provides that the receiver does not declare is normal
197
+ // (a response carrying more than a call site reads), so it is only worth
198
+ // naming beside a member the receiver is waiting on and does not get: that
199
+ // pairing is what a renamed or relocated field looks like from the outside.
200
+ // On the common subset case — a call site reading fewer fields than the
201
+ // producer returns — nothing is absent and nothing is named.
202
+ if (absent === 0)
203
+ return;
204
+ for (const [name] of sentProps) {
205
+ if (expectedProps.has(name))
206
+ continue;
207
+ ctx.found.push({ path: join(path, name), nature: 'extra_in_sent' });
208
+ }
209
+ }
210
+ function join(path, name) {
211
+ return path === '' ? name : `${path}.${name}`;
212
+ }
213
+ function isOptional(symbol) {
214
+ return (symbol.flags & ts.SymbolFlags.Optional) !== 0;
215
+ }
216
+ function memberType(symbol, ctx) {
217
+ return ctx.checker.getTypeOfSymbolAtLocation(symbol, ctx.at);
218
+ }
219
+ function printType(type, ctx) {
220
+ const text = ctx.checker.typeToString(type, undefined, ts.TypeFormatFlags.NoTruncation | ts.TypeFormatFlags.InTypeAlias);
221
+ const flat = text.replace(/\s+/g, ' ').trim();
222
+ return flat.length > MAX_TYPE_TEXT ? `${flat.slice(0, MAX_TYPE_TEXT - 1)}…` : flat;
223
+ }
224
+ /**
225
+ * The sentence appended to a mismatch diagnostic. Names the two sides as
226
+ * producer and consumer (never the probe's internal sent/expected), so the
227
+ * reader knows which repo to change.
228
+ */
229
+ export function describeFieldReport(report, sentSide, expectedSide) {
230
+ const parts = [];
231
+ if (report.wireApplied) {
232
+ parts.push(`The ${sentSide}'s type is compared in the form JSON puts on the wire: a value with a toJSON() method (a Date, for example) travels as what it serialises to.`);
233
+ }
234
+ if (report.differences.length > 0) {
235
+ const named = report.differences
236
+ .map((d) => describeDifference(d, sentSide, expectedSide))
237
+ .join('; ');
238
+ const more = report.truncated > 0
239
+ ? `; and ${report.truncated} further field${report.truncated === 1 ? '' : 's'} differ${report.truncated === 1 ? 's' : ''} (${MAX_NAMED_FIELDS} named here)`
240
+ : '';
241
+ parts.push(`Fields that differ: ${named}${more}.`);
242
+ }
243
+ return parts.length === 0 ? '' : ` ${parts.join(' ')}`;
244
+ }
245
+ function describeDifference(difference, sentSide, expectedSide) {
246
+ const at = `'${difference.path}'`;
247
+ switch (difference.nature) {
248
+ case 'missing_in_sent':
249
+ return `${at} is required by the ${expectedSide} and the ${sentSide} does not send it`;
250
+ case 'extra_in_sent':
251
+ return `${at} is sent by the ${sentSide} and the ${expectedSide} declares no such field`;
252
+ case 'optional_in_sent':
253
+ return `${at} is optional on the ${sentSide} and required by the ${expectedSide}`;
254
+ case 'optional_in_expected':
255
+ return `${at} is always sent by the ${sentSide} and optional on the ${expectedSide}`;
256
+ case 'type_differs':
257
+ return `${at} is ${difference.sentText} on the ${sentSide} and ${difference.expectedText} on the ${expectedSide}`;
258
+ }
259
+ }
@@ -7,6 +7,12 @@
7
7
  * conditional-type relation diverges around `any`, and because the compiler's
8
8
  * elaborated assignment error is the user-facing mismatch report.
9
9
  *
10
+ * An `http` pair carries a SECOND assignment of the same value in the form JSON
11
+ * puts on the wire, and that one decides the bucket (see `wireAssignmentLine`):
12
+ * a payload is serialised before it travels, so the declared types are not what
13
+ * meet each other. Both questions go to the same judge; nothing here decides a
14
+ * verdict.
15
+ *
10
16
  * GraphQL pairs additionally unwrap the producer's resolver-return ENVELOPE
11
17
  * before the assignment (see the `graphql` branch in `buildProbe`): a GraphQL
12
18
  * producer's captured type is the resolver function's return type with
@@ -59,9 +65,23 @@ export interface ProbePlan {
59
65
  importLines: number[];
60
66
  /** 1-based line -> gate name. TS2344 here => baked-any / unverifiable. */
61
67
  gateLines: Map<number, GateName>;
62
- /** 1-based line of the value-level assignment. Errors here => incompatible. */
68
+ /** 1-based line of the value-level assignment of the DECLARED sent type. */
63
69
  assignmentLine: number;
70
+ /**
71
+ * 1-based line of the second assignment, which sends the same value in the
72
+ * form JSON puts on the wire (carrick-tools/carrick-cloud#1119). Present for
73
+ * `http` pairs, whose payload is serialised; absent for every other
74
+ * protocol, where the declared form is what travels.
75
+ *
76
+ * This is the DECISIVE line when present: the comparand short-circuits to
77
+ * the declared type whenever that already assigns, so an error here means
78
+ * the shapes disagree in both forms, and no error here means they agree in
79
+ * the form that actually travels.
80
+ */
81
+ wireAssignmentLine?: number;
64
82
  }
83
+ /** The assignment line whose diagnostic decides the bucket. */
84
+ export declare function decisiveAssignmentLine(plan: ProbePlan): number;
65
85
  /**
66
86
  * Build one probe, recording the exact line of every gate and the assignment so
67
87
  * the classifier never depends on hard-coded offsets. `packageOf` maps a
@@ -7,6 +7,12 @@
7
7
  * conditional-type relation diverges around `any`, and because the compiler's
8
8
  * elaborated assignment error is the user-facing mismatch report.
9
9
  *
10
+ * An `http` pair carries a SECOND assignment of the same value in the form JSON
11
+ * puts on the wire, and that one decides the bucket (see `wireAssignmentLine`):
12
+ * a payload is serialised before it travels, so the declared types are not what
13
+ * meet each other. Both questions go to the same judge; nothing here decides a
14
+ * verdict.
15
+ *
10
16
  * GraphQL pairs additionally unwrap the producer's resolver-return ENVELOPE
11
17
  * before the assignment (see the `graphql` branch in `buildProbe`): a GraphQL
12
18
  * producer's captured type is the resolver function's return type with
@@ -63,6 +69,10 @@ export function pairId(spec) {
63
69
  ].join('|');
64
70
  return fnv1a(key);
65
71
  }
72
+ /** The assignment line whose diagnostic decides the bucket. */
73
+ export function decisiveAssignmentLine(plan) {
74
+ return plan.wireAssignmentLine ?? plan.assignmentLine;
75
+ }
66
76
  /**
67
77
  * Build one probe, recording the exact line of every gate and the assignment so
68
78
  * the classifier never depends on hard-coded offsets. `packageOf` maps a
@@ -148,6 +158,34 @@ export function buildProbe(spec, packageOf) {
148
158
  else {
149
159
  assignmentLine = push(`const expected: Expected = sent;`);
150
160
  }
161
+ // The JSON wire line (carrick-tools/carrick-cloud#1119). An `http` payload is
162
+ // serialised before it travels, and `JSON.stringify` writes a value's
163
+ // `toJSON()` RESULT: a producer's `Date` arrives at the consumer as the
164
+ // string it serialises to, so comparing the DECLARED `Date` against a
165
+ // correctly-declared `string` reports a drift that cannot happen. The
166
+ // transform is applied to the SENT side in BOTH directions, which is where
167
+ // serialisation happens: a consumer that sends a `Date` in a request body
168
+ // likewise delivers a string, so a producer declaring `Date` there is a real
169
+ // mismatch and stays one.
170
+ //
171
+ // `WireSent` short-circuits to the declared type when that already assigns,
172
+ // so a pair that agrees as declared never instantiates the mapped type (no
173
+ // cost, and no way for the transform to turn an agreeing pair into a
174
+ // disagreeing one). tsc stays the judge of both forms.
175
+ let wireAssignmentLine;
176
+ if (spec.protocol === 'http') {
177
+ push(`type JsonWireDepth = [never, 0, 1, 2, 3, 4, 5, 6];`);
178
+ push(`type JsonWire<T, D extends number = 6> = [D] extends [never] ? T : T extends { toJSON: (...args: any[]) => infer R } ? JsonWire<R, JsonWireDepth[D]> : T extends (...args: any[]) => any ? T : T extends object ? { [K in keyof T]: JsonWire<T[K], JsonWireDepth[D]> } : T;`);
179
+ // Keep the DECLARED type whenever serialising changes nothing observable,
180
+ // so the compiler's headline still names the real surface alias (the probe
181
+ // prints `Sent`, which the scrub rewrites) instead of expanding a mapped
182
+ // type structurally. Only a pair whose payload really is transformed loses
183
+ // that name — and there the declared name no longer describes what travels.
184
+ push(`type JsonWireSame<A, B> = [A] extends [B] ? ([B] extends [A] ? true : false) : false;`);
185
+ push(`type WireSent = [Sent] extends [Expected] ? Sent : (JsonWireSame<JsonWire<Sent>, Sent> extends true ? Sent : JsonWire<Sent>);`);
186
+ push(`declare const sentWire: WireSent;`);
187
+ wireAssignmentLine = push(`const expectedWire: Expected = sentWire;`);
188
+ }
151
189
  return {
152
190
  pairId: id,
153
191
  spec,
@@ -159,5 +197,6 @@ export function buildProbe(spec, packageOf) {
159
197
  importLines,
160
198
  gateLines,
161
199
  assignmentLine,
200
+ wireAssignmentLine,
162
201
  };
163
202
  }
@@ -24,7 +24,8 @@ import { classifyPair, parseTscOutput, } from './check-classify.js';
24
24
  import { scrubPaths } from './check-scrub.js';
25
25
  import { assembleWorkspace, writeProbes, } from './check-workspace.js';
26
26
  import { buildPoisonIndexes } from './check-poison.js';
27
- import { probeDeepFindings } from './check-deep.js';
27
+ import { openProbeProgram, probeDeepFindings } from './check-deep.js';
28
+ import { pairFieldReports } from './check-fields.js';
28
29
  function runProcess(command, args, cwd) {
29
30
  return new Promise((resolve, reject) => {
30
31
  const child = spawn(command, args, { cwd, stdio: ['ignore', 'pipe', 'pipe'] });
@@ -279,7 +280,12 @@ export async function runCheck(opts, onProgress) {
279
280
  // carries a member-level any/unknown HERE — after install, where the capture
280
281
  // could not look (carrick#450). It sets `resolved` and nothing else; no
281
282
  // bucket depends on it.
282
- const deepByPair = probeDeepFindings(ws.probesDir, probing);
283
+ const probeProgram = openProbeProgram(ws.probesDir, probing);
284
+ const deepByPair = probeDeepFindings(probeProgram, probing);
285
+ // The same program answers which FIELDS differ on a pair the judge called
286
+ // incompatible (carrick-tools/carrick-cloud#1118). It names what the verdict
287
+ // is about; it never decides one.
288
+ const fieldsByPair = pairFieldReports(probeProgram, probing);
283
289
  const verdicts = sortVerdicts([
284
290
  ...probing.map((plan) => classifyPair({
285
291
  plan,
@@ -287,6 +293,7 @@ export async function runCheck(opts, onProgress) {
287
293
  poisonReason,
288
294
  scrubCtx,
289
295
  deepFindings: deepByPair.get(plan.pairId),
296
+ fieldReport: fieldsByPair.get(plan.pairId),
290
297
  })),
291
298
  ...preGated,
292
299
  ...unresolved,
@@ -12,31 +12,37 @@ does.
12
12
 
13
13
  ## 1. Two wordings
14
14
 
15
- Search twice. One query says what the code is for, the other says how it does
15
+ Ask in two wordings. One says what the code is for, the other says how it does
16
16
  it. A purpose wording misses a helper whose description names only its
17
17
  mechanism, and a mechanism wording misses one described only by its job.
18
18
 
19
19
  ```
20
- search_by_intent({{SCOPE}}, query: "<what it is for>", compact: true, top_k: 20)
21
- search_by_intent({{SCOPE}}, query: "<how it does it>", compact: true, top_k: 20)
20
+ search_by_intent({{SCOPE}}, query: "<what it is for>", also_phrased_as: ["<how it does it>"], compact: true, top_k: 20)
22
21
  ```
23
22
 
24
23
  `compact: true` returns locator-only rows, which is the shape a census needs.
25
- Page each query while `has_more` is true:
24
+ One answer holds both wordings, deduped, with `phrasings` naming them and
25
+ `matched_phrasings` on each row saying which ones reached it. Page it while
26
+ `has_more` is true:
26
27
 
27
28
  ```
28
- search_by_intent({{SCOPE}}, query: "<the same query>", compact: true, top_k: 20, offset: <next_offset>)
29
+ search_by_intent({{SCOPE}}, query: "<the same query>", also_phrased_as: ["<the same second wording>"], compact: true, top_k: 20, offset: <next_offset>)
29
30
  ```
30
31
 
32
+ Where the answer carries no `phrasings`, or the field comes back refused, the
33
+ server answering takes one wording per call. Run the second wording as its own
34
+ paged search.
35
+
31
36
  Stop when `has_more` is false. Lower `similarity_threshold` where the tail of a
32
37
  page is still on topic.
33
38
 
34
39
  ## 2. Union
35
40
 
36
- Join the two result sets on `file_path` and `line_number`. A row both wordings
37
- found is one row. Keep `retrieved_by` and `similarity` on each row, and keep
38
- which wording found it: a row only one wording reached is the row a single
39
- search would have lost.
41
+ One answer over both wordings is already deduped, and `matched_phrasings` says
42
+ which wordings reached each row. Two answers are joined here, on `file_path` and
43
+ `line_number`, and a row both wordings found is one row. Either way keep
44
+ `retrieved_by` and `similarity`, and keep which wording found each row. A row
45
+ only one wording reached is the row a single search would have lost.
40
46
 
41
47
  ## 3. Receipt
42
48
 
@@ -44,6 +50,9 @@ Report these numbers before the list, per query where the field is per query:
44
50
 
45
51
  - rows read, which is the count you paged to;
46
52
  - `total_candidates`, the exact size of the ranked list for that query;
53
+ - `hidden_by_threshold` where the answer carries it: how many rows the floor
54
+ removed, and `best` where it names the closest of them. A count above zero is
55
+ the case for one more search at a lower `similarity_threshold`;
47
56
  - `total_without_intent`, which is index-wide: functions carrying no intent
48
57
  text, which no search looked at;
49
58
  - `total_intent_carried_forward` where the response states it, which are intents
@@ -74,7 +74,13 @@ against a `string` on the other is not a difference.
74
74
  | POST /api/orders | UNRESOLVED | NewOrder | OrderDraft | web/src/orders.ts:52 | unresolved; type texts differ on `note` |
75
75
 
76
76
  Class words, and only these: MATCH, DRIFT, UNRESOLVED, NOT JUDGED, CONSUMER
77
- UNTYPED, PRODUCER UNTYPED. A reading of the two type texts goes in the verdict
77
+ UNTYPED, PRODUCER UNTYPED. Every operation the answer returned carries one of
78
+ them, the class column is never empty, and the report states operations returned
79
+ against operations classed. Where more than one word fits an operation, the row
80
+ takes the first that applies of DRIFT, PRODUCER UNTYPED, CONSUMER UNTYPED,
81
+ UNRESOLVED, NOT JUDGED, MATCH, because a stored incompatible verdict is the
82
+ finding and a side carrying no type is why nothing past it could be judged.
83
+ A reading of the two type texts goes in the verdict
78
84
  column beside the stored state, in the words "type texts differ", so nothing in
79
85
  the table reads as a verdict the index did not give you.
80
86
 
@@ -83,7 +83,9 @@ One table, then the detail.
83
83
  | admin-ui | src/api/orders.ts:44 | INCOMPATIBLE | fact |
84
84
 
85
85
  Verdict words, and only these: COMPATIBLE, INCOMPATIBLE, UNRESOLVED,
86
- NOT COMPARED. In the same message, state:
86
+ NOT COMPARED. Every consumer call site the answers returned carries one of them
87
+ and the verdict column is never empty, and the report states call sites returned
88
+ against call sites given a verdict. In the same message, state:
87
89
 
88
90
  - the producers, with file and line;
89
91
  - unmatched calls and near misses, listed apart from consumers;
@@ -7,8 +7,8 @@ description: Use at the end of a task that added or changed functions, and whene
7
7
 
8
8
  {{SCOPE_NOTE}}
9
9
 
10
- `find_similar` does the comparison. Your work is to read both spans and class
11
- each pair.
10
+ `find_similar` does the comparison. Your work is to read the spans it names
11
+ and class every row it returned.
12
12
 
13
13
  ## Targeted: the functions this task added or changed
14
14
 
@@ -26,10 +26,13 @@ An entry is either a `name` with a `file` for a function the index holds, or a
26
26
  the other in an entry, never both. A `name` that matches more than one
27
27
  definition comes back with its candidates on that entry's `error`.
28
28
 
29
- The two kinds are scored on different scales and the response states both
30
- floors: 0.85 between two indexed functions, 0.45 for a description. A stored
31
- vector carries the function's name in front of its intent and a bare sentence
32
- does not, so a description scoring 0.5 is a hit worth reading.
29
+ The two kinds are scored on different scales, and each result states the floor
30
+ it was ranked against. `vector_basis` says what the cosines are over. On
31
+ `intent` the vector is the intent sentence alone, and a copy somebody renamed
32
+ scores as close as one that kept its name. On `name_anchored` the function's
33
+ name sits in front of the sentence, a renamed copy scores lower, and
34
+ `intent_text` is the signal that still finds it. Read every score against the
35
+ floor and the basis in the answer you got.
33
36
 
34
37
  ## Audit: the whole project
35
38
 
@@ -38,9 +41,8 @@ find_similar({{SCOPE}})
38
41
  ```
39
42
 
40
43
  `clusters` groups functions that describe the same behaviour, ordered by size.
41
- Page with `offset: <next_offset>` while `has_more` is true. A group is
42
- transitive, so `lowest_similarity` can sit under the floor and a large group can
43
- hold more than one idea.
44
+ Call again with `offset: <next_offset>` for as long as the response carries
45
+ `has_more`, and class what every page returned.
44
46
 
45
47
  Where the project is larger than one pass, the response carries `error` in place
46
48
  of clusters and names the two routes under the ceiling: a `service`, or a higher
@@ -48,15 +50,28 @@ of clusters and names the two routes under the ceiling: a `service`, or a higher
48
50
  `truncated` is present the audit is partial, and its `scanned_functions` of `of`
49
51
  says by how much.
50
52
 
51
- ## Class each pair
52
-
53
- Read both spans in source, then class:
54
-
55
- - **DUPLICATE**: the same behaviour, and one call site could use the other.
56
- - **VARIANT**: near neighbours that cannot share an implementation. Say in one
57
- line why they cannot.
58
- - **FALSE POSITIVE**: the index describes them alike and the code does different
59
- work.
53
+ ## Class every row
54
+
55
+ Every row the answer returned is classed here. In an audit the first member of a
56
+ cluster is what the rest of that cluster is classed against; in a targeted call
57
+ it is the function you asked about. Read that span at the file and line the
58
+ response gave, read each other row the same way, and take the first of these
59
+ that holds:
60
+
61
+ - **FALSE POSITIVE**: the two contracts differ. Different inputs, a different
62
+ result, or a different effect, and the intent sentences alone brought them
63
+ together.
64
+ - **VARIANT**: one contract, and a behavioural difference you can name in a
65
+ clause. A different normalisation, a different error path, a different
66
+ default. Write the clause in the row. Where a comment on the member or at the
67
+ head of its file names the file it mirrors, the clause is "documented
68
+ mirror".
69
+ - **DUPLICATE**: one contract, and nothing left to name. Two bodies that run
70
+ the same once the identifiers are renamed land here.
71
+
72
+ A cluster is transitive, so `lowest_similarity` can sit under the floor and a
73
+ large group can hold more than one idea. A member that shares no contract with
74
+ the first is FALSE POSITIVE on its own row, and stays in the table.
60
75
 
61
76
  `matched_on` says which signal joined a row. `similarity` is the intent vectors;
62
77
  `intent_text` is two identical intent sentences, which is the signal that still
@@ -64,17 +79,19 @@ finds a copy somebody renamed.
64
79
 
65
80
  ## Report
66
81
 
67
- | class | function | file:line | pair | why |
82
+ One row per match, and per cluster member beyond the first. The class column
83
+ carries one of the three words and is never empty.
84
+
85
+ | class | member | file:line | against | why |
68
86
  |---|---|---|---|---|
69
- | DUPLICATE | slugify | src/text.ts:12 | src/util/url.ts:4 | same replacement rules |
87
+ | DUPLICATE | slugify | src/util/url.ts:4 | src/text.ts:12 | same replacement rules |
88
+ | VARIANT | slugTag | src/tags.ts:20 | src/text.ts:12 | documented mirror |
70
89
 
71
- Then relay the counts the response stated, in its numbers:
90
+ State `total_clusters` from the response against the number of clusters carrying
91
+ rows above. Where the two differ, name the clusters left out.
72
92
 
73
- - `compared_functions`, and `total_clusters` on an audit;
74
- - `not_compared`: `without_intent`, `intent_not_embedded`, `awaiting_embedding`,
75
- `model_mismatch`;
76
- - `excluded`: `below_min_lines`, `tests`, `generated`, `callbacks`,
77
- `other_service`.
93
+ Then relay the counts the response stated, in its numbers: `compared_functions`,
94
+ and every key the answer carries under `not_compared` and under `excluded`.
78
95
 
79
96
  Rows outside the comparison were not looked at, so an empty answer covers what
80
97
  was compared and nothing further.