yarramate 1.36.0 → 1.37.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,7 +19,7 @@ import type { Diagnostic, ResolvedProfileContext, SemanticGraph, WorkspaceSource
19
19
  * a fixture and fails if evaluation moves without this bumping, so the rule is
20
20
  * enforced rather than remembered.
21
21
  */
22
- export declare const INTERROGATION_SEMANTICS_VERSION = "1";
22
+ export declare const INTERROGATION_SEMANTICS_VERSION = "2";
23
23
  export interface CatalogueSelector {
24
24
  /**
25
25
  * Kinds to select. Absent selects every concept, which is what a
@@ -22,7 +22,7 @@ import { validateCatalogue } from './schema-validation.js';
22
22
  * a fixture and fails if evaluation moves without this bumping, so the rule is
23
23
  * enforced rather than remembered.
24
24
  */
25
- export const INTERROGATION_SEMANTICS_VERSION = '1';
25
+ export const INTERROGATION_SEMANTICS_VERSION = '2';
26
26
  const indexGraph = (graph) => {
27
27
  const relationshipIds = new Set(graph.subjects
28
28
  .filter(({ type }) => type === 'relationship')
@@ -72,11 +72,20 @@ export const normalizeLabel = (label) => label
72
72
  .split(/[^a-z0-9]+/)
73
73
  .filter((token) => token.length > 0)
74
74
  .map(singularize);
75
+ // Only the trailing run is stripped, because a role noun qualifies the name
76
+ // it follows: `order-gateway`, `orders-service`, `payment-api`. A type word
77
+ // anywhere else is part of the name rather than a label on it, and stripping
78
+ // it deletes the very thing that tells two subjects apart:
79
+ // `visual-session-server-source` and `visual-session-store-source` disagree
80
+ // on precisely one word, and removing it from both scored them identical.
81
+ //
75
82
  // Removing every token would erase a subject genuinely called "Gateway", so
76
83
  // a label that is nothing but type nouns keeps them.
77
84
  export const headTokens = (tokens) => {
78
- const head = tokens.filter((token) => !typeTokens.has(token));
79
- return head.length === 0 ? tokens : head;
85
+ let end = tokens.length;
86
+ while (end > 0 && typeTokens.has(tokens[end - 1]))
87
+ end -= 1;
88
+ return end === 0 ? tokens : tokens.slice(0, end);
80
89
  };
81
90
  const levenshtein = (left, right) => {
82
91
  if (left === right)
@@ -116,11 +125,61 @@ const jaccard = (left, right) => {
116
125
  // agree before the question opens.
117
126
  export const strongLexicalThreshold = 0.95;
118
127
  export const moderateLexicalThreshold = 0.8;
128
+ /**
129
+ * Pair the two token lists one-to-one, strongest first. Two tokens correspond
130
+ * when they are the same word: identical after normalization, or close enough
131
+ * that the difference reads as a misspelling ("component" beside
132
+ * "componant"). The bar is `moderateLexicalThreshold` rather than a second
133
+ * published number, because the judgment is the one the threshold already
134
+ * names, asked of a shorter string.
135
+ *
136
+ * Greedy is enough and optimal pairing is not worth its cost: ties break on
137
+ * position, so the result is the same on any machine.
138
+ */
139
+ const correspondence = (left, right) => {
140
+ const candidates = [];
141
+ for (const [leftIndex, leftToken] of left.entries()) {
142
+ for (const [rightIndex, rightToken] of right.entries()) {
143
+ const score = similarity(leftToken, rightToken);
144
+ if (score >= moderateLexicalThreshold) {
145
+ candidates.push({ score, left: leftIndex, right: rightIndex });
146
+ }
147
+ }
148
+ }
149
+ candidates.sort((first, second) => second.score - first.score ||
150
+ first.left - second.left ||
151
+ first.right - second.right);
152
+ const leftTaken = new Set();
153
+ const rightTaken = new Set();
154
+ for (const candidate of candidates) {
155
+ if (leftTaken.has(candidate.left) || rightTaken.has(candidate.right)) {
156
+ continue;
157
+ }
158
+ leftTaken.add(candidate.left);
159
+ rightTaken.add(candidate.right);
160
+ }
161
+ return { leftMatched: leftTaken.size, rightMatched: rightTaken.size };
162
+ };
119
163
  const labelScore = (left, right) => {
120
164
  const leftHead = headTokens(normalizeLabel(left));
121
165
  const rightHead = headTokens(normalizeLabel(right));
122
166
  if (leftHead.length === 0 || rightHead.length === 0)
123
167
  return 0;
168
+ const pairing = correspondence(leftHead, rightHead);
169
+ // Each side says a word the other never says. That is a family naming its
170
+ // members apart, not one subject recorded twice: "Add command" beside "Ask
171
+ // command", `session-server` beside `session-store`, `likec4-check-result`
172
+ // beside `likec4-diagnostic-result`. Character overlap between such labels
173
+ // measures how elaborate the shared convention is rather than how alike the
174
+ // subjects are, and the more disciplined the naming the higher it scores,
175
+ // so the pair is never offered however close the strings look.
176
+ if (pairing.leftMatched < leftHead.length &&
177
+ pairing.rightMatched < rightHead.length) {
178
+ return 0;
179
+ }
180
+ // One list is the other, or the other plus words. Same words means the same
181
+ // subject named twice; extra words are what a copy looks like
182
+ // (`payment-batch-processor` beside `payment-batch-processor-v2`).
124
183
  return Math.max(jaccard(new Set(leftHead), new Set(rightHead)), similarity(leftHead.join(' '), rightHead.join(' ')));
125
184
  };
126
185
  // The best matching pair of labels decides, because any one alias matching