@titan-design/style-analyzer 0.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js ADDED
@@ -0,0 +1,2218 @@
1
+ // src/index.ts
2
+ import {
3
+ shouldIncludeFile,
4
+ getLanguageFromPath,
5
+ parseFile,
6
+ getSupportedLanguages
7
+ } from "@titan-design/code-parser";
8
+
9
+ // src/extractors/naming.ts
10
+ var NAMING_PATTERNS = {
11
+ camelCase: /^[a-z][a-zA-Z0-9]*$/,
12
+ PascalCase: /^[A-Z][a-zA-Z0-9]*$/,
13
+ snake_case: /^[a-z][a-z0-9]*(_[a-z0-9]+)+$/,
14
+ SCREAMING_SNAKE: /^[A-Z][A-Z0-9]*(_[A-Z0-9]+)+$/,
15
+ "kebab-case": /^[a-z][a-z0-9]*(-[a-z0-9]+)+$/
16
+ };
17
+ var BOOLEAN_PREFIXES = /^(is|has|should|can|will|did|was)[A-Z_]/;
18
+ var PYTHON_BOOLEAN_PREFIXES = /^(is|has|should|can|will|did|was)_/;
19
+ function detectConvention(name) {
20
+ for (const [convention, pattern] of Object.entries(NAMING_PATTERNS)) {
21
+ if (pattern.test(name)) return convention;
22
+ }
23
+ if (/^[a-z][a-z0-9]*$/.test(name)) return "camelCase";
24
+ return null;
25
+ }
26
+ function detectBooleanPrefix(name, language) {
27
+ const pattern = language === "python" ? PYTHON_BOOLEAN_PREFIXES : BOOLEAN_PREFIXES;
28
+ const match = name.match(pattern);
29
+ return match ? match[1] : null;
30
+ }
31
+ var NamingExtractor = class {
32
+ name = "naming";
33
+ extract(file) {
34
+ const observations = [];
35
+ const visit = (node) => {
36
+ this.processNode(node, file, observations);
37
+ for (const child of node.children) {
38
+ visit(child);
39
+ }
40
+ };
41
+ visit(file.tree.rootNode);
42
+ return observations;
43
+ }
44
+ processNode(node, file, observations) {
45
+ switch (file.language) {
46
+ case "typescript":
47
+ case "tsx":
48
+ this.processTypeScriptNode(node, file, observations);
49
+ break;
50
+ case "python":
51
+ this.processPythonNode(node, file, observations);
52
+ break;
53
+ }
54
+ }
55
+ processTypeScriptNode(node, file, observations) {
56
+ switch (node.type) {
57
+ case "variable_declarator":
58
+ this.processTypeScriptVariable(node, file, observations);
59
+ break;
60
+ case "function_declaration":
61
+ this.observeDeclarationName(node, "naming.function", file, observations);
62
+ break;
63
+ case "interface_declaration":
64
+ case "type_alias_declaration":
65
+ this.observeDeclarationName(node, "naming.type", file, observations);
66
+ break;
67
+ case "enum_declaration":
68
+ this.observeDeclarationName(node, "naming.enum", file, observations);
69
+ break;
70
+ case "class_declaration":
71
+ if (this.observeDeclarationName(node, "naming.type", file, observations)) {
72
+ this.detectPrivateMembers(node, file, observations);
73
+ }
74
+ break;
75
+ case "required_parameter":
76
+ case "optional_parameter":
77
+ this.processTypeScriptParameter(node, file, observations);
78
+ break;
79
+ }
80
+ }
81
+ processTypeScriptVariable(node, file, observations) {
82
+ const nameNode = node.childForFieldName("name");
83
+ if (!nameNode || nameNode.type !== "identifier") return;
84
+ const name = nameNode.text;
85
+ const declKind = node.parent?.type === "lexical_declaration" ? node.parent.children[0]?.text : null;
86
+ if (declKind === "const" && NAMING_PATTERNS.SCREAMING_SNAKE.test(name)) {
87
+ this.addObservation(observations, "naming.constant", "SCREAMING_SNAKE", file, node);
88
+ return;
89
+ }
90
+ this.observeVariable(name, node, file, observations);
91
+ }
92
+ processTypeScriptParameter(node, file, observations) {
93
+ const nameNode = node.childForFieldName("pattern") ?? node.childForFieldName("name");
94
+ if (!nameNode || nameNode.type !== "identifier") return;
95
+ if (nameNode.text === "this") return;
96
+ const convention = detectConvention(nameNode.text);
97
+ if (convention) {
98
+ this.addObservation(observations, "naming.parameter", convention, file, node);
99
+ }
100
+ }
101
+ processPythonNode(node, file, observations) {
102
+ switch (node.type) {
103
+ case "assignment":
104
+ this.processPythonAssignment(node, file, observations);
105
+ break;
106
+ case "function_definition": {
107
+ const nameNode = node.childForFieldName("name");
108
+ if (!nameNode) break;
109
+ if (nameNode.text.startsWith("__") && nameNode.text.endsWith("__")) break;
110
+ this.observeDeclarationName(node, "naming.function", file, observations);
111
+ break;
112
+ }
113
+ case "class_definition":
114
+ this.observeDeclarationName(node, "naming.type", file, observations);
115
+ break;
116
+ case "parameters":
117
+ this.processPythonParameters(node, file, observations);
118
+ break;
119
+ }
120
+ }
121
+ processPythonAssignment(node, file, observations) {
122
+ const left = node.childForFieldName("left");
123
+ if (!left || left.type !== "identifier") return;
124
+ const name = left.text;
125
+ if (node.parent?.type === "module" && NAMING_PATTERNS.SCREAMING_SNAKE.test(name)) {
126
+ this.addObservation(observations, "naming.constant", "SCREAMING_SNAKE", file, node);
127
+ return;
128
+ }
129
+ this.observeVariable(name, node, file, observations);
130
+ }
131
+ processPythonParameters(node, file, observations) {
132
+ for (const child of node.children) {
133
+ if (child.type === "identifier" && child.text !== "self" && child.text !== "cls") {
134
+ const convention = detectConvention(child.text);
135
+ if (convention) {
136
+ this.addObservation(observations, "naming.parameter", convention, file, child);
137
+ }
138
+ }
139
+ if (child.type === "typed_parameter") {
140
+ const paramName = child.childForFieldName("name") ?? child.children[0];
141
+ if (paramName && paramName.type === "identifier" && paramName.text !== "self") {
142
+ const convention = detectConvention(paramName.text);
143
+ if (convention) {
144
+ this.addObservation(observations, "naming.parameter", convention, file, child);
145
+ }
146
+ }
147
+ }
148
+ }
149
+ }
150
+ observeVariable(name, node, file, observations) {
151
+ const prefix = detectBooleanPrefix(name, file.language);
152
+ if (prefix) {
153
+ this.addObservation(observations, "naming.boolean", prefix, file, node);
154
+ }
155
+ const convention = detectConvention(name);
156
+ if (convention) {
157
+ this.addObservation(observations, "naming.variable", convention, file, node);
158
+ }
159
+ }
160
+ // Returns whether the node had a name, because a nameless class skips its private-member scan.
161
+ observeDeclarationName(node, type, file, observations) {
162
+ const nameNode = node.childForFieldName("name");
163
+ if (!nameNode) return false;
164
+ const convention = detectConvention(nameNode.text);
165
+ if (convention) {
166
+ this.addObservation(observations, type, convention, file, node);
167
+ }
168
+ return true;
169
+ }
170
+ detectPrivateMembers(classNode, file, observations) {
171
+ const body = classNode.childForFieldName("body");
172
+ if (!body) return;
173
+ for (const member of body.children) {
174
+ if (member.type === "public_field_definition") {
175
+ const nameNode = member.childForFieldName("name");
176
+ if (!nameNode) continue;
177
+ const name = nameNode.text;
178
+ if (name.startsWith("#")) {
179
+ this.addObservation(observations, "naming.private-member", "hash-prefix", file, member);
180
+ } else if (name.startsWith("_") && !name.startsWith("__")) {
181
+ this.addObservation(observations, "naming.private-member", "underscore-prefix", file, member);
182
+ }
183
+ }
184
+ }
185
+ }
186
+ addObservation(observations, type, value, file, node) {
187
+ observations.push({
188
+ type,
189
+ category: "naming",
190
+ value,
191
+ file: file.filePath,
192
+ line: node.startPosition.row + 1
193
+ });
194
+ }
195
+ };
196
+
197
+ // src/extractors/structure.ts
198
+ var PYTHON_BUILTINS = /* @__PURE__ */ new Set([
199
+ "os",
200
+ "sys",
201
+ "re",
202
+ "json",
203
+ "math",
204
+ "time",
205
+ "datetime",
206
+ "pathlib",
207
+ "collections",
208
+ "itertools",
209
+ "functools",
210
+ "typing",
211
+ "io",
212
+ "abc",
213
+ "dataclasses",
214
+ "enum",
215
+ "logging",
216
+ "unittest",
217
+ "hashlib",
218
+ "subprocess",
219
+ "argparse",
220
+ "copy",
221
+ "glob",
222
+ "shutil",
223
+ "tempfile",
224
+ "textwrap",
225
+ "contextlib",
226
+ "operator",
227
+ "string",
228
+ "struct",
229
+ "csv",
230
+ "xml"
231
+ ]);
232
+ function classifyImportSource(source, language) {
233
+ if (language === "python") {
234
+ if (source.startsWith(".")) return "relative";
235
+ const topModule = source.split(".")[0];
236
+ if (PYTHON_BUILTINS.has(topModule)) return "builtin";
237
+ return "external";
238
+ }
239
+ if (source.startsWith("node:")) return "builtin";
240
+ if (source.startsWith(".") || source.startsWith("..")) return "relative";
241
+ if (source.startsWith("@")) {
242
+ const scope = source.split("/")[0];
243
+ if (["@app", "@lib", "@src", "@internal", "@modules"].includes(scope)) {
244
+ return "internal";
245
+ }
246
+ }
247
+ return "external";
248
+ }
249
+ var EXPORT_DECLARATION_TYPES = [
250
+ "function_declaration",
251
+ "class_declaration",
252
+ "lexical_declaration",
253
+ "interface_declaration",
254
+ "type_alias_declaration",
255
+ "enum_declaration"
256
+ ];
257
+ function isBarrelFile(root) {
258
+ let exportFromCount = 0;
259
+ let otherStatements = 0;
260
+ for (const child of root.children) {
261
+ if (child.type === "export_statement") {
262
+ const source = child.childForFieldName("source");
263
+ if (source) {
264
+ exportFromCount++;
265
+ } else {
266
+ otherStatements++;
267
+ }
268
+ } else if (child.isNamed && child.type !== "comment") {
269
+ otherStatements++;
270
+ }
271
+ }
272
+ return exportFromCount > 0 && otherStatements <= 1;
273
+ }
274
+ var StructureExtractor = class {
275
+ name = "structure";
276
+ extract(file) {
277
+ const observations = [];
278
+ const root = file.tree.rootNode;
279
+ this.extractImports(root, file, observations);
280
+ this.extractExports(root, file, observations);
281
+ this.detectBarrelFile(root, file, observations);
282
+ return observations;
283
+ }
284
+ extractImports(root, file, observations) {
285
+ const groupSequence = [];
286
+ for (const child of root.children) {
287
+ const source = this.importSource(child, file.language);
288
+ if (source) {
289
+ const group = classifyImportSource(source, file.language);
290
+ groupSequence.push(group);
291
+ observations.push({
292
+ type: "structure.import-group",
293
+ category: "structure",
294
+ value: group,
295
+ file: file.filePath,
296
+ line: child.startPosition.row + 1,
297
+ metadata: { source }
298
+ });
299
+ }
300
+ }
301
+ this.addImportOrder(groupSequence, file, observations);
302
+ }
303
+ addImportOrder(groupSequence, file, observations) {
304
+ const uniqueOrder = [...new Set(groupSequence)];
305
+ if (uniqueOrder.length > 0) {
306
+ observations.push({
307
+ type: "structure.import-order",
308
+ category: "structure",
309
+ value: JSON.stringify(uniqueOrder),
310
+ file: file.filePath,
311
+ line: 1,
312
+ metadata: { groupCount: uniqueOrder.length }
313
+ });
314
+ }
315
+ }
316
+ importSource(child, language) {
317
+ let source = null;
318
+ if (language === "python") {
319
+ if (child.type === "import_statement") {
320
+ const nameNode = child.childForFieldName("name");
321
+ source = nameNode?.text ?? null;
322
+ } else if (child.type === "import_from_statement") {
323
+ const moduleNode = child.childForFieldName("module_name");
324
+ const dots = child.children.filter((c) => c.type === "." || c.type === "relative_import").map((c) => c.text).join("");
325
+ source = dots + (moduleNode?.text ?? "");
326
+ }
327
+ } else {
328
+ if (child.type === "import_statement") {
329
+ const sourceNode = child.childForFieldName("source");
330
+ source = sourceNode?.text?.replace(/['"]/g, "") ?? null;
331
+ }
332
+ }
333
+ return source;
334
+ }
335
+ extractExports(root, file, observations) {
336
+ if (file.language === "python") return;
337
+ for (const child of root.children) {
338
+ if (child.type === "export_statement") {
339
+ this.processExport(child, file, observations);
340
+ }
341
+ }
342
+ }
343
+ processExport(child, file, observations) {
344
+ const isDefault = child.children.some((c) => c.type === "default");
345
+ const style = isDefault ? "default" : "named";
346
+ observations.push({
347
+ type: "structure.export-style",
348
+ category: "structure",
349
+ value: style,
350
+ file: file.filePath,
351
+ line: child.startPosition.row + 1
352
+ });
353
+ const proximity = this.exportProximity(child, isDefault);
354
+ if (proximity) {
355
+ observations.push({
356
+ type: "structure.export-proximity",
357
+ category: "structure",
358
+ value: proximity,
359
+ file: file.filePath,
360
+ line: child.startPosition.row + 1
361
+ });
362
+ }
363
+ }
364
+ exportProximity(child, isDefault) {
365
+ const hasDeclaration = child.children.some(
366
+ (c) => EXPORT_DECLARATION_TYPES.includes(c.type)
367
+ );
368
+ const isReExport = child.childForFieldName("source") !== null;
369
+ if (hasDeclaration || isDefault) return "inline";
370
+ if (!isReExport) return "trailing";
371
+ return null;
372
+ }
373
+ detectBarrelFile(root, file, observations) {
374
+ if (file.language === "python") return;
375
+ if (isBarrelFile(root)) {
376
+ observations.push({
377
+ type: "structure.barrel-file",
378
+ category: "structure",
379
+ value: true,
380
+ file: file.filePath,
381
+ line: 1
382
+ });
383
+ }
384
+ }
385
+ };
386
+
387
+ // src/extractors/control-flow.ts
388
+ var ARRAY_METHODS = /* @__PURE__ */ new Set([
389
+ "map",
390
+ "filter",
391
+ "reduce",
392
+ "forEach",
393
+ "find",
394
+ "some",
395
+ "every",
396
+ "flatMap",
397
+ "findIndex"
398
+ ]);
399
+ var ControlFlowExtractor = class {
400
+ name = "control-flow";
401
+ extract(file) {
402
+ const observations = [];
403
+ this.walk(file.tree.rootNode, file, observations);
404
+ return observations;
405
+ }
406
+ walk(node, file, observations) {
407
+ this.processNode(node, file, observations);
408
+ for (const child of node.children) {
409
+ this.walk(child, file, observations);
410
+ }
411
+ }
412
+ processNode(node, file, observations) {
413
+ this.processBranch(node, file, observations);
414
+ this.processLoop(node, file, observations);
415
+ this.processCall(node, file, observations);
416
+ }
417
+ processBranch(node, file, observations) {
418
+ if (node.type === "ternary_expression" || node.type === "conditional_expression") {
419
+ this.emit(observations, "control-flow.ternary", true, file, node);
420
+ }
421
+ if (node.type === "if_statement") {
422
+ this.emit(observations, "control-flow.if-else", true, file, node);
423
+ this.detectGuardClause(node, file, observations);
424
+ this.detectElseAfterReturn(node, file, observations);
425
+ }
426
+ }
427
+ processLoop(node, file, observations) {
428
+ if (node.type === "for_statement" && file.language !== "python") {
429
+ this.emit(observations, "control-flow.for-loop", true, file, node);
430
+ }
431
+ if (node.type === "for_in_statement") {
432
+ const isForOf = node.children.some((c) => c.type === "of");
433
+ const loopType = isForOf ? "control-flow.for-of" : "control-flow.for-in";
434
+ this.emit(observations, loopType, true, file, node);
435
+ }
436
+ if (node.type === "for_statement" && file.language === "python") {
437
+ this.emit(observations, "control-flow.for-loop", true, file, node);
438
+ }
439
+ if (node.type === "list_comprehension" || node.type === "set_comprehension" || node.type === "dictionary_comprehension" || node.type === "generator_expression") {
440
+ this.emit(observations, "control-flow.array-method", true, file, node);
441
+ }
442
+ }
443
+ processCall(node, file, observations) {
444
+ if (node.type === "call_expression") {
445
+ const fn = node.childForFieldName("function");
446
+ if (fn?.type === "member_expression") {
447
+ this.processMemberCall(fn, node, file, observations);
448
+ }
449
+ }
450
+ if (node.type === "await_expression") {
451
+ this.emit(observations, "control-flow.async-await", true, file, node);
452
+ }
453
+ }
454
+ processMemberCall(fn, node, file, observations) {
455
+ const property = fn.childForFieldName("property");
456
+ if (property) {
457
+ const methodName = property.text;
458
+ if (ARRAY_METHODS.has(methodName)) {
459
+ this.emit(
460
+ observations,
461
+ "control-flow.array-method",
462
+ methodName,
463
+ file,
464
+ node
465
+ );
466
+ }
467
+ if (methodName === "then") {
468
+ this.emit(
469
+ observations,
470
+ "control-flow.promise-then",
471
+ true,
472
+ file,
473
+ node
474
+ );
475
+ }
476
+ }
477
+ }
478
+ detectGuardClause(node, file, observations) {
479
+ const parent = node.parent;
480
+ if (!parent) return;
481
+ if (!this.isFunctionBody(parent)) return;
482
+ const siblings = parent.children.filter(
483
+ (c) => c.type !== "comment" && c.type !== "{" && c.type !== "}"
484
+ );
485
+ const nodeIndex = siblings.indexOf(node);
486
+ if (nodeIndex > 2) return;
487
+ const consequent = node.childForFieldName("consequence") ?? node.childForFieldName("body");
488
+ if (!consequent) return;
489
+ const hasReturn = this.containsReturn(consequent);
490
+ const hasElse = node.childForFieldName("alternative") !== null;
491
+ if (hasReturn && !hasElse) {
492
+ this.emit(observations, "control-flow.guard-clause", true, file, node);
493
+ }
494
+ }
495
+ isFunctionBody(parent) {
496
+ const isFunctionBody = parent.type === "statement_block" && (parent.parent?.type === "function_declaration" || parent.parent?.type === "method_definition" || parent.parent?.type === "arrow_function");
497
+ const isPythonFunctionBody = parent.type === "block" && parent.parent?.type === "function_definition";
498
+ return isFunctionBody || isPythonFunctionBody;
499
+ }
500
+ detectElseAfterReturn(node, file, observations) {
501
+ const consequent = node.childForFieldName("consequence") ?? node.childForFieldName("body");
502
+ const alternative = node.childForFieldName("alternative");
503
+ if (!consequent || !alternative) return;
504
+ if (this.containsReturn(consequent)) {
505
+ this.emit(
506
+ observations,
507
+ "control-flow.else-after-return",
508
+ true,
509
+ file,
510
+ node
511
+ );
512
+ }
513
+ }
514
+ containsReturn(node) {
515
+ if (node.type === "return_statement") return true;
516
+ for (const child of node.children) {
517
+ if (child.type === "return_statement") return true;
518
+ }
519
+ return false;
520
+ }
521
+ emit(observations, type, value, file, node) {
522
+ observations.push({
523
+ type,
524
+ category: "control-flow",
525
+ value,
526
+ file: file.filePath,
527
+ line: node.startPosition.row + 1
528
+ });
529
+ }
530
+ };
531
+
532
+ // src/extractors/documentation.ts
533
+ var JSDOC_TAG_PATTERN = /@(param|returns?|throws?|example|deprecated|see|since|type|typedef|template|callback|async)\b/g;
534
+ var PYTHON_DOC_TAG_PATTERN = /^[ \t]*(Args|Returns?|Raises?|Yields?|Note|Notes|Example|Attributes|Todo|References):/gm;
535
+ var DocumentationExtractor = class {
536
+ name = "documentation";
537
+ extract(file) {
538
+ const observations = [];
539
+ this.walkDeclarations(file.tree.rootNode, file, observations);
540
+ this.walkForComments(file.tree.rootNode, file, observations);
541
+ return observations;
542
+ }
543
+ walkDeclarations(node, file, observations) {
544
+ for (const child of node.children) {
545
+ if (this.isDeclaration(child, file.language)) {
546
+ this.processDeclaration(child, file, observations);
547
+ }
548
+ if (child.type === "class_declaration" || child.type === "class_definition" || child.type === "class_body" || child.type === "block") {
549
+ this.walkDeclarations(child, file, observations);
550
+ }
551
+ if (child.type === "export_statement") {
552
+ this.walkDeclarations(child, file, observations);
553
+ }
554
+ }
555
+ }
556
+ isDeclaration(node, language) {
557
+ if (language === "python") {
558
+ return node.type === "function_definition" || node.type === "class_definition";
559
+ }
560
+ return node.type === "function_declaration" || node.type === "method_definition" || node.type === "class_declaration" || node.type === "interface_declaration" || node.type === "type_alias_declaration";
561
+ }
562
+ processDeclaration(node, file, observations) {
563
+ const isExported = this.isExported(node, file.language);
564
+ const hasDoc = this.hasLeadingDoc(node, file.language);
565
+ this.emit(observations, "documentation.jsdoc-presence", hasDoc, file, node);
566
+ const coverageType = isExported ? "documentation.public-coverage" : "documentation.private-coverage";
567
+ this.emit(observations, coverageType, hasDoc, file, node);
568
+ if (hasDoc) {
569
+ this.extractTags(node, file, observations);
570
+ }
571
+ }
572
+ hasLeadingDoc(node, language) {
573
+ if (language === "python") {
574
+ return this.hasPythonDocstring(node);
575
+ }
576
+ return this.hasJSDoc(node);
577
+ }
578
+ hasJSDoc(node) {
579
+ const prev = node.previousSibling;
580
+ if (prev?.type === "comment" && prev.text.startsWith("/**")) {
581
+ return true;
582
+ }
583
+ if (node.parent?.type === "export_statement") {
584
+ const exportPrev = node.parent.previousSibling;
585
+ if (exportPrev?.type === "comment" && exportPrev.text.startsWith("/**")) {
586
+ return true;
587
+ }
588
+ }
589
+ return false;
590
+ }
591
+ hasPythonDocstring(node) {
592
+ const body = node.childForFieldName("body");
593
+ if (!body) return false;
594
+ const firstStatement = body.children.find(
595
+ (c) => c.type !== "comment" && c.type !== "newline"
596
+ );
597
+ if (!firstStatement) return false;
598
+ if (firstStatement.type === "expression_statement") {
599
+ const expr = firstStatement.children[0];
600
+ return expr?.type === "string" || expr?.type === "concatenated_string";
601
+ }
602
+ return false;
603
+ }
604
+ extractTags(node, file, observations) {
605
+ if (file.language === "python") {
606
+ this.extractPythonDocTags(node, file, observations);
607
+ return;
608
+ }
609
+ const commentNode = this.getLeadingComment(node);
610
+ if (!commentNode) return;
611
+ const text = commentNode.text;
612
+ const tagMatches = text.matchAll(JSDOC_TAG_PATTERN);
613
+ const seenTags = /* @__PURE__ */ new Set();
614
+ for (const match of tagMatches) {
615
+ const tag = `@${match[1]}`;
616
+ const normalized = tag.replace(/^@return$/, "@returns").replace(/^@throw$/, "@throws");
617
+ if (seenTags.has(normalized)) continue;
618
+ seenTags.add(normalized);
619
+ this.emit(observations, "documentation.jsdoc-tag", normalized, file, commentNode);
620
+ }
621
+ }
622
+ extractPythonDocTags(node, file, observations) {
623
+ const body = node.childForFieldName("body");
624
+ if (!body) return;
625
+ const firstStatement = body.children.find(
626
+ (c) => c.type === "expression_statement"
627
+ );
628
+ if (!firstStatement) return;
629
+ const expr = firstStatement.children[0];
630
+ if (!expr) return;
631
+ const text = expr.text;
632
+ const tagMatches = text.matchAll(PYTHON_DOC_TAG_PATTERN);
633
+ const seenTags = /* @__PURE__ */ new Set();
634
+ for (const match of tagMatches) {
635
+ const tag = match[1];
636
+ if (seenTags.has(tag)) continue;
637
+ seenTags.add(tag);
638
+ this.emit(observations, "documentation.jsdoc-tag", tag, file, expr);
639
+ }
640
+ }
641
+ walkForComments(node, file, observations) {
642
+ if (node.type === "comment") {
643
+ if (!node.text.startsWith("/**")) {
644
+ this.emit(observations, "documentation.inline-comment", true, file, node);
645
+ const placement = this.getCommentPlacement(node);
646
+ this.emit(observations, "documentation.comment-placement", placement, file, node);
647
+ }
648
+ }
649
+ for (const child of node.children) {
650
+ this.walkForComments(child, file, observations);
651
+ }
652
+ }
653
+ getCommentPlacement(node) {
654
+ const prev = node.previousSibling;
655
+ if (prev && prev.endPosition.row === node.startPosition.row) {
656
+ return "trailing";
657
+ }
658
+ return "leading";
659
+ }
660
+ isExported(node, language) {
661
+ if (language === "python") {
662
+ const nameNode = node.childForFieldName("name");
663
+ return node.parent?.type === "module" && !!nameNode && !nameNode.text.startsWith("_");
664
+ }
665
+ return node.parent?.type === "export_statement";
666
+ }
667
+ getLeadingComment(node) {
668
+ const prev = node.previousSibling;
669
+ if (prev?.type === "comment" && prev.text.startsWith("/**")) {
670
+ return prev;
671
+ }
672
+ if (node.parent?.type === "export_statement") {
673
+ const exportPrev = node.parent.previousSibling;
674
+ if (exportPrev?.type === "comment" && exportPrev.text.startsWith("/**")) {
675
+ return exportPrev;
676
+ }
677
+ }
678
+ return null;
679
+ }
680
+ emit(observations, type, value, file, node) {
681
+ observations.push({
682
+ type,
683
+ category: "documentation",
684
+ value,
685
+ file: file.filePath,
686
+ line: node.startPosition.row + 1
687
+ });
688
+ }
689
+ };
690
+
691
+ // src/extractors/error-handling.ts
692
+ var RESULT_TYPE_NAMES = /* @__PURE__ */ new Set([
693
+ "Result",
694
+ "Either",
695
+ "Ok",
696
+ "Err",
697
+ "Success",
698
+ "Failure"
699
+ ]);
700
+ var GENERIC_CATCH_TYPES = /* @__PURE__ */ new Set([
701
+ "Error",
702
+ "Exception",
703
+ "unknown"
704
+ ]);
705
+ var ErrorHandlingExtractor = class {
706
+ name = "error-handling";
707
+ extract(file) {
708
+ const observations = [];
709
+ this.walk(file.tree.rootNode, file, observations);
710
+ return observations;
711
+ }
712
+ walk(node, file, observations) {
713
+ this.processNode(node, file, observations);
714
+ for (const child of node.children) {
715
+ this.walk(child, file, observations);
716
+ }
717
+ }
718
+ processNode(node, file, observations) {
719
+ this.processDeclaration(node, file, observations);
720
+ this.processFunctionCheck(node, file, observations);
721
+ }
722
+ processDeclaration(node, file, observations) {
723
+ if (node.type === "try_statement") {
724
+ this.emit(observations, "error-handling.try-catch", true, file, node);
725
+ this.analyzeCatchClauses(node, file, observations);
726
+ }
727
+ if (node.type === "class_declaration" || node.type === "class_definition") {
728
+ this.detectCustomErrorClass(node, file, observations);
729
+ }
730
+ if (node.type === "type_alias_declaration" && (file.language === "typescript" || file.language === "tsx")) {
731
+ this.detectResultType(node, file, observations);
732
+ }
733
+ }
734
+ processFunctionCheck(node, file, observations) {
735
+ if (node.type === "function_declaration" || node.type === "method_definition") {
736
+ this.detectResultReturnType(node, file, observations);
737
+ }
738
+ if (node.type === "function_declaration") {
739
+ this.detectAssertNever(node, file, observations);
740
+ }
741
+ if (node.type === "switch_statement") {
742
+ this.detectExhaustiveSwitch(node, file, observations);
743
+ }
744
+ }
745
+ analyzeCatchClauses(tryNode, file, observations) {
746
+ for (const child of tryNode.children) {
747
+ if (child.type === "catch_clause") {
748
+ const body = child.childForFieldName("body");
749
+ if (body && this.hasInstanceofCheck(body)) {
750
+ this.emit(observations, "error-handling.catch-specificity", "specific", file, child);
751
+ } else {
752
+ this.emit(observations, "error-handling.catch-specificity", "generic", file, child);
753
+ }
754
+ }
755
+ if (child.type === "except_clause") {
756
+ const typeNode = child.children.find(
757
+ (c) => c.type === "identifier" || c.type === "attribute"
758
+ );
759
+ if (typeNode && !GENERIC_CATCH_TYPES.has(typeNode.text)) {
760
+ this.emit(observations, "error-handling.catch-specificity", "specific", file, child);
761
+ } else {
762
+ this.emit(observations, "error-handling.catch-specificity", "generic", file, child);
763
+ }
764
+ }
765
+ }
766
+ }
767
+ hasInstanceofCheck(body) {
768
+ return body.text.includes("instanceof");
769
+ }
770
+ detectCustomErrorClass(node, file, observations) {
771
+ if (file.language === "python") {
772
+ const superclasses = node.childForFieldName("superclasses");
773
+ if (!superclasses) return;
774
+ const bases = superclasses.text;
775
+ if (bases.includes("Error") || bases.includes("Exception")) {
776
+ this.emit(observations, "error-handling.custom-error-class", true, file, node);
777
+ }
778
+ return;
779
+ }
780
+ const heritage = node.children.find(
781
+ (c) => c.type === "class_heritage"
782
+ );
783
+ if (!heritage) return;
784
+ if (heritage.text.includes("Error")) {
785
+ this.emit(observations, "error-handling.custom-error-class", true, file, node);
786
+ }
787
+ }
788
+ detectResultType(node, file, observations) {
789
+ const nameNode = node.childForFieldName("name");
790
+ if (!nameNode) return;
791
+ if (RESULT_TYPE_NAMES.has(nameNode.text)) {
792
+ this.emit(observations, "error-handling.result-type", nameNode.text, file, node);
793
+ }
794
+ }
795
+ detectResultReturnType(node, file, observations) {
796
+ const returnType = node.childForFieldName("return_type");
797
+ if (!returnType) return;
798
+ const text = returnType.text;
799
+ for (const name of RESULT_TYPE_NAMES) {
800
+ if (text.includes(name)) {
801
+ this.emit(observations, "error-handling.result-type", name, file, node);
802
+ break;
803
+ }
804
+ }
805
+ }
806
+ detectAssertNever(node, file, observations) {
807
+ const nameNode = node.childForFieldName("name");
808
+ if (!nameNode) return;
809
+ const name = nameNode.text;
810
+ if (name !== "assertNever" && name !== "absurd") return;
811
+ const params = node.childForFieldName("parameters");
812
+ const returnType = node.childForFieldName("return_type");
813
+ const hasNeverParam = params?.text.includes("never") ?? false;
814
+ const returnsNever = returnType?.text.includes("never") ?? false;
815
+ if (hasNeverParam || returnsNever) {
816
+ this.emit(observations, "error-handling.assert-never", true, file, node);
817
+ }
818
+ }
819
+ detectExhaustiveSwitch(node, file, observations) {
820
+ const body = node.childForFieldName("body");
821
+ if (!body) return;
822
+ let defaultCallsAssertNever = false;
823
+ for (const child of body.children) {
824
+ if (child.type === "switch_default") {
825
+ const text = child.text;
826
+ if (text.includes("assertNever") || text.includes("absurd")) {
827
+ defaultCallsAssertNever = true;
828
+ }
829
+ }
830
+ }
831
+ this.emit(
832
+ observations,
833
+ "error-handling.exhaustive-switch",
834
+ defaultCallsAssertNever,
835
+ file,
836
+ node
837
+ );
838
+ }
839
+ emit(observations, type, value, file, node) {
840
+ observations.push({
841
+ type,
842
+ category: "error-handling",
843
+ value,
844
+ file: file.filePath,
845
+ line: node.startPosition.row + 1
846
+ });
847
+ }
848
+ };
849
+
850
+ // src/extractors/formatting.ts
851
+ import { readFile } from "fs/promises";
852
+
853
+ // src/extractors/formatting-config.ts
854
+ function parsePrettierConfig(raw, configPath) {
855
+ const config = JSON.parse(raw);
856
+ return [
857
+ ...prettierSyntaxObservations(config, configPath),
858
+ ...prettierIndentObservations(config, configPath)
859
+ ];
860
+ }
861
+ function prettierSyntaxObservations(config, configPath) {
862
+ const observations = [];
863
+ if (config.semi !== void 0) {
864
+ observations.push(makeFormattingObs(
865
+ "formatting.semicolons",
866
+ config.semi,
867
+ configPath,
868
+ 1,
869
+ "config"
870
+ ));
871
+ }
872
+ if (config.singleQuote !== void 0) {
873
+ observations.push(makeFormattingObs(
874
+ "formatting.quoteStyle",
875
+ config.singleQuote ? "single" : "double",
876
+ configPath,
877
+ 1,
878
+ "config"
879
+ ));
880
+ }
881
+ if (config.trailingComma !== void 0) {
882
+ observations.push(makeFormattingObs(
883
+ "formatting.trailingCommas",
884
+ config.trailingComma !== "none",
885
+ configPath,
886
+ 1,
887
+ "config"
888
+ ));
889
+ }
890
+ return observations;
891
+ }
892
+ function prettierIndentObservations(config, configPath) {
893
+ const observations = [];
894
+ if (config.tabWidth !== void 0) {
895
+ observations.push(makeFormattingObs(
896
+ "formatting.indentSize",
897
+ config.tabWidth,
898
+ configPath,
899
+ 1,
900
+ "config"
901
+ ));
902
+ }
903
+ if (config.useTabs !== void 0) {
904
+ observations.push(makeFormattingObs(
905
+ "formatting.indentStyle",
906
+ config.useTabs ? "tab" : "space",
907
+ configPath,
908
+ 1,
909
+ "config"
910
+ ));
911
+ }
912
+ return observations;
913
+ }
914
+ function parseEditorConfig(raw, configPath) {
915
+ const observations = [];
916
+ const section = parseEditorConfigGlobal(raw);
917
+ if (section.indent_style) {
918
+ observations.push(makeFormattingObs(
919
+ "formatting.indentStyle",
920
+ section.indent_style,
921
+ configPath,
922
+ 1,
923
+ "config"
924
+ ));
925
+ }
926
+ if (section.indent_size) {
927
+ observations.push(makeFormattingObs(
928
+ "formatting.indentSize",
929
+ parseInt(section.indent_size, 10),
930
+ configPath,
931
+ 1,
932
+ "config"
933
+ ));
934
+ }
935
+ if (section.insert_final_newline) {
936
+ observations.push(makeFormattingObs(
937
+ "formatting.trailingNewline",
938
+ section.insert_final_newline === "true",
939
+ configPath,
940
+ 1,
941
+ "config"
942
+ ));
943
+ }
944
+ return observations;
945
+ }
946
+ function parseEditorConfigGlobal(raw) {
947
+ const result = {};
948
+ const lines = raw.split("\n");
949
+ for (const line of lines) {
950
+ const trimmed = line.trim();
951
+ if (trimmed.startsWith("#") || trimmed.startsWith("[") || !trimmed) {
952
+ continue;
953
+ }
954
+ const eqIndex = trimmed.indexOf("=");
955
+ if (eqIndex === -1) continue;
956
+ const key = trimmed.substring(0, eqIndex).trim();
957
+ const value = trimmed.substring(eqIndex + 1).trim();
958
+ if (key && value) {
959
+ result[key] = value;
960
+ }
961
+ }
962
+ return result;
963
+ }
964
+ function makeFormattingObs(type, value, file, line, source) {
965
+ return {
966
+ type,
967
+ category: "formatting",
968
+ value,
969
+ file,
970
+ line,
971
+ metadata: { source }
972
+ };
973
+ }
974
+
975
+ // src/extractors/formatting.ts
976
+ var FormattingExtractor = class {
977
+ name = "formatting";
978
+ extract(file) {
979
+ return this.extractFromSource(file.content, file.filePath);
980
+ }
981
+ async extractFromConfig(configPath) {
982
+ try {
983
+ const raw = await readFile(configPath, "utf-8");
984
+ if (configPath.endsWith(".editorconfig")) {
985
+ return parseEditorConfig(raw, configPath);
986
+ }
987
+ if (configPath.includes(".prettierrc") || configPath.includes("prettier.config")) {
988
+ return parsePrettierConfig(raw, configPath);
989
+ }
990
+ return [];
991
+ } catch {
992
+ return [];
993
+ }
994
+ }
995
+ extractFromSource(source, filePath) {
996
+ const observations = [];
997
+ const lines = source.split("\n");
998
+ observations.push(...this.detectSemicolons(lines, filePath));
999
+ observations.push(...this.detectQuoteStyle(source, filePath));
1000
+ observations.push(...this.detectTrailingCommas(source, filePath));
1001
+ observations.push(...this.detectBraceStyle(source, filePath));
1002
+ observations.push(...this.detectIndentation(lines, filePath));
1003
+ return observations;
1004
+ }
1005
+ detectSemicolons(lines, filePath) {
1006
+ let withSemi = 0;
1007
+ let withoutSemi = 0;
1008
+ for (const line of lines) {
1009
+ const trimmed = line.trim();
1010
+ if (!trimmed || this.isComment(trimmed)) continue;
1011
+ if (this.isStructuralLine(trimmed)) continue;
1012
+ if (trimmed.endsWith(";")) {
1013
+ withSemi++;
1014
+ } else if (this.isStatementStart(trimmed)) {
1015
+ withoutSemi++;
1016
+ }
1017
+ }
1018
+ const total = withSemi + withoutSemi;
1019
+ if (total === 0) return [];
1020
+ return [makeFormattingObs(
1021
+ "formatting.semicolons",
1022
+ withSemi / total > 0.5,
1023
+ filePath,
1024
+ 1,
1025
+ "frequency"
1026
+ )];
1027
+ }
1028
+ detectQuoteStyle(source, filePath) {
1029
+ let singleQuotes = 0;
1030
+ let doubleQuotes = 0;
1031
+ const stringPattern = /(?<!=)(?<!\\)(['"])((?:(?!\1|\\).|\\.)*)\1/g;
1032
+ let match;
1033
+ while ((match = stringPattern.exec(source)) !== null) {
1034
+ if (match[1] === "'") {
1035
+ singleQuotes++;
1036
+ } else {
1037
+ doubleQuotes++;
1038
+ }
1039
+ }
1040
+ const total = singleQuotes + doubleQuotes;
1041
+ if (total === 0) return [];
1042
+ return [makeFormattingObs(
1043
+ "formatting.quoteStyle",
1044
+ singleQuotes > doubleQuotes ? "single" : "double",
1045
+ filePath,
1046
+ 1,
1047
+ "frequency"
1048
+ )];
1049
+ }
1050
+ detectTrailingCommas(source, filePath) {
1051
+ const trailingCommaPattern = /,\s*[\n\r]\s*[}\]]/g;
1052
+ const noTrailingPattern = /[^,\s]\s*[\n\r]\s*[}\]]/g;
1053
+ const trailing = (source.match(trailingCommaPattern) || []).length;
1054
+ const noTrailing = (source.match(noTrailingPattern) || []).length;
1055
+ const total = trailing + noTrailing;
1056
+ if (total === 0) return [];
1057
+ return [makeFormattingObs(
1058
+ "formatting.trailingCommas",
1059
+ trailing / total > 0.5,
1060
+ filePath,
1061
+ 1,
1062
+ "frequency"
1063
+ )];
1064
+ }
1065
+ detectBraceStyle(source, filePath) {
1066
+ const sameLine = (source.match(/\)[^\S\n]*\{/g) || []).length;
1067
+ const nextLine = (source.match(/\)\s*\n\s*\{/g) || []).length;
1068
+ const total = sameLine + nextLine;
1069
+ if (total === 0) return [];
1070
+ return [makeFormattingObs(
1071
+ "formatting.braceStyle",
1072
+ nextLine / total > 0.5 ? "allman" : "1tbs",
1073
+ filePath,
1074
+ 1,
1075
+ "frequency"
1076
+ )];
1077
+ }
1078
+ detectIndentation(lines, filePath) {
1079
+ const { tabCount, spaceCount, spaceSizes } = this.countIndentation(lines);
1080
+ const observations = [];
1081
+ const total = tabCount + spaceCount;
1082
+ if (total === 0) return observations;
1083
+ observations.push(makeFormattingObs(
1084
+ "formatting.indentStyle",
1085
+ tabCount > spaceCount ? "tab" : "space",
1086
+ filePath,
1087
+ 1,
1088
+ "frequency"
1089
+ ));
1090
+ if (spaceCount > tabCount && spaceSizes.length > 0) {
1091
+ const gcd = this.findGcdOfArray(spaceSizes.filter((s) => s > 0));
1092
+ observations.push(makeFormattingObs(
1093
+ "formatting.indentSize",
1094
+ gcd,
1095
+ filePath,
1096
+ 1,
1097
+ "frequency"
1098
+ ));
1099
+ }
1100
+ return observations;
1101
+ }
1102
+ countIndentation(lines) {
1103
+ let tabCount = 0;
1104
+ let spaceCount = 0;
1105
+ const spaceSizes = [];
1106
+ for (const line of lines) {
1107
+ if (!line || line.trim() === "") continue;
1108
+ const leadingWhitespace = line.match(/^(\s+)/);
1109
+ if (!leadingWhitespace) continue;
1110
+ const ws = leadingWhitespace[1];
1111
+ if (ws.includes(" ")) {
1112
+ tabCount++;
1113
+ } else if (ws.length > 0) {
1114
+ spaceCount++;
1115
+ spaceSizes.push(ws.length);
1116
+ }
1117
+ }
1118
+ return { tabCount, spaceCount, spaceSizes };
1119
+ }
1120
+ isComment(trimmed) {
1121
+ return trimmed.startsWith("//") || trimmed.startsWith("/*") || trimmed.startsWith("*");
1122
+ }
1123
+ isStructuralLine(trimmed) {
1124
+ return trimmed.endsWith("{") || trimmed.endsWith("}") || trimmed.endsWith("(") || trimmed.endsWith(",");
1125
+ }
1126
+ isStatementStart(trimmed) {
1127
+ return trimmed.startsWith("const ") || trimmed.startsWith("let ") || trimmed.startsWith("var ") || trimmed.startsWith("return ") || trimmed.startsWith("import ") || trimmed.startsWith("export ");
1128
+ }
1129
+ findGcdOfArray(nums) {
1130
+ if (nums.length === 0) return 2;
1131
+ return nums.reduce((a, b) => this.gcd(a, b));
1132
+ }
1133
+ gcd(a, b) {
1134
+ while (b) {
1135
+ [a, b] = [b, a % b];
1136
+ }
1137
+ return a;
1138
+ }
1139
+ };
1140
+
1141
+ // src/extractors/complexity.ts
1142
+ var TS_FUNCTION_TYPES = /* @__PURE__ */ new Set([
1143
+ "function_declaration",
1144
+ "method_definition"
1145
+ ]);
1146
+ var PY_FUNCTION_TYPES = /* @__PURE__ */ new Set([
1147
+ "function_definition"
1148
+ ]);
1149
+ var TS_NESTING_TYPES = /* @__PURE__ */ new Set([
1150
+ "if_statement",
1151
+ "for_statement",
1152
+ "for_in_statement",
1153
+ "while_statement",
1154
+ "do_statement",
1155
+ "switch_statement",
1156
+ "try_statement"
1157
+ ]);
1158
+ var PY_NESTING_TYPES = /* @__PURE__ */ new Set([
1159
+ "if_statement",
1160
+ "for_statement",
1161
+ "while_statement",
1162
+ "try_statement"
1163
+ ]);
1164
+ var TS_BRANCH_TYPES = /* @__PURE__ */ new Set([
1165
+ "if_statement",
1166
+ "for_statement",
1167
+ "for_in_statement",
1168
+ "while_statement",
1169
+ "do_statement",
1170
+ "switch_case",
1171
+ "catch_clause",
1172
+ "ternary_expression"
1173
+ ]);
1174
+ var PY_BRANCH_TYPES = /* @__PURE__ */ new Set([
1175
+ "if_statement",
1176
+ "elif_clause",
1177
+ "for_statement",
1178
+ "while_statement",
1179
+ "except_clause",
1180
+ "conditional_expression"
1181
+ ]);
1182
+ var ComplexityExtractor = class {
1183
+ name = "complexity";
1184
+ extract(file) {
1185
+ const observations = [];
1186
+ const lineCount = file.content.split("\n").filter((l) => l.trim() !== "").length;
1187
+ observations.push({
1188
+ type: "complexity.fileLength",
1189
+ category: "complexity",
1190
+ value: lineCount,
1191
+ file: file.filePath,
1192
+ line: 1
1193
+ });
1194
+ const functionTypes = this.getFunctionTypes(file.language);
1195
+ const functions = this.findFunctions(
1196
+ file.tree.rootNode,
1197
+ functionTypes,
1198
+ file.language
1199
+ );
1200
+ for (const fn of functions) {
1201
+ observations.push(...this.functionObservations(fn, file.filePath));
1202
+ }
1203
+ return observations;
1204
+ }
1205
+ functionObservations(fn, filePath) {
1206
+ const metrics = [
1207
+ ["complexity.functionLength", fn.statementCount],
1208
+ ["complexity.nestingDepth", fn.maxNestingDepth],
1209
+ ["complexity.cyclomatic", fn.cyclomaticComplexity]
1210
+ ];
1211
+ return metrics.map(([type, value]) => ({
1212
+ type,
1213
+ category: "complexity",
1214
+ value,
1215
+ file: filePath,
1216
+ line: fn.line,
1217
+ metadata: { functionName: fn.name }
1218
+ }));
1219
+ }
1220
+ getFunctionTypes(language) {
1221
+ return language === "python" ? PY_FUNCTION_TYPES : TS_FUNCTION_TYPES;
1222
+ }
1223
+ findFunctions(root, functionTypes, language) {
1224
+ const functions = [];
1225
+ const visit = (node) => {
1226
+ if (functionTypes.has(node.type)) {
1227
+ const info = this.analyzeFunctionNode(node, language);
1228
+ if (info) functions.push(info);
1229
+ }
1230
+ for (const child of node.children) {
1231
+ visit(child);
1232
+ }
1233
+ };
1234
+ visit(root);
1235
+ return functions;
1236
+ }
1237
+ analyzeFunctionNode(node, language) {
1238
+ const nameNode = node.childForFieldName("name");
1239
+ if (!nameNode) return null;
1240
+ const name = nameNode.text;
1241
+ const body = node.childForFieldName("body");
1242
+ if (!body) return null;
1243
+ const statementCount = this.countStatements(body, language);
1244
+ const maxNestingDepth = this.measureNestingDepth(body, language, 0);
1245
+ const cyclomaticComplexity = this.measureCyclomaticComplexity(
1246
+ body,
1247
+ language
1248
+ );
1249
+ return {
1250
+ name,
1251
+ statementCount,
1252
+ maxNestingDepth,
1253
+ cyclomaticComplexity,
1254
+ line: node.startPosition.row + 1
1255
+ };
1256
+ }
1257
+ countStatements(body, language) {
1258
+ let count = 0;
1259
+ for (const child of body.namedChildren) {
1260
+ if (this.isStatement(child, language)) {
1261
+ count++;
1262
+ }
1263
+ }
1264
+ return count;
1265
+ }
1266
+ isStatement(node, language) {
1267
+ if (language === "python") {
1268
+ return this.isPythonStatement(node);
1269
+ }
1270
+ return this.isTypeScriptStatement(node);
1271
+ }
1272
+ isTypeScriptStatement(node) {
1273
+ const statementTypes = /* @__PURE__ */ new Set([
1274
+ "lexical_declaration",
1275
+ "variable_declaration",
1276
+ "expression_statement",
1277
+ "return_statement",
1278
+ "if_statement",
1279
+ "for_statement",
1280
+ "for_in_statement",
1281
+ "while_statement",
1282
+ "do_statement",
1283
+ "switch_statement",
1284
+ "try_statement",
1285
+ "throw_statement",
1286
+ "break_statement",
1287
+ "continue_statement"
1288
+ ]);
1289
+ return statementTypes.has(node.type);
1290
+ }
1291
+ isPythonStatement(node) {
1292
+ const statementTypes = /* @__PURE__ */ new Set([
1293
+ "expression_statement",
1294
+ "return_statement",
1295
+ "if_statement",
1296
+ "for_statement",
1297
+ "while_statement",
1298
+ "try_statement",
1299
+ "raise_statement",
1300
+ "assert_statement",
1301
+ "pass_statement",
1302
+ "break_statement",
1303
+ "continue_statement",
1304
+ "assignment",
1305
+ "augmented_assignment"
1306
+ ]);
1307
+ return statementTypes.has(node.type);
1308
+ }
1309
+ measureNestingDepth(node, language, depth) {
1310
+ const nestingTypes = language === "python" ? PY_NESTING_TYPES : TS_NESTING_TYPES;
1311
+ let maxDepth = depth;
1312
+ for (const child of node.namedChildren) {
1313
+ if (nestingTypes.has(child.type)) {
1314
+ const childMax = this.measureNestingDepth(
1315
+ child,
1316
+ language,
1317
+ depth + 1
1318
+ );
1319
+ if (childMax > maxDepth) maxDepth = childMax;
1320
+ } else {
1321
+ const childMax = this.measureNestingDepth(
1322
+ child,
1323
+ language,
1324
+ depth
1325
+ );
1326
+ if (childMax > maxDepth) maxDepth = childMax;
1327
+ }
1328
+ }
1329
+ return maxDepth;
1330
+ }
1331
+ measureCyclomaticComplexity(body, language) {
1332
+ let complexity = 1;
1333
+ const branchTypes = language === "python" ? PY_BRANCH_TYPES : TS_BRANCH_TYPES;
1334
+ const visit = (node) => {
1335
+ complexity += this.branchIncrement(node, language, branchTypes);
1336
+ for (const child of node.namedChildren) {
1337
+ visit(child);
1338
+ }
1339
+ };
1340
+ visit(body);
1341
+ return complexity;
1342
+ }
1343
+ branchIncrement(node, language, branchTypes) {
1344
+ let increment = 0;
1345
+ if (branchTypes.has(node.type)) {
1346
+ increment++;
1347
+ }
1348
+ if (node.type === "binary_expression") {
1349
+ const operator = node.childForFieldName("operator");
1350
+ if (operator) {
1351
+ const text = operator.text;
1352
+ if (text === "&&" || text === "||") {
1353
+ increment++;
1354
+ }
1355
+ }
1356
+ }
1357
+ if (language === "python" && node.type === "boolean_operator") {
1358
+ increment++;
1359
+ }
1360
+ return increment;
1361
+ }
1362
+ };
1363
+
1364
+ // src/extractors/idioms.ts
1365
+ var IdiomsExtractor = class {
1366
+ name = "idioms";
1367
+ minLines;
1368
+ minTokens;
1369
+ constructor(options) {
1370
+ this.minLines = options?.minLines ?? 3;
1371
+ this.minTokens = options?.minTokens ?? 25;
1372
+ }
1373
+ extract(_file) {
1374
+ return [];
1375
+ }
1376
+ async extractFromSources(sources) {
1377
+ const observations = [];
1378
+ const clones = await this.detectClones(sources);
1379
+ const groups = this.groupClones(clones);
1380
+ for (const group of groups.values()) {
1381
+ const frequency = group.instances.length;
1382
+ if (frequency < 2) continue;
1383
+ observations.push(this.cloneObservation(group));
1384
+ }
1385
+ return observations;
1386
+ }
1387
+ cloneObservation(group) {
1388
+ const frequency = group.instances.length;
1389
+ const first = group.instances[0];
1390
+ return {
1391
+ type: "idiom.clone",
1392
+ category: "idioms",
1393
+ value: this.summarizeClone(first.fragment),
1394
+ file: first.sourceFile,
1395
+ line: first.startLine,
1396
+ metadata: {
1397
+ frequency,
1398
+ fragment: first.fragment,
1399
+ linesCount: group.linesCount,
1400
+ locations: group.instances.map((inst) => ({
1401
+ file: inst.sourceFile,
1402
+ startLine: inst.startLine,
1403
+ endLine: inst.endLine
1404
+ }))
1405
+ }
1406
+ };
1407
+ }
1408
+ async detectClones(sources) {
1409
+ const { allClones, sourceMap } = await this.runDetector(sources);
1410
+ return allClones.map((clone) => this.toCloneGroup(clone, sourceMap));
1411
+ }
1412
+ async runDetector(sources) {
1413
+ const { Detector, MemoryStore } = await import("@jscpd/core");
1414
+ const { Tokenizer } = await import("@jscpd/tokenizer");
1415
+ const options = {
1416
+ minLines: this.minLines,
1417
+ minTokens: this.minTokens
1418
+ };
1419
+ const store = new MemoryStore();
1420
+ const tokenizer = new Tokenizer();
1421
+ const detector = new Detector(tokenizer, store, [], options);
1422
+ const allClones = [];
1423
+ const sourceMap = /* @__PURE__ */ new Map();
1424
+ for (const source of sources) {
1425
+ sourceMap.set(source.path, source.content);
1426
+ const format = this.languageToFormat(source.language);
1427
+ const detected = await detector.detect(
1428
+ source.path,
1429
+ source.content,
1430
+ format
1431
+ );
1432
+ allClones.push(...detected);
1433
+ }
1434
+ return { allClones, sourceMap };
1435
+ }
1436
+ toCloneGroup(clone, sourceMap) {
1437
+ return {
1438
+ instances: [
1439
+ this.toInstance(clone.duplicationA, sourceMap),
1440
+ this.toInstance(clone.duplicationB, sourceMap)
1441
+ ],
1442
+ linesCount: clone.duplicationA.end.line - clone.duplicationA.start.line + 1
1443
+ };
1444
+ }
1445
+ toInstance(duplication, sourceMap) {
1446
+ const content = sourceMap.get(duplication.sourceId) ?? "";
1447
+ return {
1448
+ sourceFile: duplication.sourceId,
1449
+ startLine: duplication.start.line,
1450
+ endLine: duplication.end.line,
1451
+ fragment: this.extractFragment(
1452
+ content,
1453
+ duplication.start.line,
1454
+ duplication.end.line
1455
+ )
1456
+ };
1457
+ }
1458
+ extractFragment(content, startLine, endLine) {
1459
+ const lines = content.split("\n");
1460
+ return lines.slice(startLine - 1, endLine).join("\n");
1461
+ }
1462
+ languageToFormat(language) {
1463
+ const mapping = {
1464
+ typescript: "typescript",
1465
+ javascript: "javascript",
1466
+ python: "python",
1467
+ tsx: "tsx",
1468
+ jsx: "jsx"
1469
+ };
1470
+ return mapping[language] ?? language;
1471
+ }
1472
+ groupClones(clones) {
1473
+ const groups = /* @__PURE__ */ new Map();
1474
+ for (const clone of clones) {
1475
+ const key = this.normalizeFragment(clone.instances[0]?.fragment ?? "");
1476
+ const existing = groups.get(key);
1477
+ if (existing) {
1478
+ this.mergeInstances(existing, clone);
1479
+ } else {
1480
+ groups.set(key, {
1481
+ instances: [...clone.instances],
1482
+ linesCount: clone.linesCount
1483
+ });
1484
+ }
1485
+ }
1486
+ return groups;
1487
+ }
1488
+ mergeInstances(existing, clone) {
1489
+ for (const inst of clone.instances) {
1490
+ const alreadyTracked = existing.instances.some(
1491
+ (e) => e.sourceFile === inst.sourceFile && e.startLine === inst.startLine
1492
+ );
1493
+ if (!alreadyTracked) {
1494
+ existing.instances.push(inst);
1495
+ }
1496
+ }
1497
+ }
1498
+ normalizeFragment(fragment) {
1499
+ return fragment.replace(/\s+/g, " ").trim().substring(0, 200);
1500
+ }
1501
+ summarizeClone(fragment) {
1502
+ const firstLine = fragment.split("\n")[0]?.trim() ?? "";
1503
+ if (firstLine.length > 80) {
1504
+ return firstLine.substring(0, 77) + "...";
1505
+ }
1506
+ return firstLine;
1507
+ }
1508
+ };
1509
+
1510
+ // src/extractors/review-voice.ts
1511
+ var TOPIC_PATTERNS = [
1512
+ {
1513
+ topic: "naming",
1514
+ patterns: [
1515
+ /\bnam(?:e|ing)\b/i,
1516
+ /\brenam(?:e|ing)\b/i,
1517
+ /\bcamelCase\b/i,
1518
+ /\bsnake_case\b/i,
1519
+ /\bdescriptive\b/i,
1520
+ /\bconfusing\b/i,
1521
+ /\bprefer\s+\S+\s+over\b/i
1522
+ ]
1523
+ },
1524
+ {
1525
+ topic: "error-handling",
1526
+ patterns: [
1527
+ /\berror\s*handl/i,
1528
+ /\bmissing\s+error/i,
1529
+ /\btry\s*\/?\s*catch\b/i,
1530
+ /\bwhat\s+happens\s+if\b/i,
1531
+ /\bedge\s*case/i,
1532
+ /\bnull\b/i,
1533
+ /\bfails?\b/i
1534
+ ]
1535
+ },
1536
+ {
1537
+ topic: "complexity",
1538
+ patterns: [
1539
+ /\bcomplex\b/i,
1540
+ /\btoo\s+long\b/i,
1541
+ /\bsplit\b.*\bup\b/i,
1542
+ /\bextract\b.*\bfunction/i,
1543
+ /\bsmaller\s+functions?\b/i,
1544
+ /\bsimplif/i,
1545
+ /\bnesting\b/i
1546
+ ]
1547
+ },
1548
+ {
1549
+ topic: "style",
1550
+ patterns: [
1551
+ /\bstyle\b/i,
1552
+ /\bformat/i,
1553
+ /\bconsistenc/i,
1554
+ /\bnit\b/i,
1555
+ /\bquotes?\b/i,
1556
+ /\bsemicolon/i,
1557
+ /\bindent/i,
1558
+ /\bwhitespace\b/i
1559
+ ]
1560
+ },
1561
+ {
1562
+ topic: "performance",
1563
+ patterns: [
1564
+ /\bperforman/i,
1565
+ /\bmemoiz/i,
1566
+ /\bre-?render/i,
1567
+ /\boptimiz/i,
1568
+ /\bexpensive\b/i,
1569
+ /\befficien/i,
1570
+ /\bcach/i
1571
+ ]
1572
+ },
1573
+ {
1574
+ topic: "documentation",
1575
+ patterns: [
1576
+ /\bdocument/i,
1577
+ /\bjsdoc\b/i
1578
+ ]
1579
+ },
1580
+ {
1581
+ topic: "testing",
1582
+ patterns: [
1583
+ /\btest/i,
1584
+ /\bcover(?:age)?\b/i,
1585
+ /\bassert/i,
1586
+ /\bmock/i,
1587
+ /\bspec\b/i
1588
+ ]
1589
+ },
1590
+ {
1591
+ topic: "security",
1592
+ patterns: [
1593
+ /\bsecur/i,
1594
+ /\bvulnerab/i,
1595
+ /\bsaniti[zs]/i,
1596
+ /\binject/i,
1597
+ /\bescap/i,
1598
+ /\bxss\b/i
1599
+ ]
1600
+ },
1601
+ {
1602
+ topic: "readability",
1603
+ patterns: [
1604
+ /\breadab/i,
1605
+ /\bearly\s+return/i,
1606
+ /\bguard\s+clause/i
1607
+ ]
1608
+ },
1609
+ {
1610
+ topic: "structure",
1611
+ patterns: [
1612
+ /\bstructur/i,
1613
+ /\barchitect/i,
1614
+ /\borganiz/i,
1615
+ /\bseparati/i
1616
+ ]
1617
+ }
1618
+ ];
1619
+ var KEYWORD_PATTERNS = [
1620
+ { keyword: "early-return", pattern: /\bearly\s+return\b/i },
1621
+ { keyword: "guard-clause", pattern: /\bguard\s+clause\b/i },
1622
+ {
1623
+ keyword: "single-responsibility",
1624
+ pattern: /\bsingle\s+responsib/i
1625
+ },
1626
+ { keyword: "dry", pattern: /\b(?:DRY|don'?t\s+repeat)\b/i },
1627
+ { keyword: "immutability", pattern: /\bimmutab/i },
1628
+ { keyword: "type-safety", pattern: /\btype[\s-]*safe/i },
1629
+ { keyword: "null-check", pattern: /\bnull\s+check/i },
1630
+ { keyword: "magic-number", pattern: /\bmagic\s+number/i }
1631
+ ];
1632
+ var ReviewVoiceExtractor = class {
1633
+ name = "reviewVoice";
1634
+ extract(_file) {
1635
+ return [];
1636
+ }
1637
+ extractFromComments(comments) {
1638
+ if (comments.length === 0) return [];
1639
+ const observations = [];
1640
+ observations.push(...this.categorizeTopics(comments));
1641
+ observations.push(...this.extractKeywords(comments));
1642
+ return observations;
1643
+ }
1644
+ categorizeTopics(comments) {
1645
+ const { topicCounts, topicExamples } = this.countTopics(comments);
1646
+ const observations = [];
1647
+ for (const [topic, count] of topicCounts) {
1648
+ observations.push({
1649
+ type: "reviewVoice.topicFrequency",
1650
+ category: "reviewVoice",
1651
+ value: topic,
1652
+ file: "_reviews",
1653
+ line: 0,
1654
+ metadata: {
1655
+ count,
1656
+ total: comments.length,
1657
+ ratio: count / comments.length,
1658
+ examples: topicExamples.get(topic) ?? []
1659
+ }
1660
+ });
1661
+ }
1662
+ observations.sort(
1663
+ (a, b) => (b.metadata?.count ?? 0) - (a.metadata?.count ?? 0)
1664
+ );
1665
+ return observations;
1666
+ }
1667
+ countTopics(comments) {
1668
+ const topicCounts = /* @__PURE__ */ new Map();
1669
+ const topicExamples = /* @__PURE__ */ new Map();
1670
+ for (const comment of comments) {
1671
+ for (const topic of this.matchTopics(comment.body)) {
1672
+ topicCounts.set(topic, (topicCounts.get(topic) ?? 0) + 1);
1673
+ const examples = topicExamples.get(topic) ?? [];
1674
+ if (examples.length < 3) {
1675
+ examples.push(comment.body);
1676
+ topicExamples.set(topic, examples);
1677
+ }
1678
+ }
1679
+ }
1680
+ return { topicCounts, topicExamples };
1681
+ }
1682
+ matchTopics(body) {
1683
+ const matchedTopics = /* @__PURE__ */ new Set();
1684
+ for (const { topic, patterns } of TOPIC_PATTERNS) {
1685
+ for (const pattern of patterns) {
1686
+ if (pattern.test(body)) {
1687
+ matchedTopics.add(topic);
1688
+ break;
1689
+ }
1690
+ }
1691
+ }
1692
+ return matchedTopics;
1693
+ }
1694
+ extractKeywords(comments) {
1695
+ const keywordCounts = /* @__PURE__ */ new Map();
1696
+ for (const comment of comments) {
1697
+ for (const { keyword, pattern } of KEYWORD_PATTERNS) {
1698
+ if (pattern.test(comment.body)) {
1699
+ keywordCounts.set(keyword, (keywordCounts.get(keyword) ?? 0) + 1);
1700
+ }
1701
+ }
1702
+ }
1703
+ const observations = [];
1704
+ for (const [keyword, count] of keywordCounts) {
1705
+ observations.push({
1706
+ type: "reviewVoice.keyword",
1707
+ category: "reviewVoice",
1708
+ value: keyword,
1709
+ file: "_reviews",
1710
+ line: 0,
1711
+ metadata: {
1712
+ count,
1713
+ total: comments.length
1714
+ }
1715
+ });
1716
+ }
1717
+ return observations;
1718
+ }
1719
+ };
1720
+
1721
+ // src/extractors/factory.ts
1722
+ function createStyleExtractors() {
1723
+ return [
1724
+ new NamingExtractor(),
1725
+ new StructureExtractor(),
1726
+ new ControlFlowExtractor(),
1727
+ new DocumentationExtractor(),
1728
+ new ErrorHandlingExtractor(),
1729
+ new FormattingExtractor(),
1730
+ new ComplexityExtractor(),
1731
+ new IdiomsExtractor(),
1732
+ new ReviewVoiceExtractor()
1733
+ ];
1734
+ }
1735
+
1736
+ // src/aggregator/confidence.ts
1737
+ import { DEFAULT_SEVERITY_THRESHOLDS } from "@titan-design/style-profile";
1738
+ var DEFAULT_STABILITY_WEIGHTS = {
1739
+ high: 1,
1740
+ medium: 0.85,
1741
+ low: 0.7
1742
+ };
1743
+ function computeConfidence(consistency, stability, weights = DEFAULT_STABILITY_WEIGHTS) {
1744
+ return Math.min(1, consistency * weights[stability]);
1745
+ }
1746
+ function mapSeverity(confidence, thresholds = DEFAULT_SEVERITY_THRESHOLDS) {
1747
+ if (confidence >= thresholds.error) return "error";
1748
+ if (confidence >= thresholds.warn) return "warn";
1749
+ if (confidence >= thresholds.info) return "info";
1750
+ return "off";
1751
+ }
1752
+
1753
+ // src/aggregator/frequency.ts
1754
+ function groupByType(observations) {
1755
+ const groups = /* @__PURE__ */ new Map();
1756
+ for (const obs of observations) {
1757
+ const existing = groups.get(obs.type) ?? [];
1758
+ existing.push(obs);
1759
+ groups.set(obs.type, existing);
1760
+ }
1761
+ return groups;
1762
+ }
1763
+ function computeDistribution(observations) {
1764
+ const valueCounts = /* @__PURE__ */ new Map();
1765
+ for (const obs of observations) {
1766
+ const key = normalizeValue(obs.value);
1767
+ valueCounts.set(key, (valueCounts.get(key) ?? 0) + 1);
1768
+ }
1769
+ const total = observations.length;
1770
+ let dominant = "";
1771
+ let maxCount = 0;
1772
+ for (const [value, count] of valueCounts) {
1773
+ if (count > maxCount) {
1774
+ maxCount = count;
1775
+ dominant = value;
1776
+ }
1777
+ }
1778
+ return {
1779
+ values: valueCounts,
1780
+ total,
1781
+ dominant,
1782
+ consistency: total > 0 ? maxCount / total : 0
1783
+ };
1784
+ }
1785
+ function normalizeValue(value) {
1786
+ if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") {
1787
+ return value;
1788
+ }
1789
+ if (Array.isArray(value)) {
1790
+ return JSON.stringify(value);
1791
+ }
1792
+ return String(value);
1793
+ }
1794
+ function selectExamples(observations, maxExamples) {
1795
+ if (observations.length <= maxExamples) {
1796
+ return [...observations];
1797
+ }
1798
+ const step = Math.floor(observations.length / maxExamples);
1799
+ const examples = [];
1800
+ for (let i = 0; i < observations.length && examples.length < maxExamples; i += step) {
1801
+ examples.push(observations[i]);
1802
+ }
1803
+ return examples;
1804
+ }
1805
+
1806
+ // src/aggregator/stability.ts
1807
+ var STABILITY_MAP = {
1808
+ // Category 1: Naming Conventions
1809
+ "naming.variables": "high",
1810
+ "naming.functions": "high",
1811
+ "naming.types": "high",
1812
+ "naming.constants": "high",
1813
+ "naming.files": "high",
1814
+ "naming.booleans": "medium",
1815
+ "naming.abbreviations": "medium",
1816
+ "naming.parameters": "medium",
1817
+ "naming.enums": "high",
1818
+ "naming.privateMembers": "high",
1819
+ // Category 2: Code Structure
1820
+ "structure.import-group": "high",
1821
+ "structure.import-order": "high",
1822
+ "structure.importPathStyle": "medium",
1823
+ "structure.typeImportSeparation": "medium",
1824
+ "structure.export-style": "high",
1825
+ "structure.barrel-file": "medium",
1826
+ "structure.export-proximity": "medium",
1827
+ "structure.functionLength": "high",
1828
+ "structure.nestingDepth": "high",
1829
+ "structure.fileLength": "medium",
1830
+ "structure.moduleTopology": "low",
1831
+ "structure.fileOrganization": "low",
1832
+ // Category 3: Control Flow Patterns
1833
+ "controlFlow.guardClauses": "high",
1834
+ "controlFlow.earlyReturn": "high",
1835
+ "controlFlow.ternaryPreference": "medium",
1836
+ "controlFlow.arrayMethods": "high",
1837
+ "controlFlow.forStyle": "medium",
1838
+ "controlFlow.asyncAwait": "high",
1839
+ "controlFlow.switchVsIf": "medium",
1840
+ "controlFlow.optionalChaining": "medium",
1841
+ "controlFlow.nullishCoalescing": "low",
1842
+ // Category 4: Error Handling
1843
+ "errorHandling.tryCatchFrequency": "high",
1844
+ "errorHandling.catchSpecificity": "medium",
1845
+ "errorHandling.resultType": "high",
1846
+ "errorHandling.errorReturnTuples": "medium",
1847
+ "errorHandling.customErrorClasses": "medium",
1848
+ "errorHandling.exhaustiveSwitch": "high",
1849
+ "errorHandling.assertNever": "high",
1850
+ "errorHandling.floatingPromises": "medium",
1851
+ "errorHandling.errorBoundary": "low",
1852
+ // Category 5: Documentation
1853
+ "documentation.jsdocPresence": "high",
1854
+ "documentation.publicPrivateCoverage": "medium",
1855
+ "documentation.inlineCommentDensity": "medium",
1856
+ "documentation.commentPlacement": "medium",
1857
+ "documentation.sectionComments": "low",
1858
+ "documentation.moduleHeaders": "medium",
1859
+ "documentation.jsdocTags": "medium",
1860
+ "documentation.voice": "low",
1861
+ "documentation.whyVsWhat": "low",
1862
+ "documentation.redundancy": "low",
1863
+ // Category 6: Type System Usage
1864
+ "typeSystem.annotationDensity": "high",
1865
+ "typeSystem.explicitReturn": "high",
1866
+ "typeSystem.moduleBoundaryTypes": "medium",
1867
+ "typeSystem.inferrableTypes": "medium",
1868
+ "typeSystem.interfaceVsType": "medium",
1869
+ "typeSystem.genericUsage": "low",
1870
+ "typeSystem.readonlyUsage": "medium",
1871
+ "typeSystem.discriminatedUnions": "medium",
1872
+ "typeSystem.utilityTypes": "low",
1873
+ // Category 7: Formatting & Layout
1874
+ "formatting.indentStyle": "high",
1875
+ "formatting.indentSize": "high",
1876
+ "formatting.semicolons": "high",
1877
+ "formatting.quoteStyle": "high",
1878
+ "formatting.trailingCommas": "high",
1879
+ "formatting.braceStyle": "high",
1880
+ "formatting.lineLength": "medium",
1881
+ "formatting.blankLines": "medium",
1882
+ "formatting.destructuring": "medium",
1883
+ "formatting.defaultParams": "low",
1884
+ "formatting.arrowVsFunction": "medium",
1885
+ "formatting.trailingNewline": "high",
1886
+ // Category 8: Higher-Level Patterns
1887
+ "patterns.compositionVsInheritance": "medium",
1888
+ "patterns.classVsFunctional": "high",
1889
+ "patterns.pureFunctions": "low",
1890
+ "patterns.immutability": "medium",
1891
+ "patterns.explicitVsImplicit": "medium",
1892
+ "patterns.dryAdherence": "medium",
1893
+ // Category 9: Habitual Idioms
1894
+ "idiom.clone": "high",
1895
+ "idiom.errorHandlingShape": "medium",
1896
+ "idiom.dataTransformation": "medium",
1897
+ "idiom.apiCallPattern": "medium",
1898
+ "idiom.testStructure": "medium",
1899
+ // Category 10: Review Voice
1900
+ "reviewVoice.topicFrequency": "medium",
1901
+ "reviewVoice.keyword": "medium",
1902
+ "reviewVoice.tone": "low",
1903
+ "reviewVoice.themes": "low",
1904
+ "reviewVoice.values": "low",
1905
+ // Complexity (from task-07)
1906
+ "complexity.functionLength": "high",
1907
+ "complexity.nestingDepth": "high",
1908
+ "complexity.cyclomatic": "high",
1909
+ "complexity.fileLength": "medium"
1910
+ };
1911
+ function lookupStability(type) {
1912
+ if (STABILITY_MAP[type]) return STABILITY_MAP[type];
1913
+ const category = type.indexOf(".") > 0 ? type.substring(0, type.indexOf(".")) : type;
1914
+ if (STABILITY_MAP[category]) return STABILITY_MAP[category];
1915
+ return "medium";
1916
+ }
1917
+
1918
+ // src/aggregator/aggregator.ts
1919
+ var Aggregator = class {
1920
+ stabilityWeights;
1921
+ severityThresholds;
1922
+ reviewThreshold;
1923
+ maxExamples;
1924
+ constructor(config) {
1925
+ this.stabilityWeights = config?.stabilityWeights ?? DEFAULT_STABILITY_WEIGHTS;
1926
+ this.severityThresholds = config?.severityThresholds ?? DEFAULT_SEVERITY_THRESHOLDS;
1927
+ this.reviewThreshold = config?.reviewThreshold ?? 0.6;
1928
+ this.maxExamples = config?.maxExamples ?? 5;
1929
+ }
1930
+ aggregate(observations) {
1931
+ const grouped = groupByType(observations);
1932
+ const features = /* @__PURE__ */ new Map();
1933
+ const reviewQueue = [];
1934
+ for (const [type, typeObservations] of grouped) {
1935
+ const feature = this.buildFeature(type, typeObservations);
1936
+ features.set(type, feature);
1937
+ if (feature.needsReview) {
1938
+ reviewQueue.push(feature);
1939
+ }
1940
+ }
1941
+ reviewQueue.sort((a, b) => a.confidence - b.confidence);
1942
+ return {
1943
+ features,
1944
+ reviewQueue,
1945
+ summary: {
1946
+ totalObservations: observations.length,
1947
+ totalFeatures: features.size,
1948
+ avgConfidence: averageConfidence(features),
1949
+ featuresNeedingReview: reviewQueue.length
1950
+ }
1951
+ };
1952
+ }
1953
+ buildFeature(type, typeObservations) {
1954
+ const distribution = computeDistribution(typeObservations);
1955
+ const stability = lookupStability(type);
1956
+ const confidence = computeConfidence(
1957
+ distribution.consistency,
1958
+ stability,
1959
+ this.stabilityWeights
1960
+ );
1961
+ return {
1962
+ type,
1963
+ category: this.extractCategory(type),
1964
+ convention: distribution.dominant,
1965
+ distribution,
1966
+ confidence,
1967
+ stability,
1968
+ severity: mapSeverity(confidence, this.severityThresholds),
1969
+ needsReview: confidence < this.reviewThreshold,
1970
+ examples: selectExamples(typeObservations, this.maxExamples)
1971
+ };
1972
+ }
1973
+ extractCategory(type) {
1974
+ const dotIndex = type.indexOf(".");
1975
+ return dotIndex > 0 ? type.substring(0, dotIndex) : type;
1976
+ }
1977
+ };
1978
+ function averageConfidence(features) {
1979
+ const confidences = Array.from(features.values()).map((f) => f.confidence);
1980
+ return confidences.length > 0 ? confidences.reduce((a, b) => a + b, 0) / confidences.length : 0;
1981
+ }
1982
+
1983
+ // src/enricher/llm-runner.ts
1984
+ var LlmRunner = class {
1985
+ provider;
1986
+ totalTokenBudget;
1987
+ constructor(config) {
1988
+ this.provider = config.provider;
1989
+ this.totalTokenBudget = config.totalTokenBudget ?? 2e4;
1990
+ }
1991
+ async run(jobs) {
1992
+ const results = [];
1993
+ const errors = [];
1994
+ let totalTokensUsed = 0;
1995
+ let budgetExceeded = false;
1996
+ for (const job of jobs) {
1997
+ if (totalTokensUsed >= this.totalTokenBudget) {
1998
+ budgetExceeded = true;
1999
+ break;
2000
+ }
2001
+ totalTokensUsed += await this.runJob(job, results, errors);
2002
+ }
2003
+ return { results, errors, totalTokensUsed, budgetExceeded };
2004
+ }
2005
+ // Returns the tokens the job consumed, which is zero when the provider throws.
2006
+ async runJob(job, results, errors) {
2007
+ try {
2008
+ const response = await this.provider.generate(job.messages, {
2009
+ maxTokens: job.maxTokens
2010
+ });
2011
+ results.push({
2012
+ key: job.key,
2013
+ content: response.content,
2014
+ tokensUsed: response.tokensUsed
2015
+ });
2016
+ return response.tokensUsed;
2017
+ } catch (error) {
2018
+ errors.push({
2019
+ key: job.key,
2020
+ error: error instanceof Error ? error.message : String(error)
2021
+ });
2022
+ return 0;
2023
+ }
2024
+ }
2025
+ };
2026
+
2027
+ // src/enricher/prompts.ts
2028
+ var DESCRIPTION_PROMPT = {
2029
+ featureTypes: [
2030
+ "documentation.voice",
2031
+ "documentation.whyVsWhat",
2032
+ "documentation.redundancy",
2033
+ "patterns.pureFunctions",
2034
+ "patterns.explicitVsImplicit",
2035
+ "errorHandling.errorBoundary",
2036
+ "structure.fileOrganization"
2037
+ ],
2038
+ system: "You are a code style analyst. Given statistical observations about a developer's coding patterns, write a concise, actionable style rule description. Output ONLY the description text (1-3 sentences). Do not include markdown formatting or headers.",
2039
+ buildUserMessage: (input) => {
2040
+ const lines = [
2041
+ `Feature: ${input.featureType}`,
2042
+ `Dominant pattern: ${JSON.stringify(input.convention)}`,
2043
+ `Confidence: ${(input.confidence * 100).toFixed(0)}%`,
2044
+ `Consistency: ${(input.consistency * 100).toFixed(0)}%`
2045
+ ];
2046
+ if (input.distribution) {
2047
+ lines.push(
2048
+ `Distribution: ${JSON.stringify(input.distribution)}`
2049
+ );
2050
+ }
2051
+ if (input.examples.length > 0) {
2052
+ lines.push("", "Representative code samples:");
2053
+ for (const example of input.examples.slice(0, 5)) {
2054
+ lines.push("```", example, "```");
2055
+ }
2056
+ }
2057
+ lines.push(
2058
+ "",
2059
+ "Write a concise style rule description for this pattern."
2060
+ );
2061
+ return lines.join("\n");
2062
+ },
2063
+ maxTokens: 300
2064
+ };
2065
+ var REVIEW_VOICE_PROMPT = {
2066
+ featureTypes: [
2067
+ "reviewVoice.tone",
2068
+ "reviewVoice.themes",
2069
+ "reviewVoice.values"
2070
+ ],
2071
+ system: "You are analyzing a developer's code review comments to understand their review voice and priorities. Given topic frequencies and example comments, synthesize a brief description of what this developer cares about in code reviews. Output ONLY the synthesis text (2-4 sentences). Do not include markdown formatting or headers.",
2072
+ buildUserMessage: (input) => {
2073
+ const lines = [
2074
+ `Review topic: ${input.featureType}`,
2075
+ `Pattern: ${JSON.stringify(input.convention)}`
2076
+ ];
2077
+ if (input.distribution) {
2078
+ lines.push(
2079
+ `Topic frequencies: ${JSON.stringify(input.distribution)}`
2080
+ );
2081
+ }
2082
+ if (input.examples.length > 0) {
2083
+ lines.push("", "Example review comments:");
2084
+ for (const example of input.examples.slice(0, 5)) {
2085
+ lines.push(`- "${example}"`);
2086
+ }
2087
+ }
2088
+ lines.push(
2089
+ "",
2090
+ "Synthesize what this developer values in code reviews."
2091
+ );
2092
+ return lines.join("\n");
2093
+ },
2094
+ maxTokens: 400
2095
+ };
2096
+ var AI_ENRICHED_FEATURES = [
2097
+ ...DESCRIPTION_PROMPT.featureTypes,
2098
+ ...REVIEW_VOICE_PROMPT.featureTypes
2099
+ ];
2100
+ function getPromptForFeature(featureType) {
2101
+ if (DESCRIPTION_PROMPT.featureTypes.includes(featureType)) {
2102
+ return DESCRIPTION_PROMPT;
2103
+ }
2104
+ if (REVIEW_VOICE_PROMPT.featureTypes.includes(featureType)) {
2105
+ return REVIEW_VOICE_PROMPT;
2106
+ }
2107
+ return null;
2108
+ }
2109
+ function needsAiEnrichment(featureType) {
2110
+ return AI_ENRICHED_FEATURES.includes(featureType);
2111
+ }
2112
+
2113
+ // src/enricher/enricher.ts
2114
+ var Enricher = class {
2115
+ runner;
2116
+ enabled;
2117
+ constructor(config) {
2118
+ this.runner = new LlmRunner({
2119
+ provider: config.provider,
2120
+ totalTokenBudget: config.totalTokenBudget
2121
+ });
2122
+ this.enabled = config.enabled ?? true;
2123
+ }
2124
+ async enrich(features) {
2125
+ if (!this.enabled) {
2126
+ return {
2127
+ enriched: /* @__PURE__ */ new Map(),
2128
+ errors: [],
2129
+ totalTokensUsed: 0,
2130
+ budgetExceeded: false,
2131
+ skipped: true
2132
+ };
2133
+ }
2134
+ const runResult = await this.runner.run(this.buildJobs(features));
2135
+ return toEnrichmentResult(runResult);
2136
+ }
2137
+ buildJobs(features) {
2138
+ const jobs = [];
2139
+ for (const [type, feature] of features) {
2140
+ if (!needsAiEnrichment(type)) continue;
2141
+ const promptTemplate = getPromptForFeature(type);
2142
+ if (!promptTemplate) continue;
2143
+ const input = this.buildPromptInput(type, feature);
2144
+ const messages = [
2145
+ { role: "system", content: promptTemplate.system },
2146
+ {
2147
+ role: "user",
2148
+ content: promptTemplate.buildUserMessage(input)
2149
+ }
2150
+ ];
2151
+ jobs.push({ key: type, messages, maxTokens: promptTemplate.maxTokens });
2152
+ }
2153
+ return jobs;
2154
+ }
2155
+ buildPromptInput(type, feature) {
2156
+ const distributionRecord = {};
2157
+ for (const [key, count] of feature.distribution.values) {
2158
+ distributionRecord[String(key)] = count;
2159
+ }
2160
+ const exampleTexts = feature.examples.map((obs) => {
2161
+ if (typeof obs.value === "string") return obs.value;
2162
+ return JSON.stringify(obs.value);
2163
+ });
2164
+ return {
2165
+ category: feature.category,
2166
+ featureType: type,
2167
+ convention: feature.convention,
2168
+ confidence: feature.confidence,
2169
+ consistency: feature.distribution.consistency,
2170
+ examples: exampleTexts,
2171
+ distribution: distributionRecord
2172
+ };
2173
+ }
2174
+ };
2175
+ function toEnrichmentResult(runResult) {
2176
+ const enriched = /* @__PURE__ */ new Map();
2177
+ for (const r of runResult.results) {
2178
+ enriched.set(r.key, {
2179
+ featureType: r.key,
2180
+ description: r.content,
2181
+ tokensUsed: r.tokensUsed
2182
+ });
2183
+ }
2184
+ return {
2185
+ enriched,
2186
+ errors: runResult.errors.map((e) => ({
2187
+ featureType: e.key,
2188
+ error: e.error
2189
+ })),
2190
+ totalTokensUsed: runResult.totalTokensUsed,
2191
+ budgetExceeded: runResult.budgetExceeded,
2192
+ skipped: false
2193
+ };
2194
+ }
2195
+ export {
2196
+ AI_ENRICHED_FEATURES,
2197
+ Aggregator,
2198
+ ComplexityExtractor,
2199
+ ControlFlowExtractor,
2200
+ DocumentationExtractor,
2201
+ Enricher,
2202
+ ErrorHandlingExtractor,
2203
+ FormattingExtractor,
2204
+ IdiomsExtractor,
2205
+ NamingExtractor,
2206
+ ReviewVoiceExtractor,
2207
+ StructureExtractor,
2208
+ computeConfidence,
2209
+ createStyleExtractors,
2210
+ getLanguageFromPath,
2211
+ getSupportedLanguages,
2212
+ lookupStability,
2213
+ mapSeverity,
2214
+ needsAiEnrichment,
2215
+ parseFile,
2216
+ shouldIncludeFile
2217
+ };
2218
+ //# sourceMappingURL=index.js.map