code-gauge 1.13.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"crossFileDuplication.cjs","names":["selectMaximalGroups"],"sources":["../src/crossFileDuplication.ts"],"sourcesContent":["import { selectMaximalGroups } from './duplicateSelection.js';\nimport type { CrossFileDuplicateCandidate } from './duplication.js';\n\nexport interface CrossFileDuplicationSourceFile {\n file: string;\n candidates: CrossFileDuplicateCandidate[];\n}\n\nexport interface CrossFileDuplicateOccurrence {\n endLine: number;\n file: string;\n startLine: number;\n}\n\nexport interface CrossFileDuplicateBlockGroup {\n files: string[];\n occurrences: CrossFileDuplicateOccurrence[];\n /** Normalized token count of one occurrence (all occurrences share it). */\n tokenCount: number;\n}\n\nexport interface CrossFileDuplicationMetrics {\n /** Number of redundant copies across all groups, i.e. sum of (occurrenceCount - 1). */\n duplicateBlockCount: number;\n /** Groups the file participates in, keyed by the file name passed in. */\n duplicateBlockGroupCountByFile: Record<string, number>;\n groups: CrossFileDuplicateBlockGroup[];\n}\n\ninterface SelectableCandidate extends CrossFileDuplicateCandidate {\n regionBucket: number;\n file: string;\n}\n\n/**\n * Detects code regions duplicated across files: per-file candidates (whole block subtrees and\n * full container runs, fingerprinted with the same normalization as within-file duplication) are\n * grouped by fingerprint, and only maximal, non-overlapping regions whose group spans at least two\n * files are counted. Groups that shrink to a single file during selection are shed — a\n * within-file repeat is already reported by that file's own duplication metrics.\n */\nexport function measureCrossFileDuplication(files: CrossFileDuplicationSourceFile[]): CrossFileDuplicationMetrics {\n const candidates: SelectableCandidate[] = files.flatMap(({ file, candidates }, fileIndex) =>\n candidates.map((candidate) => ({ ...candidate, regionBucket: fileIndex, file }))\n );\n const counted = selectMaximalGroups(\n candidates,\n spansMultipleFiles,\n // File index and position break coverage ties deterministically.\n (left, right) => left.regionBucket - right.regionBucket || left.startIndex - right.startIndex\n );\n return summarize(counted);\n}\n\nfunction spansMultipleFiles(group: SelectableCandidate[]): boolean {\n return group.length >= 2 && new Set(group.map((candidate) => candidate.regionBucket)).size >= 2;\n}\n\nfunction summarize(counted: Map<string, SelectableCandidate[]>): CrossFileDuplicationMetrics {\n const groups: CrossFileDuplicateBlockGroup[] = [];\n // Accumulated in a Map: file names are arbitrary strings, and a plain object would read\n // inherited properties for names like \"constructor\".\n const groupCountByFile = new Map<string, number>();\n let duplicateBlockCount = 0;\n for (const group of counted.values()) {\n duplicateBlockCount += group.length - 1;\n const occurrences = group\n .map(({ file, startLine, endLine }) => ({ file, startLine, endLine }))\n .toSorted((left, right) => left.file.localeCompare(right.file) || left.startLine - right.startLine);\n const files = [...new Set(occurrences.map(({ file }) => file))];\n for (const file of files) {\n groupCountByFile.set(file, (groupCountByFile.get(file) ?? 0) + 1);\n }\n groups.push({ files, occurrences, tokenCount: group[0]?.tokenCount ?? 0 });\n }\n groups.sort(\n (left, right) =>\n right.tokenCount - left.tokenCount ||\n (left.occurrences[0]?.file ?? '').localeCompare(right.occurrences[0]?.file ?? '') ||\n (left.occurrences[0]?.startLine ?? 0) - (right.occurrences[0]?.startLine ?? 0)\n );\n return { duplicateBlockCount, duplicateBlockGroupCountByFile: Object.fromEntries(groupCountByFile), groups };\n}\n"],"mappings":"yDAyCA,SAAgB,EAA4B,EAAsE,CAChH,IAAM,EAAoC,EAAM,SAAS,CAAE,OAAM,cAAc,IAC7E,EAAW,IAAK,IAAe,CAAE,GAAG,EAAW,aAAc,EAAW,MAAK,EAAE,CACjF,EAOA,OAAO,EANSA,EAAAA,oBACd,EACA,GAEC,EAAM,IAAU,EAAK,aAAe,EAAM,cAAgB,EAAK,WAAa,EAAM,UAE9D,CAAC,CAC1B,CAEA,SAAS,EAAmB,EAAuC,CACjE,OAAO,EAAM,QAAU,GAAK,IAAI,IAAI,EAAM,IAAK,GAAc,EAAU,YAAY,CAAC,CAAC,CAAC,MAAQ,CAChG,CAEA,SAAS,EAAU,EAA0E,CAC3F,IAAM,EAAyC,CAAC,EAG1C,EAAmB,IAAI,IACzB,EAAsB,EAC1B,IAAK,IAAM,KAAS,EAAQ,OAAO,EAAG,CACpC,GAAuB,EAAM,OAAS,EACtC,IAAM,EAAc,EACjB,KAAK,CAAE,OAAM,YAAW,cAAe,CAAE,OAAM,YAAW,SAAQ,EAAE,CAAC,CACrE,UAAU,EAAM,IAAU,EAAK,KAAK,cAAc,EAAM,IAAI,GAAK,EAAK,UAAY,EAAM,SAAS,EAC9F,EAAQ,CAAC,GAAG,IAAI,IAAI,EAAY,KAAK,CAAE,UAAW,CAAI,CAAC,CAAC,EAC9D,IAAK,IAAM,KAAQ,EACjB,EAAiB,IAAI,GAAO,EAAiB,IAAI,CAAI,GAAK,GAAK,CAAC,EAElE,EAAO,KAAK,CAAE,QAAO,cAAa,WAAY,EAAM,EAAE,EAAE,YAAc,CAAE,CAAC,CAC3E,CAOA,OANA,EAAO,MACJ,EAAM,IACL,EAAM,WAAa,EAAK,aACvB,EAAK,YAAY,EAAE,EAAE,MAAQ,GAAA,CAAI,cAAc,EAAM,YAAY,EAAE,EAAE,MAAQ,EAAE,IAC/E,EAAK,YAAY,EAAE,EAAE,WAAa,IAAM,EAAM,YAAY,EAAE,EAAE,WAAa,EAChF,EACO,CAAE,sBAAqB,+BAAgC,OAAO,YAAY,CAAgB,EAAG,QAAO,CAC7G"}
1
+ {"version":3,"file":"crossFileDuplication.cjs","names":["defaultDuplicationOptions","selectMaximalGroups","buildLiteralCountPrefix","collectSequenceWindowCandidates","mergeAdjacentGroups","countRedundantFragments"],"sources":["../src/crossFileDuplication.ts"],"sourcesContent":["import { selectMaximalGroups } from './duplicateSelection.js';\nimport {\n buildLiteralCountPrefix,\n collectSequenceWindowCandidates,\n countRedundantFragments,\n defaultDuplicationOptions,\n mergeAdjacentGroups,\n type CountedOccurrence,\n type CrossFileDuplicateCandidate,\n type CrossFileDuplicationFileData,\n type SequenceWindowContext,\n} from './duplication.js';\nimport type { DuplicationOptions } from './types.js';\n\nexport interface CrossFileDuplicationSourceFile extends Partial<CrossFileDuplicationFileData> {\n file: string;\n candidates: CrossFileDuplicateCandidate[];\n}\n\nexport interface CrossFileDuplicateOccurrence {\n endLine: number;\n file: string;\n startLine: number;\n}\n\nexport interface CrossFileDuplicateBlockGroup {\n files: string[];\n occurrences: CrossFileDuplicateOccurrence[];\n /** Matched token count of one occurrence (all occurrences share it; gaps are not counted). */\n tokenCount: number;\n}\n\nexport interface CrossFileDuplicationMetrics {\n /** Number of redundant copies across all groups, counted per matched fragment like within-file. */\n duplicateBlockCount: number;\n /** Groups the file participates in, keyed by the file name passed in. */\n duplicateBlockGroupCountByFile: Record<string, number>;\n groups: CrossFileDuplicateBlockGroup[];\n}\n\ninterface SelectableCandidate extends CrossFileDuplicateCandidate {\n regionBucket: number;\n file: string;\n}\n\n/** A cross-file occurrence: a within-file occurrence in the project-wide token index space. */\ninterface CrossFileOccurrence extends CountedOccurrence {\n file: string;\n}\n\n/**\n * Detects code regions duplicated across files. Per-file candidates (whole block subtrees and full\n * container runs, fingerprinted with the same normalization as within-file duplication) are joined\n * by a project-level window index over per-statement fingerprint sequences (CPD-style), so a\n * copy-pasted partial statement run embedded in different surrounding code is matched even though\n * no single file can know it repeats elsewhere. Candidates are grouped by fingerprint, and only\n * maximal, non-overlapping regions whose group spans at least two files are counted. Groups that\n * shrink to a single file during selection are shed — a within-file repeat is already reported by\n * that file's own duplication metrics. Groups separated by a small token gap within each file then\n * merge into gapped (Type-3) clone groups under `maxGapTokens`, exactly like within-file merging.\n */\nexport function measureCrossFileDuplication(\n files: CrossFileDuplicationSourceFile[],\n options?: DuplicationOptions\n): CrossFileDuplicationMetrics {\n const minTokens = options?.minTokens ?? defaultDuplicationOptions.minTokens;\n const maxGapTokens = options?.maxGapTokens ?? defaultDuplicationOptions.maxGapTokens;\n const candidates: SelectableCandidate[] = files.flatMap(({ file, candidates }, fileIndex) =>\n candidates.map((candidate) => ({ ...candidate, regionBucket: fileIndex, file }))\n );\n // Pushed one by one: spreading the project-scale window-candidate array as call arguments\n // overflows V8's argument limit (~124k) and crashes on Node, though Bun/JSC tolerates it.\n for (const candidate of collectWindowCandidates(files, minTokens)) {\n candidates.push(candidate);\n }\n const counted = selectMaximalGroups(\n candidates,\n spansMultipleFiles,\n // File index and position break coverage ties deterministically.\n (left, right) => left.regionBucket - right.regionBucket || left.startIndex - right.startIndex\n );\n return summarize(mergeGapAdjacentGroups([...counted.values()], files, maxGapTokens));\n}\n\n/** Repeated sub-windows of sibling statements matched across the whole project's files. */\nfunction collectWindowCandidates(files: CrossFileDuplicationSourceFile[], minTokens: number): SelectableCandidate[] {\n const fileIndexByContext: number[] = [];\n const contexts: SequenceWindowContext[] = [];\n for (const [fileIndex, { tokens, containerStatements }] of files.entries()) {\n if (tokens && containerStatements) {\n fileIndexByContext.push(fileIndex);\n contexts.push({ tokens, literalCountPrefix: buildLiteralCountPrefix(tokens), containers: containerStatements });\n }\n }\n if (contexts.length < 2) {\n return [];\n }\n return collectSequenceWindowCandidates(contexts, minTokens, true).flatMap(({ candidate, contextIndex }) => {\n const fileIndex = fileIndexByContext[contextIndex];\n const file = fileIndex === undefined ? undefined : files[fileIndex];\n return fileIndex === undefined || file === undefined\n ? []\n : [{ ...candidate, regionBucket: fileIndex, file: file.file }];\n });\n}\n\nfunction spansMultipleFiles(group: SelectableCandidate[]): boolean {\n return group.length >= 2 && new Set(group.map((candidate) => candidate.regionBucket)).size >= 2;\n}\n\n/**\n * Reuses the within-file gapped (Type-3) merging by mapping every occurrence into one project-wide\n * token index space: each file's tokens are offset by more than `maxGapTokens` past the previous\n * file's, so occurrences in different files are never gap-adjacent and pairs always stay within\n * one file.\n */\nfunction mergeGapAdjacentGroups(\n groups: SelectableCandidate[][],\n files: CrossFileDuplicationSourceFile[],\n maxGapTokens: number\n): CrossFileOccurrence[][] {\n const tokenOffsets: number[] = [];\n let offset = 0;\n for (const { tokens, candidates } of files) {\n tokenOffsets.push(offset);\n // Accumulated in a loop: spreading a project-scale candidate array as call arguments would\n // overflow V8's argument limit (~124k) and crash on Node.\n let tokenCount = tokens?.length ?? 0;\n if (!tokens) {\n for (const candidate of candidates) {\n tokenCount = Math.max(tokenCount, candidate.endTokenIndex);\n }\n }\n offset += tokenCount + maxGapTokens + 1;\n }\n const occurrenceGroups = groups.map((group) =>\n group\n .map((candidate): CrossFileOccurrence => {\n const start = candidate.startTokenIndex + (tokenOffsets[candidate.regionBucket] ?? 0);\n const end = candidate.endTokenIndex + (tokenOffsets[candidate.regionBucket] ?? 0);\n return {\n file: candidate.file,\n segments: [{ startTokenIndex: start, endTokenIndex: end }],\n tokenCount: candidate.tokenCount,\n startTokenIndex: start,\n endTokenIndex: end,\n startIndex: candidate.startIndex,\n endIndex: candidate.endIndex,\n startLine: candidate.startLine,\n endLine: candidate.endLine,\n };\n })\n .toSorted((left, right) => left.startTokenIndex - right.startTokenIndex)\n );\n return mergeAdjacentGroups(occurrenceGroups, maxGapTokens);\n}\n\nfunction summarize(groups: CrossFileOccurrence[][]): CrossFileDuplicationMetrics {\n const reported: CrossFileDuplicateBlockGroup[] = [];\n // Accumulated in a Map: file names are arbitrary strings, and a plain object would read\n // inherited properties for names like \"constructor\".\n const groupCountByFile = new Map<string, number>();\n let duplicateBlockCount = 0;\n for (const group of groups) {\n // Mirrors within-file counting: each redundant occurrence contributes one count per matched\n // fragment, gapped merging consolidates the grouping without halving the count, and spans a\n // partial merge shares between a retained group and the merged group count once.\n duplicateBlockCount += countRedundantFragments(group);\n const occurrences = group\n .map(({ file, startLine, endLine }) => ({ file, startLine, endLine }))\n .toSorted((left, right) => left.file.localeCompare(right.file) || left.startLine - right.startLine);\n const files = [...new Set(occurrences.map(({ file }) => file))];\n for (const file of files) {\n groupCountByFile.set(file, (groupCountByFile.get(file) ?? 0) + 1);\n }\n reported.push({ files, occurrences, tokenCount: group[0]?.tokenCount ?? 0 });\n }\n reported.sort(\n (left, right) =>\n right.tokenCount - left.tokenCount ||\n (left.occurrences[0]?.file ?? '').localeCompare(right.occurrences[0]?.file ?? '') ||\n (left.occurrences[0]?.startLine ?? 0) - (right.occurrences[0]?.startLine ?? 0)\n );\n return {\n duplicateBlockCount,\n duplicateBlockGroupCountByFile: Object.fromEntries(groupCountByFile),\n groups: reported,\n };\n}\n"],"mappings":"wFA6DA,SAAgB,EACd,EACA,EAC6B,CAC7B,IAAM,EAAY,GAAS,WAAaA,EAAAA,0BAA0B,UAC5D,EAAe,GAAS,cAAgBA,EAAAA,0BAA0B,aAClE,EAAoC,EAAM,SAAS,CAAE,OAAM,cAAc,IAC7E,EAAW,IAAK,IAAe,CAAE,GAAG,EAAW,aAAc,EAAW,MAAK,EAAE,CACjF,EAGA,IAAK,IAAM,KAAa,EAAwB,EAAO,CAAS,EAC9D,EAAW,KAAK,CAAS,EAQ3B,OAAO,EAAU,EAAuB,CAAC,GANzBC,EAAAA,oBACd,EACA,GAEC,EAAM,IAAU,EAAK,aAAe,EAAM,cAAgB,EAAK,WAAa,EAAM,UAEnC,CAAC,CAAC,OAAO,CAAC,EAAG,EAAO,CAAY,CAAC,CACrF,CAGA,SAAS,EAAwB,EAAyC,EAA0C,CAClH,IAAM,EAA+B,CAAC,EAChC,EAAoC,CAAC,EAC3C,IAAK,GAAM,CAAC,EAAW,CAAE,SAAQ,0BAA0B,EAAM,QAAQ,EACnE,GAAU,IACZ,EAAmB,KAAK,CAAS,EACjC,EAAS,KAAK,CAAE,SAAQ,mBAAoBC,EAAAA,wBAAwB,CAAM,EAAG,WAAY,CAAoB,CAAC,GAMlH,OAHI,EAAS,OAAS,EACb,CAAC,EAEHC,EAAAA,gCAAgC,EAAU,EAAW,EAAI,CAAC,CAAC,SAAS,CAAE,YAAW,kBAAmB,CACzG,IAAM,EAAY,EAAmB,GAC/B,EAAO,IAAc,IAAA,GAAY,IAAA,GAAY,EAAM,GACzD,OAAO,IAAc,IAAA,IAAa,IAAS,IAAA,GACvC,CAAC,EACD,CAAC,CAAE,GAAG,EAAW,aAAc,EAAW,KAAM,EAAK,IAAK,CAAC,CACjE,CAAC,CACH,CAEA,SAAS,EAAmB,EAAuC,CACjE,OAAO,EAAM,QAAU,GAAK,IAAI,IAAI,EAAM,IAAK,GAAc,EAAU,YAAY,CAAC,CAAC,CAAC,MAAQ,CAChG,CAQA,SAAS,EACP,EACA,EACA,EACyB,CACzB,IAAM,EAAyB,CAAC,EAC5B,EAAS,EACb,IAAK,GAAM,CAAE,SAAQ,gBAAgB,EAAO,CAC1C,EAAa,KAAK,CAAM,EAGxB,IAAI,EAAa,GAAQ,QAAU,EACnC,GAAI,CAAC,EACH,IAAK,IAAM,KAAa,EACtB,EAAa,KAAK,IAAI,EAAY,EAAU,aAAa,EAG7D,GAAU,EAAa,EAAe,CACxC,CACA,IAAM,EAAmB,EAAO,IAAK,GACnC,EACG,IAAK,GAAmC,CACvC,IAAM,EAAQ,EAAU,iBAAmB,EAAa,EAAU,eAAiB,GAC7E,EAAM,EAAU,eAAiB,EAAa,EAAU,eAAiB,GAC/E,MAAO,CACL,KAAM,EAAU,KAChB,SAAU,CAAC,CAAE,gBAAiB,EAAO,cAAe,CAAI,CAAC,EACzD,WAAY,EAAU,WACtB,gBAAiB,EACjB,cAAe,EACf,WAAY,EAAU,WACtB,SAAU,EAAU,SACpB,UAAW,EAAU,UACrB,QAAS,EAAU,OACrB,CACF,CAAC,CAAC,CACD,UAAU,EAAM,IAAU,EAAK,gBAAkB,EAAM,eAAe,CAC3E,EACA,OAAOC,EAAAA,oBAAoB,EAAkB,CAAY,CAC3D,CAEA,SAAS,EAAU,EAA8D,CAC/E,IAAM,EAA2C,CAAC,EAG5C,EAAmB,IAAI,IACzB,EAAsB,EAC1B,IAAK,IAAM,KAAS,EAAQ,CAI1B,GAAuBC,EAAAA,wBAAwB,CAAK,EACpD,IAAM,EAAc,EACjB,KAAK,CAAE,OAAM,YAAW,cAAe,CAAE,OAAM,YAAW,SAAQ,EAAE,CAAC,CACrE,UAAU,EAAM,IAAU,EAAK,KAAK,cAAc,EAAM,IAAI,GAAK,EAAK,UAAY,EAAM,SAAS,EAC9F,EAAQ,CAAC,GAAG,IAAI,IAAI,EAAY,KAAK,CAAE,UAAW,CAAI,CAAC,CAAC,EAC9D,IAAK,IAAM,KAAQ,EACjB,EAAiB,IAAI,GAAO,EAAiB,IAAI,CAAI,GAAK,GAAK,CAAC,EAElE,EAAS,KAAK,CAAE,QAAO,cAAa,WAAY,EAAM,EAAE,EAAE,YAAc,CAAE,CAAC,CAC7E,CAOA,OANA,EAAS,MACN,EAAM,IACL,EAAM,WAAa,EAAK,aACvB,EAAK,YAAY,EAAE,EAAE,MAAQ,GAAA,CAAI,cAAc,EAAM,YAAY,EAAE,EAAE,MAAQ,EAAE,IAC/E,EAAK,YAAY,EAAE,EAAE,WAAa,IAAM,EAAM,YAAY,EAAE,EAAE,WAAa,EAChF,EACO,CACL,sBACA,+BAAgC,OAAO,YAAY,CAAgB,EACnE,OAAQ,CACV,CACF"}
@@ -1,5 +1,6 @@
1
- import type { CrossFileDuplicateCandidate } from './duplication.js';
2
- export interface CrossFileDuplicationSourceFile {
1
+ import { type CrossFileDuplicateCandidate, type CrossFileDuplicationFileData } from './duplication.js';
2
+ import type { DuplicationOptions } from './types.js';
3
+ export interface CrossFileDuplicationSourceFile extends Partial<CrossFileDuplicationFileData> {
3
4
  file: string;
4
5
  candidates: CrossFileDuplicateCandidate[];
5
6
  }
@@ -11,21 +12,25 @@ export interface CrossFileDuplicateOccurrence {
11
12
  export interface CrossFileDuplicateBlockGroup {
12
13
  files: string[];
13
14
  occurrences: CrossFileDuplicateOccurrence[];
14
- /** Normalized token count of one occurrence (all occurrences share it). */
15
+ /** Matched token count of one occurrence (all occurrences share it; gaps are not counted). */
15
16
  tokenCount: number;
16
17
  }
17
18
  export interface CrossFileDuplicationMetrics {
18
- /** Number of redundant copies across all groups, i.e. sum of (occurrenceCount - 1). */
19
+ /** Number of redundant copies across all groups, counted per matched fragment like within-file. */
19
20
  duplicateBlockCount: number;
20
21
  /** Groups the file participates in, keyed by the file name passed in. */
21
22
  duplicateBlockGroupCountByFile: Record<string, number>;
22
23
  groups: CrossFileDuplicateBlockGroup[];
23
24
  }
24
25
  /**
25
- * Detects code regions duplicated across files: per-file candidates (whole block subtrees and
26
- * full container runs, fingerprinted with the same normalization as within-file duplication) are
27
- * grouped by fingerprint, and only maximal, non-overlapping regions whose group spans at least two
28
- * files are counted. Groups that shrink to a single file during selection are shed — a
29
- * within-file repeat is already reported by that file's own duplication metrics.
26
+ * Detects code regions duplicated across files. Per-file candidates (whole block subtrees and full
27
+ * container runs, fingerprinted with the same normalization as within-file duplication) are joined
28
+ * by a project-level window index over per-statement fingerprint sequences (CPD-style), so a
29
+ * copy-pasted partial statement run embedded in different surrounding code is matched even though
30
+ * no single file can know it repeats elsewhere. Candidates are grouped by fingerprint, and only
31
+ * maximal, non-overlapping regions whose group spans at least two files are counted. Groups that
32
+ * shrink to a single file during selection are shed — a within-file repeat is already reported by
33
+ * that file's own duplication metrics. Groups separated by a small token gap within each file then
34
+ * merge into gapped (Type-3) clone groups under `maxGapTokens`, exactly like within-file merging.
30
35
  */
31
- export declare function measureCrossFileDuplication(files: CrossFileDuplicationSourceFile[]): CrossFileDuplicationMetrics;
36
+ export declare function measureCrossFileDuplication(files: CrossFileDuplicationSourceFile[], options?: DuplicationOptions): CrossFileDuplicationMetrics;
@@ -1,2 +1,2 @@
1
- import{selectMaximalGroups as e}from"./duplicateSelection.js";function t(t){let i=t.flatMap(({file:e,candidates:t},n)=>t.map(t=>({...t,regionBucket:n,file:e})));return r(e(i,n,(e,t)=>e.regionBucket-t.regionBucket||e.startIndex-t.startIndex))}function n(e){return e.length>=2&&new Set(e.map(e=>e.regionBucket)).size>=2}function r(e){let t=[],n=new Map,r=0;for(let i of e.values()){r+=i.length-1;let e=i.map(({file:e,startLine:t,endLine:n})=>({file:e,startLine:t,endLine:n})).toSorted((e,t)=>e.file.localeCompare(t.file)||e.startLine-t.startLine),a=[...new Set(e.map(({file:e})=>e))];for(let e of a)n.set(e,(n.get(e)??0)+1);t.push({files:a,occurrences:e,tokenCount:i[0]?.tokenCount??0})}return t.sort((e,t)=>t.tokenCount-e.tokenCount||(e.occurrences[0]?.file??``).localeCompare(t.occurrences[0]?.file??``)||(e.occurrences[0]?.startLine??0)-(t.occurrences[0]?.startLine??0)),{duplicateBlockCount:r,duplicateBlockGroupCountByFile:Object.fromEntries(n),groups:t}}export{t as measureCrossFileDuplication};
1
+ import{selectMaximalGroups as e}from"./duplicateSelection.js";import{buildLiteralCountPrefix as t,collectSequenceWindowCandidates as n,countRedundantFragments as r,defaultDuplicationOptions as i,mergeAdjacentGroups as a}from"./duplication.js";function o(t,n){let r=n?.minTokens??i.minTokens,a=n?.maxGapTokens??i.maxGapTokens,o=t.flatMap(({file:e,candidates:t},n)=>t.map(t=>({...t,regionBucket:n,file:e})));for(let e of s(t,r))o.push(e);return u(l([...e(o,c,(e,t)=>e.regionBucket-t.regionBucket||e.startIndex-t.startIndex).values()],t,a))}function s(e,r){let i=[],a=[];for(let[n,{tokens:r,containerStatements:o}]of e.entries())r&&o&&(i.push(n),a.push({tokens:r,literalCountPrefix:t(r),containers:o}));return a.length<2?[]:n(a,r,!0).flatMap(({candidate:t,contextIndex:n})=>{let r=i[n],a=r===void 0?void 0:e[r];return r===void 0||a===void 0?[]:[{...t,regionBucket:r,file:a.file}]})}function c(e){return e.length>=2&&new Set(e.map(e=>e.regionBucket)).size>=2}function l(e,t,n){let r=[],i=0;for(let{tokens:e,candidates:a}of t){r.push(i);let t=e?.length??0;if(!e)for(let e of a)t=Math.max(t,e.endTokenIndex);i+=t+n+1}let o=e.map(e=>e.map(e=>{let t=e.startTokenIndex+(r[e.regionBucket]??0),n=e.endTokenIndex+(r[e.regionBucket]??0);return{file:e.file,segments:[{startTokenIndex:t,endTokenIndex:n}],tokenCount:e.tokenCount,startTokenIndex:t,endTokenIndex:n,startIndex:e.startIndex,endIndex:e.endIndex,startLine:e.startLine,endLine:e.endLine}}).toSorted((e,t)=>e.startTokenIndex-t.startTokenIndex));return a(o,n)}function u(e){let t=[],n=new Map,i=0;for(let a of e){i+=r(a);let e=a.map(({file:e,startLine:t,endLine:n})=>({file:e,startLine:t,endLine:n})).toSorted((e,t)=>e.file.localeCompare(t.file)||e.startLine-t.startLine),o=[...new Set(e.map(({file:e})=>e))];for(let e of o)n.set(e,(n.get(e)??0)+1);t.push({files:o,occurrences:e,tokenCount:a[0]?.tokenCount??0})}return t.sort((e,t)=>t.tokenCount-e.tokenCount||(e.occurrences[0]?.file??``).localeCompare(t.occurrences[0]?.file??``)||(e.occurrences[0]?.startLine??0)-(t.occurrences[0]?.startLine??0)),{duplicateBlockCount:i,duplicateBlockGroupCountByFile:Object.fromEntries(n),groups:t}}export{o as measureCrossFileDuplication};
2
2
  //# sourceMappingURL=crossFileDuplication.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"crossFileDuplication.js","names":[],"sources":["../src/crossFileDuplication.ts"],"sourcesContent":["import { selectMaximalGroups } from './duplicateSelection.js';\nimport type { CrossFileDuplicateCandidate } from './duplication.js';\n\nexport interface CrossFileDuplicationSourceFile {\n file: string;\n candidates: CrossFileDuplicateCandidate[];\n}\n\nexport interface CrossFileDuplicateOccurrence {\n endLine: number;\n file: string;\n startLine: number;\n}\n\nexport interface CrossFileDuplicateBlockGroup {\n files: string[];\n occurrences: CrossFileDuplicateOccurrence[];\n /** Normalized token count of one occurrence (all occurrences share it). */\n tokenCount: number;\n}\n\nexport interface CrossFileDuplicationMetrics {\n /** Number of redundant copies across all groups, i.e. sum of (occurrenceCount - 1). */\n duplicateBlockCount: number;\n /** Groups the file participates in, keyed by the file name passed in. */\n duplicateBlockGroupCountByFile: Record<string, number>;\n groups: CrossFileDuplicateBlockGroup[];\n}\n\ninterface SelectableCandidate extends CrossFileDuplicateCandidate {\n regionBucket: number;\n file: string;\n}\n\n/**\n * Detects code regions duplicated across files: per-file candidates (whole block subtrees and\n * full container runs, fingerprinted with the same normalization as within-file duplication) are\n * grouped by fingerprint, and only maximal, non-overlapping regions whose group spans at least two\n * files are counted. Groups that shrink to a single file during selection are shed — a\n * within-file repeat is already reported by that file's own duplication metrics.\n */\nexport function measureCrossFileDuplication(files: CrossFileDuplicationSourceFile[]): CrossFileDuplicationMetrics {\n const candidates: SelectableCandidate[] = files.flatMap(({ file, candidates }, fileIndex) =>\n candidates.map((candidate) => ({ ...candidate, regionBucket: fileIndex, file }))\n );\n const counted = selectMaximalGroups(\n candidates,\n spansMultipleFiles,\n // File index and position break coverage ties deterministically.\n (left, right) => left.regionBucket - right.regionBucket || left.startIndex - right.startIndex\n );\n return summarize(counted);\n}\n\nfunction spansMultipleFiles(group: SelectableCandidate[]): boolean {\n return group.length >= 2 && new Set(group.map((candidate) => candidate.regionBucket)).size >= 2;\n}\n\nfunction summarize(counted: Map<string, SelectableCandidate[]>): CrossFileDuplicationMetrics {\n const groups: CrossFileDuplicateBlockGroup[] = [];\n // Accumulated in a Map: file names are arbitrary strings, and a plain object would read\n // inherited properties for names like \"constructor\".\n const groupCountByFile = new Map<string, number>();\n let duplicateBlockCount = 0;\n for (const group of counted.values()) {\n duplicateBlockCount += group.length - 1;\n const occurrences = group\n .map(({ file, startLine, endLine }) => ({ file, startLine, endLine }))\n .toSorted((left, right) => left.file.localeCompare(right.file) || left.startLine - right.startLine);\n const files = [...new Set(occurrences.map(({ file }) => file))];\n for (const file of files) {\n groupCountByFile.set(file, (groupCountByFile.get(file) ?? 0) + 1);\n }\n groups.push({ files, occurrences, tokenCount: group[0]?.tokenCount ?? 0 });\n }\n groups.sort(\n (left, right) =>\n right.tokenCount - left.tokenCount ||\n (left.occurrences[0]?.file ?? '').localeCompare(right.occurrences[0]?.file ?? '') ||\n (left.occurrences[0]?.startLine ?? 0) - (right.occurrences[0]?.startLine ?? 0)\n );\n return { duplicateBlockCount, duplicateBlockGroupCountByFile: Object.fromEntries(groupCountByFile), groups };\n}\n"],"mappings":"8DAyCA,SAAgB,EAA4B,EAAsE,CAChH,IAAM,EAAoC,EAAM,SAAS,CAAE,OAAM,cAAc,IAC7E,EAAW,IAAK,IAAe,CAAE,GAAG,EAAW,aAAc,EAAW,MAAK,EAAE,CACjF,EAOA,OAAO,EANS,EACd,EACA,GAEC,EAAM,IAAU,EAAK,aAAe,EAAM,cAAgB,EAAK,WAAa,EAAM,UAE9D,CAAC,CAC1B,CAEA,SAAS,EAAmB,EAAuC,CACjE,OAAO,EAAM,QAAU,GAAK,IAAI,IAAI,EAAM,IAAK,GAAc,EAAU,YAAY,CAAC,CAAC,CAAC,MAAQ,CAChG,CAEA,SAAS,EAAU,EAA0E,CAC3F,IAAM,EAAyC,CAAC,EAG1C,EAAmB,IAAI,IACzB,EAAsB,EAC1B,IAAK,IAAM,KAAS,EAAQ,OAAO,EAAG,CACpC,GAAuB,EAAM,OAAS,EACtC,IAAM,EAAc,EACjB,KAAK,CAAE,OAAM,YAAW,cAAe,CAAE,OAAM,YAAW,SAAQ,EAAE,CAAC,CACrE,UAAU,EAAM,IAAU,EAAK,KAAK,cAAc,EAAM,IAAI,GAAK,EAAK,UAAY,EAAM,SAAS,EAC9F,EAAQ,CAAC,GAAG,IAAI,IAAI,EAAY,KAAK,CAAE,UAAW,CAAI,CAAC,CAAC,EAC9D,IAAK,IAAM,KAAQ,EACjB,EAAiB,IAAI,GAAO,EAAiB,IAAI,CAAI,GAAK,GAAK,CAAC,EAElE,EAAO,KAAK,CAAE,QAAO,cAAa,WAAY,EAAM,EAAE,EAAE,YAAc,CAAE,CAAC,CAC3E,CAOA,OANA,EAAO,MACJ,EAAM,IACL,EAAM,WAAa,EAAK,aACvB,EAAK,YAAY,EAAE,EAAE,MAAQ,GAAA,CAAI,cAAc,EAAM,YAAY,EAAE,EAAE,MAAQ,EAAE,IAC/E,EAAK,YAAY,EAAE,EAAE,WAAa,IAAM,EAAM,YAAY,EAAE,EAAE,WAAa,EAChF,EACO,CAAE,sBAAqB,+BAAgC,OAAO,YAAY,CAAgB,EAAG,QAAO,CAC7G"}
1
+ {"version":3,"file":"crossFileDuplication.js","names":[],"sources":["../src/crossFileDuplication.ts"],"sourcesContent":["import { selectMaximalGroups } from './duplicateSelection.js';\nimport {\n buildLiteralCountPrefix,\n collectSequenceWindowCandidates,\n countRedundantFragments,\n defaultDuplicationOptions,\n mergeAdjacentGroups,\n type CountedOccurrence,\n type CrossFileDuplicateCandidate,\n type CrossFileDuplicationFileData,\n type SequenceWindowContext,\n} from './duplication.js';\nimport type { DuplicationOptions } from './types.js';\n\nexport interface CrossFileDuplicationSourceFile extends Partial<CrossFileDuplicationFileData> {\n file: string;\n candidates: CrossFileDuplicateCandidate[];\n}\n\nexport interface CrossFileDuplicateOccurrence {\n endLine: number;\n file: string;\n startLine: number;\n}\n\nexport interface CrossFileDuplicateBlockGroup {\n files: string[];\n occurrences: CrossFileDuplicateOccurrence[];\n /** Matched token count of one occurrence (all occurrences share it; gaps are not counted). */\n tokenCount: number;\n}\n\nexport interface CrossFileDuplicationMetrics {\n /** Number of redundant copies across all groups, counted per matched fragment like within-file. */\n duplicateBlockCount: number;\n /** Groups the file participates in, keyed by the file name passed in. */\n duplicateBlockGroupCountByFile: Record<string, number>;\n groups: CrossFileDuplicateBlockGroup[];\n}\n\ninterface SelectableCandidate extends CrossFileDuplicateCandidate {\n regionBucket: number;\n file: string;\n}\n\n/** A cross-file occurrence: a within-file occurrence in the project-wide token index space. */\ninterface CrossFileOccurrence extends CountedOccurrence {\n file: string;\n}\n\n/**\n * Detects code regions duplicated across files. Per-file candidates (whole block subtrees and full\n * container runs, fingerprinted with the same normalization as within-file duplication) are joined\n * by a project-level window index over per-statement fingerprint sequences (CPD-style), so a\n * copy-pasted partial statement run embedded in different surrounding code is matched even though\n * no single file can know it repeats elsewhere. Candidates are grouped by fingerprint, and only\n * maximal, non-overlapping regions whose group spans at least two files are counted. Groups that\n * shrink to a single file during selection are shed — a within-file repeat is already reported by\n * that file's own duplication metrics. Groups separated by a small token gap within each file then\n * merge into gapped (Type-3) clone groups under `maxGapTokens`, exactly like within-file merging.\n */\nexport function measureCrossFileDuplication(\n files: CrossFileDuplicationSourceFile[],\n options?: DuplicationOptions\n): CrossFileDuplicationMetrics {\n const minTokens = options?.minTokens ?? defaultDuplicationOptions.minTokens;\n const maxGapTokens = options?.maxGapTokens ?? defaultDuplicationOptions.maxGapTokens;\n const candidates: SelectableCandidate[] = files.flatMap(({ file, candidates }, fileIndex) =>\n candidates.map((candidate) => ({ ...candidate, regionBucket: fileIndex, file }))\n );\n // Pushed one by one: spreading the project-scale window-candidate array as call arguments\n // overflows V8's argument limit (~124k) and crashes on Node, though Bun/JSC tolerates it.\n for (const candidate of collectWindowCandidates(files, minTokens)) {\n candidates.push(candidate);\n }\n const counted = selectMaximalGroups(\n candidates,\n spansMultipleFiles,\n // File index and position break coverage ties deterministically.\n (left, right) => left.regionBucket - right.regionBucket || left.startIndex - right.startIndex\n );\n return summarize(mergeGapAdjacentGroups([...counted.values()], files, maxGapTokens));\n}\n\n/** Repeated sub-windows of sibling statements matched across the whole project's files. */\nfunction collectWindowCandidates(files: CrossFileDuplicationSourceFile[], minTokens: number): SelectableCandidate[] {\n const fileIndexByContext: number[] = [];\n const contexts: SequenceWindowContext[] = [];\n for (const [fileIndex, { tokens, containerStatements }] of files.entries()) {\n if (tokens && containerStatements) {\n fileIndexByContext.push(fileIndex);\n contexts.push({ tokens, literalCountPrefix: buildLiteralCountPrefix(tokens), containers: containerStatements });\n }\n }\n if (contexts.length < 2) {\n return [];\n }\n return collectSequenceWindowCandidates(contexts, minTokens, true).flatMap(({ candidate, contextIndex }) => {\n const fileIndex = fileIndexByContext[contextIndex];\n const file = fileIndex === undefined ? undefined : files[fileIndex];\n return fileIndex === undefined || file === undefined\n ? []\n : [{ ...candidate, regionBucket: fileIndex, file: file.file }];\n });\n}\n\nfunction spansMultipleFiles(group: SelectableCandidate[]): boolean {\n return group.length >= 2 && new Set(group.map((candidate) => candidate.regionBucket)).size >= 2;\n}\n\n/**\n * Reuses the within-file gapped (Type-3) merging by mapping every occurrence into one project-wide\n * token index space: each file's tokens are offset by more than `maxGapTokens` past the previous\n * file's, so occurrences in different files are never gap-adjacent and pairs always stay within\n * one file.\n */\nfunction mergeGapAdjacentGroups(\n groups: SelectableCandidate[][],\n files: CrossFileDuplicationSourceFile[],\n maxGapTokens: number\n): CrossFileOccurrence[][] {\n const tokenOffsets: number[] = [];\n let offset = 0;\n for (const { tokens, candidates } of files) {\n tokenOffsets.push(offset);\n // Accumulated in a loop: spreading a project-scale candidate array as call arguments would\n // overflow V8's argument limit (~124k) and crash on Node.\n let tokenCount = tokens?.length ?? 0;\n if (!tokens) {\n for (const candidate of candidates) {\n tokenCount = Math.max(tokenCount, candidate.endTokenIndex);\n }\n }\n offset += tokenCount + maxGapTokens + 1;\n }\n const occurrenceGroups = groups.map((group) =>\n group\n .map((candidate): CrossFileOccurrence => {\n const start = candidate.startTokenIndex + (tokenOffsets[candidate.regionBucket] ?? 0);\n const end = candidate.endTokenIndex + (tokenOffsets[candidate.regionBucket] ?? 0);\n return {\n file: candidate.file,\n segments: [{ startTokenIndex: start, endTokenIndex: end }],\n tokenCount: candidate.tokenCount,\n startTokenIndex: start,\n endTokenIndex: end,\n startIndex: candidate.startIndex,\n endIndex: candidate.endIndex,\n startLine: candidate.startLine,\n endLine: candidate.endLine,\n };\n })\n .toSorted((left, right) => left.startTokenIndex - right.startTokenIndex)\n );\n return mergeAdjacentGroups(occurrenceGroups, maxGapTokens);\n}\n\nfunction summarize(groups: CrossFileOccurrence[][]): CrossFileDuplicationMetrics {\n const reported: CrossFileDuplicateBlockGroup[] = [];\n // Accumulated in a Map: file names are arbitrary strings, and a plain object would read\n // inherited properties for names like \"constructor\".\n const groupCountByFile = new Map<string, number>();\n let duplicateBlockCount = 0;\n for (const group of groups) {\n // Mirrors within-file counting: each redundant occurrence contributes one count per matched\n // fragment, gapped merging consolidates the grouping without halving the count, and spans a\n // partial merge shares between a retained group and the merged group count once.\n duplicateBlockCount += countRedundantFragments(group);\n const occurrences = group\n .map(({ file, startLine, endLine }) => ({ file, startLine, endLine }))\n .toSorted((left, right) => left.file.localeCompare(right.file) || left.startLine - right.startLine);\n const files = [...new Set(occurrences.map(({ file }) => file))];\n for (const file of files) {\n groupCountByFile.set(file, (groupCountByFile.get(file) ?? 0) + 1);\n }\n reported.push({ files, occurrences, tokenCount: group[0]?.tokenCount ?? 0 });\n }\n reported.sort(\n (left, right) =>\n right.tokenCount - left.tokenCount ||\n (left.occurrences[0]?.file ?? '').localeCompare(right.occurrences[0]?.file ?? '') ||\n (left.occurrences[0]?.startLine ?? 0) - (right.occurrences[0]?.startLine ?? 0)\n );\n return {\n duplicateBlockCount,\n duplicateBlockGroupCountByFile: Object.fromEntries(groupCountByFile),\n groups: reported,\n };\n}\n"],"mappings":"mPA6DA,SAAgB,EACd,EACA,EAC6B,CAC7B,IAAM,EAAY,GAAS,WAAa,EAA0B,UAC5D,EAAe,GAAS,cAAgB,EAA0B,aAClE,EAAoC,EAAM,SAAS,CAAE,OAAM,cAAc,IAC7E,EAAW,IAAK,IAAe,CAAE,GAAG,EAAW,aAAc,EAAW,MAAK,EAAE,CACjF,EAGA,IAAK,IAAM,KAAa,EAAwB,EAAO,CAAS,EAC9D,EAAW,KAAK,CAAS,EAQ3B,OAAO,EAAU,EAAuB,CAAC,GANzB,EACd,EACA,GAEC,EAAM,IAAU,EAAK,aAAe,EAAM,cAAgB,EAAK,WAAa,EAAM,UAEnC,CAAC,CAAC,OAAO,CAAC,EAAG,EAAO,CAAY,CAAC,CACrF,CAGA,SAAS,EAAwB,EAAyC,EAA0C,CAClH,IAAM,EAA+B,CAAC,EAChC,EAAoC,CAAC,EAC3C,IAAK,GAAM,CAAC,EAAW,CAAE,SAAQ,0BAA0B,EAAM,QAAQ,EACnE,GAAU,IACZ,EAAmB,KAAK,CAAS,EACjC,EAAS,KAAK,CAAE,SAAQ,mBAAoB,EAAwB,CAAM,EAAG,WAAY,CAAoB,CAAC,GAMlH,OAHI,EAAS,OAAS,EACb,CAAC,EAEH,EAAgC,EAAU,EAAW,EAAI,CAAC,CAAC,SAAS,CAAE,YAAW,kBAAmB,CACzG,IAAM,EAAY,EAAmB,GAC/B,EAAO,IAAc,IAAA,GAAY,IAAA,GAAY,EAAM,GACzD,OAAO,IAAc,IAAA,IAAa,IAAS,IAAA,GACvC,CAAC,EACD,CAAC,CAAE,GAAG,EAAW,aAAc,EAAW,KAAM,EAAK,IAAK,CAAC,CACjE,CAAC,CACH,CAEA,SAAS,EAAmB,EAAuC,CACjE,OAAO,EAAM,QAAU,GAAK,IAAI,IAAI,EAAM,IAAK,GAAc,EAAU,YAAY,CAAC,CAAC,CAAC,MAAQ,CAChG,CAQA,SAAS,EACP,EACA,EACA,EACyB,CACzB,IAAM,EAAyB,CAAC,EAC5B,EAAS,EACb,IAAK,GAAM,CAAE,SAAQ,gBAAgB,EAAO,CAC1C,EAAa,KAAK,CAAM,EAGxB,IAAI,EAAa,GAAQ,QAAU,EACnC,GAAI,CAAC,EACH,IAAK,IAAM,KAAa,EACtB,EAAa,KAAK,IAAI,EAAY,EAAU,aAAa,EAG7D,GAAU,EAAa,EAAe,CACxC,CACA,IAAM,EAAmB,EAAO,IAAK,GACnC,EACG,IAAK,GAAmC,CACvC,IAAM,EAAQ,EAAU,iBAAmB,EAAa,EAAU,eAAiB,GAC7E,EAAM,EAAU,eAAiB,EAAa,EAAU,eAAiB,GAC/E,MAAO,CACL,KAAM,EAAU,KAChB,SAAU,CAAC,CAAE,gBAAiB,EAAO,cAAe,CAAI,CAAC,EACzD,WAAY,EAAU,WACtB,gBAAiB,EACjB,cAAe,EACf,WAAY,EAAU,WACtB,SAAU,EAAU,SACpB,UAAW,EAAU,UACrB,QAAS,EAAU,OACrB,CACF,CAAC,CAAC,CACD,UAAU,EAAM,IAAU,EAAK,gBAAkB,EAAM,eAAe,CAC3E,EACA,OAAO,EAAoB,EAAkB,CAAY,CAC3D,CAEA,SAAS,EAAU,EAA8D,CAC/E,IAAM,EAA2C,CAAC,EAG5C,EAAmB,IAAI,IACzB,EAAsB,EAC1B,IAAK,IAAM,KAAS,EAAQ,CAI1B,GAAuB,EAAwB,CAAK,EACpD,IAAM,EAAc,EACjB,KAAK,CAAE,OAAM,YAAW,cAAe,CAAE,OAAM,YAAW,SAAQ,EAAE,CAAC,CACrE,UAAU,EAAM,IAAU,EAAK,KAAK,cAAc,EAAM,IAAI,GAAK,EAAK,UAAY,EAAM,SAAS,EAC9F,EAAQ,CAAC,GAAG,IAAI,IAAI,EAAY,KAAK,CAAE,UAAW,CAAI,CAAC,CAAC,EAC9D,IAAK,IAAM,KAAQ,EACjB,EAAiB,IAAI,GAAO,EAAiB,IAAI,CAAI,GAAK,GAAK,CAAC,EAElE,EAAS,KAAK,CAAE,QAAO,cAAa,WAAY,EAAM,EAAE,EAAE,YAAc,CAAE,CAAC,CAC7E,CAOA,OANA,EAAS,MACN,EAAM,IACL,EAAM,WAAa,EAAK,aACvB,EAAK,YAAY,EAAE,EAAE,MAAQ,GAAA,CAAI,cAAc,EAAM,YAAY,EAAE,EAAE,MAAQ,EAAE,IAC/E,EAAK,YAAY,EAAE,EAAE,WAAa,IAAM,EAAM,YAAY,EAAE,EAAE,WAAa,EAChF,EACO,CACL,sBACA,+BAAgC,OAAO,YAAY,CAAgB,EACnE,OAAQ,CACV,CACF"}
@@ -1,2 +1,2 @@
1
- "use strict";const e=require("./duplicateSelection.cjs"),t=new Set(`statement_block.block.compound_statement.body_statement.constructor_body.do_block.if_statement.for_statement.for_in_statement.enhanced_for_statement.for_range_loop.while_statement.do_statement.try_statement.try_with_resources_statement.with_statement.switch_statement.switch_expression.switch_case.switch_block_statement_group.switch_rule.case_clause.case_statement.match_statement.match_arm.except_clause.catch_clause.finally_clause.elif_clause.ensure.expression_statement.return_statement.return_expression.if_expression.for_expression.while_expression.loop_expression.match_expression.jsx_element.jsx_self_closing_element.if.unless.case.case_match.while.until.for.begin.when`.split(`.`)),n=new Set([`program`,`source_file`,`translation_unit`,`module`,`statement_block`,`block`,`compound_statement`,`body_statement`,`constructor_body`,`class_body`,`block_body`,`do_block`,`do`,`ensure`,`then`,`else`,`case_statement`,`switch_block_statement_group`,`switch_rule`,`expression_case`,`type_case`,`communication_case`,`default_case`]),r=new Set([`identifier`,`constant`,`instance_variable`,`class_variable`,`global_variable`]),i=new Set([`shorthand_property_identifier`,`shorthand_property_identifier_pattern`]),a=new Map([[`number`,`#num`],[`number_literal`,`#num`],[`integer`,`#num`],[`float`,`#num`],[`integer_literal`,`#num`],[`float_literal`,`#num`],[`int_literal`,`#num`],[`rune_literal`,`#char`],[`imaginary_literal`,`#num`],[`decimal_integer_literal`,`#num`],[`hex_integer_literal`,`#num`],[`octal_integer_literal`,`#num`],[`binary_integer_literal`,`#num`],[`decimal_floating_point_literal`,`#num`],[`hex_floating_point_literal`,`#num`],[`string_fragment`,`#str`],[`multiline_string_fragment`,`#str`],[`string_content`,`#str`],[`raw_string_content`,`#str`],[`heredoc_content`,`#str`],[`heredoc_beginning`,`#heredoc`],[`heredoc_end`,`#heredoc`],[`string`,`#str`],[`template_string`,`#str`],[`string_literal`,`#str`],[`interpreted_string_literal`,`#str`],[`raw_string_literal`,`#str`],[`raw_string`,`#str`],[`escape_sequence`,`#str`],[`char_literal`,`#char`],[`character_literal`,`#char`],[`character`,`#char`],[`regex_pattern`,`#regex`]]),o=new Set([`#num`,`#str`,`#char`,`#regex`]),s=new Set([`comment`,`line_comment`,`block_comment`]),c=new Set([`string_fragment`,`multiline_string_fragment`,`string_content`,`raw_string_content`,`escape_sequence`,`heredoc_content`]),l=new Set([`string_fragment`,`multiline_string_fragment`,`string_content`,`raw_string_content`,`escape_sequence`,`heredoc_content`,`string_start`,`string_end`]),u=new Map([[`call_expression`,`function`],[`method_invocation`,`name`],[`call`,`method`],[`attribute`,`attribute`],[`macro_invocation`,`macro`],[`field_access`,`field`],[`new_expression`,`constructor`],[`keyword_argument`,`name`],[`element_value_pair`,`key`],[`generic_function`,`function`],[`template_function`,`name`]]),d={minTokens:40,maxGapTokens:30};function f(e,t){return e*5>=t}function p(t,n,r){let i=r?.minTokens??d.minTokens,a=r?.maxGapTokens??d.maxGapTokens,o=[],s=[],c=[];h(t,o,s,c);let l=S(o),u=[...w(o,l,s,i),...T(o,l,c,i)];return U(B(z(e.selectMaximalGroups(u,e=>e.length>=2)),a),n,o)}function m(t,n){let r=n?.minTokens??d.minTokens,i=[],a=[],o=[];h(t,i,a,o);let s=S(i),c=w(i,s,a,r);for(let e of o){let t=e[0],n=e.at(-1);!t||!n||n.endTokenIndex-t.startTokenIndex<r||c.push(O(`s:${N(i,s,t.startTokenIndex,n.endTokenIndex)}`,t.startTokenIndex,n.endTokenIndex,t.node,n.node))}return e.dedupeByRegion(c).map(({fingerprint:e,tokenCount:t,startIndex:n,endIndex:r,startLine:i,endLine:a})=>({fingerprint:e,tokenCount:t,startIndex:n,endIndex:r,startLine:i,endLine:a}))}function h(e,r,i,a){function o(e){let c=r.length,l=e.childCount===0?void 0:g(e);if(e.childCount===0)_(e,r);else if(l!==void 0)r.push(v(l,y(e,l),e.startPosition.row,e.endPosition.row));else if(!s.has(e.type)){let t=[],r=e.isNamed&&n.has(e.type);for(let n of e.children){let e=o(n);r&&n.isNamed&&!s.has(n.type)&&t.push(e)}r&&t.length>0&&a.push(t)}let u={startTokenIndex:c,endTokenIndex:r.length,node:e};return e.isNamed&&t.has(e.type)&&i.push(u),u}o(e)}function g(e){let t=e.isNamed?a.get(e.type):void 0;if(t!==void 0)return e.namedChildren.every(e=>l.has(e.type))?t:void 0}function _(e,t){if(s.has(e.type))return;let n=e.startPosition.row,o=e.endPosition.row;if(e.isNamed&&i.has(e.type)){t.push(v(e.text,void 0,n,o),v(`:`,void 0,n,o),{kind:`id`,text:e.text,textHash:0,textHash2:0,startRow:n,endRow:o});return}if(e.isNamed&&r.has(e.type)&&!C(e)){t.push({kind:`id`,text:e.text,textHash:0,textHash2:0,startRow:n,endRow:o});return}let c=e.isNamed?a.get(e.type):void 0;c===void 0?t.push(v(e.text,void 0,n,o)):t.push(v(c,y(e,c),n,o))}function v(e,t,n,r){let i={kind:`text`,text:e,textHash:I(e),textHash2:L(e),startRow:n,endRow:r};return t!==void 0&&o.has(e)&&(i.literalHash=I(t),i.literalHash2=L(t)),i}function y(e,t){if(t!==`#str`&&t!==`#char`||c.has(e.type))return e.text;let n=e.namedChildren.filter(e=>c.has(e.type));return n.length>0?n.map(e=>e.text).join(``):x(e.text)}const b=new Set([`"`,`'`,"`"]);function x(e){let t=e[0];return e.length>=2&&t!==void 0&&b.has(t)&&e.endsWith(t)?e.slice(1,-1):e}function S(e){let t=new Int32Array(e.length+1);for(let[n,r]of e.entries())t[n+1]=(t[n]??0)+(r.literalHash===void 0?0:1);return t}function C(e){let t=e.parent;if(!t)return!1;if(t.type===`method_reference`||t.type===`call`&&t.childForFieldName(`function`)?.id===e.id||e.type===`constant`&&t.type===`call`&&t.childForFieldName(`receiver`)?.id===e.id||t.type===`method_invocation`&&t.childForFieldName(`object`)?.id===e.id&&/^\p{Lu}/u.test(e.text))return!0;if((t.type===`scoped_identifier`||t.type===`qualified_identifier`)&&(t.childForFieldName(`name`)?.id===e.id||t.childForFieldName(`path`)?.id===e.id)){let e=t;for(;e.parent&&(e.parent.type===`scoped_identifier`||e.parent.type===`qualified_identifier`||e.parent.type===`generic_function`||e.parent.type===`template_function`);)e=e.parent;if(e.parent?.type===`call_expression`&&e.parent.childForFieldName(`function`)?.id===e.id)return!0}if(t.type===`literal_element`&&t.parent?.type===`keyed_element`&&t.parent.namedChild(0)?.id===t.id)return!0;let n=u.get(t.type);return n!==void 0&&t.childForFieldName(n)?.id===e.id}function w(e,t,n,r){let i=[];for(let a of n)a.endTokenIndex-a.startTokenIndex<r||i.push(O(`b:${N(e,t,a.startTokenIndex,a.endTokenIndex)}`,a.startTokenIndex,a.endTokenIndex,a.node,a.node));return i}function T(e,t,n,r){let i=[],a=new Map,o=n.map(t=>D(e,t,r));for(let[e,t]of o.entries())for(let[n,r]of t.windowKeysByStart.entries())for(let t of r){if(t===void 0)continue;let r=a.get(t);r?(r.count+=1,r.containerIndex!==e&&(r.containerIndex=-1),r.minStart=Math.min(r.minStart,n),r.maxStart=Math.max(r.maxStart,n)):a.set(t,{count:1,containerIndex:e,minStart:n,maxStart:n})}let s=(e,t)=>{if(e===void 0)return!1;let n=a.get(e);return n!==void 0&&n.count>=2&&(n.containerIndex===-1||n.maxStart-n.minStart>=t)},c=e=>{let t=o[e.containerIndex]?.statementHashes??[],n=t[e.start];for(let r=e.start+1;r<e.start+e.length;r+=1)if(t[r]!==n)return!0;return!1},l=[];for(let[e,t]of o.entries())for(let[n,r]of t.windowKeysByStart.entries())for(let[i,a]of r.entries()){if(!s(a,i)||!c({containerIndex:e,start:n,length:i}))continue;let r=t.windowKeysByStart[n]?.[i+1],o=t.windowKeysByStart[n-1]?.[i+1];s(r,i+1)||s(o,i+1)||l.push({containerIndex:e,start:n,length:i})}let u=new Set(l.map(E)),d=l;for(;d.length>0;){let r=[];for(let a of d){let o=n[a.containerIndex],s=o?.[a.start],c=o?.[a.start+a.length-1];if(!s||!c)continue;let l=`s:${N(e,t,s.startTokenIndex,c.endTokenIndex)}`;i.push(O(l,s.startTokenIndex,c.endTokenIndex,s.node,c.node)),r.push(a)}d=[];for(let e of r)for(let t of[e.start,e.start+1]){let n={containerIndex:e.containerIndex,start:t,length:e.length-1},r=o[e.containerIndex]?.windowKeysByStart[t]?.[n.length];u.has(E(n))||!s(r,n.length)||!c(n)||(u.add(E(n)),d.push(n))}}return i}function E(e){return`${e.containerIndex}:${e.start}:${e.length}`}function D(e,t,n){let r=t.map(t=>P(e,t.startTokenIndex,t.endTokenIndex)),i=[];for(let e=0;e<t.length;e+=1){let a=[],o=5381,s=0,c=Math.min(t.length,e+100);for(let i=e;i<c;i+=1){let c=t[i],l=r[i];if(!c||l===void 0)break;o=R(o,l),s+=c.endTokenIndex-c.startTokenIndex;let u=i-e+1;a[u]=u>=2&&s>=n?R(o,u):void 0}i.push(a)}return{windowKeysByStart:i,statementHashes:r}}function O(e,t,n,r,i){return{fingerprint:e,tokenCount:n-t,startTokenIndex:t,endTokenIndex:n,startIndex:r.startIndex,endIndex:i.endIndex,startLine:r.startPosition.row+1,endLine:i.endPosition.row+1}}const k=[],A=[];function j(e){let t=k[e];return t===void 0&&(t=I(`$${e}`),k[e]=t),t}function M(e){let t=A[e];return t===void 0&&(t=L(`$${e}`),A[e]=t),t}function N(e,t,n,r){let[i,a]=F(e,n,r,f((t[r]??0)-(t[n]??0),r-n));return`${i}:${a}:${r-n}`}function P(e,t,n){let[r,i]=F(e,t,n,!1);return r^Math.imul(i,31)}function F(e,t,n,r){let i=new Map,a=5381,o=52711;for(let s=t;s<n;s+=1){let t=e[s];if(!t)continue;let n,c;if(t.kind===`id`){let e=i.get(t.text);e===void 0&&(e=i.size,i.set(t.text,e)),n=j(e),c=M(e)}else n=t.textHash,c=t.textHash2;a=Math.imul(a,31)+n|0,o=Math.imul(o,37)^c,r&&t.literalHash!==void 0&&t.literalHash2!==void 0&&(a=Math.imul(a,31)+t.literalHash|0,o=Math.imul(o,37)^t.literalHash2)}return[a,o]}function I(e){let t=5381;for(let n=0;n<e.length;n+=1)t=Math.imul(t,33)^e.charCodeAt(n);return t}function L(e){let t=-2128831035;for(let n=0;n<e.length;n+=1)t=Math.imul(t^e.charCodeAt(n),16777619);return t}function R(e,t){return Math.imul(e,31)+t}function z(e){let t=[];for(let n of e.values()){let e=n.map(e=>({segments:[{startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex}],tokenCount:e.tokenCount,startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex,startIndex:e.startIndex,endIndex:e.endIndex,startLine:e.startLine,endLine:e.endLine}));e.sort((e,t)=>e.startTokenIndex-t.startTokenIndex||e.endTokenIndex-t.endTokenIndex),t.push(e)}return t}function B(e,t){if(t<=0||e.length<2)return e;e.sort(V);for(let n=!0;n;){n=!1;for(let r=0;r<e.length&&!n;r+=1)for(let i=r+1;i<e.length;i+=1){let a=e[r],o=e[i];if(!a||!o)continue;let s=H(a,o,t)??H(o,a,t);if(s){e[r]=s,e.splice(i,1),e.sort(V),n=!0;break}}}return e}function V(e,t){let n=e[0],r=t[0];return(n?.startTokenIndex??0)-(r?.startTokenIndex??0)||(n?.endTokenIndex??0)-(r?.endTokenIndex??0)}function H(e,t,n){if(e.length===t.length){for(let[r,i]of e.entries()){let a=t[r];if(!a)return;let o=a.startTokenIndex-i.endTokenIndex;if(o<0||o>n)return;let s=e[r+1];if(s&&a.endTokenIndex>s.startTokenIndex)return}return e.map((e,n)=>{let r=t[n];return r?{segments:[...e.segments,...r.segments],tokenCount:e.tokenCount+r.tokenCount,startTokenIndex:e.startTokenIndex,endTokenIndex:r.endTokenIndex,startIndex:e.startIndex,endIndex:r.endIndex,startLine:e.startLine,endLine:r.endLine}:e})}}function U(e,t,n){let r=0,i=0,a=[],o=new Set;for(let s of e){r+=(s.length-1)*(s[0]?.segments.length??1);for(let e of s){i=Math.max(i,e.tokenCount);for(let r of e.segments)for(let e=r.startTokenIndex;e<r.endTokenIndex;e+=1){let r=n[e];for(let e=r?.startRow??0;e<=(r?.endRow??-1);e+=1)t.has(e+1)&&o.add(e+1)}}a.push(s.map(({startLine:e,endLine:t})=>({startLine:e,endLine:t})).toSorted((e,t)=>e.startLine-t.startLine))}return a.sort((e,t)=>(e[0]?.startLine??0)-(t[0]?.startLine??0)),{duplicateBlockCount:r,duplicateBlockGroupCount:e.length,duplicateBlockGroups:a,duplicateLineCount:o.size,duplicationRatio:t.size===0?0:o.size/t.size,maxDuplicateBlockSize:i}}exports.collectCrossFileDuplicateCandidates=m,exports.defaultDuplicationOptions=d,exports.measureDuplication=p;
1
+ "use strict";const e=require("./duplicateSelection.cjs"),t=new Set(`statement_block.block.compound_statement.body_statement.constructor_body.do_block.if_statement.for_statement.for_in_statement.enhanced_for_statement.for_range_loop.while_statement.do_statement.try_statement.try_with_resources_statement.with_statement.switch_statement.switch_expression.switch_case.switch_block_statement_group.switch_rule.case_clause.case_statement.match_statement.match_arm.except_clause.catch_clause.finally_clause.elif_clause.ensure.expression_statement.return_statement.return_expression.if_expression.for_expression.while_expression.loop_expression.match_expression.jsx_element.jsx_self_closing_element.if.unless.case.case_match.while.until.for.begin.when`.split(`.`)),n=new Set([`program`,`source_file`,`translation_unit`,`module`,`statement_block`,`block`,`compound_statement`,`body_statement`,`constructor_body`,`class_body`,`block_body`,`do_block`,`do`,`ensure`,`then`,`else`,`case_statement`,`switch_block_statement_group`,`switch_rule`,`expression_case`,`type_case`,`communication_case`,`default_case`]),r=new Set([`identifier`,`constant`,`instance_variable`,`class_variable`,`global_variable`]),i=new Set([`shorthand_property_identifier`,`shorthand_property_identifier_pattern`]),a=new Map([[`number`,`#num`],[`number_literal`,`#num`],[`integer`,`#num`],[`float`,`#num`],[`integer_literal`,`#num`],[`float_literal`,`#num`],[`int_literal`,`#num`],[`rune_literal`,`#char`],[`imaginary_literal`,`#num`],[`decimal_integer_literal`,`#num`],[`hex_integer_literal`,`#num`],[`octal_integer_literal`,`#num`],[`binary_integer_literal`,`#num`],[`decimal_floating_point_literal`,`#num`],[`hex_floating_point_literal`,`#num`],[`string_fragment`,`#str`],[`multiline_string_fragment`,`#str`],[`string_content`,`#str`],[`raw_string_content`,`#str`],[`heredoc_content`,`#str`],[`heredoc_beginning`,`#heredoc`],[`heredoc_end`,`#heredoc`],[`string`,`#str`],[`template_string`,`#str`],[`string_literal`,`#str`],[`interpreted_string_literal`,`#str`],[`raw_string_literal`,`#str`],[`raw_string`,`#str`],[`escape_sequence`,`#str`],[`char_literal`,`#char`],[`character_literal`,`#char`],[`character`,`#char`],[`regex_pattern`,`#regex`]]),o=new Set([`#num`,`#str`,`#char`,`#regex`]),s=new Set([`comment`,`line_comment`,`block_comment`]),c=new Set([`string_fragment`,`multiline_string_fragment`,`string_content`,`raw_string_content`,`escape_sequence`,`heredoc_content`]),l=new Set([`string_fragment`,`multiline_string_fragment`,`string_content`,`raw_string_content`,`escape_sequence`,`heredoc_content`,`string_start`,`string_end`]),u=new Map([[`call_expression`,`function`],[`method_invocation`,`name`],[`call`,`method`],[`attribute`,`attribute`],[`macro_invocation`,`macro`],[`field_access`,`field`],[`new_expression`,`constructor`],[`keyword_argument`,`name`],[`element_value_pair`,`key`],[`generic_function`,`function`],[`template_function`,`name`]]),d={minTokens:40,maxGapTokens:30,minSimilarityPercent:70};function f(e,t){return e*5>=t}function p(t,n,r){let i=r?.minTokens??d.minTokens,a=r?.maxGapTokens??d.maxGapTokens,o=r?.minSimilarityPercent??d.minSimilarityPercent,s=[],c=[],l=[];h(t,s,c,l);let u=S(s),f=[...w(s,u,c,i),...T(s,u,l,i)],p=X(Y(e.selectMaximalGroups(f,e=>e.length>=2)),a),m=k(s,u,c,i,o,p);return re([...p.filter(e=>e.length>0),...m],n,s)}function m(t,n){let r=n?.minTokens??d.minTokens,i=[],a=[],o=[];h(t,i,a,o);let s=S(i),c=w(i,s,a,r);for(let e of o){let t=e[0],n=e.at(-1);!t||!n||n.endTokenIndex-t.startTokenIndex<r||c.push(R(`s:${U(i,s,t.startTokenIndex,n.endTokenIndex)}`,t.startTokenIndex,n.endTokenIndex,t,n))}return{candidates:e.dedupeByRegion(c),tokens:i,containerStatements:o}}function h(e,r,i,a){function o(e){let c=r.length,l=e.childCount===0?void 0:g(e);if(e.childCount===0)_(e,r);else if(l!==void 0)r.push(v(l,y(e,l),e.startPosition.row,e.endPosition.row));else if(!s.has(e.type)){let t=[],r=e.isNamed&&n.has(e.type);for(let n of e.children){let e=o(n);r&&n.isNamed&&!s.has(n.type)&&t.push(e)}r&&t.length>0&&a.push(t)}let u={startTokenIndex:c,endTokenIndex:r.length,startIndex:e.startIndex,endIndex:e.endIndex,startLine:e.startPosition.row+1,endLine:e.endPosition.row+1};return e.isNamed&&t.has(e.type)&&i.push(u),u}o(e)}function g(e){let t=e.isNamed?a.get(e.type):void 0;if(t!==void 0)return e.namedChildren.every(e=>l.has(e.type))?t:void 0}function _(e,t){if(s.has(e.type))return;let n=e.startPosition.row,o=e.endPosition.row;if(e.isNamed&&i.has(e.type)){t.push(v(e.text,void 0,n,o,!0),v(`:`,void 0,n,o),{kind:`id`,text:e.text,textHash:0,textHash2:0,startRow:n,endRow:o});return}if(e.isNamed&&r.has(e.type)&&!C(e)){t.push({kind:`id`,text:e.text,textHash:0,textHash2:0,startRow:n,endRow:o});return}let c=e.isNamed?a.get(e.type):void 0;c===void 0?t.push(v(e.text,void 0,n,o,e.isNamed)):t.push(v(c,y(e,c),n,o))}function v(e,t,n,r,i=!1){let a={kind:`text`,text:e,textHash:K(e),textHash2:q(e),startRow:n,endRow:r};return t!==void 0&&o.has(e)&&(a.literalHash=K(t),a.literalHash2=q(t)),i&&(a.isName=!0),a}function y(e,t){if(t!==`#str`&&t!==`#char`||c.has(e.type))return e.text;let n=e.namedChildren.filter(e=>c.has(e.type));return n.length>0?n.map(e=>e.text).join(``):x(e.text)}const b=new Set([`"`,`'`,"`"]);function x(e){let t=e[0];return e.length>=2&&t!==void 0&&b.has(t)&&e.endsWith(t)?e.slice(1,-1):e}function S(e){let t=new Int32Array(e.length+1);for(let[n,r]of e.entries())t[n+1]=(t[n]??0)+(r.literalHash===void 0?0:1);return t}function C(e){let t=e.parent;if(!t)return!1;if(t.type===`method_reference`||t.type===`call`&&t.childForFieldName(`function`)?.id===e.id||e.type===`constant`&&t.type===`call`&&t.childForFieldName(`receiver`)?.id===e.id||t.type===`method_invocation`&&t.childForFieldName(`object`)?.id===e.id&&/^\p{Lu}/u.test(e.text))return!0;if((t.type===`scoped_identifier`||t.type===`qualified_identifier`)&&(t.childForFieldName(`name`)?.id===e.id||t.childForFieldName(`path`)?.id===e.id)){let e=t;for(;e.parent&&(e.parent.type===`scoped_identifier`||e.parent.type===`qualified_identifier`||e.parent.type===`generic_function`||e.parent.type===`template_function`);)e=e.parent;if(e.parent?.type===`call_expression`&&e.parent.childForFieldName(`function`)?.id===e.id)return!0}if(t.type===`literal_element`&&t.parent?.type===`keyed_element`&&t.parent.namedChild(0)?.id===t.id)return!0;let n=u.get(t.type);return n!==void 0&&t.childForFieldName(n)?.id===e.id}function w(e,t,n,r){let i=[];for(let a of n)a.endTokenIndex-a.startTokenIndex<r||i.push(R(`b:${U(e,t,a.startTokenIndex,a.endTokenIndex)}`,a.startTokenIndex,a.endTokenIndex,a,a));return i}function T(e,t,n,r){return E([{tokens:e,literalCountPrefix:t,containers:n}],r,!1).map(({candidate:e})=>e)}function E(e,t,n){let r=[],i=[],a=[];for(let[t,n]of e.entries())for(let e of n.containers)i.push(t),a.push(e);let o=t=>e[i[t]??0],s=new Map,c=a.map((e,n)=>O(o(n)?.tokens??[],e,t));for(let[e,t]of c.entries()){let n=i[e]??0;for(let[r,i]of t.windowKeysByStart.entries())for(let t of i){if(t===void 0)continue;let i=s.get(t);i?(i.count+=1,i.containerIndex!==e&&(i.containerIndex=-1),i.contextIndex!==n&&(i.contextIndex=-1),i.minStart=Math.min(i.minStart,r),i.maxStart=Math.max(i.maxStart,r)):s.set(t,{count:1,containerIndex:e,contextIndex:n,minStart:r,maxStart:r})}}let l=(e,t)=>{if(e===void 0)return!1;let r=s.get(e);return r===void 0||r.count<2?!1:n?r.contextIndex===-1:r.containerIndex===-1||r.maxStart-r.minStart>=t},u=e=>{let t=c[e.containerIndex]?.statementHashes??[],n=t[e.start];for(let r=e.start+1;r<e.start+e.length;r+=1)if(t[r]!==n)return!0;return!1},d=[];for(let[e,t]of c.entries())for(let[n,r]of t.windowKeysByStart.entries())for(let[i,a]of r.entries()){if(!l(a,i)||!u({containerIndex:e,start:n,length:i}))continue;let r=t.windowKeysByStart[n]?.[i+1],o=t.windowKeysByStart[n-1]?.[i+1];l(r,i+1)||l(o,i+1)||d.push({containerIndex:e,start:n,length:i})}let f=new Set(d.map(D)),p=d;for(;p.length>0;){let e=[];for(let t of p){let n=a[t.containerIndex],s=n?.[t.start],c=n?.[t.start+t.length-1],l=o(t.containerIndex);if(!s||!c||!l)continue;let u=`s:${U(l.tokens,l.literalCountPrefix,s.startTokenIndex,c.endTokenIndex)}`;r.push({candidate:R(u,s.startTokenIndex,c.endTokenIndex,s,c),contextIndex:i[t.containerIndex]??0}),e.push(t)}p=[];for(let t of e)for(let e of[t.start,t.start+1]){let n={containerIndex:t.containerIndex,start:e,length:t.length-1},r=c[t.containerIndex]?.windowKeysByStart[e]?.[n.length];f.has(D(n))||!l(r,n.length)||!u(n)||(f.add(D(n)),p.push(n))}}return r}function D(e){return`${e.containerIndex}:${e.start}:${e.length}`}function O(e,t,n){let r=t.map(t=>W(e,t.startTokenIndex,t.endTokenIndex)),i=[];for(let e=0;e<t.length;e+=1){let a=[],o=5381,s=0,c=Math.min(t.length,e+100);for(let i=e;i<c;i+=1){let c=t[i],l=r[i];if(!c||l===void 0)break;o=J(o,l),s+=c.endTokenIndex-c.startTokenIndex;let u=i-e+1;a[u]=u>=2&&s>=n?J(o,u):void 0}i.push(a)}return{windowKeysByStart:i,statementHashes:r}}function k(e,t,n,r,i,a){if(i>=100)return[];let o=A(n.filter(e=>{let n=e.endTokenIndex-e.startTokenIndex,i=(t[e.endTokenIndex]??0)-(t[e.startTokenIndex]??0);return n>=r&&!f(i,n)}).toSorted((e,t)=>e.startTokenIndex-t.startTokenIndex||t.endTokenIndex-e.endTokenIndex));if(o.length<2)return[];let s=o.map(e=>{let t=[];for(let[n,r]of a.entries())r.some(t=>t.startTokenIndex<e.endTokenIndex&&e.startTokenIndex<t.endTokenIndex)&&t.push(n);return t}),c=new Map,l=o.map(t=>F(e,t,c)),u=l.map(({sequence:e})=>ee(e)),d=te(u),p=o.map((e,t)=>t),m=e=>{let t=e;for(;p[t]!==t;)t=p[t]??t;for(;p[e]!==t;){let n=p[e]??t;p[e]=t,e=n}return t};for(let[e,t]of d){let n=Math.floor(e/u.length),r=e%u.length,a=l[n],o=l[r],c=u[n],d=u[r];if(!(!a||!o||!c||!d)&&!((s[n]?.length??0)>0&&(s[r]?.length??0)>0)&&!(t*100<10*Math.min(c.size,d.size))&&!(I(a,o)*100<=50*Math.max(a.contentTotal,o.contentTotal))&&ne(a.sequence,o.sequence)*100>=i*Math.max(a.sequence.length,o.sequence.length)){let e=m(n),t=m(r);p[Math.max(e,t)]=Math.min(e,t)}}let h=new Map;for(let e of o.keys()){let t=m(e),n=h.get(t)??[];n.push(e),h.set(t,n)}let g=[];for(let e of h.values()){if(e.length<2)continue;let t=e.filter(e=>(s[e]?.length??0)===0),n=e.filter(e=>(s[e]?.length??0)>0);if(n.length===0){g.push(e.flatMap(e=>o[e]?[P(o[e])]:[]));continue}if(t.length===0)continue;let r=t=>e.some(e=>{let n=o[e];return n!==void 0&&t.startTokenIndex<n.endTokenIndex&&n.startTokenIndex<t.endTokenIndex}),i=[...new Set(n.flatMap(e=>s[e]??[]))].filter(e=>{let t=a[e];return t!==void 0&&t.length>0&&t.every(r)}).toSorted((e,t)=>e-t),[c,...l]=i;if(c!==void 0){let t=new Set,n=[];for(let r of e){let e=o[r];if(!e)continue;let c=[];for(let n of i)for(let r of a[n]??[])!t.has(r)&&r.startTokenIndex<e.endTokenIndex&&e.startTokenIndex<r.endTokenIndex&&(t.add(r),c.push({occurrence:r,groupIndex:n}));c.sort((e,t)=>e.occurrence.startTokenIndex-t.occurrence.startTokenIndex||e.occurrence.endTokenIndex-t.occurrence.endTokenIndex);let l=[],u=new Set;for(let{occurrence:e,groupIndex:t}of c)u.has(t)&&(n.push(M(l)),l=[],u.clear()),l.push(e),u.add(t);l.length>0&&n.push(M(l)),c.length===0&&(s[r]?.length??0)===0&&n.push(P(e))}n.sort((e,t)=>e.startTokenIndex-t.startTokenIndex||e.endTokenIndex-t.endTokenIndex);for(let e of n)e.sharedWithMergedGroup=void 0;a[c]=n;for(let e of l)a[e]=[]}else t.length>=2&&g.push(t.flatMap(e=>o[e]?[P(o[e])]:[]))}return g.sort(Z),g}function A(e){let t=[],n=[];for(let r of e){for(;n.length>0&&n.at(-1).range.endTokenIndex<=r.startTokenIndex;)n.pop();let e=n.at(-1);if(e&&e.range.startTokenIndex===r.startTokenIndex&&e.range.endTokenIndex===r.endTokenIndex)continue;let i={range:r,children:[]};e?e.children.push(i):t.push(i),n.push(i)}let r=[],i=e=>{if(j(e))for(let t of e.children)i(t);else r.push(e.range)};for(let e of t)i(e);return r}function j(e){let[t]=e.children;return e.children.length>=2||t!==void 0&&j(t)}function M(e){let t=e[0];if(!t||e.length===1)return t??N();let n=[];for(let t of e.flatMap(e=>e.segments).toSorted((e,t)=>e.startTokenIndex-t.startTokenIndex||e.endTokenIndex-t.endTokenIndex)){let e=n.at(-1);e&&t.startTokenIndex<e.endTokenIndex?e.endTokenIndex=Math.max(e.endTokenIndex,t.endTokenIndex):n.push({...t})}return{segments:n,tokenCount:n.reduce((e,t)=>e+t.endTokenIndex-t.startTokenIndex,0),startTokenIndex:Math.min(...e.map(e=>e.startTokenIndex)),endTokenIndex:Math.max(...e.map(e=>e.endTokenIndex)),startIndex:Math.min(...e.map(e=>e.startIndex)),endIndex:Math.max(...e.map(e=>e.endIndex)),startLine:Math.min(...e.map(e=>e.startLine)),endLine:Math.max(...e.map(e=>e.endLine))}}function N(){return{segments:[],tokenCount:0,startTokenIndex:0,endTokenIndex:0,startIndex:0,endIndex:0,startLine:0,endLine:0}}function P(e){return{segments:[{startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex}],tokenCount:e.endTokenIndex-e.startTokenIndex,startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex,startIndex:e.startIndex,endIndex:e.endIndex,startLine:e.startLine,endLine:e.endLine}}function F(e,t,n){let r=new Int32Array(t.endTokenIndex-t.startTokenIndex),i=new Map,a=new Map,o=0;for(let s=t.startTokenIndex;s<t.endTokenIndex;s+=1){let c=e[s];if(!c)continue;let l;if(c.kind===`id`){let e=i.get(c.text);e===void 0&&(e=i.size,i.set(c.text,e)),l=-(e+1)}else{let e=`${c.textHash}:${c.textHash2}:${c.literalHash??0}:${c.literalHash2??0}`,t=n.get(e);t===void 0&&(t=n.size,n.set(e,t)),l=t,(c.isName===!0||c.literalHash!==void 0)&&(a.set(l,(a.get(l)??0)+1),o+=1)}r[s-t.startTokenIndex]=l}return{sequence:r,contentCountBySymbol:a,contentTotal:o}}function I(e,t){let[n,r]=e.contentCountBySymbol.size<=t.contentCountBySymbol.size?[e,t]:[t,e],i=0;for(let[e,t]of n.contentCountBySymbol)i+=Math.min(t,r.contentCountBySymbol.get(e)??0);return i}function ee(e){let t=new Set;for(let n=0;n+5<=e.length;n+=1){let r=5381;for(let t=0;t<5;t+=1)r=Math.imul(r,31)+(e[n+t]??0)|0;t.add(r)}return t}function te(e){let t=new Map;for(let[n,r]of e.entries())for(let e of r){let r=t.get(e);r||(r=[],t.set(e,r)),r.push(n)}let n=new Map;for(let r of t.values())for(let t=0;t<r.length;t+=1){let i=r[t]??0;for(let a=t+1;a<r.length;a+=1){let t=i*e.length+(r[a]??0);n.set(t,(n.get(t)??0)+1)}}return n}function ne(e,t){let n=e.length+31>>>5,r=new Map;for(let[t,i]of e.entries()){let e=r.get(i);e||(e=new Uint32Array(n),r.set(i,e));let a=t>>>5;e[a]=(e[a]??0)|1<<(t&31)}let i=new Uint32Array(n);for(let e of t){let t=r.get(e),a=1,o=0;for(let e=0;e<n;e+=1){let n=i[e]??0,r=((t?.[e]??0)|n)>>>0,s=(n<<1|a)>>>0;a=n>>>31;let c=r-s-o;o=+(c<0),i[e]=r&~c}}let a=0;for(let e of i)a+=L(e);return a}function L(e){let t=e-(e>>>1&1431655765);return t=(t&858993459)+(t>>>2&858993459),Math.imul(t+(t>>>4)&252645135,16843009)>>>24&255}function R(e,t,n,r,i){return{fingerprint:e,tokenCount:n-t,startTokenIndex:t,endTokenIndex:n,startIndex:r.startIndex,endIndex:i.endIndex,startLine:r.startLine,endLine:i.endLine}}const z=[],B=[];function V(e){let t=z[e];return t===void 0&&(t=K(`$${e}`),z[e]=t),t}function H(e){let t=B[e];return t===void 0&&(t=q(`$${e}`),B[e]=t),t}function U(e,t,n,r){let[i,a]=G(e,n,r,f((t[r]??0)-(t[n]??0),r-n));return`${i}:${a}:${r-n}`}function W(e,t,n){let[r,i]=G(e,t,n,!1);return r^Math.imul(i,31)}function G(e,t,n,r){let i=new Map,a=5381,o=52711;for(let s=t;s<n;s+=1){let t=e[s];if(!t)continue;let n,c;if(t.kind===`id`){let e=i.get(t.text);e===void 0&&(e=i.size,i.set(t.text,e)),n=V(e),c=H(e)}else n=t.textHash,c=t.textHash2;a=Math.imul(a,31)+n|0,o=Math.imul(o,37)^c,r&&t.literalHash!==void 0&&t.literalHash2!==void 0&&(a=Math.imul(a,31)+t.literalHash|0,o=Math.imul(o,37)^t.literalHash2)}return[a,o]}function K(e){let t=5381;for(let n=0;n<e.length;n+=1)t=Math.imul(t,33)^e.charCodeAt(n);return t}function q(e){let t=-2128831035;for(let n=0;n<e.length;n+=1)t=Math.imul(t^e.charCodeAt(n),16777619);return t}function J(e,t){return Math.imul(e,31)+t}function Y(e){let t=[];for(let n of e.values()){let e=n.map(e=>({segments:[{startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex}],tokenCount:e.tokenCount,startTokenIndex:e.startTokenIndex,endTokenIndex:e.endTokenIndex,startIndex:e.startIndex,endIndex:e.endIndex,startLine:e.startLine,endLine:e.endLine}));e.sort((e,t)=>e.startTokenIndex-t.startTokenIndex||e.endTokenIndex-t.endTokenIndex),t.push(e)}return t}function X(e,t){if(t<=0||e.length<2)return e;e.sort(Z);for(let n=!0;n;){n=!1;for(let r=0;r<e.length&&!n;r+=1)for(let i=r+1;i<e.length;i+=1){let a=e[r],o=e[i];if(!a||!o)continue;let s=Q(a,o,t),c=s??Q(o,a,t);if(!c)continue;let l=s?c.firstConsumed:c.secondConsumed,u=s?c.secondConsumed:c.firstConsumed;l&&u?(e[r]=c.merged,e.splice(i,1)):u?e[i]=c.merged:e[r]=c.merged;for(let e of c.pairedRetained)e.sharedWithMergedGroup=!0;e.sort(Z),n=!0;break}}return e}function Z(e,t){let n=e[0],r=t[0];return(n?.startTokenIndex??0)-(r?.startTokenIndex??0)||(n?.endTokenIndex??0)-(r?.endTokenIndex??0)}function Q(e,t,n){let r=e.filter(e=>!e.sharedWithMergedGroup),i=t.filter(e=>!e.sharedWithMergedGroup),a=[],o=0,s=-1;for(let e of i){for(;o<r.length;){let t=r[o];if(t&&t.endTokenIndex+n<e.startTokenIndex)o+=1;else break}let t=r[o];t&&t.endTokenIndex<=e.startTokenIndex&&t.startTokenIndex>=s&&(a.push([t,e]),s=e.endTokenIndex,o+=1)}let c=a.length===e.length,l=a.length===t.length;if(!(a.length<2||!c&&!l))return{merged:a.map(([e,t])=>({...e,sharedWithMergedGroup:void 0,segments:[...e.segments,...t.segments],tokenCount:e.tokenCount+t.tokenCount,endTokenIndex:t.endTokenIndex,endIndex:t.endIndex,endLine:t.endLine})),firstConsumed:c,secondConsumed:l,pairedRetained:c===l?[]:a.map(([e,t])=>c?t:e)}}function $(e){let t=0,n=0,r=!1;for(let i of e){if(i.sharedWithMergedGroup){r=!0;continue}t+=i.segments.length,n=Math.max(n,i.segments.length)}return r?t:t-n}function re(e,t,n){let r=0,i=0,a=[],o=new Set;for(let s of e){r+=$(s);for(let e of s){i=Math.max(i,e.tokenCount);for(let r of e.segments)for(let e=r.startTokenIndex;e<r.endTokenIndex;e+=1){let r=n[e];for(let e=r?.startRow??0;e<=(r?.endRow??-1);e+=1)t.has(e+1)&&o.add(e+1)}}a.push(s.map(({startLine:e,endLine:t})=>({startLine:e,endLine:t})).toSorted((e,t)=>e.startLine-t.startLine))}return a.sort((e,t)=>(e[0]?.startLine??0)-(t[0]?.startLine??0)),{duplicateBlockCount:r,duplicateBlockGroupCount:e.length,duplicateBlockGroups:a,duplicateLineCount:o.size,duplicationRatio:t.size===0?0:o.size/t.size,maxDuplicateBlockSize:i}}exports.buildLiteralCountPrefix=S,exports.collectCrossFileDuplicateCandidates=m,exports.collectSequenceWindowCandidates=E,exports.countRedundantFragments=$,exports.defaultDuplicationOptions=d,exports.measureDuplication=p,exports.mergeAdjacentGroups=X;
2
2
  //# sourceMappingURL=duplication.cjs.map