@omfalos/mokosh 0.3.1 → 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +31 -29
- package/dist/cli.js.map +1 -1
- package/dist/cli.mjs +31 -29
- package/dist/cli.mjs.map +1 -1
- package/dist/duplication-worker.d.mts +25 -0
- package/dist/duplication-worker.d.ts +25 -0
- package/dist/duplication-worker.js +5 -0
- package/dist/duplication-worker.js.map +1 -0
- package/dist/duplication-worker.mjs +5 -0
- package/dist/duplication-worker.mjs.map +1 -0
- package/dist/index.d.mts +95 -18
- package/dist/index.d.ts +95 -18
- package/dist/index.js +22 -20
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +22 -20
- package/dist/index.mjs.map +1 -1
- package/dist/mcp.js +25 -23
- package/dist/mcp.js.map +1 -1
- package/dist/mcp.mjs +25 -23
- package/dist/mcp.mjs.map +1 -1
- package/dist/parse-CAKctgb6.d.mts +6 -0
- package/dist/parse-CAKctgb6.d.ts +6 -0
- package/dist/parse-worker.d.mts +2 -1
- package/dist/parse-worker.d.ts +2 -1
- package/dist/{types-C9fLCS45.d.mts → types-BlN-U5AM.d.ts} +2 -5
- package/dist/{types-C9fLCS45.d.ts → types-BtSqoqbZ.d.mts} +2 -5
- package/package.json +7 -7
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { F as FileType } from './parse-CAKctgb6.mjs';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Language-agnostic source tokenizer for duplicate-code detection. Strips per-language
|
|
5
|
+
* comment syntax, then splits what remains into a normalized token stream shared by every
|
|
6
|
+
* language `DEFAULT_EXTENSIONS` covers — one generic tokenizer rather than a per-language
|
|
7
|
+
* lexer, so `findDuplicates` works uniformly across TS/JS, Python, Go, CoffeeScript,
|
|
8
|
+
* LiveScript, Lua, Gherkin, style files, and Markdown.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** One normalized token plus the 1-based source line it came from. */
|
|
12
|
+
interface NormalizedToken {
|
|
13
|
+
text: string;
|
|
14
|
+
line: number;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** Piscina task handler: tokenizes a single file's content in a worker thread, for `findDuplicates`. */
|
|
18
|
+
|
|
19
|
+
declare function tokenizeInWorker(payload: {
|
|
20
|
+
source: string;
|
|
21
|
+
fileType: FileType;
|
|
22
|
+
ignoreLiterals: boolean;
|
|
23
|
+
}): NormalizedToken[];
|
|
24
|
+
|
|
25
|
+
export { tokenizeInWorker as default };
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { F as FileType } from './parse-CAKctgb6.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Language-agnostic source tokenizer for duplicate-code detection. Strips per-language
|
|
5
|
+
* comment syntax, then splits what remains into a normalized token stream shared by every
|
|
6
|
+
* language `DEFAULT_EXTENSIONS` covers — one generic tokenizer rather than a per-language
|
|
7
|
+
* lexer, so `findDuplicates` works uniformly across TS/JS, Python, Go, CoffeeScript,
|
|
8
|
+
* LiveScript, Lua, Gherkin, style files, and Markdown.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** One normalized token plus the 1-based source line it came from. */
|
|
12
|
+
interface NormalizedToken {
|
|
13
|
+
text: string;
|
|
14
|
+
line: number;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** Piscina task handler: tokenizes a single file's content in a worker thread, for `findDuplicates`. */
|
|
18
|
+
|
|
19
|
+
declare function tokenizeInWorker(payload: {
|
|
20
|
+
source: string;
|
|
21
|
+
fileType: FileType;
|
|
22
|
+
ignoreLiterals: boolean;
|
|
23
|
+
}): NormalizedToken[];
|
|
24
|
+
|
|
25
|
+
export { tokenizeInWorker as default };
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
"use strict";var p=Object.defineProperty;var u=Object.getOwnPropertyDescriptor;var h=Object.getOwnPropertyNames;var x=Object.prototype.hasOwnProperty;var b=(e,t)=>{for(var n in t)p(e,n,{get:t[n],enumerable:!0})},y=(e,t,n,i)=>{if(t&&typeof t=="object"||typeof t=="function")for(let o of h(t))!x.call(e,o)&&o!==n&&p(e,o,{get:()=>t[o],enumerable:!(i=u(t,o))||i.enumerable});return e};var N=e=>y(p({},"__esModule",{value:!0}),e);var $={};b($,{default:()=>T});module.exports=N($);var R={typescript:{line:["//"],block:[["/*","*/"]]},javascript:{line:["//"],block:[["/*","*/"]]},go:{line:["//"],block:[["/*","*/"]]},css:{block:[["/*","*/"]]},scss:{line:["//"],block:[["/*","*/"]]},less:{line:["//"],block:[["/*","*/"]]},stylus:{line:["//"],block:[["/*","*/"]]},python:{line:["#"]},gherkin:{line:["#"]},coffeescript:{line:["#"],block:[["###","###"]]},livescript:{line:["#"],block:[["###","###"]]},lua:{line:["--"],block:[["--[[","]]"]]},markdown:{block:[["<!--","-->"]]}};function m(e,t,n){let i=e.slice(t,n).replace(/[^\n]/g," ");return e.slice(0,t)+i+e.slice(n)}function z(e,t){if(!t)return e;let n=e;for(let[i,o]of t.block??[]){let c=0;for(;;){let l=n.indexOf(i,c);if(l===-1)break;let r=n.indexOf(o,l+i.length),s=r===-1?n.length:r+o.length;n=m(n,l,s),c=s}}for(let i of t.line??[]){let o=0;for(;;){let c=n.indexOf(i,o);if(c===-1)break;let l=n.indexOf(`
|
|
2
|
+
`,c),r=l===-1?n.length:l;n=m(n,c,r),o=r}}return n}var A=["===","!==","...","=>","==","!=","<=",">=","&&","||","::","->","..","+=","-=","*=","/="],d=new RegExp(`${A.map(e=>e.replace(/[.*+?^${}()|[\]\\]/g,"\\$&")).join("|")}|[A-Za-z_$][A-Za-z0-9_$]*|\\d+(?:\\.\\d+)?|"(?:\\\\.|[^"\\\\])*"|'(?:\\\\.|[^'\\\\])*'|\`(?:\\\\.|[^\`\\\\])*\`|\\S`,"g"),_=/^[A-Za-z_$][A-Za-z0-9_$]*$/,k=/^\d+(?:\.\d+)?$/,w=/^(".*"|'.*'|`.*`)$/s,S=new Set(["if","else","elif","for","while","do","switch","case","default","break","continue","return","function","def","class","struct","interface","const","let","var","local","import","export","from","package","try","catch","except","finally","throw","raise","new","this","self","end","then","yield","async","await","nil","null","None","true","false","True","False","and","or","not","in","of","is"]);function g(e,t,n=!0){let i=z(e,R[t]),o=[],c=1,l=0;d.lastIndex=0;let r=d.exec(i);for(;r!==null;){for(let a=l;a<r.index;a++)i[a]===`
|
|
3
|
+
`&&c++;l=r.index;let s=r[0],f=s;_.test(s)&&!S.has(s)?f="ID":n&&(k.test(s)||w.test(s))&&(f=k.test(s)?"NUM":"STR"),o.push({text:f,line:c});for(let a=l;a<r.index+s.length;a++)i[a]===`
|
|
4
|
+
`&&c++;l=r.index+s.length,r=d.exec(i)}return o}function T(e){return g(e.source,e.fileType,e.ignoreLiterals)}
|
|
5
|
+
//# sourceMappingURL=duplication-worker.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/duplication-worker.ts","../src/graph/duplication/tokenizer.ts"],"sourcesContent":["/** Piscina task handler: tokenizes a single file's content in a worker thread, for `findDuplicates`. */\n\nimport type { NormalizedToken } from \"./graph/duplication/tokenizer.js\";\nimport { tokenize } from \"./graph/duplication/tokenizer.js\";\nimport type { FileType } from \"./types/parse\";\n\nexport default function tokenizeInWorker(payload: {\n source: string;\n fileType: FileType;\n ignoreLiterals: boolean;\n}): NormalizedToken[] {\n return tokenize(payload.source, payload.fileType, payload.ignoreLiterals);\n}\n","/**\n * Language-agnostic source tokenizer for duplicate-code detection. Strips per-language\n * comment syntax, then splits what remains into a normalized token stream shared by every\n * language `DEFAULT_EXTENSIONS` covers — one generic tokenizer rather than a per-language\n * lexer, so `findDuplicates` works uniformly across TS/JS, Python, Go, CoffeeScript,\n * LiveScript, Lua, Gherkin, style files, and Markdown.\n */\nimport type { FileType } from \"../../types/parse\";\n\n/** One normalized token plus the 1-based source line it came from. */\nexport interface NormalizedToken {\n text: string;\n line: number;\n}\n\ninterface CommentSyntax {\n line?: string[];\n block?: Array<[string, string]>;\n}\n\n/**\n * Per-`FileType` comment markers used to mask out comment text before tokenizing, so comment\n * wording never contributes to a duplicate match. Deliberately not string-literal-aware — a\n * `//` inside a string is rare and the cost of occasionally over-stripping is low for a\n * heuristic duplicate finder, same trade-off the complexity scorers make elsewhere.\n */\nconst COMMENT_SYNTAX: Partial<Record<FileType, CommentSyntax>> = {\n typescript: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n javascript: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n go: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n css: { block: [[\"/*\", \"*/\"]] },\n scss: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n less: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n stylus: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n python: { line: [\"#\"] },\n gherkin: { line: [\"#\"] },\n coffeescript: { line: [\"#\"], block: [[\"###\", \"###\"]] },\n livescript: { line: [\"#\"], block: [[\"###\", \"###\"]] },\n lua: { line: [\"--\"], block: [[\"--[[\", \"]]\"]] },\n markdown: { block: [[\"<!--\", \"-->\"]] },\n};\n\n/**\n * @description Replaces every non-newline character in `source[start, end)` with a space,\n * so downstream line-number tracking stays correct while the masked text can no longer\n * match anything.\n * @param source - Full file source.\n * @param start - Start offset (inclusive) of the range to mask.\n * @param end - End offset (exclusive) of the range to mask.\n * @returns `source` with the range masked.\n */\nfunction maskRange(source: string, start: number, end: number): string {\n const masked = source.slice(start, end).replace(/[^\\n]/g, \" \");\n return source.slice(0, start) + masked + source.slice(end);\n}\n\n/**\n * @description Masks out every line and block comment matching `syntax`, preserving line\n * breaks and overall string length so line numbers computed later stay accurate.\n * @param source - Full file source.\n * @param syntax - Comment markers for the file's language; absent for languages with no\n * configured comment syntax, in which case the source passes through unchanged.\n * @returns The source with comment text masked to spaces.\n */\nfunction stripComments(source: string, syntax: CommentSyntax | undefined): string {\n if (!syntax) return source;\n let result = source;\n\n for (const [open, close] of syntax.block ?? []) {\n let searchFrom = 0;\n for (;;) {\n const start = result.indexOf(open, searchFrom);\n if (start === -1) break;\n const end = result.indexOf(close, start + open.length);\n const rangeEnd = end === -1 ? result.length : end + close.length;\n result = maskRange(result, start, rangeEnd);\n searchFrom = rangeEnd;\n }\n }\n\n for (const marker of syntax.line ?? []) {\n let searchFrom = 0;\n for (;;) {\n const start = result.indexOf(marker, searchFrom);\n if (start === -1) break;\n const lineEnd = result.indexOf(\"\\n\", start);\n const rangeEnd = lineEnd === -1 ? result.length : lineEnd;\n result = maskRange(result, start, rangeEnd);\n searchFrom = rangeEnd;\n }\n }\n\n return result;\n}\n\nconst MULTI_CHAR_OPERATORS = [\n \"===\",\n \"!==\",\n \"...\",\n \"=>\",\n \"==\",\n \"!=\",\n \"<=\",\n \">=\",\n \"&&\",\n \"||\",\n \"::\",\n \"->\",\n \"..\",\n \"+=\",\n \"-=\",\n \"*=\",\n \"/=\",\n];\n\n/** Matches, in priority order: multi-char operators, identifiers, numbers, and quoted strings. */\nconst TOKEN_PATTERN = new RegExp(\n `${MULTI_CHAR_OPERATORS.map((op) => op.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\")).join(\"|\")}|[A-Za-z_$][A-Za-z0-9_$]*|\\\\d+(?:\\\\.\\\\d+)?|\"(?:\\\\\\\\.|[^\"\\\\\\\\])*\"|'(?:\\\\\\\\.|[^'\\\\\\\\])*'|\\`(?:\\\\\\\\.|[^\\`\\\\\\\\])*\\`|\\\\S`,\n \"g\",\n);\n\nconst IDENTIFIER_PATTERN = /^[A-Za-z_$][A-Za-z0-9_$]*$/;\nconst NUMBER_PATTERN = /^\\d+(?:\\.\\d+)?$/;\nconst STRING_PATTERN = /^(\".*\"|'.*'|`.*`)$/s;\n\n/**\n * Keywords kept verbatim rather than collapsed to the `ID` placeholder, shared across every\n * language rather than a per-language keyword table — deliberately coarse. Without this,\n * structurally different code (an `if` vs. a `for`, a `class` vs. a `function`) would tokenize\n * identically once every identifier-shaped word became `ID`, which would make the shingle\n * matcher far too eager. This list only needs to cover the keywords common enough across\n * TS/JS/Python/Go/CoffeeScript/LiveScript/Lua/Gherkin to matter for shape-preservation — it\n * doesn't need to be exhaustive or language-precise, since keeping a non-keyword here just\n * costs a little precision, not correctness.\n */\nconst KEYWORDS = new Set([\n \"if\",\n \"else\",\n \"elif\",\n \"for\",\n \"while\",\n \"do\",\n \"switch\",\n \"case\",\n \"default\",\n \"break\",\n \"continue\",\n \"return\",\n \"function\",\n \"def\",\n \"class\",\n \"struct\",\n \"interface\",\n \"const\",\n \"let\",\n \"var\",\n \"local\",\n \"import\",\n \"export\",\n \"from\",\n \"package\",\n \"try\",\n \"catch\",\n \"except\",\n \"finally\",\n \"throw\",\n \"raise\",\n \"new\",\n \"this\",\n \"self\",\n \"end\",\n \"then\",\n \"yield\",\n \"async\",\n \"await\",\n \"nil\",\n \"null\",\n \"None\",\n \"true\",\n \"false\",\n \"True\",\n \"False\",\n \"and\",\n \"or\",\n \"not\",\n \"in\",\n \"of\",\n \"is\",\n]);\n\n/**\n * @description Tokenizes source text into a normalized stream for shingle-based duplicate\n * matching: identifiers always collapse to a single placeholder (so renamed-variable clones\n * still hash identically), and literals optionally collapse too (`ignoreLiterals`, default\n * on) so only structural shape — not the specific values used — drives the match.\n * @param source - Raw file content.\n * @param fileType - Selects the comment-stripping rule; unrecognised types skip stripping.\n * @param ignoreLiterals - When true (default), string/number literal tokens are also\n * normalized to a placeholder rather than compared verbatim.\n * @returns Normalized tokens in source order, each carrying its 1-based source line.\n */\nexport function tokenize(\n source: string,\n fileType: FileType,\n ignoreLiterals = true,\n): NormalizedToken[] {\n const stripped = stripComments(source, COMMENT_SYNTAX[fileType]);\n const tokens: NormalizedToken[] = [];\n\n let line = 1;\n let lastIndex = 0;\n TOKEN_PATTERN.lastIndex = 0;\n let match = TOKEN_PATTERN.exec(stripped);\n\n while (match !== null) {\n for (let i = lastIndex; i < match.index; i++) {\n if (stripped[i] === \"\\n\") line++;\n }\n lastIndex = match.index;\n\n const raw = match[0];\n let text = raw;\n if (IDENTIFIER_PATTERN.test(raw) && !KEYWORDS.has(raw)) {\n text = \"ID\";\n } else if (ignoreLiterals && (NUMBER_PATTERN.test(raw) || STRING_PATTERN.test(raw))) {\n text = NUMBER_PATTERN.test(raw) ? \"NUM\" : \"STR\";\n }\n tokens.push({ text, line });\n\n for (let i = lastIndex; i < match.index + raw.length; i++) {\n if (stripped[i] === \"\\n\") line++;\n }\n lastIndex = match.index + raw.length;\n match = TOKEN_PATTERN.exec(stripped);\n }\n\n return tokens;\n}\n"],"mappings":"yaAAA,IAAAA,EAAA,GAAAC,EAAAD,EAAA,aAAAE,IAAA,eAAAC,EAAAH,GC0BA,IAAMI,EAA2D,CAC/D,WAAY,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAClD,WAAY,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAClD,GAAI,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC1C,IAAK,CAAE,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC7B,KAAM,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC5C,KAAM,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC5C,OAAQ,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC9C,OAAQ,CAAE,KAAM,CAAC,GAAG,CAAE,EACtB,QAAS,CAAE,KAAM,CAAC,GAAG,CAAE,EACvB,aAAc,CAAE,KAAM,CAAC,GAAG,EAAG,MAAO,CAAC,CAAC,MAAO,KAAK,CAAC,CAAE,EACrD,WAAY,CAAE,KAAM,CAAC,GAAG,EAAG,MAAO,CAAC,CAAC,MAAO,KAAK,CAAC,CAAE,EACnD,IAAK,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,OAAQ,IAAI,CAAC,CAAE,EAC7C,SAAU,CAAE,MAAO,CAAC,CAAC,OAAQ,KAAK,CAAC,CAAE,CACvC,EAWA,SAASC,EAAUC,EAAgBC,EAAeC,EAAqB,CACrE,IAAMC,EAASH,EAAO,MAAMC,EAAOC,CAAG,EAAE,QAAQ,SAAU,GAAG,EAC7D,OAAOF,EAAO,MAAM,EAAGC,CAAK,EAAIE,EAASH,EAAO,MAAME,CAAG,CAC3D,CAUA,SAASE,EAAcJ,EAAgBK,EAA2C,CAChF,GAAI,CAACA,EAAQ,OAAOL,EACpB,IAAIM,EAASN,EAEb,OAAW,CAACO,EAAMC,CAAK,IAAKH,EAAO,OAAS,CAAC,EAAG,CAC9C,IAAII,EAAa,EACjB,OAAS,CACP,IAAMR,EAAQK,EAAO,QAAQC,EAAME,CAAU,EAC7C,GAAIR,IAAU,GAAI,MAClB,IAAMC,EAAMI,EAAO,QAAQE,EAAOP,EAAQM,EAAK,MAAM,EAC/CG,EAAWR,IAAQ,GAAKI,EAAO,OAASJ,EAAMM,EAAM,OAC1DF,EAASP,EAAUO,EAAQL,EAAOS,CAAQ,EAC1CD,EAAaC,CACf,CACF,CAEA,QAAWC,KAAUN,EAAO,MAAQ,CAAC,EAAG,CACtC,IAAII,EAAa,EACjB,OAAS,CACP,IAAMR,EAAQK,EAAO,QAAQK,EAAQF,CAAU,EAC/C,GAAIR,IAAU,GAAI,MAClB,IAAMW,EAAUN,EAAO,QAAQ;AAAA,EAAML,CAAK,EACpCS,EAAWE,IAAY,GAAKN,EAAO,OAASM,EAClDN,EAASP,EAAUO,EAAQL,EAAOS,CAAQ,EAC1CD,EAAaC,CACf,CACF,CAEA,OAAOJ,CACT,CAEA,IAAMO,EAAuB,CAC3B,MACA,MACA,MACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,IACF,EAGMC,EAAgB,IAAI,OACxB,GAAGD,EAAqB,IAAKE,GAAOA,EAAG,QAAQ,sBAAuB,MAAM,CAAC,EAAE,KAAK,GAAG,CAAC,sHACxF,GACF,EAEMC,EAAqB,6BACrBC,EAAiB,kBACjBC,EAAiB,sBAYjBC,EAAW,IAAI,IAAI,CACvB,KACA,OACA,OACA,MACA,QACA,KACA,SACA,OACA,UACA,QACA,WACA,SACA,WACA,MACA,QACA,SACA,YACA,QACA,MACA,MACA,QACA,SACA,SACA,OACA,UACA,MACA,QACA,SACA,UACA,QACA,QACA,MACA,OACA,OACA,MACA,OACA,QACA,QACA,QACA,MACA,OACA,OACA,OACA,QACA,OACA,QACA,MACA,KACA,MACA,KACA,KACA,IACF,CAAC,EAaM,SAASC,EACdpB,EACAqB,EACAC,EAAiB,GACE,CACnB,IAAMC,EAAWnB,EAAcJ,EAAQF,EAAeuB,CAAQ,CAAC,EACzDG,EAA4B,CAAC,EAE/BC,EAAO,EACPC,EAAY,EAChBZ,EAAc,UAAY,EAC1B,IAAIa,EAAQb,EAAc,KAAKS,CAAQ,EAEvC,KAAOI,IAAU,MAAM,CACrB,QAASC,EAAIF,EAAWE,EAAID,EAAM,MAAOC,IACnCL,EAASK,CAAC,IAAM;AAAA,GAAMH,IAE5BC,EAAYC,EAAM,MAElB,IAAME,EAAMF,EAAM,CAAC,EACfG,EAAOD,EACPb,EAAmB,KAAKa,CAAG,GAAK,CAACV,EAAS,IAAIU,CAAG,EACnDC,EAAO,KACER,IAAmBL,EAAe,KAAKY,CAAG,GAAKX,EAAe,KAAKW,CAAG,KAC/EC,EAAOb,EAAe,KAAKY,CAAG,EAAI,MAAQ,OAE5CL,EAAO,KAAK,CAAE,KAAAM,EAAM,KAAAL,CAAK,CAAC,EAE1B,QAASG,EAAIF,EAAWE,EAAID,EAAM,MAAQE,EAAI,OAAQD,IAChDL,EAASK,CAAC,IAAM;AAAA,GAAMH,IAE5BC,EAAYC,EAAM,MAAQE,EAAI,OAC9BF,EAAQb,EAAc,KAAKS,CAAQ,CACrC,CAEA,OAAOC,CACT,CDvOe,SAARO,EAAkCC,EAInB,CACpB,OAAOC,EAASD,EAAQ,OAAQA,EAAQ,SAAUA,EAAQ,cAAc,CAC1E","names":["duplication_worker_exports","__export","tokenizeInWorker","__toCommonJS","COMMENT_SYNTAX","maskRange","source","start","end","masked","stripComments","syntax","result","open","close","searchFrom","rangeEnd","marker","lineEnd","MULTI_CHAR_OPERATORS","TOKEN_PATTERN","op","IDENTIFIER_PATTERN","NUMBER_PATTERN","STRING_PATTERN","KEYWORDS","tokenize","fileType","ignoreLiterals","stripped","tokens","line","lastIndex","match","i","raw","text","tokenizeInWorker","payload","tokenize"]}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
var u={typescript:{line:["//"],block:[["/*","*/"]]},javascript:{line:["//"],block:[["/*","*/"]]},go:{line:["//"],block:[["/*","*/"]]},css:{block:[["/*","*/"]]},scss:{line:["//"],block:[["/*","*/"]]},less:{line:["//"],block:[["/*","*/"]]},stylus:{line:["//"],block:[["/*","*/"]]},python:{line:["#"]},gherkin:{line:["#"]},coffeescript:{line:["#"],block:[["###","###"]]},livescript:{line:["#"],block:[["###","###"]]},lua:{line:["--"],block:[["--[[","]]"]]},markdown:{block:[["<!--","-->"]]}};function d(n,s,e){let i=n.slice(s,e).replace(/[^\n]/g," ");return n.slice(0,s)+i+n.slice(e)}function h(n,s){if(!s)return n;let e=n;for(let[i,c]of s.block??[]){let l=0;for(;;){let o=e.indexOf(i,l);if(o===-1)break;let t=e.indexOf(c,o+i.length),r=t===-1?e.length:t+c.length;e=d(e,o,r),l=r}}for(let i of s.line??[]){let c=0;for(;;){let l=e.indexOf(i,c);if(l===-1)break;let o=e.indexOf(`
|
|
2
|
+
`,l),t=o===-1?e.length:o;e=d(e,l,t),c=t}}return e}var x=["===","!==","...","=>","==","!=","<=",">=","&&","||","::","->","..","+=","-=","*=","/="],p=new RegExp(`${x.map(n=>n.replace(/[.*+?^${}()|[\]\\]/g,"\\$&")).join("|")}|[A-Za-z_$][A-Za-z0-9_$]*|\\d+(?:\\.\\d+)?|"(?:\\\\.|[^"\\\\])*"|'(?:\\\\.|[^'\\\\])*'|\`(?:\\\\.|[^\`\\\\])*\`|\\S`,"g"),b=/^[A-Za-z_$][A-Za-z0-9_$]*$/,m=/^\d+(?:\.\d+)?$/,y=/^(".*"|'.*'|`.*`)$/s,N=new Set(["if","else","elif","for","while","do","switch","case","default","break","continue","return","function","def","class","struct","interface","const","let","var","local","import","export","from","package","try","catch","except","finally","throw","raise","new","this","self","end","then","yield","async","await","nil","null","None","true","false","True","False","and","or","not","in","of","is"]);function k(n,s,e=!0){let i=h(n,u[s]),c=[],l=1,o=0;p.lastIndex=0;let t=p.exec(i);for(;t!==null;){for(let a=o;a<t.index;a++)i[a]===`
|
|
3
|
+
`&&l++;o=t.index;let r=t[0],f=r;b.test(r)&&!N.has(r)?f="ID":e&&(m.test(r)||y.test(r))&&(f=m.test(r)?"NUM":"STR"),c.push({text:f,line:l});for(let a=o;a<t.index+r.length;a++)i[a]===`
|
|
4
|
+
`&&l++;o=t.index+r.length,t=p.exec(i)}return c}function E(n){return k(n.source,n.fileType,n.ignoreLiterals)}export{E as default};
|
|
5
|
+
//# sourceMappingURL=duplication-worker.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/graph/duplication/tokenizer.ts","../src/duplication-worker.ts"],"sourcesContent":["/**\n * Language-agnostic source tokenizer for duplicate-code detection. Strips per-language\n * comment syntax, then splits what remains into a normalized token stream shared by every\n * language `DEFAULT_EXTENSIONS` covers — one generic tokenizer rather than a per-language\n * lexer, so `findDuplicates` works uniformly across TS/JS, Python, Go, CoffeeScript,\n * LiveScript, Lua, Gherkin, style files, and Markdown.\n */\nimport type { FileType } from \"../../types/parse\";\n\n/** One normalized token plus the 1-based source line it came from. */\nexport interface NormalizedToken {\n text: string;\n line: number;\n}\n\ninterface CommentSyntax {\n line?: string[];\n block?: Array<[string, string]>;\n}\n\n/**\n * Per-`FileType` comment markers used to mask out comment text before tokenizing, so comment\n * wording never contributes to a duplicate match. Deliberately not string-literal-aware — a\n * `//` inside a string is rare and the cost of occasionally over-stripping is low for a\n * heuristic duplicate finder, same trade-off the complexity scorers make elsewhere.\n */\nconst COMMENT_SYNTAX: Partial<Record<FileType, CommentSyntax>> = {\n typescript: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n javascript: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n go: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n css: { block: [[\"/*\", \"*/\"]] },\n scss: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n less: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n stylus: { line: [\"//\"], block: [[\"/*\", \"*/\"]] },\n python: { line: [\"#\"] },\n gherkin: { line: [\"#\"] },\n coffeescript: { line: [\"#\"], block: [[\"###\", \"###\"]] },\n livescript: { line: [\"#\"], block: [[\"###\", \"###\"]] },\n lua: { line: [\"--\"], block: [[\"--[[\", \"]]\"]] },\n markdown: { block: [[\"<!--\", \"-->\"]] },\n};\n\n/**\n * @description Replaces every non-newline character in `source[start, end)` with a space,\n * so downstream line-number tracking stays correct while the masked text can no longer\n * match anything.\n * @param source - Full file source.\n * @param start - Start offset (inclusive) of the range to mask.\n * @param end - End offset (exclusive) of the range to mask.\n * @returns `source` with the range masked.\n */\nfunction maskRange(source: string, start: number, end: number): string {\n const masked = source.slice(start, end).replace(/[^\\n]/g, \" \");\n return source.slice(0, start) + masked + source.slice(end);\n}\n\n/**\n * @description Masks out every line and block comment matching `syntax`, preserving line\n * breaks and overall string length so line numbers computed later stay accurate.\n * @param source - Full file source.\n * @param syntax - Comment markers for the file's language; absent for languages with no\n * configured comment syntax, in which case the source passes through unchanged.\n * @returns The source with comment text masked to spaces.\n */\nfunction stripComments(source: string, syntax: CommentSyntax | undefined): string {\n if (!syntax) return source;\n let result = source;\n\n for (const [open, close] of syntax.block ?? []) {\n let searchFrom = 0;\n for (;;) {\n const start = result.indexOf(open, searchFrom);\n if (start === -1) break;\n const end = result.indexOf(close, start + open.length);\n const rangeEnd = end === -1 ? result.length : end + close.length;\n result = maskRange(result, start, rangeEnd);\n searchFrom = rangeEnd;\n }\n }\n\n for (const marker of syntax.line ?? []) {\n let searchFrom = 0;\n for (;;) {\n const start = result.indexOf(marker, searchFrom);\n if (start === -1) break;\n const lineEnd = result.indexOf(\"\\n\", start);\n const rangeEnd = lineEnd === -1 ? result.length : lineEnd;\n result = maskRange(result, start, rangeEnd);\n searchFrom = rangeEnd;\n }\n }\n\n return result;\n}\n\nconst MULTI_CHAR_OPERATORS = [\n \"===\",\n \"!==\",\n \"...\",\n \"=>\",\n \"==\",\n \"!=\",\n \"<=\",\n \">=\",\n \"&&\",\n \"||\",\n \"::\",\n \"->\",\n \"..\",\n \"+=\",\n \"-=\",\n \"*=\",\n \"/=\",\n];\n\n/** Matches, in priority order: multi-char operators, identifiers, numbers, and quoted strings. */\nconst TOKEN_PATTERN = new RegExp(\n `${MULTI_CHAR_OPERATORS.map((op) => op.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\")).join(\"|\")}|[A-Za-z_$][A-Za-z0-9_$]*|\\\\d+(?:\\\\.\\\\d+)?|\"(?:\\\\\\\\.|[^\"\\\\\\\\])*\"|'(?:\\\\\\\\.|[^'\\\\\\\\])*'|\\`(?:\\\\\\\\.|[^\\`\\\\\\\\])*\\`|\\\\S`,\n \"g\",\n);\n\nconst IDENTIFIER_PATTERN = /^[A-Za-z_$][A-Za-z0-9_$]*$/;\nconst NUMBER_PATTERN = /^\\d+(?:\\.\\d+)?$/;\nconst STRING_PATTERN = /^(\".*\"|'.*'|`.*`)$/s;\n\n/**\n * Keywords kept verbatim rather than collapsed to the `ID` placeholder, shared across every\n * language rather than a per-language keyword table — deliberately coarse. Without this,\n * structurally different code (an `if` vs. a `for`, a `class` vs. a `function`) would tokenize\n * identically once every identifier-shaped word became `ID`, which would make the shingle\n * matcher far too eager. This list only needs to cover the keywords common enough across\n * TS/JS/Python/Go/CoffeeScript/LiveScript/Lua/Gherkin to matter for shape-preservation — it\n * doesn't need to be exhaustive or language-precise, since keeping a non-keyword here just\n * costs a little precision, not correctness.\n */\nconst KEYWORDS = new Set([\n \"if\",\n \"else\",\n \"elif\",\n \"for\",\n \"while\",\n \"do\",\n \"switch\",\n \"case\",\n \"default\",\n \"break\",\n \"continue\",\n \"return\",\n \"function\",\n \"def\",\n \"class\",\n \"struct\",\n \"interface\",\n \"const\",\n \"let\",\n \"var\",\n \"local\",\n \"import\",\n \"export\",\n \"from\",\n \"package\",\n \"try\",\n \"catch\",\n \"except\",\n \"finally\",\n \"throw\",\n \"raise\",\n \"new\",\n \"this\",\n \"self\",\n \"end\",\n \"then\",\n \"yield\",\n \"async\",\n \"await\",\n \"nil\",\n \"null\",\n \"None\",\n \"true\",\n \"false\",\n \"True\",\n \"False\",\n \"and\",\n \"or\",\n \"not\",\n \"in\",\n \"of\",\n \"is\",\n]);\n\n/**\n * @description Tokenizes source text into a normalized stream for shingle-based duplicate\n * matching: identifiers always collapse to a single placeholder (so renamed-variable clones\n * still hash identically), and literals optionally collapse too (`ignoreLiterals`, default\n * on) so only structural shape — not the specific values used — drives the match.\n * @param source - Raw file content.\n * @param fileType - Selects the comment-stripping rule; unrecognised types skip stripping.\n * @param ignoreLiterals - When true (default), string/number literal tokens are also\n * normalized to a placeholder rather than compared verbatim.\n * @returns Normalized tokens in source order, each carrying its 1-based source line.\n */\nexport function tokenize(\n source: string,\n fileType: FileType,\n ignoreLiterals = true,\n): NormalizedToken[] {\n const stripped = stripComments(source, COMMENT_SYNTAX[fileType]);\n const tokens: NormalizedToken[] = [];\n\n let line = 1;\n let lastIndex = 0;\n TOKEN_PATTERN.lastIndex = 0;\n let match = TOKEN_PATTERN.exec(stripped);\n\n while (match !== null) {\n for (let i = lastIndex; i < match.index; i++) {\n if (stripped[i] === \"\\n\") line++;\n }\n lastIndex = match.index;\n\n const raw = match[0];\n let text = raw;\n if (IDENTIFIER_PATTERN.test(raw) && !KEYWORDS.has(raw)) {\n text = \"ID\";\n } else if (ignoreLiterals && (NUMBER_PATTERN.test(raw) || STRING_PATTERN.test(raw))) {\n text = NUMBER_PATTERN.test(raw) ? \"NUM\" : \"STR\";\n }\n tokens.push({ text, line });\n\n for (let i = lastIndex; i < match.index + raw.length; i++) {\n if (stripped[i] === \"\\n\") line++;\n }\n lastIndex = match.index + raw.length;\n match = TOKEN_PATTERN.exec(stripped);\n }\n\n return tokens;\n}\n","/** Piscina task handler: tokenizes a single file's content in a worker thread, for `findDuplicates`. */\n\nimport type { NormalizedToken } from \"./graph/duplication/tokenizer.js\";\nimport { tokenize } from \"./graph/duplication/tokenizer.js\";\nimport type { FileType } from \"./types/parse\";\n\nexport default function tokenizeInWorker(payload: {\n source: string;\n fileType: FileType;\n ignoreLiterals: boolean;\n}): NormalizedToken[] {\n return tokenize(payload.source, payload.fileType, payload.ignoreLiterals);\n}\n"],"mappings":"AA0BA,IAAMA,EAA2D,CAC/D,WAAY,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAClD,WAAY,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAClD,GAAI,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC1C,IAAK,CAAE,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC7B,KAAM,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC5C,KAAM,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC5C,OAAQ,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,KAAM,IAAI,CAAC,CAAE,EAC9C,OAAQ,CAAE,KAAM,CAAC,GAAG,CAAE,EACtB,QAAS,CAAE,KAAM,CAAC,GAAG,CAAE,EACvB,aAAc,CAAE,KAAM,CAAC,GAAG,EAAG,MAAO,CAAC,CAAC,MAAO,KAAK,CAAC,CAAE,EACrD,WAAY,CAAE,KAAM,CAAC,GAAG,EAAG,MAAO,CAAC,CAAC,MAAO,KAAK,CAAC,CAAE,EACnD,IAAK,CAAE,KAAM,CAAC,IAAI,EAAG,MAAO,CAAC,CAAC,OAAQ,IAAI,CAAC,CAAE,EAC7C,SAAU,CAAE,MAAO,CAAC,CAAC,OAAQ,KAAK,CAAC,CAAE,CACvC,EAWA,SAASC,EAAUC,EAAgBC,EAAeC,EAAqB,CACrE,IAAMC,EAASH,EAAO,MAAMC,EAAOC,CAAG,EAAE,QAAQ,SAAU,GAAG,EAC7D,OAAOF,EAAO,MAAM,EAAGC,CAAK,EAAIE,EAASH,EAAO,MAAME,CAAG,CAC3D,CAUA,SAASE,EAAcJ,EAAgBK,EAA2C,CAChF,GAAI,CAACA,EAAQ,OAAOL,EACpB,IAAIM,EAASN,EAEb,OAAW,CAACO,EAAMC,CAAK,IAAKH,EAAO,OAAS,CAAC,EAAG,CAC9C,IAAII,EAAa,EACjB,OAAS,CACP,IAAMR,EAAQK,EAAO,QAAQC,EAAME,CAAU,EAC7C,GAAIR,IAAU,GAAI,MAClB,IAAMC,EAAMI,EAAO,QAAQE,EAAOP,EAAQM,EAAK,MAAM,EAC/CG,EAAWR,IAAQ,GAAKI,EAAO,OAASJ,EAAMM,EAAM,OAC1DF,EAASP,EAAUO,EAAQL,EAAOS,CAAQ,EAC1CD,EAAaC,CACf,CACF,CAEA,QAAWC,KAAUN,EAAO,MAAQ,CAAC,EAAG,CACtC,IAAII,EAAa,EACjB,OAAS,CACP,IAAMR,EAAQK,EAAO,QAAQK,EAAQF,CAAU,EAC/C,GAAIR,IAAU,GAAI,MAClB,IAAMW,EAAUN,EAAO,QAAQ;AAAA,EAAML,CAAK,EACpCS,EAAWE,IAAY,GAAKN,EAAO,OAASM,EAClDN,EAASP,EAAUO,EAAQL,EAAOS,CAAQ,EAC1CD,EAAaC,CACf,CACF,CAEA,OAAOJ,CACT,CAEA,IAAMO,EAAuB,CAC3B,MACA,MACA,MACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,KACA,IACF,EAGMC,EAAgB,IAAI,OACxB,GAAGD,EAAqB,IAAKE,GAAOA,EAAG,QAAQ,sBAAuB,MAAM,CAAC,EAAE,KAAK,GAAG,CAAC,sHACxF,GACF,EAEMC,EAAqB,6BACrBC,EAAiB,kBACjBC,EAAiB,sBAYjBC,EAAW,IAAI,IAAI,CACvB,KACA,OACA,OACA,MACA,QACA,KACA,SACA,OACA,UACA,QACA,WACA,SACA,WACA,MACA,QACA,SACA,YACA,QACA,MACA,MACA,QACA,SACA,SACA,OACA,UACA,MACA,QACA,SACA,UACA,QACA,QACA,MACA,OACA,OACA,MACA,OACA,QACA,QACA,QACA,MACA,OACA,OACA,OACA,QACA,OACA,QACA,MACA,KACA,MACA,KACA,KACA,IACF,CAAC,EAaM,SAASC,EACdpB,EACAqB,EACAC,EAAiB,GACE,CACnB,IAAMC,EAAWnB,EAAcJ,EAAQF,EAAeuB,CAAQ,CAAC,EACzDG,EAA4B,CAAC,EAE/BC,EAAO,EACPC,EAAY,EAChBZ,EAAc,UAAY,EAC1B,IAAIa,EAAQb,EAAc,KAAKS,CAAQ,EAEvC,KAAOI,IAAU,MAAM,CACrB,QAASC,EAAIF,EAAWE,EAAID,EAAM,MAAOC,IACnCL,EAASK,CAAC,IAAM;AAAA,GAAMH,IAE5BC,EAAYC,EAAM,MAElB,IAAME,EAAMF,EAAM,CAAC,EACfG,EAAOD,EACPb,EAAmB,KAAKa,CAAG,GAAK,CAACV,EAAS,IAAIU,CAAG,EACnDC,EAAO,KACER,IAAmBL,EAAe,KAAKY,CAAG,GAAKX,EAAe,KAAKW,CAAG,KAC/EC,EAAOb,EAAe,KAAKY,CAAG,EAAI,MAAQ,OAE5CL,EAAO,KAAK,CAAE,KAAAM,EAAM,KAAAL,CAAK,CAAC,EAE1B,QAASG,EAAIF,EAAWE,EAAID,EAAM,MAAQE,EAAI,OAAQD,IAChDL,EAASK,CAAC,IAAM;AAAA,GAAMH,IAE5BC,EAAYC,EAAM,MAAQE,EAAI,OAC9BF,EAAQb,EAAc,KAAKS,CAAQ,CACrC,CAEA,OAAOC,CACT,CCvOe,SAARO,EAAkCC,EAInB,CACpB,OAAOC,EAASD,EAAQ,OAAQA,EAAQ,SAAUA,EAAQ,cAAc,CAC1E","names":["COMMENT_SYNTAX","maskRange","source","start","end","masked","stripComments","syntax","result","open","close","searchFrom","rangeEnd","marker","lineEnd","MULTI_CHAR_OPERATORS","TOKEN_PATTERN","op","IDENTIFIER_PATTERN","NUMBER_PATTERN","STRING_PATTERN","KEYWORDS","tokenize","fileType","ignoreLiterals","stripped","tokens","line","lastIndex","match","i","raw","text","tokenizeInWorker","payload","tokenize"]}
|
package/dist/index.d.mts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { F as FileNode, C as CallEdge, I as ImportEdge,
|
|
2
|
-
export { E as ExportedSymbol
|
|
1
|
+
import { F as FileNode, C as CallEdge, I as ImportEdge, P as ParseResult, S as StructuredTag } from './types-BtSqoqbZ.mjs';
|
|
2
|
+
export { E as ExportedSymbol } from './types-BtSqoqbZ.mjs';
|
|
3
|
+
import { F as FileType, N as NodeCategory } from './parse-CAKctgb6.mjs';
|
|
4
|
+
export { I as ImportType, T as TagKind } from './parse-CAKctgb6.mjs';
|
|
3
5
|
|
|
4
6
|
interface SerializedGraph {
|
|
5
7
|
nodes: FileNode[];
|
|
@@ -442,10 +444,30 @@ declare function saveChangeImpactCache(cache: ChangeImpactCache, cachePath: stri
|
|
|
442
444
|
*/
|
|
443
445
|
declare function loadChangeImpactCache(cachePath: string): ChangeImpactCache | null;
|
|
444
446
|
|
|
447
|
+
/**
|
|
448
|
+
* Language-family partitioning for duplicate detection. `findDuplicates` never compares files
|
|
449
|
+
* across a family boundary — see docs/adr-013-duplicate-detection-noise-reduction.md. The
|
|
450
|
+
* immediate driver is CSS-family declarations: they share a small, finite vocabulary (property
|
|
451
|
+
* names, common value keywords) that the shared `KEYWORDS` denylist in `tokenizer.ts` doesn't
|
|
452
|
+
* cover, so unrelated style rules with the same declaration *shape* but different selectors/
|
|
453
|
+
* properties/values were hashing identically to unrelated TS/JS/Python shapes too. Splitting by
|
|
454
|
+
* family also lets later phases tune `windowSize`/`minLines`/vocabulary per family without one
|
|
455
|
+
* family's tuning affecting another's.
|
|
456
|
+
*/
|
|
457
|
+
|
|
458
|
+
/** One partition of `findDuplicates` matching — files in different families are never compared. */
|
|
459
|
+
type DuplicateFamily = "style" | "code";
|
|
460
|
+
|
|
445
461
|
/**
|
|
446
462
|
* Sliding-window (shingle) hashing over a normalized token stream, plus chain-merging of
|
|
447
|
-
* consecutive matching windows into contiguous duplicate blocks
|
|
448
|
-
*
|
|
463
|
+
* consecutive matching windows into contiguous duplicate blocks, a structural-punctuation-density
|
|
464
|
+
* gate that drops blocks that are mostly object/array-literal shape (e.g. MCP tool `inputSchema`
|
|
465
|
+
* boilerplate) rather than substantive shared logic, and exact-occurrence clustering of pair-
|
|
466
|
+
* matches into one N-occurrence group instead of reporting C(N,2) near-identical pairs for a block
|
|
467
|
+
* repeated N times. Shared by every language `tokenize()` supports — the shingling step itself has
|
|
468
|
+
* no language awareness at all. See docs/adr-013-duplicate-detection-noise-reduction.md for the
|
|
469
|
+
* noise this addresses, and docs/adr-014-duplicate-detection-scale.md for why oversized hash
|
|
470
|
+
* buckets are capped below.
|
|
449
471
|
*/
|
|
450
472
|
|
|
451
473
|
interface DuplicateOccurrence {
|
|
@@ -454,14 +476,29 @@ interface DuplicateOccurrence {
|
|
|
454
476
|
endLine: number;
|
|
455
477
|
}
|
|
456
478
|
interface DuplicateGroup {
|
|
457
|
-
/**
|
|
458
|
-
|
|
459
|
-
|
|
479
|
+
/** Every location sharing this duplicated block (two or more) — locations that pairwise
|
|
480
|
+
* chain-match are clustered into one group instead of being reported once per pair, so a
|
|
481
|
+
* block repeated N times produces one N-occurrence group, not C(N,2) near-identical ones. */
|
|
482
|
+
occurrences: DuplicateOccurrence[];
|
|
483
|
+
/** Line span of the largest single pairwise match clustered into this group (each pair's own
|
|
484
|
+
* span is the shorter of its two occurrences) — the best-verified size for this block, not an
|
|
485
|
+
* average or a value shrunk by a more weakly-matching cluster member. */
|
|
460
486
|
lines: number;
|
|
461
487
|
/** Token-window length backing this block, after chain-merging adjacent windows. */
|
|
462
488
|
tokens: number;
|
|
489
|
+
/** Which language family both occurrences belong to (set by `findDuplicates`, which never
|
|
490
|
+
* matches across families — see docs/adr-013-duplicate-detection-noise-reduction.md).
|
|
491
|
+
* Absent when called directly with a token stream that isn't family-scoped, e.g. in tests. */
|
|
492
|
+
family?: DuplicateFamily | undefined;
|
|
463
493
|
}
|
|
464
494
|
|
|
495
|
+
/** Configures whether/how tokenizing is offloaded to a `piscina` worker pool. `false` always
|
|
496
|
+
* tokenizes in-process. */
|
|
497
|
+
type ParallelTokenizingOption = boolean | {
|
|
498
|
+
minFiles?: number;
|
|
499
|
+
maxThreads?: number;
|
|
500
|
+
};
|
|
501
|
+
|
|
465
502
|
interface FindDuplicatesOptions {
|
|
466
503
|
/** Minimum duplicated block size, in source lines, to report (default 6). */
|
|
467
504
|
minLines?: number | undefined;
|
|
@@ -470,27 +507,67 @@ interface FindDuplicatesOptions {
|
|
|
470
507
|
/** When true (default), string/number literals are normalized too, so only structural shape
|
|
471
508
|
* — not the specific values used — drives a match. Set false for stricter, Type-1-only matching. */
|
|
472
509
|
ignoreLiterals?: boolean | undefined;
|
|
510
|
+
/** Maximum fraction of a token-shingled block's window that may be object/array-literal
|
|
511
|
+
* structural punctuation (`{ } : , [ ]`) (default 0.5) — gates out blocks that are mostly
|
|
512
|
+
* schema/object-literal shape (e.g. MCP tool `inputSchema` boilerplate repeated across
|
|
513
|
+
* unrelated tool definitions) rather than substantive shared logic. Does not apply to the
|
|
514
|
+
* CSS/Less/SCSS structural comparator, which already matches on literal declaration content.
|
|
515
|
+
* Set to 1 to disable. See docs/adr-013-duplicate-detection-noise-reduction.md. */
|
|
516
|
+
maxPunctuationRatio?: number | undefined;
|
|
517
|
+
/** Skip a hash bucket's O(k²) pairwise comparison once it holds more than this many locations
|
|
518
|
+
* (default 400) — bounds worst-case scan time on large repos where a single ubiquitous token
|
|
519
|
+
* window (a common import line, a boilerplate header) would otherwise blow past what a single
|
|
520
|
+
* scan can finish in. Set `Infinity` to disable. See docs/adr-014-duplicate-detection-scale.md. */
|
|
521
|
+
maxBucketSize?: number | undefined;
|
|
473
522
|
/** Caps the number of duplicate blocks returned, largest-first (default 50). */
|
|
474
523
|
limit?: number | undefined;
|
|
475
524
|
/** Directory names to exclude, matched against any path segment (default `DEFAULT_IGNORE_DIRS`
|
|
476
525
|
* — `node_modules`, `dist`, `.git`, `mokosh-cache`, `coverage`, etc.). Pass `[]` to disable. */
|
|
477
526
|
ignoreDirs?: readonly string[] | undefined;
|
|
527
|
+
/** Controls worker-pool offloading of per-file tokenizing (default `true`): offloads once the
|
|
528
|
+
* candidate file count reaches `minFiles` (default 20, matching `GraphBuilder`'s parse pool);
|
|
529
|
+
* `false` always tokenizes in-process; an object overrides `minFiles`/`maxThreads`. See
|
|
530
|
+
* docs/adr-014-duplicate-detection-scale.md. */
|
|
531
|
+
parallelTokenizing?: ParallelTokenizingOption | undefined;
|
|
532
|
+
}
|
|
533
|
+
interface FindDuplicatesResult {
|
|
534
|
+
/** Duplicate blocks, largest-first, capped at `limit`. */
|
|
535
|
+
groups: DuplicateGroup[];
|
|
536
|
+
/** How many hash buckets were skipped for exceeding `maxBucketSize` — a non-zero count means
|
|
537
|
+
* results may under-report duplication that's unusually widespread (see `maxBucketSize`). */
|
|
538
|
+
skippedBuckets: number;
|
|
478
539
|
}
|
|
479
540
|
/**
|
|
480
541
|
* @description Scans every file already present in `graph` for cross-file (and within-file)
|
|
481
|
-
* duplicated code
|
|
482
|
-
*
|
|
483
|
-
*
|
|
484
|
-
*
|
|
542
|
+
* duplicated code. CSS/Less/SCSS files are compared structurally — by their rule bodies'
|
|
543
|
+
* literal, ordered `property: value` declarations, independent of selector name — via
|
|
544
|
+
* {@link findStyleBlockDuplicates}. Every other language (TS/JS, Python, Go, CoffeeScript,
|
|
545
|
+
* LiveScript, Lua, Gherkin, Markdown, and Stylus, which has no shared PostCSS AST here) runs
|
|
546
|
+
* the generic token-shingling pipeline instead: comments are stripped per `FileType`,
|
|
547
|
+
* identifiers (and, by default, literals) are normalized to placeholders so renamed-variable
|
|
548
|
+
* copies still match, then a sliding token window is hashed and chain-merged into contiguous
|
|
549
|
+
* blocks. Token-shingled files are additionally partitioned into language families
|
|
550
|
+
* ({@link getDuplicateFamily} — `"style"` for Stylus, `"code"` for everything else) so
|
|
551
|
+
* matching never crosses that boundary — see docs/adr-013-duplicate-detection-noise-reduction.md.
|
|
485
552
|
* @param graph - The graph to scan; its node paths (already ignore-rule-filtered) are the file
|
|
486
553
|
* list, re-read from disk since duplication data isn't cached on `FileNode`.
|
|
487
554
|
* @param rootDir - Absolute project root that graph paths are relative to.
|
|
488
|
-
* @param options - `minLines`/`windowSize` tune sensitivity
|
|
489
|
-
*
|
|
490
|
-
*
|
|
491
|
-
*
|
|
492
|
-
|
|
493
|
-
|
|
555
|
+
* @param options - `minLines`/`windowSize` tune token-shingle sensitivity (`minLines` also caps
|
|
556
|
+
* CSS/Less/SCSS block size); `ignoreLiterals` toggles Type-2 vs Type-1 matching for the
|
|
557
|
+
* token-shingle path only (CSS/Less/SCSS always match on literal declaration content);
|
|
558
|
+
* `maxPunctuationRatio` gates out token-shingle blocks that are mostly object/array-literal
|
|
559
|
+
* structural punctuation (e.g. schema/object-literal boilerplate) rather than substantive
|
|
560
|
+
* shared logic; `maxBucketSize` bounds worst-case scan time on large repos by skipping
|
|
561
|
+
* pathologically common hash buckets; `ignoreDirs` excludes files under matching directory
|
|
562
|
+
* names; `limit` caps results; `parallelTokenizing` offloads per-file tokenizing to a worker
|
|
563
|
+
* pool once the candidate file count is large enough to be worth it. Lock files are always
|
|
564
|
+
* excluded, independent of `ignoreDirs`.
|
|
565
|
+
* @returns `groups` — duplicate blocks (each tagged with its `family`), two or more occurrences
|
|
566
|
+
* per block, every block that pairwise chain-matches another clustered into one group instead
|
|
567
|
+
* of one per pair, sorted largest-first across all families — plus `skippedBuckets` (see
|
|
568
|
+
* {@link FindDuplicatesResult}).
|
|
569
|
+
*/
|
|
570
|
+
declare function findDuplicates(graph: Graph, rootDir: string, options?: FindDuplicatesOptions): Promise<FindDuplicatesResult>;
|
|
494
571
|
|
|
495
572
|
/** Controls how aggressively `detectFeatures` promotes files to features. */
|
|
496
573
|
interface FeatureDetectionOptions {
|
|
@@ -1353,4 +1430,4 @@ declare function createWorkspaceGraph(rootDir: string, options?: {
|
|
|
1353
1430
|
*/
|
|
1354
1431
|
declare function getAllProjectFiles(rootDir: string, options?: ScanOptions): string[];
|
|
1355
1432
|
|
|
1356
|
-
export { type ApiSurface, type ApplyTagsFileResult, type ApplyTagsResult, CALL_EDGE_TYPES, CallEdge, type CalleeEntry, type CallerEntry$1 as CallerEntry, type ChangeImpactCache, type ComplexFunctionEntry, DEFAULT_EXTENSIONS, DEFAULT_IGNORE_DIRS, type DependencyGraph, type DuplicateGroup, type DuplicateOccurrence, EXPORT_TRACKING_TYPES, type ExportKind, type FeatureDetectionOptions, type FeatureDomain, type FeatureGraph, type FeatureGraphOptions, type FeatureInfo, FileNode, FileType, type FindComplexFunctionsOptions, type FindDuplicatesOptions, type FunctionCallInfo, type GetAffectedOptions, type GetCallersOptions, Graph, type CallerEntry as GraphCallerEntry, type GraphExporter, IMPORT_SYMBOL_TYPES, ImportEdge, type LanguageCoverage, MermaidExporter, type ModuleResponsibility, type ModuleRole, type MokoshConfig, type MonorepoDetector, type MonorepoLayout, NodeCategory, type NodeMeta, type NodeQuery, type ParallelParsingOption, type PathWithSymbols, type ProposeTagsOptions, type PublicExport, type ResponsibilityGraph, type ScanOptions, type SerializedGraph, type SerializedWorkspaceGraph, type SlimNode, type SlimSerializedGraph, StructuredTag, type SymbolCaller, type SymbolImporter, type SymbolMatch, type SymbolPrecision, SymbolTraversalContext, type TestNodeIdentifier, type TraversalOptions, type TraversalVisitor, type TypeEdge, type TypeGraph, type TypeKind, type TypeNode, type TypeQueryResult, WorkspaceGraph, type WorkspacePackage, type WorkspacePackageSummary, type WorkspacePackagesSummary, applyConfig, applyTags, buildApiSurface, buildChangeImpactCache, buildFeatureGraph, buildResponsibilityGraph, buildTypeGraph, computeGraphHash, configToGraphOptions, createImportMap, createWorkspaceGraph, detectAllEntryPoints, detectEntryPoint, detectFeatures, detectMonorepo, filterGraph, findComplexFunctions, findDuplicates, findSymbol, getAffected, getAllProjectFiles, getCallers, getDependencies, getDependents, getLanguageCoverage, getNodeMeta, hasCoverageData, isChangeImpactCacheValid, loadChangeImpactCache, loadCoverageMap, loadMokoshConfig, parseQuery, proposeAffectedTests, proposeTags, queryCallGraph, queryChangeImpact, queryTypeGraph, registerConfigMatcher, registerMonorepoDetector, registerParser, registerTestLibrary, registerTestPattern, saveChangeImpactCache, slimSerialize, summarizeWorkspacePackages, toMermaid };
|
|
1433
|
+
export { type ApiSurface, type ApplyTagsFileResult, type ApplyTagsResult, CALL_EDGE_TYPES, CallEdge, type CalleeEntry, type CallerEntry$1 as CallerEntry, type ChangeImpactCache, type ComplexFunctionEntry, DEFAULT_EXTENSIONS, DEFAULT_IGNORE_DIRS, type DependencyGraph, type DuplicateFamily, type DuplicateGroup, type DuplicateOccurrence, EXPORT_TRACKING_TYPES, type ExportKind, type FeatureDetectionOptions, type FeatureDomain, type FeatureGraph, type FeatureGraphOptions, type FeatureInfo, FileNode, FileType, type FindComplexFunctionsOptions, type FindDuplicatesOptions, type FunctionCallInfo, type GetAffectedOptions, type GetCallersOptions, Graph, type CallerEntry as GraphCallerEntry, type GraphExporter, IMPORT_SYMBOL_TYPES, ImportEdge, type LanguageCoverage, MermaidExporter, type ModuleResponsibility, type ModuleRole, type MokoshConfig, type MonorepoDetector, type MonorepoLayout, NodeCategory, type NodeMeta, type NodeQuery, type ParallelParsingOption, type PathWithSymbols, type ProposeTagsOptions, type PublicExport, type ResponsibilityGraph, type ScanOptions, type SerializedGraph, type SerializedWorkspaceGraph, type SlimNode, type SlimSerializedGraph, StructuredTag, type SymbolCaller, type SymbolImporter, type SymbolMatch, type SymbolPrecision, SymbolTraversalContext, type TestNodeIdentifier, type TraversalOptions, type TraversalVisitor, type TypeEdge, type TypeGraph, type TypeKind, type TypeNode, type TypeQueryResult, WorkspaceGraph, type WorkspacePackage, type WorkspacePackageSummary, type WorkspacePackagesSummary, applyConfig, applyTags, buildApiSurface, buildChangeImpactCache, buildFeatureGraph, buildResponsibilityGraph, buildTypeGraph, computeGraphHash, configToGraphOptions, createImportMap, createWorkspaceGraph, detectAllEntryPoints, detectEntryPoint, detectFeatures, detectMonorepo, filterGraph, findComplexFunctions, findDuplicates, findSymbol, getAffected, getAllProjectFiles, getCallers, getDependencies, getDependents, getLanguageCoverage, getNodeMeta, hasCoverageData, isChangeImpactCacheValid, loadChangeImpactCache, loadCoverageMap, loadMokoshConfig, parseQuery, proposeAffectedTests, proposeTags, queryCallGraph, queryChangeImpact, queryTypeGraph, registerConfigMatcher, registerMonorepoDetector, registerParser, registerTestLibrary, registerTestPattern, saveChangeImpactCache, slimSerialize, summarizeWorkspacePackages, toMermaid };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { F as FileNode, C as CallEdge, I as ImportEdge,
|
|
2
|
-
export { E as ExportedSymbol
|
|
1
|
+
import { F as FileNode, C as CallEdge, I as ImportEdge, P as ParseResult, S as StructuredTag } from './types-BlN-U5AM.js';
|
|
2
|
+
export { E as ExportedSymbol } from './types-BlN-U5AM.js';
|
|
3
|
+
import { F as FileType, N as NodeCategory } from './parse-CAKctgb6.js';
|
|
4
|
+
export { I as ImportType, T as TagKind } from './parse-CAKctgb6.js';
|
|
3
5
|
|
|
4
6
|
interface SerializedGraph {
|
|
5
7
|
nodes: FileNode[];
|
|
@@ -442,10 +444,30 @@ declare function saveChangeImpactCache(cache: ChangeImpactCache, cachePath: stri
|
|
|
442
444
|
*/
|
|
443
445
|
declare function loadChangeImpactCache(cachePath: string): ChangeImpactCache | null;
|
|
444
446
|
|
|
447
|
+
/**
|
|
448
|
+
* Language-family partitioning for duplicate detection. `findDuplicates` never compares files
|
|
449
|
+
* across a family boundary — see docs/adr-013-duplicate-detection-noise-reduction.md. The
|
|
450
|
+
* immediate driver is CSS-family declarations: they share a small, finite vocabulary (property
|
|
451
|
+
* names, common value keywords) that the shared `KEYWORDS` denylist in `tokenizer.ts` doesn't
|
|
452
|
+
* cover, so unrelated style rules with the same declaration *shape* but different selectors/
|
|
453
|
+
* properties/values were hashing identically to unrelated TS/JS/Python shapes too. Splitting by
|
|
454
|
+
* family also lets later phases tune `windowSize`/`minLines`/vocabulary per family without one
|
|
455
|
+
* family's tuning affecting another's.
|
|
456
|
+
*/
|
|
457
|
+
|
|
458
|
+
/** One partition of `findDuplicates` matching — files in different families are never compared. */
|
|
459
|
+
type DuplicateFamily = "style" | "code";
|
|
460
|
+
|
|
445
461
|
/**
|
|
446
462
|
* Sliding-window (shingle) hashing over a normalized token stream, plus chain-merging of
|
|
447
|
-
* consecutive matching windows into contiguous duplicate blocks
|
|
448
|
-
*
|
|
463
|
+
* consecutive matching windows into contiguous duplicate blocks, a structural-punctuation-density
|
|
464
|
+
* gate that drops blocks that are mostly object/array-literal shape (e.g. MCP tool `inputSchema`
|
|
465
|
+
* boilerplate) rather than substantive shared logic, and exact-occurrence clustering of pair-
|
|
466
|
+
* matches into one N-occurrence group instead of reporting C(N,2) near-identical pairs for a block
|
|
467
|
+
* repeated N times. Shared by every language `tokenize()` supports — the shingling step itself has
|
|
468
|
+
* no language awareness at all. See docs/adr-013-duplicate-detection-noise-reduction.md for the
|
|
469
|
+
* noise this addresses, and docs/adr-014-duplicate-detection-scale.md for why oversized hash
|
|
470
|
+
* buckets are capped below.
|
|
449
471
|
*/
|
|
450
472
|
|
|
451
473
|
interface DuplicateOccurrence {
|
|
@@ -454,14 +476,29 @@ interface DuplicateOccurrence {
|
|
|
454
476
|
endLine: number;
|
|
455
477
|
}
|
|
456
478
|
interface DuplicateGroup {
|
|
457
|
-
/**
|
|
458
|
-
|
|
459
|
-
|
|
479
|
+
/** Every location sharing this duplicated block (two or more) — locations that pairwise
|
|
480
|
+
* chain-match are clustered into one group instead of being reported once per pair, so a
|
|
481
|
+
* block repeated N times produces one N-occurrence group, not C(N,2) near-identical ones. */
|
|
482
|
+
occurrences: DuplicateOccurrence[];
|
|
483
|
+
/** Line span of the largest single pairwise match clustered into this group (each pair's own
|
|
484
|
+
* span is the shorter of its two occurrences) — the best-verified size for this block, not an
|
|
485
|
+
* average or a value shrunk by a more weakly-matching cluster member. */
|
|
460
486
|
lines: number;
|
|
461
487
|
/** Token-window length backing this block, after chain-merging adjacent windows. */
|
|
462
488
|
tokens: number;
|
|
489
|
+
/** Which language family both occurrences belong to (set by `findDuplicates`, which never
|
|
490
|
+
* matches across families — see docs/adr-013-duplicate-detection-noise-reduction.md).
|
|
491
|
+
* Absent when called directly with a token stream that isn't family-scoped, e.g. in tests. */
|
|
492
|
+
family?: DuplicateFamily | undefined;
|
|
463
493
|
}
|
|
464
494
|
|
|
495
|
+
/** Configures whether/how tokenizing is offloaded to a `piscina` worker pool. `false` always
|
|
496
|
+
* tokenizes in-process. */
|
|
497
|
+
type ParallelTokenizingOption = boolean | {
|
|
498
|
+
minFiles?: number;
|
|
499
|
+
maxThreads?: number;
|
|
500
|
+
};
|
|
501
|
+
|
|
465
502
|
interface FindDuplicatesOptions {
|
|
466
503
|
/** Minimum duplicated block size, in source lines, to report (default 6). */
|
|
467
504
|
minLines?: number | undefined;
|
|
@@ -470,27 +507,67 @@ interface FindDuplicatesOptions {
|
|
|
470
507
|
/** When true (default), string/number literals are normalized too, so only structural shape
|
|
471
508
|
* — not the specific values used — drives a match. Set false for stricter, Type-1-only matching. */
|
|
472
509
|
ignoreLiterals?: boolean | undefined;
|
|
510
|
+
/** Maximum fraction of a token-shingled block's window that may be object/array-literal
|
|
511
|
+
* structural punctuation (`{ } : , [ ]`) (default 0.5) — gates out blocks that are mostly
|
|
512
|
+
* schema/object-literal shape (e.g. MCP tool `inputSchema` boilerplate repeated across
|
|
513
|
+
* unrelated tool definitions) rather than substantive shared logic. Does not apply to the
|
|
514
|
+
* CSS/Less/SCSS structural comparator, which already matches on literal declaration content.
|
|
515
|
+
* Set to 1 to disable. See docs/adr-013-duplicate-detection-noise-reduction.md. */
|
|
516
|
+
maxPunctuationRatio?: number | undefined;
|
|
517
|
+
/** Skip a hash bucket's O(k²) pairwise comparison once it holds more than this many locations
|
|
518
|
+
* (default 400) — bounds worst-case scan time on large repos where a single ubiquitous token
|
|
519
|
+
* window (a common import line, a boilerplate header) would otherwise blow past what a single
|
|
520
|
+
* scan can finish in. Set `Infinity` to disable. See docs/adr-014-duplicate-detection-scale.md. */
|
|
521
|
+
maxBucketSize?: number | undefined;
|
|
473
522
|
/** Caps the number of duplicate blocks returned, largest-first (default 50). */
|
|
474
523
|
limit?: number | undefined;
|
|
475
524
|
/** Directory names to exclude, matched against any path segment (default `DEFAULT_IGNORE_DIRS`
|
|
476
525
|
* — `node_modules`, `dist`, `.git`, `mokosh-cache`, `coverage`, etc.). Pass `[]` to disable. */
|
|
477
526
|
ignoreDirs?: readonly string[] | undefined;
|
|
527
|
+
/** Controls worker-pool offloading of per-file tokenizing (default `true`): offloads once the
|
|
528
|
+
* candidate file count reaches `minFiles` (default 20, matching `GraphBuilder`'s parse pool);
|
|
529
|
+
* `false` always tokenizes in-process; an object overrides `minFiles`/`maxThreads`. See
|
|
530
|
+
* docs/adr-014-duplicate-detection-scale.md. */
|
|
531
|
+
parallelTokenizing?: ParallelTokenizingOption | undefined;
|
|
532
|
+
}
|
|
533
|
+
interface FindDuplicatesResult {
|
|
534
|
+
/** Duplicate blocks, largest-first, capped at `limit`. */
|
|
535
|
+
groups: DuplicateGroup[];
|
|
536
|
+
/** How many hash buckets were skipped for exceeding `maxBucketSize` — a non-zero count means
|
|
537
|
+
* results may under-report duplication that's unusually widespread (see `maxBucketSize`). */
|
|
538
|
+
skippedBuckets: number;
|
|
478
539
|
}
|
|
479
540
|
/**
|
|
480
541
|
* @description Scans every file already present in `graph` for cross-file (and within-file)
|
|
481
|
-
* duplicated code
|
|
482
|
-
*
|
|
483
|
-
*
|
|
484
|
-
*
|
|
542
|
+
* duplicated code. CSS/Less/SCSS files are compared structurally — by their rule bodies'
|
|
543
|
+
* literal, ordered `property: value` declarations, independent of selector name — via
|
|
544
|
+
* {@link findStyleBlockDuplicates}. Every other language (TS/JS, Python, Go, CoffeeScript,
|
|
545
|
+
* LiveScript, Lua, Gherkin, Markdown, and Stylus, which has no shared PostCSS AST here) runs
|
|
546
|
+
* the generic token-shingling pipeline instead: comments are stripped per `FileType`,
|
|
547
|
+
* identifiers (and, by default, literals) are normalized to placeholders so renamed-variable
|
|
548
|
+
* copies still match, then a sliding token window is hashed and chain-merged into contiguous
|
|
549
|
+
* blocks. Token-shingled files are additionally partitioned into language families
|
|
550
|
+
* ({@link getDuplicateFamily} — `"style"` for Stylus, `"code"` for everything else) so
|
|
551
|
+
* matching never crosses that boundary — see docs/adr-013-duplicate-detection-noise-reduction.md.
|
|
485
552
|
* @param graph - The graph to scan; its node paths (already ignore-rule-filtered) are the file
|
|
486
553
|
* list, re-read from disk since duplication data isn't cached on `FileNode`.
|
|
487
554
|
* @param rootDir - Absolute project root that graph paths are relative to.
|
|
488
|
-
* @param options - `minLines`/`windowSize` tune sensitivity
|
|
489
|
-
*
|
|
490
|
-
*
|
|
491
|
-
*
|
|
492
|
-
|
|
493
|
-
|
|
555
|
+
* @param options - `minLines`/`windowSize` tune token-shingle sensitivity (`minLines` also caps
|
|
556
|
+
* CSS/Less/SCSS block size); `ignoreLiterals` toggles Type-2 vs Type-1 matching for the
|
|
557
|
+
* token-shingle path only (CSS/Less/SCSS always match on literal declaration content);
|
|
558
|
+
* `maxPunctuationRatio` gates out token-shingle blocks that are mostly object/array-literal
|
|
559
|
+
* structural punctuation (e.g. schema/object-literal boilerplate) rather than substantive
|
|
560
|
+
* shared logic; `maxBucketSize` bounds worst-case scan time on large repos by skipping
|
|
561
|
+
* pathologically common hash buckets; `ignoreDirs` excludes files under matching directory
|
|
562
|
+
* names; `limit` caps results; `parallelTokenizing` offloads per-file tokenizing to a worker
|
|
563
|
+
* pool once the candidate file count is large enough to be worth it. Lock files are always
|
|
564
|
+
* excluded, independent of `ignoreDirs`.
|
|
565
|
+
* @returns `groups` — duplicate blocks (each tagged with its `family`), two or more occurrences
|
|
566
|
+
* per block, every block that pairwise chain-matches another clustered into one group instead
|
|
567
|
+
* of one per pair, sorted largest-first across all families — plus `skippedBuckets` (see
|
|
568
|
+
* {@link FindDuplicatesResult}).
|
|
569
|
+
*/
|
|
570
|
+
declare function findDuplicates(graph: Graph, rootDir: string, options?: FindDuplicatesOptions): Promise<FindDuplicatesResult>;
|
|
494
571
|
|
|
495
572
|
/** Controls how aggressively `detectFeatures` promotes files to features. */
|
|
496
573
|
interface FeatureDetectionOptions {
|
|
@@ -1353,4 +1430,4 @@ declare function createWorkspaceGraph(rootDir: string, options?: {
|
|
|
1353
1430
|
*/
|
|
1354
1431
|
declare function getAllProjectFiles(rootDir: string, options?: ScanOptions): string[];
|
|
1355
1432
|
|
|
1356
|
-
export { type ApiSurface, type ApplyTagsFileResult, type ApplyTagsResult, CALL_EDGE_TYPES, CallEdge, type CalleeEntry, type CallerEntry$1 as CallerEntry, type ChangeImpactCache, type ComplexFunctionEntry, DEFAULT_EXTENSIONS, DEFAULT_IGNORE_DIRS, type DependencyGraph, type DuplicateGroup, type DuplicateOccurrence, EXPORT_TRACKING_TYPES, type ExportKind, type FeatureDetectionOptions, type FeatureDomain, type FeatureGraph, type FeatureGraphOptions, type FeatureInfo, FileNode, FileType, type FindComplexFunctionsOptions, type FindDuplicatesOptions, type FunctionCallInfo, type GetAffectedOptions, type GetCallersOptions, Graph, type CallerEntry as GraphCallerEntry, type GraphExporter, IMPORT_SYMBOL_TYPES, ImportEdge, type LanguageCoverage, MermaidExporter, type ModuleResponsibility, type ModuleRole, type MokoshConfig, type MonorepoDetector, type MonorepoLayout, NodeCategory, type NodeMeta, type NodeQuery, type ParallelParsingOption, type PathWithSymbols, type ProposeTagsOptions, type PublicExport, type ResponsibilityGraph, type ScanOptions, type SerializedGraph, type SerializedWorkspaceGraph, type SlimNode, type SlimSerializedGraph, StructuredTag, type SymbolCaller, type SymbolImporter, type SymbolMatch, type SymbolPrecision, SymbolTraversalContext, type TestNodeIdentifier, type TraversalOptions, type TraversalVisitor, type TypeEdge, type TypeGraph, type TypeKind, type TypeNode, type TypeQueryResult, WorkspaceGraph, type WorkspacePackage, type WorkspacePackageSummary, type WorkspacePackagesSummary, applyConfig, applyTags, buildApiSurface, buildChangeImpactCache, buildFeatureGraph, buildResponsibilityGraph, buildTypeGraph, computeGraphHash, configToGraphOptions, createImportMap, createWorkspaceGraph, detectAllEntryPoints, detectEntryPoint, detectFeatures, detectMonorepo, filterGraph, findComplexFunctions, findDuplicates, findSymbol, getAffected, getAllProjectFiles, getCallers, getDependencies, getDependents, getLanguageCoverage, getNodeMeta, hasCoverageData, isChangeImpactCacheValid, loadChangeImpactCache, loadCoverageMap, loadMokoshConfig, parseQuery, proposeAffectedTests, proposeTags, queryCallGraph, queryChangeImpact, queryTypeGraph, registerConfigMatcher, registerMonorepoDetector, registerParser, registerTestLibrary, registerTestPattern, saveChangeImpactCache, slimSerialize, summarizeWorkspacePackages, toMermaid };
|
|
1433
|
+
export { type ApiSurface, type ApplyTagsFileResult, type ApplyTagsResult, CALL_EDGE_TYPES, CallEdge, type CalleeEntry, type CallerEntry$1 as CallerEntry, type ChangeImpactCache, type ComplexFunctionEntry, DEFAULT_EXTENSIONS, DEFAULT_IGNORE_DIRS, type DependencyGraph, type DuplicateFamily, type DuplicateGroup, type DuplicateOccurrence, EXPORT_TRACKING_TYPES, type ExportKind, type FeatureDetectionOptions, type FeatureDomain, type FeatureGraph, type FeatureGraphOptions, type FeatureInfo, FileNode, FileType, type FindComplexFunctionsOptions, type FindDuplicatesOptions, type FunctionCallInfo, type GetAffectedOptions, type GetCallersOptions, Graph, type CallerEntry as GraphCallerEntry, type GraphExporter, IMPORT_SYMBOL_TYPES, ImportEdge, type LanguageCoverage, MermaidExporter, type ModuleResponsibility, type ModuleRole, type MokoshConfig, type MonorepoDetector, type MonorepoLayout, NodeCategory, type NodeMeta, type NodeQuery, type ParallelParsingOption, type PathWithSymbols, type ProposeTagsOptions, type PublicExport, type ResponsibilityGraph, type ScanOptions, type SerializedGraph, type SerializedWorkspaceGraph, type SlimNode, type SlimSerializedGraph, StructuredTag, type SymbolCaller, type SymbolImporter, type SymbolMatch, type SymbolPrecision, SymbolTraversalContext, type TestNodeIdentifier, type TraversalOptions, type TraversalVisitor, type TypeEdge, type TypeGraph, type TypeKind, type TypeNode, type TypeQueryResult, WorkspaceGraph, type WorkspacePackage, type WorkspacePackageSummary, type WorkspacePackagesSummary, applyConfig, applyTags, buildApiSurface, buildChangeImpactCache, buildFeatureGraph, buildResponsibilityGraph, buildTypeGraph, computeGraphHash, configToGraphOptions, createImportMap, createWorkspaceGraph, detectAllEntryPoints, detectEntryPoint, detectFeatures, detectMonorepo, filterGraph, findComplexFunctions, findDuplicates, findSymbol, getAffected, getAllProjectFiles, getCallers, getDependencies, getDependents, getLanguageCoverage, getNodeMeta, hasCoverageData, isChangeImpactCacheValid, loadChangeImpactCache, loadCoverageMap, loadMokoshConfig, parseQuery, proposeAffectedTests, proposeTags, queryCallGraph, queryChangeImpact, queryTypeGraph, registerConfigMatcher, registerMonorepoDetector, registerParser, registerTestLibrary, registerTestPattern, saveChangeImpactCache, slimSerialize, summarizeWorkspacePackages, toMermaid };
|