cg-code-graph 0.10.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. cg_code_graph-0.10.1.dist-info/METADATA +678 -0
  2. cg_code_graph-0.10.1.dist-info/RECORD +174 -0
  3. cg_code_graph-0.10.1.dist-info/WHEEL +5 -0
  4. cg_code_graph-0.10.1.dist-info/entry_points.txt +3 -0
  5. cg_code_graph-0.10.1.dist-info/licenses/LICENSE +21 -0
  6. cg_code_graph-0.10.1.dist-info/top_level.txt +1 -0
  7. codegraph/__init__.py +2 -0
  8. codegraph/aitools.py +129 -0
  9. codegraph/apps.py +76 -0
  10. codegraph/blindspots.py +428 -0
  11. codegraph/bridges.py +1701 -0
  12. codegraph/cli.py +725 -0
  13. codegraph/concepts.py +362 -0
  14. codegraph/config.py +559 -0
  15. codegraph/core/__init__.py +0 -0
  16. codegraph/core/cache.py +375 -0
  17. codegraph/core/detect.py +80 -0
  18. codegraph/core/extractors.py +187 -0
  19. codegraph/core/fsutil.py +61 -0
  20. codegraph/core/generated.py +575 -0
  21. codegraph/core/model.py +174 -0
  22. codegraph/core/paths.py +175 -0
  23. codegraph/core/plugin.py +160 -0
  24. codegraph/core/store.py +80 -0
  25. codegraph/core/syntax_errors.py +132 -0
  26. codegraph/coverage.py +928 -0
  27. codegraph/doctor.py +453 -0
  28. codegraph/external.py +613 -0
  29. codegraph/indexer.py +336 -0
  30. codegraph/link.py +434 -0
  31. codegraph/lint_async.py +524 -0
  32. codegraph/mcp_server.py +1303 -0
  33. codegraph/parity.py +473 -0
  34. codegraph/parity_structure.py +307 -0
  35. codegraph/payload.py +321 -0
  36. codegraph/plans.py +1285 -0
  37. codegraph/platform_scan.py +643 -0
  38. codegraph/platforms.py +1369 -0
  39. codegraph/plugins/__init__.py +0 -0
  40. codegraph/plugins/cfamily/__init__.py +0 -0
  41. codegraph/plugins/cfamily/plugin.py +930 -0
  42. codegraph/plugins/cfamily/syntax.py +881 -0
  43. codegraph/plugins/dart/__init__.py +0 -0
  44. codegraph/plugins/dart/bridges.py +345 -0
  45. codegraph/plugins/dart/extractor/bin/extract.dart +717 -0
  46. codegraph/plugins/dart/extractor/pubspec.lock +149 -0
  47. codegraph/plugins/dart/extractor/pubspec.yaml +7 -0
  48. codegraph/plugins/dart/http.py +904 -0
  49. codegraph/plugins/dart/models.py +308 -0
  50. codegraph/plugins/dart/plugin.py +625 -0
  51. codegraph/plugins/dart/program.py +907 -0
  52. codegraph/plugins/django/__init__.py +0 -0
  53. codegraph/plugins/django/extras.py +378 -0
  54. codegraph/plugins/django/models.py +508 -0
  55. codegraph/plugins/django/plugin.py +728 -0
  56. codegraph/plugins/django/schemas.py +339 -0
  57. codegraph/plugins/django/shapes.py +216 -0
  58. codegraph/plugins/django/urls.py +603 -0
  59. codegraph/plugins/express/__init__.py +0 -0
  60. codegraph/plugins/express/plugin.py +428 -0
  61. codegraph/plugins/flutter/__init__.py +0 -0
  62. codegraph/plugins/flutter/plugin.py +538 -0
  63. codegraph/plugins/kotlin/__init__.py +0 -0
  64. codegraph/plugins/kotlin/exact.py +457 -0
  65. codegraph/plugins/kotlin/plugin.py +1961 -0
  66. codegraph/plugins/kotlin/reparse.py +234 -0
  67. codegraph/plugins/laravel/__init__.py +0 -0
  68. codegraph/plugins/laravel/broadcast.py +351 -0
  69. codegraph/plugins/laravel/plugin.py +863 -0
  70. codegraph/plugins/laravel/tests.py +262 -0
  71. codegraph/plugins/laravel/values.py +728 -0
  72. codegraph/plugins/native/__init__.py +0 -0
  73. codegraph/plugins/native/gates.py +286 -0
  74. codegraph/plugins/native/runner.py +183 -0
  75. codegraph/plugins/native/scipread.py +194 -0
  76. codegraph/plugins/native/ts.py +54 -0
  77. codegraph/plugins/nest/__init__.py +0 -0
  78. codegraph/plugins/nest/plugin.py +654 -0
  79. codegraph/plugins/nextjs/__init__.py +0 -0
  80. codegraph/plugins/nextjs/plugin.py +336 -0
  81. codegraph/plugins/nuxt/__init__.py +0 -0
  82. codegraph/plugins/nuxt/plugin.py +308 -0
  83. codegraph/plugins/php/__init__.py +0 -0
  84. codegraph/plugins/php/extractor/composer.json +5 -0
  85. codegraph/plugins/php/extractor/composer.lock +76 -0
  86. codegraph/plugins/php/extractor/extract.php +743 -0
  87. codegraph/plugins/php/gating.py +573 -0
  88. codegraph/plugins/php/plugin.py +668 -0
  89. codegraph/plugins/php/strings.py +197 -0
  90. codegraph/plugins/python/__init__.py +0 -0
  91. codegraph/plugins/python/aitools.py +664 -0
  92. codegraph/plugins/python/external.py +245 -0
  93. codegraph/plugins/python/fields.py +107 -0
  94. codegraph/plugins/python/plugin.py +1733 -0
  95. codegraph/plugins/python/refs.py +485 -0
  96. codegraph/plugins/python/roots.py +412 -0
  97. codegraph/plugins/python/socketio.py +210 -0
  98. codegraph/plugins/python/subproc.py +864 -0
  99. codegraph/plugins/python/tests.py +1040 -0
  100. codegraph/plugins/python/values.py +179 -0
  101. codegraph/plugins/pyweb/__init__.py +0 -0
  102. codegraph/plugins/pyweb/plugin.py +1334 -0
  103. codegraph/plugins/pyweb/values.py +68 -0
  104. codegraph/plugins/rust/__init__.py +0 -0
  105. codegraph/plugins/rust/cargo.py +226 -0
  106. codegraph/plugins/rust/plugin.py +980 -0
  107. codegraph/plugins/rust/syntax.py +678 -0
  108. codegraph/plugins/scip/__init__.py +0 -0
  109. codegraph/plugins/scip/importer.py +129 -0
  110. codegraph/plugins/scip/scip.proto +962 -0
  111. codegraph/plugins/scip/scip_pb2.py +97 -0
  112. codegraph/plugins/stubs/__init__.py +0 -0
  113. codegraph/plugins/stubs/plugins.py +38 -0
  114. codegraph/plugins/swift/__init__.py +0 -0
  115. codegraph/plugins/swift/baseurl.py +109 -0
  116. codegraph/plugins/swift/exact.py +415 -0
  117. codegraph/plugins/swift/indexstore.py +209 -0
  118. codegraph/plugins/swift/packages.py +174 -0
  119. codegraph/plugins/swift/plugin.py +2890 -0
  120. codegraph/plugins/ts/__init__.py +0 -0
  121. codegraph/plugins/ts/baseurl.py +185 -0
  122. codegraph/plugins/ts/extractor/extract.mjs +2652 -0
  123. codegraph/plugins/ts/extractor/fw.mjs +685 -0
  124. codegraph/plugins/ts/extractor/package-lock.json +205 -0
  125. codegraph/plugins/ts/extractor/package.json +9 -0
  126. codegraph/plugins/ts/plugin.py +480 -0
  127. codegraph/plugins/tsweb/__init__.py +0 -0
  128. codegraph/plugins/tsweb/common.py +290 -0
  129. codegraph/plugins/tsweb/data.py +276 -0
  130. codegraph/presets/__init__.py +146 -0
  131. codegraph/presets/c_cpp.yaml +9 -0
  132. codegraph/presets/common.yaml +66 -0
  133. codegraph/presets/dart.yaml +9 -0
  134. codegraph/presets/django-ninja.yaml +15 -0
  135. codegraph/presets/django.yaml +25 -0
  136. codegraph/presets/djangorestframework.yaml +17 -0
  137. codegraph/presets/express.yaml +17 -0
  138. codegraph/presets/kotlin.yaml +11 -0
  139. codegraph/presets/laravel.yaml +40 -0
  140. codegraph/presets/nest.yaml +11 -0
  141. codegraph/presets/nextjs.yaml +15 -0
  142. codegraph/presets/nuxt.yaml +9 -0
  143. codegraph/presets/php.yaml +5 -0
  144. codegraph/presets/python.yaml +10 -0
  145. codegraph/presets/rust.yaml +5 -0
  146. codegraph/presets/swift.yaml +10 -0
  147. codegraph/presets/typescript.yaml +13 -0
  148. codegraph/process_runs.py +328 -0
  149. codegraph/protocols/__init__.py +299 -0
  150. codegraph/protocols/builtin.py +67 -0
  151. codegraph/protocols/matchers.py +144 -0
  152. codegraph/protocols/view.py +334 -0
  153. codegraph/query.py +2089 -0
  154. codegraph/realtime.py +260 -0
  155. codegraph/roundtrip.py +346 -0
  156. codegraph/routes.py +442 -0
  157. codegraph/starters.py +218 -0
  158. codegraph/tests_index.py +117 -0
  159. codegraph/viz/__init__.py +0 -0
  160. codegraph/viz/graph.py +369 -0
  161. codegraph/viz/server.py +198 -0
  162. codegraph/viz/static/app.css +148 -0
  163. codegraph/viz/static/app.js +1082 -0
  164. codegraph/viz/static/index.html +81 -0
  165. codegraph/viz/static/layered.js +237 -0
  166. codegraph/viz/static/vendor/VERSIONS.txt +4 -0
  167. codegraph/viz/static/vendor/cose-base.js +3214 -0
  168. codegraph/viz/static/vendor/cytoscape-fcose.js +1549 -0
  169. codegraph/viz/static/vendor/cytoscape.min.js +31 -0
  170. codegraph/viz/static/vendor/layout-base.js +5230 -0
  171. codegraph/viz/tools/package-lock.json +303 -0
  172. codegraph/viz/tools/package.json +7 -0
  173. codegraph/viz/tools/shoot.mjs +165 -0
  174. codegraph/xcode.py +251 -0
@@ -0,0 +1,61 @@
1
+ """File discovery helpers that never abort on odd entries (dangling symlinks, links out of the repo, races).
2
+
3
+ A dangling symlink (e.g. a committed link to a file on the author's machine) is skipped with a warning by
4
+ discovery, and hashed by its link target in cache fingerprints, so one bad entry never stops a language."""
5
+ from __future__ import annotations
6
+
7
+ import hashlib
8
+ import os
9
+ import sys
10
+
11
+ _warned: set[str] = set()
12
+
13
+
14
+ def is_real_file(path: str | os.PathLike) -> bool:
15
+ """True for a readable regular file (following symlinks); False for dangling links, dirs, sockets..."""
16
+ try:
17
+ return os.path.isfile(path)
18
+ except OSError:
19
+ return False
20
+
21
+
22
+ def warn_skip(path: str | os.PathLike, why: str = "dangling symlink") -> None:
23
+ p = str(path)
24
+ if p not in _warned:
25
+ _warned.add(p)
26
+ print(f"codegraph: skipping {p}: {why}", file=sys.stderr)
27
+
28
+
29
+ def keep_file(path: str | os.PathLike) -> bool:
30
+ """Discovery filter: real files pass; a broken entry is reported once and skipped."""
31
+ if is_real_file(path):
32
+ return True
33
+ if os.path.islink(path):
34
+ warn_skip(path)
35
+ return False
36
+
37
+
38
+ # Bumped whenever the cache key scheme changes: every extractor / SCIP cache key includes it, so entries written by
39
+ # an older cg are never reused (TS and Dart drop them on the next write, the SCIP cache prunes them).
40
+ # 2: keys hash file content (was size + mtime, which returned stale facts after a same-size edit with a restored mtime).
41
+ CACHE_VERSION = 3
42
+
43
+
44
+ def content_key(path: str | os.PathLike) -> str:
45
+ """'size|blake2b of the bytes' for cache fingerprints. Content, not mtime: an edit that keeps the size and the old
46
+ mtime must still invalidate. A dangling symlink hashes as 'link|<target>', anything else unreadable by a marker."""
47
+ if is_real_file(path): # regular files only: never open a FIFO or device found in the tree
48
+ try:
49
+ h = hashlib.blake2b(digest_size=16)
50
+ n = 0
51
+ with open(path, "rb") as fh:
52
+ for chunk in iter(lambda: fh.read(1 << 20), b""):
53
+ h.update(chunk)
54
+ n += len(chunk)
55
+ return f"{n}|{h.hexdigest()}"
56
+ except OSError:
57
+ return "unreadable"
58
+ try:
59
+ return f"link|{os.readlink(path)}"
60
+ except OSError:
61
+ return "missing" if not os.path.lexists(path) else "special"
@@ -0,0 +1,575 @@
1
+ """Generated, copied and vendored files: one classifier for every language plugin and the coverage scan.
2
+
3
+ A file nobody edits by hand gets a classification (`generated`, `copied` or `vendored`) and the reason it was
4
+ recognised, from (first match wins, in this order):
5
+
6
+ 1. the project's `.cg.yaml`: `generated.keep` (never classified), `generated.paths`, `generated.vendored`
7
+ 2. `.gitattributes` at any depth: `linguist-generated` / `linguist-vendored`, with git's precedence (a deeper file and
8
+ a later line win; `-attr` / `attr=false` marks a file as hand-written, `!attr` returns it to detection)
9
+ 3. copy targets: Capacitor's `webDir` copied into `android/app/src/main/assets/public/` and `ios/App/App/public/`
10
+ (plus the `capacitor.config.json` / `capacitor.plugins.json` copies and the Cordova plugin projects that
11
+ `cap sync` writes), Cordova's `www/` copied into `platforms/<platform>/www/`; each copied file maps back to its
12
+ source (`copy_of`)
13
+ 4. files listed in `.openapi-generator/FILES` (OpenAPI Generator output)
14
+ 5. framework build output and generator file names (codegraph/presets/common.yaml `generated`): `.nuxt/`, `.next/`,
15
+ `*.g.dart`, `*_pb2.py`, `*.pb.go`, `GeneratedPluginRegistrant.java` ...
16
+ 6. a generator banner in the file's leading comment block: `@generated`, `Code generated ... DO NOT EDIT.`,
17
+ `GENERATED CODE - DO NOT MODIFY BY HAND`, `<auto-generated>`, `This file is automatically generated`, protoc /
18
+ gRPC and OpenAPI Generator / Swagger Codegen banners, Sourcery / SwiftGen / swift-openapi-generator / Mockolo
19
+
20
+ By default these files are kept out of the graph (every walk asks `PathRules.excluded`, and a final pass drops nodes
21
+ whose file is classified) and `cg coverage` lists them by reason. Files that serve as resolution input (Nuxt's
22
+ `.nuxt/` types, `*.g.dart` JSON keys) are still read by their plugin. With `cg index --include-generated` (or
23
+ `generated.include: true`) they are indexed and every node from them carries `attrs.generated`."""
24
+ from __future__ import annotations
25
+
26
+ import fnmatch
27
+ import json
28
+ import os
29
+ import re
30
+ from collections import Counter
31
+ from pathlib import Path
32
+
33
+ from .. import presets
34
+
35
+ GENERATED, COPIED, VENDORED = "generated", "copied", "vendored"
36
+ HEADER_BYTES = 4096 # the leading comment block is read from the first 4 KB of a file
37
+ MAX_PATHS = 500 # file paths stored per reason in the index stats (counts are always exact)
38
+ SHOW_PATHS = 5
39
+
40
+ # generator banners, matched in comment lines of the file's leading comment block
41
+ HEADER_MARKERS: tuple[tuple[str, re.Pattern], ...] = (
42
+ ("@generated header", re.compile(r"(^|[\s*/#(\"'])@generated\b")),
43
+ ("'Code generated ... DO NOT EDIT.' header", re.compile(r"\bCode generated\b.*\bDO NOT EDIT\b", re.I)),
44
+ ("'GENERATED CODE - DO NOT MODIFY BY HAND' header", re.compile(r"GENERATED CODE\s*-\s*DO NOT MODIFY BY HAND", re.I)),
45
+ ("<auto-generated> header", re.compile(r"<auto-generated\b", re.I)),
46
+ ("protoc / gRPC banner", re.compile(r"\bgenerated by\b.*\b(protocol buffer compiler|protoc|grpc)\b", re.I)),
47
+ ("OpenAPI Generator banner", re.compile(r"\bauto[- ]?generated by (the )?(OpenAPI[- ]Generator|Swagger[- ]Codegen)"
48
+ r"|\bGenerated by:? https?://(openapi-generator\.tech|github\.com/swagger-api/swagger-codegen)", re.I)),
49
+ # Swift generators (#100): Sourcery templates, SwiftGen assets / strings, swift-openapi-generator, Mockolo mocks
50
+ ("Sourcery banner", re.compile(r"\bGenerated using Sourcery\b")),
51
+ ("SwiftGen banner", re.compile(r"\bGenerated using SwiftGen\b")),
52
+ ("swift-openapi-generator banner", re.compile(r"\bGenerated by swift-openapi-generator\b")),
53
+ ("Mockolo banner", re.compile(r"@Generated by Mockolo\b")),
54
+ ("'automatically generated' header", re.compile(r"\b(this|the) (file|code|class|module|source) (is|was|has been) "
55
+ r"(automatically|auto)[- ]?generated\b", re.I)),
56
+ )
57
+ COMMENT_LEADERS = ("//", "#", "/*", "*", "<!--", "--", ";", "(*", "{-")
58
+ SOURCE_EXTS: frozenset[str] = frozenset() # filled from coverage on first use (all source extensions)
59
+
60
+ CAPACITOR_CONFIGS = ("capacitor.config.ts", "capacitor.config.js", "capacitor.config.json", "capacitor.config.mts")
61
+ # where `npx cap copy` / `cap sync` writes, relative to the Capacitor app directory
62
+ CAPACITOR_COPY_TARGETS = ("android/app/src/main/assets/public", "ios/App/App/public")
63
+ CAPACITOR_SYNC_FILES = ("android/app/src/main/assets/capacitor.config.json", "android/app/src/main/assets/capacitor.plugins.json",
64
+ "android/app/src/main/res/xml/config.xml", "ios/App/App/capacitor.config.json", "ios/App/App/config.xml")
65
+ CAPACITOR_SYNC_DIRS = ("android/capacitor-cordova-android-plugins", "ios/capacitor-cordova-ios-plugins")
66
+
67
+
68
+ def _source_exts() -> frozenset[str]:
69
+ global SOURCE_EXTS
70
+ if not SOURCE_EXTS:
71
+ from ..coverage import SUPPORTED, UNSUPPORTED
72
+ SOURCE_EXTS = frozenset({e for v in SUPPORTED.values() for e in v} | set(UNSUPPORTED)
73
+ | {".java", ".kt", ".swift", ".m", ".mm", ".h", ".go", ".cs"})
74
+ return SOURCE_EXTS
75
+
76
+
77
+ def header_marker(path: str | os.PathLike) -> str | None:
78
+ """Reason of the first generator banner in the file's leading comment block, None without one. Only comment lines
79
+ before the first line of code count, so a string or docstring that mentions a banner does not."""
80
+ try:
81
+ with open(path, "rb") as fh:
82
+ head = fh.read(HEADER_BYTES)
83
+ except OSError:
84
+ return None
85
+ if not head:
86
+ return None
87
+ text = head.decode("utf-8", "replace")
88
+ if text.startswith("\ufeff"):
89
+ text = text[1:]
90
+ in_block = None
91
+ for i, raw in enumerate(text.splitlines()):
92
+ line = raw.strip()
93
+ if not line:
94
+ continue
95
+ if i == 0 and line.startswith("#!"):
96
+ continue
97
+ if in_block:
98
+ comment = True
99
+ if in_block in line:
100
+ in_block = None
101
+ else:
102
+ low = line.lower()
103
+ comment = low.startswith(COMMENT_LEADERS)
104
+ if line.startswith("/*") and "*/" not in line[2:]:
105
+ in_block = "*/"
106
+ elif line.startswith("<!--") and "-->" not in line[4:]:
107
+ in_block = "-->"
108
+ elif line.startswith("{-") and "-}" not in line[2:]:
109
+ in_block = "-}"
110
+ elif line.startswith("(*") and "*)" not in line[2:]:
111
+ in_block = "*)"
112
+ if not comment:
113
+ # `package foo;` / `<?php` / `import` ... ends the header; PHP's opening tag is still part of it
114
+ if line.startswith("<?php") or line == "<?":
115
+ continue
116
+ return None
117
+ for reason, rx in HEADER_MARKERS:
118
+ if rx.search(line):
119
+ return reason
120
+ return None
121
+
122
+
123
+ # ------------------------------------------------------------------------------------------- .gitattributes
124
+ def _attr_regex(pattern: str) -> re.Pattern:
125
+ """gitattributes pattern -> regex over the path relative to the .gitattributes directory (git semantics: no
126
+ recursive match below a matched directory; use `dir/**`)."""
127
+ from .paths import glob_regex
128
+ return re.compile(glob_regex(pattern, below=False))
129
+
130
+
131
+ def parse_gitattributes(text: str) -> list[tuple[re.Pattern, str, bool | None]]:
132
+ """(pattern regex, 'generated' | 'vendored', True / False / None) per linguist attribute of each line."""
133
+ out = []
134
+ for raw in text.splitlines():
135
+ line = raw.strip()
136
+ if not line or line.startswith("#") or line.startswith("[attr]"):
137
+ continue
138
+ parts = line.split()
139
+ pat, attrs = parts[0], parts[1:]
140
+ if pat.startswith('"'):
141
+ continue # quoted patterns (C-style escapes): rare, not read
142
+ if pat.endswith("/"):
143
+ continue # a directory pattern never matches files in an attributes file
144
+ rx = None
145
+ for a in attrs:
146
+ val: bool | None
147
+ name = a
148
+ if a.startswith("-"):
149
+ name, val = a[1:], False
150
+ elif a.startswith("!"):
151
+ name, val = a[1:], None
152
+ elif "=" in a:
153
+ name, v = a.split("=", 1)
154
+ val = v.strip().lower() not in ("false", "0", "no", "off")
155
+ else:
156
+ val = True
157
+ kind = {"linguist-generated": GENERATED, "linguist-vendored": VENDORED}.get(name)
158
+ if kind:
159
+ rx = rx or _attr_regex(pat)
160
+ out.append((rx, kind, val))
161
+ return out
162
+
163
+
164
+ # ------------------------------------------------------------------------------------------- copy targets
165
+ _WEBDIR_RE = re.compile(r"""\bwebDir\s*:\s*(['"`])([^'"`]+)\1""")
166
+
167
+
168
+ def capacitor_web_dir(cfg_file: Path) -> str | None:
169
+ try:
170
+ text = cfg_file.read_text(encoding="utf-8", errors="replace")
171
+ except OSError:
172
+ return None
173
+ if cfg_file.suffix == ".json":
174
+ try:
175
+ v = (json.loads(text) or {}).get("webDir")
176
+ except (ValueError, AttributeError):
177
+ v = None
178
+ else:
179
+ m = _WEBDIR_RE.search(text)
180
+ v = m.group(2) if m else None
181
+ if not isinstance(v, str) or not v.strip():
182
+ return None
183
+ v = v.strip().replace("\\", "/").strip("/")
184
+ if v.startswith("./"):
185
+ v = v[2:]
186
+ return v or None
187
+
188
+
189
+ def _ignored_by(dir_: Path, name: str) -> bool:
190
+ """`name` (a directory of `dir_`) is listed in `dir_/.gitignore` (plain names only: `dist`, `/dist`, `dist/`)."""
191
+ try:
192
+ lines = (dir_ / ".gitignore").read_text(encoding="utf-8", errors="replace").splitlines()
193
+ except OSError:
194
+ return False
195
+ return any(ln.strip().strip("/") == name for ln in lines if ln.strip() and not ln.startswith("#"))
196
+
197
+
198
+ def _bundler_output(dir_: Path, name: str) -> str | None:
199
+ """The bundler config of `dir_` that writes into `name` (Angular's angular.json outputPath), else None."""
200
+ try:
201
+ text = (dir_ / "angular.json").read_text(encoding="utf-8", errors="replace")
202
+ except OSError:
203
+ return None
204
+ rx = r'"outputPath"\s*:\s*(?:\{[^}]*?"base"\s*:\s*)?"(?:\./)?' + re.escape(name) + r'(?:/[^"]*)?"'
205
+ return "angular.json outputPath" if re.search(rx, text) else None
206
+
207
+
208
+ class Hit(dict):
209
+ """{"kind": generated | copied | vendored, "reason": ..., optionally "copy_of": source path, "test": True}."""
210
+
211
+
212
+ # test folders and test-target directories (#100): a generated file there runs as a test or test support (Sourcery's
213
+ # preview / accessibility test lists, generated test stubs), so by default it stays in the graph as test code, with
214
+ # attrs.generated; only generated non-test sources leave it. `.cg.yaml` generated.paths still excludes anything.
215
+ TEST_DIR = re.compile(r"(?:^|/)(?:tests?|Tests?|__tests__|specs?|androidTest|testFixtures|integrationTest|unitTest|"
216
+ r"[^/]*(?:Tests|UITests))/")
217
+
218
+
219
+ def _testy(rel: str, h: "Hit | None") -> "Hit | None":
220
+ if h is not None and h.get("kind") == GENERATED and not h["reason"].startswith(".cg.yaml") and TEST_DIR.search(rel):
221
+ h = Hit(h, test=True)
222
+ return h
223
+
224
+
225
+ class Classifier:
226
+ """Classification of one project's files (see the module docstring). `scan(rel, abs)` classifies a file the
227
+ coverage scan walks (reads its header); `lookup(rel)` answers for any repo-relative path from what the scan
228
+ recorded plus the path rules."""
229
+
230
+ def __init__(self, root: str | Path, cfg: dict | None = None, include: bool = False):
231
+ from .paths import glob_regex
232
+ self.root = Path(root)
233
+ cfg = cfg or {}
234
+ g = cfg.get("generated") or {}
235
+ self.include = bool(include or g.get("include"))
236
+ self.keep = re.compile("|".join(f"(?:{glob_regex(p)})" for p in g["keep"])) if g.get("keep") else None
237
+ self.user: list[tuple[re.Pattern, Hit]] = []
238
+ for key, kind in (("paths", GENERATED), ("vendored", VENDORED)):
239
+ for p in g.get(key) or []:
240
+ self.user.append((re.compile(glob_regex(p)), Hit(kind=kind, reason=f".cg.yaml generated.{key}")))
241
+ pre = presets.values("common", "generated", default={}) or {}
242
+ self.build_dirs: dict[str, str] = dict(pre.get("build_dirs") or {})
243
+ self.file_globs: list[tuple[str, str]] = list((pre.get("files") or {}).items())
244
+ self.tool_paths = [(re.compile(glob_regex(g)), r) for g, r in (pre.get("paths") or {}).items()]
245
+ self._exact_names = {n: r for n, r in self.file_globs if not any(c in n for c in "*?[")}
246
+ self._name_globs = [(n, r) for n, r in self.file_globs if any(c in n for c in "*?[")]
247
+ self.attrs: list[tuple[str, list]] = [] # (base dir, rules) in walk order (parents first)
248
+ self.copies: list[dict] = [] # {"target", "source", "via", "kind"}
249
+ self.listed: dict[str, str] = {} # rel -> reason (.openapi-generator/FILES)
250
+ self.files: dict[str, Hit] = {} # classified files the scan saw
251
+ self.pruned: dict[str, str] = {} # build output directories the scan did not descend into
252
+ self._cache: dict[str, Hit | None] = {}
253
+
254
+ # -------------------------------------------------- per directory (called by the scan, parents first)
255
+ def visit_dir(self, rel_dir: str, dp: str, dns: list[str], fns: list[str]) -> None:
256
+ names = set(fns)
257
+ if ".gitattributes" in names:
258
+ try:
259
+ txt = Path(dp, ".gitattributes").read_text(encoding="utf-8", errors="replace")
260
+ except OSError:
261
+ txt = ""
262
+ rules = parse_gitattributes(txt)
263
+ if rules:
264
+ self.attrs.append((rel_dir, rules))
265
+ pre = f"{rel_dir}/" if rel_dir else ""
266
+ for cf in CAPACITOR_CONFIGS:
267
+ if cf in names:
268
+ web = capacitor_web_dir(Path(dp, cf))
269
+ src = f"{pre}{web}" if web else None
270
+ via = f"Capacitor webDir {web!r} ({pre}{cf})" if web else f"Capacitor ({pre}{cf})"
271
+ for t in CAPACITOR_COPY_TARGETS:
272
+ self.copies.append({"target": pre + t, "source": src, "via": via, "kind": COPIED,
273
+ "reason": f"copy of {web}/ (Capacitor webDir)" if web else "Capacitor web asset copy"})
274
+ for t in CAPACITOR_SYNC_DIRS:
275
+ self.copies.append({"target": pre + t, "source": None, "via": f"Capacitor sync ({pre}{cf})",
276
+ "kind": GENERATED, "reason": "Capacitor sync output"})
277
+ for f in CAPACITOR_SYNC_FILES:
278
+ self.listed[pre + f] = "Capacitor sync output"
279
+ why = None
280
+ if web and _ignored_by(Path(dp), web.split("/")[0]):
281
+ why = f"listed in {pre}.gitignore"
282
+ elif web:
283
+ why = _bundler_output(Path(dp), web)
284
+ if why:
285
+ self.copies.append({"target": pre + web, "source": None, "via": f"Capacitor webDir, {why}",
286
+ "kind": GENERATED, "reason": "web build output (Capacitor webDir)"})
287
+ break
288
+ if "config.xml" in names and "www" in dns and "platforms" in dns:
289
+ try:
290
+ is_cordova = "<widget" in Path(dp, "config.xml").read_text(encoding="utf-8", errors="replace")[:4096]
291
+ except OSError:
292
+ is_cordova = False
293
+ if is_cordova:
294
+ for plat in sorted(os.listdir(os.path.join(dp, "platforms"))):
295
+ for www in ("www", "app/src/main/assets/www", "platform_www"):
296
+ if os.path.isdir(os.path.join(dp, "platforms", plat, www)):
297
+ self.copies.append({"target": f"{pre}platforms/{plat}/{www}", "source": f"{pre}www",
298
+ "via": f"Cordova ({pre}config.xml)", "kind": COPIED,
299
+ "reason": "copy of www/ (Cordova platform)"})
300
+ if ".openapi-generator" in dns and os.path.isfile(os.path.join(dp, ".openapi-generator", "FILES")):
301
+ try:
302
+ for ln in Path(dp, ".openapi-generator", "FILES").read_text(encoding="utf-8", errors="replace").splitlines():
303
+ ln = ln.strip().strip("/")
304
+ if ln and not ln.startswith("#") and ".." not in ln.split("/"):
305
+ self.listed.setdefault(pre + ln, "OpenAPI Generator (.openapi-generator/FILES)")
306
+ except OSError:
307
+ pass
308
+
309
+ def skipped_dir(self, rel: str, name: str) -> None:
310
+ """The scan does not descend into `rel`: recorded when it is a framework build output directory."""
311
+ r = self.build_dirs.get(name)
312
+ if r:
313
+ self.pruned.setdefault(rel, r)
314
+
315
+ # -------------------------------------------------- classification
316
+ def _attr(self, rel: str) -> Hit | None | bool:
317
+ """Hit from .gitattributes, False when a linguist attribute marks the file as hand-written, None otherwise."""
318
+ state: dict[str, bool | None] = {}
319
+ for base, rules in self.attrs:
320
+ if base and not rel.startswith(base + "/"):
321
+ continue
322
+ sub = rel[len(base) + 1:] if base else rel
323
+ for rx, kind, val in rules:
324
+ if rx.match(sub):
325
+ state[kind] = val
326
+ if state.get(VENDORED):
327
+ return Hit(kind=VENDORED, reason="linguist-vendored (.gitattributes)")
328
+ if state.get(GENERATED):
329
+ return Hit(kind=GENERATED, reason="linguist-generated (.gitattributes)")
330
+ if state.get(GENERATED) is False or state.get(VENDORED) is False:
331
+ return False
332
+ return None
333
+
334
+ def _copy(self, rel: str) -> Hit | None:
335
+ best = None
336
+ for c in self.copies:
337
+ t = c["target"]
338
+ if rel == t or rel.startswith(t + "/"):
339
+ if best is None or len(t) > len(best["target"]):
340
+ best = c
341
+ if best is None:
342
+ return None
343
+ h = Hit(kind=best["kind"], reason=best["reason"])
344
+ if best["source"] and rel != best["target"]:
345
+ h["copy_of"] = best["source"] + rel[len(best["target"]):]
346
+ return h
347
+
348
+ def _path_rule(self, rel: str) -> Hit | None:
349
+ parts = rel.split("/")
350
+ for p in parts[:-1]:
351
+ # hidden framework directories at any depth; `dist/` / `build/` are only reported (every walk skips them by
352
+ # name, and a source package can be named `build`)
353
+ r = self.build_dirs.get(p) if p.startswith(".") else None
354
+ if r:
355
+ return Hit(kind=GENERATED, reason=r)
356
+ for rx, r in self.tool_paths:
357
+ if rx.match(rel):
358
+ return Hit(kind=GENERATED, reason=r)
359
+ name = parts[-1]
360
+ r = self._exact_names.get(name)
361
+ if r is None:
362
+ for g, rr in self._name_globs:
363
+ if fnmatch.fnmatchcase(name, g):
364
+ r = rr
365
+ break
366
+ return Hit(kind=GENERATED, reason=r) if r else None
367
+
368
+ def rules_hit(self, rel: str) -> Hit | None:
369
+ """Classification from everything but the file header (cheap; no file read)."""
370
+ if self.keep and self.keep.match(rel):
371
+ return None
372
+ for rx, h in self.user:
373
+ if rx.match(rel):
374
+ return h
375
+ a = self._attr(rel)
376
+ if a is False:
377
+ return None
378
+ if a:
379
+ return _testy(rel, a)
380
+ h = self._copy(rel)
381
+ if h:
382
+ return h
383
+ if rel in self.listed:
384
+ return _testy(rel, Hit(kind=GENERATED, reason=self.listed[rel]))
385
+ h = self._path_rule(rel)
386
+ # build output directories (.nuxt/, .next/ ...) are never test code
387
+ return h if h is None or h["reason"] in self.build_dirs.values() else _testy(rel, h)
388
+
389
+ def scan(self, rel: str, abs_path: str) -> Hit | None:
390
+ """Classify one file the coverage scan walks: the path rules, then the header of a source file."""
391
+ h = self.rules_hit(rel)
392
+ if h is None and not (self.keep and self.keep.match(rel)) and self._attr(rel) is not False:
393
+ ext = os.path.splitext(rel)[1].lower()
394
+ if ext in _source_exts():
395
+ r = header_marker(abs_path)
396
+ if r:
397
+ h = _testy(rel, Hit(kind=GENERATED, reason=r))
398
+ if h is not None:
399
+ self.files[rel] = h
400
+ return h
401
+
402
+ def lookup(self, rel: str) -> Hit | None:
403
+ """Classification of a repo-relative path: what the scan recorded, else the path rules."""
404
+ h = self.files.get(rel)
405
+ if h is not None:
406
+ return h
407
+ if rel not in self._cache:
408
+ self._cache[rel] = self.rules_hit(rel)
409
+ return self._cache[rel]
410
+
411
+ def drops(self, h: Hit | None) -> bool:
412
+ """A classified file the default mode leaves out of the graph (a generated test file stays in)."""
413
+ return h is not None and not self.include and not h.get("test")
414
+
415
+ def excludes(self, rel: str) -> bool:
416
+ return self.drops(self.lookup(rel))
417
+
418
+ def dir_excluded(self, rel_dir: str) -> bool:
419
+ """A directory every file of which is classified (a copy target, a build output directory, a user glob)."""
420
+ if self.include or (self.keep and self.keep.match(rel_dir + "/")):
421
+ return False
422
+ if any(rx.match(rel_dir + "/") for rx, _ in self.user):
423
+ return True
424
+ if any(rel_dir == c["target"] or rel_dir.startswith(c["target"] + "/") for c in self.copies):
425
+ return True
426
+ name = rel_dir.rsplit("/", 1)[-1]
427
+ return name.startswith(".") and name in self.build_dirs
428
+
429
+ def dir_regexes(self) -> list[str]:
430
+ """JavaScript-compatible regexes of the directories excluded as a whole (copy targets, Capacitor sync output,
431
+ .cg.yaml generated globs); framework build directories are left to each extractor (Nuxt reads `.nuxt/`)."""
432
+ if self.include:
433
+ return []
434
+ out = [f"(?:^{re.escape(c['target'])}(?:/.*)?$)" for c in self.copies]
435
+ if not self.keep: # with keep globs the scan's explicit file list carries the .cg.yaml rules
436
+ out += [f"(?:{rx.pattern})" for rx, _ in self.user]
437
+ return out
438
+
439
+ # -------------------------------------------------- reporting
440
+ def summary(self) -> dict:
441
+ """The coverage entry: counts by reason and kind, sample paths per reason, copy targets, build directories."""
442
+ by_reason, by_kind, kept = Counter(), Counter(), Counter()
443
+ paths: dict[str, list[str]] = {}
444
+ kept_paths: dict[str, list[str]] = {}
445
+ for rel in sorted(self.files):
446
+ h = self.files[rel]
447
+ if h.get("test") and not self.include:
448
+ kept[h["reason"]] += 1
449
+ kp = kept_paths.setdefault(h["reason"], [])
450
+ if len(kp) < MAX_PATHS:
451
+ kp.append(rel)
452
+ continue
453
+ by_reason[h["reason"]] += 1
454
+ by_kind[h["kind"]] += 1
455
+ ps = paths.setdefault(h["reason"], [])
456
+ if len(ps) < MAX_PATHS:
457
+ ps.append(rel)
458
+ copies = []
459
+ for c in self.copies:
460
+ n = sum(1 for rel in self.files if rel == c["target"] or rel.startswith(c["target"] + "/"))
461
+ if n:
462
+ copies.append({"target": c["target"], "source": c["source"], "via": c["via"], "files": n, "reason": c["reason"]})
463
+ out = {"mode": "indexed" if self.include else "excluded", "files": sum(by_reason.values()),
464
+ "by_reason": dict(by_reason.most_common()), "by_kind": dict(by_kind.most_common()), "paths": paths}
465
+ if copies:
466
+ out["copies"] = copies
467
+ if kept:
468
+ out["tests_kept"] = {"files": sum(kept.values()), "by_reason": dict(kept.most_common()), "paths": kept_paths}
469
+ if self.pruned:
470
+ out["build_dirs"] = dict(sorted(self.pruned.items())[:MAX_PATHS])
471
+ return out
472
+
473
+
474
+ def for_project(project) -> Classifier | None:
475
+ return (getattr(project, "options", None) or {}).get("generated")
476
+
477
+
478
+ def summary_text(g: dict | None) -> str:
479
+ """'generated: 412 files excluded (copy of dist/ (Capacitor webDir) 380, linguist-generated 21, @generated header 11)'."""
480
+ if not g or not (g.get("files") or g.get("build_dirs") or g.get("tests_kept")):
481
+ return ""
482
+ parts = ", ".join(f"{r} {n}" for r, n in (g.get("by_reason") or {}).items())
483
+ s = f"generated: {g['files']} file{'s' if g['files'] != 1 else ''} {g.get('mode', 'excluded')}" + (f" ({parts})" if parts else "")
484
+ tk = g.get("tests_kept") or {}
485
+ if tk.get("files"):
486
+ s += f"; {tk['files']} generated test file{'s' if tk['files'] != 1 else ''} indexed as tests"
487
+ if g.get("build_dirs"):
488
+ s += f"; build output not scanned: {', '.join(d + '/' for d in list(g['build_dirs'])[:4])}" + (
489
+ f" … +{len(g['build_dirs']) - 4}" if len(g["build_dirs"]) > 4 else "")
490
+ return s
491
+
492
+
493
+ def detail_lines(g: dict | None, all_files: bool = False, indent: str = " ") -> list[str]:
494
+ if not g or not (g.get("files") or g.get("build_dirs") or g.get("tests_kept")):
495
+ return []
496
+ out = [indent + summary_text(g)]
497
+ for c in g.get("copies") or []:
498
+ what = f"a copy of {c['source']}/" if c.get("source") else c.get("reason", "generated")
499
+ out.append(f"{indent} {c['target']}/ ({c['files']} file{'s' if c['files'] != 1 else ''}) is {what} [{c['via']}]")
500
+ for reason, ps in (g.get("paths") or {}).items():
501
+ n = (g.get("by_reason") or {}).get(reason, len(ps))
502
+ show = ps if all_files else ps[:SHOW_PATHS]
503
+ more = n - len(show)
504
+ out.append(f"{indent} {reason}: " + ", ".join(show) + (f" … +{more} more (--all-files)" if more > 0 else ""))
505
+ tk = g.get("tests_kept") or {}
506
+ for reason, ps in (tk.get("paths") or {}).items():
507
+ n = (tk.get("by_reason") or {}).get(reason, len(ps))
508
+ show = ps if all_files else ps[:SHOW_PATHS]
509
+ more = n - len(show)
510
+ out.append(f"{indent} {reason}, indexed as test code: " + ", ".join(show)
511
+ + (f" … +{more} more (--all-files)" if more > 0 else ""))
512
+ return out
513
+
514
+
515
+ def apply(builder, clf: Classifier) -> dict:
516
+ """Final pass over the graph: nodes whose file is classified leave the graph with their edges (default), or carry
517
+ `attrs.generated` (included); an included copy gets a COPY_OF edge to the module node of its source file."""
518
+ from .plugin import gc_paused
519
+ by_file: dict[str, Hit | None] = {}
520
+ drop: set[str] = set()
521
+ tagged = 0
522
+ copies = []
523
+ with gc_paused():
524
+ for nid, n in builder.nodes.items():
525
+ f = n.file
526
+ if not f:
527
+ continue
528
+ h = by_file.get(f, 0)
529
+ if h == 0:
530
+ h = by_file[f] = clf.lookup(f)
531
+ if h is None:
532
+ continue
533
+ if clf.include or h.get("test"):
534
+ n.attrs = {**(n.attrs or {}), "generated": dict(h)}
535
+ tagged += 1
536
+ if h.get("copy_of") and n.kind == "module":
537
+ copies.append((nid, n.file, h["copy_of"]))
538
+ else:
539
+ drop.add(nid)
540
+ edges_dropped = 0
541
+ if drop:
542
+ for nid in drop:
543
+ del builder.nodes[nid]
544
+ dead = [k for k, e in builder.edges.items() if e.src in drop or e.dst in drop]
545
+ for k in dead:
546
+ del builder.edges[k]
547
+ edges_dropped = len(dead)
548
+ linked = 0
549
+ if copies:
550
+ mods = {n.file: nid for nid, n in builder.nodes.items() if n.kind == "module" and n.file}
551
+ for nid, f, src in copies:
552
+ dst = mods.get(src)
553
+ if dst and dst != nid:
554
+ builder.add_edge(nid, dst, "COPY_OF", file=f, line=1)
555
+ linked += 1
556
+ with_nodes = sum(1 for h in by_file.values() if h)
557
+ out = {**classified_counts(clf), "mode": "indexed" if clf.include else "excluded"}
558
+ if clf.include:
559
+ out.update(files_with_nodes=with_nodes, nodes_labelled=tagged, copy_of_edges=linked)
560
+ else:
561
+ out.update(files_with_dropped_nodes=with_nodes, nodes_dropped=len(drop), edges_dropped=edges_dropped)
562
+ return out
563
+
564
+
565
+ def classified_counts(clf: Classifier) -> dict:
566
+ """The files the classifier labelled (the up-front scan, the same list `cg coverage` shows): the total, per
567
+ language (coverage language keys; `other` for non-source files) and per reason / kind."""
568
+ from ..coverage import UNSUPPORTED, _lang_of_file
569
+ by_lang, by_reason, by_kind = Counter(), Counter(), Counter()
570
+ for rel, h in clf.files.items():
571
+ by_lang[_lang_of_file(rel) or UNSUPPORTED.get(os.path.splitext(rel)[1].lower()) or "other"] += 1
572
+ by_reason[h["reason"]] += 1
573
+ by_kind[h["kind"]] += 1
574
+ return {"files": len(clf.files), "by_language": dict(by_lang.most_common()), "by_reason": dict(by_reason.most_common()),
575
+ "by_kind": dict(by_kind.most_common())}