cg-code-graph 0.10.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cg_code_graph-0.10.1.dist-info/METADATA +678 -0
- cg_code_graph-0.10.1.dist-info/RECORD +174 -0
- cg_code_graph-0.10.1.dist-info/WHEEL +5 -0
- cg_code_graph-0.10.1.dist-info/entry_points.txt +3 -0
- cg_code_graph-0.10.1.dist-info/licenses/LICENSE +21 -0
- cg_code_graph-0.10.1.dist-info/top_level.txt +1 -0
- codegraph/__init__.py +2 -0
- codegraph/aitools.py +129 -0
- codegraph/apps.py +76 -0
- codegraph/blindspots.py +428 -0
- codegraph/bridges.py +1701 -0
- codegraph/cli.py +725 -0
- codegraph/concepts.py +362 -0
- codegraph/config.py +559 -0
- codegraph/core/__init__.py +0 -0
- codegraph/core/cache.py +375 -0
- codegraph/core/detect.py +80 -0
- codegraph/core/extractors.py +187 -0
- codegraph/core/fsutil.py +61 -0
- codegraph/core/generated.py +575 -0
- codegraph/core/model.py +174 -0
- codegraph/core/paths.py +175 -0
- codegraph/core/plugin.py +160 -0
- codegraph/core/store.py +80 -0
- codegraph/core/syntax_errors.py +132 -0
- codegraph/coverage.py +928 -0
- codegraph/doctor.py +453 -0
- codegraph/external.py +613 -0
- codegraph/indexer.py +336 -0
- codegraph/link.py +434 -0
- codegraph/lint_async.py +524 -0
- codegraph/mcp_server.py +1303 -0
- codegraph/parity.py +473 -0
- codegraph/parity_structure.py +307 -0
- codegraph/payload.py +321 -0
- codegraph/plans.py +1285 -0
- codegraph/platform_scan.py +643 -0
- codegraph/platforms.py +1369 -0
- codegraph/plugins/__init__.py +0 -0
- codegraph/plugins/cfamily/__init__.py +0 -0
- codegraph/plugins/cfamily/plugin.py +930 -0
- codegraph/plugins/cfamily/syntax.py +881 -0
- codegraph/plugins/dart/__init__.py +0 -0
- codegraph/plugins/dart/bridges.py +345 -0
- codegraph/plugins/dart/extractor/bin/extract.dart +717 -0
- codegraph/plugins/dart/extractor/pubspec.lock +149 -0
- codegraph/plugins/dart/extractor/pubspec.yaml +7 -0
- codegraph/plugins/dart/http.py +904 -0
- codegraph/plugins/dart/models.py +308 -0
- codegraph/plugins/dart/plugin.py +625 -0
- codegraph/plugins/dart/program.py +907 -0
- codegraph/plugins/django/__init__.py +0 -0
- codegraph/plugins/django/extras.py +378 -0
- codegraph/plugins/django/models.py +508 -0
- codegraph/plugins/django/plugin.py +728 -0
- codegraph/plugins/django/schemas.py +339 -0
- codegraph/plugins/django/shapes.py +216 -0
- codegraph/plugins/django/urls.py +603 -0
- codegraph/plugins/express/__init__.py +0 -0
- codegraph/plugins/express/plugin.py +428 -0
- codegraph/plugins/flutter/__init__.py +0 -0
- codegraph/plugins/flutter/plugin.py +538 -0
- codegraph/plugins/kotlin/__init__.py +0 -0
- codegraph/plugins/kotlin/exact.py +457 -0
- codegraph/plugins/kotlin/plugin.py +1961 -0
- codegraph/plugins/kotlin/reparse.py +234 -0
- codegraph/plugins/laravel/__init__.py +0 -0
- codegraph/plugins/laravel/broadcast.py +351 -0
- codegraph/plugins/laravel/plugin.py +863 -0
- codegraph/plugins/laravel/tests.py +262 -0
- codegraph/plugins/laravel/values.py +728 -0
- codegraph/plugins/native/__init__.py +0 -0
- codegraph/plugins/native/gates.py +286 -0
- codegraph/plugins/native/runner.py +183 -0
- codegraph/plugins/native/scipread.py +194 -0
- codegraph/plugins/native/ts.py +54 -0
- codegraph/plugins/nest/__init__.py +0 -0
- codegraph/plugins/nest/plugin.py +654 -0
- codegraph/plugins/nextjs/__init__.py +0 -0
- codegraph/plugins/nextjs/plugin.py +336 -0
- codegraph/plugins/nuxt/__init__.py +0 -0
- codegraph/plugins/nuxt/plugin.py +308 -0
- codegraph/plugins/php/__init__.py +0 -0
- codegraph/plugins/php/extractor/composer.json +5 -0
- codegraph/plugins/php/extractor/composer.lock +76 -0
- codegraph/plugins/php/extractor/extract.php +743 -0
- codegraph/plugins/php/gating.py +573 -0
- codegraph/plugins/php/plugin.py +668 -0
- codegraph/plugins/php/strings.py +197 -0
- codegraph/plugins/python/__init__.py +0 -0
- codegraph/plugins/python/aitools.py +664 -0
- codegraph/plugins/python/external.py +245 -0
- codegraph/plugins/python/fields.py +107 -0
- codegraph/plugins/python/plugin.py +1733 -0
- codegraph/plugins/python/refs.py +485 -0
- codegraph/plugins/python/roots.py +412 -0
- codegraph/plugins/python/socketio.py +210 -0
- codegraph/plugins/python/subproc.py +864 -0
- codegraph/plugins/python/tests.py +1040 -0
- codegraph/plugins/python/values.py +179 -0
- codegraph/plugins/pyweb/__init__.py +0 -0
- codegraph/plugins/pyweb/plugin.py +1334 -0
- codegraph/plugins/pyweb/values.py +68 -0
- codegraph/plugins/rust/__init__.py +0 -0
- codegraph/plugins/rust/cargo.py +226 -0
- codegraph/plugins/rust/plugin.py +980 -0
- codegraph/plugins/rust/syntax.py +678 -0
- codegraph/plugins/scip/__init__.py +0 -0
- codegraph/plugins/scip/importer.py +129 -0
- codegraph/plugins/scip/scip.proto +962 -0
- codegraph/plugins/scip/scip_pb2.py +97 -0
- codegraph/plugins/stubs/__init__.py +0 -0
- codegraph/plugins/stubs/plugins.py +38 -0
- codegraph/plugins/swift/__init__.py +0 -0
- codegraph/plugins/swift/baseurl.py +109 -0
- codegraph/plugins/swift/exact.py +415 -0
- codegraph/plugins/swift/indexstore.py +209 -0
- codegraph/plugins/swift/packages.py +174 -0
- codegraph/plugins/swift/plugin.py +2890 -0
- codegraph/plugins/ts/__init__.py +0 -0
- codegraph/plugins/ts/baseurl.py +185 -0
- codegraph/plugins/ts/extractor/extract.mjs +2652 -0
- codegraph/plugins/ts/extractor/fw.mjs +685 -0
- codegraph/plugins/ts/extractor/package-lock.json +205 -0
- codegraph/plugins/ts/extractor/package.json +9 -0
- codegraph/plugins/ts/plugin.py +480 -0
- codegraph/plugins/tsweb/__init__.py +0 -0
- codegraph/plugins/tsweb/common.py +290 -0
- codegraph/plugins/tsweb/data.py +276 -0
- codegraph/presets/__init__.py +146 -0
- codegraph/presets/c_cpp.yaml +9 -0
- codegraph/presets/common.yaml +66 -0
- codegraph/presets/dart.yaml +9 -0
- codegraph/presets/django-ninja.yaml +15 -0
- codegraph/presets/django.yaml +25 -0
- codegraph/presets/djangorestframework.yaml +17 -0
- codegraph/presets/express.yaml +17 -0
- codegraph/presets/kotlin.yaml +11 -0
- codegraph/presets/laravel.yaml +40 -0
- codegraph/presets/nest.yaml +11 -0
- codegraph/presets/nextjs.yaml +15 -0
- codegraph/presets/nuxt.yaml +9 -0
- codegraph/presets/php.yaml +5 -0
- codegraph/presets/python.yaml +10 -0
- codegraph/presets/rust.yaml +5 -0
- codegraph/presets/swift.yaml +10 -0
- codegraph/presets/typescript.yaml +13 -0
- codegraph/process_runs.py +328 -0
- codegraph/protocols/__init__.py +299 -0
- codegraph/protocols/builtin.py +67 -0
- codegraph/protocols/matchers.py +144 -0
- codegraph/protocols/view.py +334 -0
- codegraph/query.py +2089 -0
- codegraph/realtime.py +260 -0
- codegraph/roundtrip.py +346 -0
- codegraph/routes.py +442 -0
- codegraph/starters.py +218 -0
- codegraph/tests_index.py +117 -0
- codegraph/viz/__init__.py +0 -0
- codegraph/viz/graph.py +369 -0
- codegraph/viz/server.py +198 -0
- codegraph/viz/static/app.css +148 -0
- codegraph/viz/static/app.js +1082 -0
- codegraph/viz/static/index.html +81 -0
- codegraph/viz/static/layered.js +237 -0
- codegraph/viz/static/vendor/VERSIONS.txt +4 -0
- codegraph/viz/static/vendor/cose-base.js +3214 -0
- codegraph/viz/static/vendor/cytoscape-fcose.js +1549 -0
- codegraph/viz/static/vendor/cytoscape.min.js +31 -0
- codegraph/viz/static/vendor/layout-base.js +5230 -0
- codegraph/viz/tools/package-lock.json +303 -0
- codegraph/viz/tools/package.json +7 -0
- codegraph/viz/tools/shoot.mjs +165 -0
- codegraph/xcode.py +251 -0
codegraph/core/fsutil.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""File discovery helpers that never abort on odd entries (dangling symlinks, links out of the repo, races).
|
|
2
|
+
|
|
3
|
+
A dangling symlink (e.g. a committed link to a file on the author's machine) is skipped with a warning by
|
|
4
|
+
discovery, and hashed by its link target in cache fingerprints, so one bad entry never stops a language."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import hashlib
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
|
|
11
|
+
_warned: set[str] = set()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def is_real_file(path: str | os.PathLike) -> bool:
|
|
15
|
+
"""True for a readable regular file (following symlinks); False for dangling links, dirs, sockets..."""
|
|
16
|
+
try:
|
|
17
|
+
return os.path.isfile(path)
|
|
18
|
+
except OSError:
|
|
19
|
+
return False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def warn_skip(path: str | os.PathLike, why: str = "dangling symlink") -> None:
|
|
23
|
+
p = str(path)
|
|
24
|
+
if p not in _warned:
|
|
25
|
+
_warned.add(p)
|
|
26
|
+
print(f"codegraph: skipping {p}: {why}", file=sys.stderr)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def keep_file(path: str | os.PathLike) -> bool:
|
|
30
|
+
"""Discovery filter: real files pass; a broken entry is reported once and skipped."""
|
|
31
|
+
if is_real_file(path):
|
|
32
|
+
return True
|
|
33
|
+
if os.path.islink(path):
|
|
34
|
+
warn_skip(path)
|
|
35
|
+
return False
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
# Bumped whenever the cache key scheme changes: every extractor / SCIP cache key includes it, so entries written by
|
|
39
|
+
# an older cg are never reused (TS and Dart drop them on the next write, the SCIP cache prunes them).
|
|
40
|
+
# 2: keys hash file content (was size + mtime, which returned stale facts after a same-size edit with a restored mtime).
|
|
41
|
+
CACHE_VERSION = 3
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def content_key(path: str | os.PathLike) -> str:
|
|
45
|
+
"""'size|blake2b of the bytes' for cache fingerprints. Content, not mtime: an edit that keeps the size and the old
|
|
46
|
+
mtime must still invalidate. A dangling symlink hashes as 'link|<target>', anything else unreadable by a marker."""
|
|
47
|
+
if is_real_file(path): # regular files only: never open a FIFO or device found in the tree
|
|
48
|
+
try:
|
|
49
|
+
h = hashlib.blake2b(digest_size=16)
|
|
50
|
+
n = 0
|
|
51
|
+
with open(path, "rb") as fh:
|
|
52
|
+
for chunk in iter(lambda: fh.read(1 << 20), b""):
|
|
53
|
+
h.update(chunk)
|
|
54
|
+
n += len(chunk)
|
|
55
|
+
return f"{n}|{h.hexdigest()}"
|
|
56
|
+
except OSError:
|
|
57
|
+
return "unreadable"
|
|
58
|
+
try:
|
|
59
|
+
return f"link|{os.readlink(path)}"
|
|
60
|
+
except OSError:
|
|
61
|
+
return "missing" if not os.path.lexists(path) else "special"
|
|
@@ -0,0 +1,575 @@
|
|
|
1
|
+
"""Generated, copied and vendored files: one classifier for every language plugin and the coverage scan.
|
|
2
|
+
|
|
3
|
+
A file nobody edits by hand gets a classification (`generated`, `copied` or `vendored`) and the reason it was
|
|
4
|
+
recognised, from (first match wins, in this order):
|
|
5
|
+
|
|
6
|
+
1. the project's `.cg.yaml`: `generated.keep` (never classified), `generated.paths`, `generated.vendored`
|
|
7
|
+
2. `.gitattributes` at any depth: `linguist-generated` / `linguist-vendored`, with git's precedence (a deeper file and
|
|
8
|
+
a later line win; `-attr` / `attr=false` marks a file as hand-written, `!attr` returns it to detection)
|
|
9
|
+
3. copy targets: Capacitor's `webDir` copied into `android/app/src/main/assets/public/` and `ios/App/App/public/`
|
|
10
|
+
(plus the `capacitor.config.json` / `capacitor.plugins.json` copies and the Cordova plugin projects that
|
|
11
|
+
`cap sync` writes), Cordova's `www/` copied into `platforms/<platform>/www/`; each copied file maps back to its
|
|
12
|
+
source (`copy_of`)
|
|
13
|
+
4. files listed in `.openapi-generator/FILES` (OpenAPI Generator output)
|
|
14
|
+
5. framework build output and generator file names (codegraph/presets/common.yaml `generated`): `.nuxt/`, `.next/`,
|
|
15
|
+
`*.g.dart`, `*_pb2.py`, `*.pb.go`, `GeneratedPluginRegistrant.java` ...
|
|
16
|
+
6. a generator banner in the file's leading comment block: `@generated`, `Code generated ... DO NOT EDIT.`,
|
|
17
|
+
`GENERATED CODE - DO NOT MODIFY BY HAND`, `<auto-generated>`, `This file is automatically generated`, protoc /
|
|
18
|
+
gRPC and OpenAPI Generator / Swagger Codegen banners, Sourcery / SwiftGen / swift-openapi-generator / Mockolo
|
|
19
|
+
|
|
20
|
+
By default these files are kept out of the graph (every walk asks `PathRules.excluded`, and a final pass drops nodes
|
|
21
|
+
whose file is classified) and `cg coverage` lists them by reason. Files that serve as resolution input (Nuxt's
|
|
22
|
+
`.nuxt/` types, `*.g.dart` JSON keys) are still read by their plugin. With `cg index --include-generated` (or
|
|
23
|
+
`generated.include: true`) they are indexed and every node from them carries `attrs.generated`."""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import fnmatch
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import re
|
|
30
|
+
from collections import Counter
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
from .. import presets
|
|
34
|
+
|
|
35
|
+
GENERATED, COPIED, VENDORED = "generated", "copied", "vendored"
|
|
36
|
+
HEADER_BYTES = 4096 # the leading comment block is read from the first 4 KB of a file
|
|
37
|
+
MAX_PATHS = 500 # file paths stored per reason in the index stats (counts are always exact)
|
|
38
|
+
SHOW_PATHS = 5
|
|
39
|
+
|
|
40
|
+
# generator banners, matched in comment lines of the file's leading comment block
|
|
41
|
+
HEADER_MARKERS: tuple[tuple[str, re.Pattern], ...] = (
|
|
42
|
+
("@generated header", re.compile(r"(^|[\s*/#(\"'])@generated\b")),
|
|
43
|
+
("'Code generated ... DO NOT EDIT.' header", re.compile(r"\bCode generated\b.*\bDO NOT EDIT\b", re.I)),
|
|
44
|
+
("'GENERATED CODE - DO NOT MODIFY BY HAND' header", re.compile(r"GENERATED CODE\s*-\s*DO NOT MODIFY BY HAND", re.I)),
|
|
45
|
+
("<auto-generated> header", re.compile(r"<auto-generated\b", re.I)),
|
|
46
|
+
("protoc / gRPC banner", re.compile(r"\bgenerated by\b.*\b(protocol buffer compiler|protoc|grpc)\b", re.I)),
|
|
47
|
+
("OpenAPI Generator banner", re.compile(r"\bauto[- ]?generated by (the )?(OpenAPI[- ]Generator|Swagger[- ]Codegen)"
|
|
48
|
+
r"|\bGenerated by:? https?://(openapi-generator\.tech|github\.com/swagger-api/swagger-codegen)", re.I)),
|
|
49
|
+
# Swift generators (#100): Sourcery templates, SwiftGen assets / strings, swift-openapi-generator, Mockolo mocks
|
|
50
|
+
("Sourcery banner", re.compile(r"\bGenerated using Sourcery\b")),
|
|
51
|
+
("SwiftGen banner", re.compile(r"\bGenerated using SwiftGen\b")),
|
|
52
|
+
("swift-openapi-generator banner", re.compile(r"\bGenerated by swift-openapi-generator\b")),
|
|
53
|
+
("Mockolo banner", re.compile(r"@Generated by Mockolo\b")),
|
|
54
|
+
("'automatically generated' header", re.compile(r"\b(this|the) (file|code|class|module|source) (is|was|has been) "
|
|
55
|
+
r"(automatically|auto)[- ]?generated\b", re.I)),
|
|
56
|
+
)
|
|
57
|
+
COMMENT_LEADERS = ("//", "#", "/*", "*", "<!--", "--", ";", "(*", "{-")
|
|
58
|
+
SOURCE_EXTS: frozenset[str] = frozenset() # filled from coverage on first use (all source extensions)
|
|
59
|
+
|
|
60
|
+
CAPACITOR_CONFIGS = ("capacitor.config.ts", "capacitor.config.js", "capacitor.config.json", "capacitor.config.mts")
|
|
61
|
+
# where `npx cap copy` / `cap sync` writes, relative to the Capacitor app directory
|
|
62
|
+
CAPACITOR_COPY_TARGETS = ("android/app/src/main/assets/public", "ios/App/App/public")
|
|
63
|
+
CAPACITOR_SYNC_FILES = ("android/app/src/main/assets/capacitor.config.json", "android/app/src/main/assets/capacitor.plugins.json",
|
|
64
|
+
"android/app/src/main/res/xml/config.xml", "ios/App/App/capacitor.config.json", "ios/App/App/config.xml")
|
|
65
|
+
CAPACITOR_SYNC_DIRS = ("android/capacitor-cordova-android-plugins", "ios/capacitor-cordova-ios-plugins")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _source_exts() -> frozenset[str]:
|
|
69
|
+
global SOURCE_EXTS
|
|
70
|
+
if not SOURCE_EXTS:
|
|
71
|
+
from ..coverage import SUPPORTED, UNSUPPORTED
|
|
72
|
+
SOURCE_EXTS = frozenset({e for v in SUPPORTED.values() for e in v} | set(UNSUPPORTED)
|
|
73
|
+
| {".java", ".kt", ".swift", ".m", ".mm", ".h", ".go", ".cs"})
|
|
74
|
+
return SOURCE_EXTS
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def header_marker(path: str | os.PathLike) -> str | None:
|
|
78
|
+
"""Reason of the first generator banner in the file's leading comment block, None without one. Only comment lines
|
|
79
|
+
before the first line of code count, so a string or docstring that mentions a banner does not."""
|
|
80
|
+
try:
|
|
81
|
+
with open(path, "rb") as fh:
|
|
82
|
+
head = fh.read(HEADER_BYTES)
|
|
83
|
+
except OSError:
|
|
84
|
+
return None
|
|
85
|
+
if not head:
|
|
86
|
+
return None
|
|
87
|
+
text = head.decode("utf-8", "replace")
|
|
88
|
+
if text.startswith("\ufeff"):
|
|
89
|
+
text = text[1:]
|
|
90
|
+
in_block = None
|
|
91
|
+
for i, raw in enumerate(text.splitlines()):
|
|
92
|
+
line = raw.strip()
|
|
93
|
+
if not line:
|
|
94
|
+
continue
|
|
95
|
+
if i == 0 and line.startswith("#!"):
|
|
96
|
+
continue
|
|
97
|
+
if in_block:
|
|
98
|
+
comment = True
|
|
99
|
+
if in_block in line:
|
|
100
|
+
in_block = None
|
|
101
|
+
else:
|
|
102
|
+
low = line.lower()
|
|
103
|
+
comment = low.startswith(COMMENT_LEADERS)
|
|
104
|
+
if line.startswith("/*") and "*/" not in line[2:]:
|
|
105
|
+
in_block = "*/"
|
|
106
|
+
elif line.startswith("<!--") and "-->" not in line[4:]:
|
|
107
|
+
in_block = "-->"
|
|
108
|
+
elif line.startswith("{-") and "-}" not in line[2:]:
|
|
109
|
+
in_block = "-}"
|
|
110
|
+
elif line.startswith("(*") and "*)" not in line[2:]:
|
|
111
|
+
in_block = "*)"
|
|
112
|
+
if not comment:
|
|
113
|
+
# `package foo;` / `<?php` / `import` ... ends the header; PHP's opening tag is still part of it
|
|
114
|
+
if line.startswith("<?php") or line == "<?":
|
|
115
|
+
continue
|
|
116
|
+
return None
|
|
117
|
+
for reason, rx in HEADER_MARKERS:
|
|
118
|
+
if rx.search(line):
|
|
119
|
+
return reason
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# ------------------------------------------------------------------------------------------- .gitattributes
|
|
124
|
+
def _attr_regex(pattern: str) -> re.Pattern:
|
|
125
|
+
"""gitattributes pattern -> regex over the path relative to the .gitattributes directory (git semantics: no
|
|
126
|
+
recursive match below a matched directory; use `dir/**`)."""
|
|
127
|
+
from .paths import glob_regex
|
|
128
|
+
return re.compile(glob_regex(pattern, below=False))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def parse_gitattributes(text: str) -> list[tuple[re.Pattern, str, bool | None]]:
|
|
132
|
+
"""(pattern regex, 'generated' | 'vendored', True / False / None) per linguist attribute of each line."""
|
|
133
|
+
out = []
|
|
134
|
+
for raw in text.splitlines():
|
|
135
|
+
line = raw.strip()
|
|
136
|
+
if not line or line.startswith("#") or line.startswith("[attr]"):
|
|
137
|
+
continue
|
|
138
|
+
parts = line.split()
|
|
139
|
+
pat, attrs = parts[0], parts[1:]
|
|
140
|
+
if pat.startswith('"'):
|
|
141
|
+
continue # quoted patterns (C-style escapes): rare, not read
|
|
142
|
+
if pat.endswith("/"):
|
|
143
|
+
continue # a directory pattern never matches files in an attributes file
|
|
144
|
+
rx = None
|
|
145
|
+
for a in attrs:
|
|
146
|
+
val: bool | None
|
|
147
|
+
name = a
|
|
148
|
+
if a.startswith("-"):
|
|
149
|
+
name, val = a[1:], False
|
|
150
|
+
elif a.startswith("!"):
|
|
151
|
+
name, val = a[1:], None
|
|
152
|
+
elif "=" in a:
|
|
153
|
+
name, v = a.split("=", 1)
|
|
154
|
+
val = v.strip().lower() not in ("false", "0", "no", "off")
|
|
155
|
+
else:
|
|
156
|
+
val = True
|
|
157
|
+
kind = {"linguist-generated": GENERATED, "linguist-vendored": VENDORED}.get(name)
|
|
158
|
+
if kind:
|
|
159
|
+
rx = rx or _attr_regex(pat)
|
|
160
|
+
out.append((rx, kind, val))
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ------------------------------------------------------------------------------------------- copy targets
|
|
165
|
+
_WEBDIR_RE = re.compile(r"""\bwebDir\s*:\s*(['"`])([^'"`]+)\1""")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def capacitor_web_dir(cfg_file: Path) -> str | None:
|
|
169
|
+
try:
|
|
170
|
+
text = cfg_file.read_text(encoding="utf-8", errors="replace")
|
|
171
|
+
except OSError:
|
|
172
|
+
return None
|
|
173
|
+
if cfg_file.suffix == ".json":
|
|
174
|
+
try:
|
|
175
|
+
v = (json.loads(text) or {}).get("webDir")
|
|
176
|
+
except (ValueError, AttributeError):
|
|
177
|
+
v = None
|
|
178
|
+
else:
|
|
179
|
+
m = _WEBDIR_RE.search(text)
|
|
180
|
+
v = m.group(2) if m else None
|
|
181
|
+
if not isinstance(v, str) or not v.strip():
|
|
182
|
+
return None
|
|
183
|
+
v = v.strip().replace("\\", "/").strip("/")
|
|
184
|
+
if v.startswith("./"):
|
|
185
|
+
v = v[2:]
|
|
186
|
+
return v or None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _ignored_by(dir_: Path, name: str) -> bool:
|
|
190
|
+
"""`name` (a directory of `dir_`) is listed in `dir_/.gitignore` (plain names only: `dist`, `/dist`, `dist/`)."""
|
|
191
|
+
try:
|
|
192
|
+
lines = (dir_ / ".gitignore").read_text(encoding="utf-8", errors="replace").splitlines()
|
|
193
|
+
except OSError:
|
|
194
|
+
return False
|
|
195
|
+
return any(ln.strip().strip("/") == name for ln in lines if ln.strip() and not ln.startswith("#"))
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _bundler_output(dir_: Path, name: str) -> str | None:
|
|
199
|
+
"""The bundler config of `dir_` that writes into `name` (Angular's angular.json outputPath), else None."""
|
|
200
|
+
try:
|
|
201
|
+
text = (dir_ / "angular.json").read_text(encoding="utf-8", errors="replace")
|
|
202
|
+
except OSError:
|
|
203
|
+
return None
|
|
204
|
+
rx = r'"outputPath"\s*:\s*(?:\{[^}]*?"base"\s*:\s*)?"(?:\./)?' + re.escape(name) + r'(?:/[^"]*)?"'
|
|
205
|
+
return "angular.json outputPath" if re.search(rx, text) else None
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
class Hit(dict):
|
|
209
|
+
"""{"kind": generated | copied | vendored, "reason": ..., optionally "copy_of": source path, "test": True}."""
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# test folders and test-target directories (#100): a generated file there runs as a test or test support (Sourcery's
|
|
213
|
+
# preview / accessibility test lists, generated test stubs), so by default it stays in the graph as test code, with
|
|
214
|
+
# attrs.generated; only generated non-test sources leave it. `.cg.yaml` generated.paths still excludes anything.
|
|
215
|
+
TEST_DIR = re.compile(r"(?:^|/)(?:tests?|Tests?|__tests__|specs?|androidTest|testFixtures|integrationTest|unitTest|"
|
|
216
|
+
r"[^/]*(?:Tests|UITests))/")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _testy(rel: str, h: "Hit | None") -> "Hit | None":
|
|
220
|
+
if h is not None and h.get("kind") == GENERATED and not h["reason"].startswith(".cg.yaml") and TEST_DIR.search(rel):
|
|
221
|
+
h = Hit(h, test=True)
|
|
222
|
+
return h
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class Classifier:
|
|
226
|
+
"""Classification of one project's files (see the module docstring). `scan(rel, abs)` classifies a file the
|
|
227
|
+
coverage scan walks (reads its header); `lookup(rel)` answers for any repo-relative path from what the scan
|
|
228
|
+
recorded plus the path rules."""
|
|
229
|
+
|
|
230
|
+
def __init__(self, root: str | Path, cfg: dict | None = None, include: bool = False):
|
|
231
|
+
from .paths import glob_regex
|
|
232
|
+
self.root = Path(root)
|
|
233
|
+
cfg = cfg or {}
|
|
234
|
+
g = cfg.get("generated") or {}
|
|
235
|
+
self.include = bool(include or g.get("include"))
|
|
236
|
+
self.keep = re.compile("|".join(f"(?:{glob_regex(p)})" for p in g["keep"])) if g.get("keep") else None
|
|
237
|
+
self.user: list[tuple[re.Pattern, Hit]] = []
|
|
238
|
+
for key, kind in (("paths", GENERATED), ("vendored", VENDORED)):
|
|
239
|
+
for p in g.get(key) or []:
|
|
240
|
+
self.user.append((re.compile(glob_regex(p)), Hit(kind=kind, reason=f".cg.yaml generated.{key}")))
|
|
241
|
+
pre = presets.values("common", "generated", default={}) or {}
|
|
242
|
+
self.build_dirs: dict[str, str] = dict(pre.get("build_dirs") or {})
|
|
243
|
+
self.file_globs: list[tuple[str, str]] = list((pre.get("files") or {}).items())
|
|
244
|
+
self.tool_paths = [(re.compile(glob_regex(g)), r) for g, r in (pre.get("paths") or {}).items()]
|
|
245
|
+
self._exact_names = {n: r for n, r in self.file_globs if not any(c in n for c in "*?[")}
|
|
246
|
+
self._name_globs = [(n, r) for n, r in self.file_globs if any(c in n for c in "*?[")]
|
|
247
|
+
self.attrs: list[tuple[str, list]] = [] # (base dir, rules) in walk order (parents first)
|
|
248
|
+
self.copies: list[dict] = [] # {"target", "source", "via", "kind"}
|
|
249
|
+
self.listed: dict[str, str] = {} # rel -> reason (.openapi-generator/FILES)
|
|
250
|
+
self.files: dict[str, Hit] = {} # classified files the scan saw
|
|
251
|
+
self.pruned: dict[str, str] = {} # build output directories the scan did not descend into
|
|
252
|
+
self._cache: dict[str, Hit | None] = {}
|
|
253
|
+
|
|
254
|
+
# -------------------------------------------------- per directory (called by the scan, parents first)
|
|
255
|
+
def visit_dir(self, rel_dir: str, dp: str, dns: list[str], fns: list[str]) -> None:
|
|
256
|
+
names = set(fns)
|
|
257
|
+
if ".gitattributes" in names:
|
|
258
|
+
try:
|
|
259
|
+
txt = Path(dp, ".gitattributes").read_text(encoding="utf-8", errors="replace")
|
|
260
|
+
except OSError:
|
|
261
|
+
txt = ""
|
|
262
|
+
rules = parse_gitattributes(txt)
|
|
263
|
+
if rules:
|
|
264
|
+
self.attrs.append((rel_dir, rules))
|
|
265
|
+
pre = f"{rel_dir}/" if rel_dir else ""
|
|
266
|
+
for cf in CAPACITOR_CONFIGS:
|
|
267
|
+
if cf in names:
|
|
268
|
+
web = capacitor_web_dir(Path(dp, cf))
|
|
269
|
+
src = f"{pre}{web}" if web else None
|
|
270
|
+
via = f"Capacitor webDir {web!r} ({pre}{cf})" if web else f"Capacitor ({pre}{cf})"
|
|
271
|
+
for t in CAPACITOR_COPY_TARGETS:
|
|
272
|
+
self.copies.append({"target": pre + t, "source": src, "via": via, "kind": COPIED,
|
|
273
|
+
"reason": f"copy of {web}/ (Capacitor webDir)" if web else "Capacitor web asset copy"})
|
|
274
|
+
for t in CAPACITOR_SYNC_DIRS:
|
|
275
|
+
self.copies.append({"target": pre + t, "source": None, "via": f"Capacitor sync ({pre}{cf})",
|
|
276
|
+
"kind": GENERATED, "reason": "Capacitor sync output"})
|
|
277
|
+
for f in CAPACITOR_SYNC_FILES:
|
|
278
|
+
self.listed[pre + f] = "Capacitor sync output"
|
|
279
|
+
why = None
|
|
280
|
+
if web and _ignored_by(Path(dp), web.split("/")[0]):
|
|
281
|
+
why = f"listed in {pre}.gitignore"
|
|
282
|
+
elif web:
|
|
283
|
+
why = _bundler_output(Path(dp), web)
|
|
284
|
+
if why:
|
|
285
|
+
self.copies.append({"target": pre + web, "source": None, "via": f"Capacitor webDir, {why}",
|
|
286
|
+
"kind": GENERATED, "reason": "web build output (Capacitor webDir)"})
|
|
287
|
+
break
|
|
288
|
+
if "config.xml" in names and "www" in dns and "platforms" in dns:
|
|
289
|
+
try:
|
|
290
|
+
is_cordova = "<widget" in Path(dp, "config.xml").read_text(encoding="utf-8", errors="replace")[:4096]
|
|
291
|
+
except OSError:
|
|
292
|
+
is_cordova = False
|
|
293
|
+
if is_cordova:
|
|
294
|
+
for plat in sorted(os.listdir(os.path.join(dp, "platforms"))):
|
|
295
|
+
for www in ("www", "app/src/main/assets/www", "platform_www"):
|
|
296
|
+
if os.path.isdir(os.path.join(dp, "platforms", plat, www)):
|
|
297
|
+
self.copies.append({"target": f"{pre}platforms/{plat}/{www}", "source": f"{pre}www",
|
|
298
|
+
"via": f"Cordova ({pre}config.xml)", "kind": COPIED,
|
|
299
|
+
"reason": "copy of www/ (Cordova platform)"})
|
|
300
|
+
if ".openapi-generator" in dns and os.path.isfile(os.path.join(dp, ".openapi-generator", "FILES")):
|
|
301
|
+
try:
|
|
302
|
+
for ln in Path(dp, ".openapi-generator", "FILES").read_text(encoding="utf-8", errors="replace").splitlines():
|
|
303
|
+
ln = ln.strip().strip("/")
|
|
304
|
+
if ln and not ln.startswith("#") and ".." not in ln.split("/"):
|
|
305
|
+
self.listed.setdefault(pre + ln, "OpenAPI Generator (.openapi-generator/FILES)")
|
|
306
|
+
except OSError:
|
|
307
|
+
pass
|
|
308
|
+
|
|
309
|
+
def skipped_dir(self, rel: str, name: str) -> None:
|
|
310
|
+
"""The scan does not descend into `rel`: recorded when it is a framework build output directory."""
|
|
311
|
+
r = self.build_dirs.get(name)
|
|
312
|
+
if r:
|
|
313
|
+
self.pruned.setdefault(rel, r)
|
|
314
|
+
|
|
315
|
+
# -------------------------------------------------- classification
|
|
316
|
+
def _attr(self, rel: str) -> Hit | None | bool:
|
|
317
|
+
"""Hit from .gitattributes, False when a linguist attribute marks the file as hand-written, None otherwise."""
|
|
318
|
+
state: dict[str, bool | None] = {}
|
|
319
|
+
for base, rules in self.attrs:
|
|
320
|
+
if base and not rel.startswith(base + "/"):
|
|
321
|
+
continue
|
|
322
|
+
sub = rel[len(base) + 1:] if base else rel
|
|
323
|
+
for rx, kind, val in rules:
|
|
324
|
+
if rx.match(sub):
|
|
325
|
+
state[kind] = val
|
|
326
|
+
if state.get(VENDORED):
|
|
327
|
+
return Hit(kind=VENDORED, reason="linguist-vendored (.gitattributes)")
|
|
328
|
+
if state.get(GENERATED):
|
|
329
|
+
return Hit(kind=GENERATED, reason="linguist-generated (.gitattributes)")
|
|
330
|
+
if state.get(GENERATED) is False or state.get(VENDORED) is False:
|
|
331
|
+
return False
|
|
332
|
+
return None
|
|
333
|
+
|
|
334
|
+
def _copy(self, rel: str) -> Hit | None:
|
|
335
|
+
best = None
|
|
336
|
+
for c in self.copies:
|
|
337
|
+
t = c["target"]
|
|
338
|
+
if rel == t or rel.startswith(t + "/"):
|
|
339
|
+
if best is None or len(t) > len(best["target"]):
|
|
340
|
+
best = c
|
|
341
|
+
if best is None:
|
|
342
|
+
return None
|
|
343
|
+
h = Hit(kind=best["kind"], reason=best["reason"])
|
|
344
|
+
if best["source"] and rel != best["target"]:
|
|
345
|
+
h["copy_of"] = best["source"] + rel[len(best["target"]):]
|
|
346
|
+
return h
|
|
347
|
+
|
|
348
|
+
def _path_rule(self, rel: str) -> Hit | None:
|
|
349
|
+
parts = rel.split("/")
|
|
350
|
+
for p in parts[:-1]:
|
|
351
|
+
# hidden framework directories at any depth; `dist/` / `build/` are only reported (every walk skips them by
|
|
352
|
+
# name, and a source package can be named `build`)
|
|
353
|
+
r = self.build_dirs.get(p) if p.startswith(".") else None
|
|
354
|
+
if r:
|
|
355
|
+
return Hit(kind=GENERATED, reason=r)
|
|
356
|
+
for rx, r in self.tool_paths:
|
|
357
|
+
if rx.match(rel):
|
|
358
|
+
return Hit(kind=GENERATED, reason=r)
|
|
359
|
+
name = parts[-1]
|
|
360
|
+
r = self._exact_names.get(name)
|
|
361
|
+
if r is None:
|
|
362
|
+
for g, rr in self._name_globs:
|
|
363
|
+
if fnmatch.fnmatchcase(name, g):
|
|
364
|
+
r = rr
|
|
365
|
+
break
|
|
366
|
+
return Hit(kind=GENERATED, reason=r) if r else None
|
|
367
|
+
|
|
368
|
+
def rules_hit(self, rel: str) -> Hit | None:
|
|
369
|
+
"""Classification from everything but the file header (cheap; no file read)."""
|
|
370
|
+
if self.keep and self.keep.match(rel):
|
|
371
|
+
return None
|
|
372
|
+
for rx, h in self.user:
|
|
373
|
+
if rx.match(rel):
|
|
374
|
+
return h
|
|
375
|
+
a = self._attr(rel)
|
|
376
|
+
if a is False:
|
|
377
|
+
return None
|
|
378
|
+
if a:
|
|
379
|
+
return _testy(rel, a)
|
|
380
|
+
h = self._copy(rel)
|
|
381
|
+
if h:
|
|
382
|
+
return h
|
|
383
|
+
if rel in self.listed:
|
|
384
|
+
return _testy(rel, Hit(kind=GENERATED, reason=self.listed[rel]))
|
|
385
|
+
h = self._path_rule(rel)
|
|
386
|
+
# build output directories (.nuxt/, .next/ ...) are never test code
|
|
387
|
+
return h if h is None or h["reason"] in self.build_dirs.values() else _testy(rel, h)
|
|
388
|
+
|
|
389
|
+
def scan(self, rel: str, abs_path: str) -> Hit | None:
|
|
390
|
+
"""Classify one file the coverage scan walks: the path rules, then the header of a source file."""
|
|
391
|
+
h = self.rules_hit(rel)
|
|
392
|
+
if h is None and not (self.keep and self.keep.match(rel)) and self._attr(rel) is not False:
|
|
393
|
+
ext = os.path.splitext(rel)[1].lower()
|
|
394
|
+
if ext in _source_exts():
|
|
395
|
+
r = header_marker(abs_path)
|
|
396
|
+
if r:
|
|
397
|
+
h = _testy(rel, Hit(kind=GENERATED, reason=r))
|
|
398
|
+
if h is not None:
|
|
399
|
+
self.files[rel] = h
|
|
400
|
+
return h
|
|
401
|
+
|
|
402
|
+
def lookup(self, rel: str) -> Hit | None:
|
|
403
|
+
"""Classification of a repo-relative path: what the scan recorded, else the path rules."""
|
|
404
|
+
h = self.files.get(rel)
|
|
405
|
+
if h is not None:
|
|
406
|
+
return h
|
|
407
|
+
if rel not in self._cache:
|
|
408
|
+
self._cache[rel] = self.rules_hit(rel)
|
|
409
|
+
return self._cache[rel]
|
|
410
|
+
|
|
411
|
+
def drops(self, h: Hit | None) -> bool:
|
|
412
|
+
"""A classified file the default mode leaves out of the graph (a generated test file stays in)."""
|
|
413
|
+
return h is not None and not self.include and not h.get("test")
|
|
414
|
+
|
|
415
|
+
def excludes(self, rel: str) -> bool:
|
|
416
|
+
return self.drops(self.lookup(rel))
|
|
417
|
+
|
|
418
|
+
def dir_excluded(self, rel_dir: str) -> bool:
|
|
419
|
+
"""A directory every file of which is classified (a copy target, a build output directory, a user glob)."""
|
|
420
|
+
if self.include or (self.keep and self.keep.match(rel_dir + "/")):
|
|
421
|
+
return False
|
|
422
|
+
if any(rx.match(rel_dir + "/") for rx, _ in self.user):
|
|
423
|
+
return True
|
|
424
|
+
if any(rel_dir == c["target"] or rel_dir.startswith(c["target"] + "/") for c in self.copies):
|
|
425
|
+
return True
|
|
426
|
+
name = rel_dir.rsplit("/", 1)[-1]
|
|
427
|
+
return name.startswith(".") and name in self.build_dirs
|
|
428
|
+
|
|
429
|
+
def dir_regexes(self) -> list[str]:
|
|
430
|
+
"""JavaScript-compatible regexes of the directories excluded as a whole (copy targets, Capacitor sync output,
|
|
431
|
+
.cg.yaml generated globs); framework build directories are left to each extractor (Nuxt reads `.nuxt/`)."""
|
|
432
|
+
if self.include:
|
|
433
|
+
return []
|
|
434
|
+
out = [f"(?:^{re.escape(c['target'])}(?:/.*)?$)" for c in self.copies]
|
|
435
|
+
if not self.keep: # with keep globs the scan's explicit file list carries the .cg.yaml rules
|
|
436
|
+
out += [f"(?:{rx.pattern})" for rx, _ in self.user]
|
|
437
|
+
return out
|
|
438
|
+
|
|
439
|
+
# -------------------------------------------------- reporting
|
|
440
|
+
def summary(self) -> dict:
|
|
441
|
+
"""The coverage entry: counts by reason and kind, sample paths per reason, copy targets, build directories."""
|
|
442
|
+
by_reason, by_kind, kept = Counter(), Counter(), Counter()
|
|
443
|
+
paths: dict[str, list[str]] = {}
|
|
444
|
+
kept_paths: dict[str, list[str]] = {}
|
|
445
|
+
for rel in sorted(self.files):
|
|
446
|
+
h = self.files[rel]
|
|
447
|
+
if h.get("test") and not self.include:
|
|
448
|
+
kept[h["reason"]] += 1
|
|
449
|
+
kp = kept_paths.setdefault(h["reason"], [])
|
|
450
|
+
if len(kp) < MAX_PATHS:
|
|
451
|
+
kp.append(rel)
|
|
452
|
+
continue
|
|
453
|
+
by_reason[h["reason"]] += 1
|
|
454
|
+
by_kind[h["kind"]] += 1
|
|
455
|
+
ps = paths.setdefault(h["reason"], [])
|
|
456
|
+
if len(ps) < MAX_PATHS:
|
|
457
|
+
ps.append(rel)
|
|
458
|
+
copies = []
|
|
459
|
+
for c in self.copies:
|
|
460
|
+
n = sum(1 for rel in self.files if rel == c["target"] or rel.startswith(c["target"] + "/"))
|
|
461
|
+
if n:
|
|
462
|
+
copies.append({"target": c["target"], "source": c["source"], "via": c["via"], "files": n, "reason": c["reason"]})
|
|
463
|
+
out = {"mode": "indexed" if self.include else "excluded", "files": sum(by_reason.values()),
|
|
464
|
+
"by_reason": dict(by_reason.most_common()), "by_kind": dict(by_kind.most_common()), "paths": paths}
|
|
465
|
+
if copies:
|
|
466
|
+
out["copies"] = copies
|
|
467
|
+
if kept:
|
|
468
|
+
out["tests_kept"] = {"files": sum(kept.values()), "by_reason": dict(kept.most_common()), "paths": kept_paths}
|
|
469
|
+
if self.pruned:
|
|
470
|
+
out["build_dirs"] = dict(sorted(self.pruned.items())[:MAX_PATHS])
|
|
471
|
+
return out
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def for_project(project) -> Classifier | None:
|
|
475
|
+
return (getattr(project, "options", None) or {}).get("generated")
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def summary_text(g: dict | None) -> str:
|
|
479
|
+
"""'generated: 412 files excluded (copy of dist/ (Capacitor webDir) 380, linguist-generated 21, @generated header 11)'."""
|
|
480
|
+
if not g or not (g.get("files") or g.get("build_dirs") or g.get("tests_kept")):
|
|
481
|
+
return ""
|
|
482
|
+
parts = ", ".join(f"{r} {n}" for r, n in (g.get("by_reason") or {}).items())
|
|
483
|
+
s = f"generated: {g['files']} file{'s' if g['files'] != 1 else ''} {g.get('mode', 'excluded')}" + (f" ({parts})" if parts else "")
|
|
484
|
+
tk = g.get("tests_kept") or {}
|
|
485
|
+
if tk.get("files"):
|
|
486
|
+
s += f"; {tk['files']} generated test file{'s' if tk['files'] != 1 else ''} indexed as tests"
|
|
487
|
+
if g.get("build_dirs"):
|
|
488
|
+
s += f"; build output not scanned: {', '.join(d + '/' for d in list(g['build_dirs'])[:4])}" + (
|
|
489
|
+
f" … +{len(g['build_dirs']) - 4}" if len(g["build_dirs"]) > 4 else "")
|
|
490
|
+
return s
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def detail_lines(g: dict | None, all_files: bool = False, indent: str = " ") -> list[str]:
|
|
494
|
+
if not g or not (g.get("files") or g.get("build_dirs") or g.get("tests_kept")):
|
|
495
|
+
return []
|
|
496
|
+
out = [indent + summary_text(g)]
|
|
497
|
+
for c in g.get("copies") or []:
|
|
498
|
+
what = f"a copy of {c['source']}/" if c.get("source") else c.get("reason", "generated")
|
|
499
|
+
out.append(f"{indent} {c['target']}/ ({c['files']} file{'s' if c['files'] != 1 else ''}) is {what} [{c['via']}]")
|
|
500
|
+
for reason, ps in (g.get("paths") or {}).items():
|
|
501
|
+
n = (g.get("by_reason") or {}).get(reason, len(ps))
|
|
502
|
+
show = ps if all_files else ps[:SHOW_PATHS]
|
|
503
|
+
more = n - len(show)
|
|
504
|
+
out.append(f"{indent} {reason}: " + ", ".join(show) + (f" … +{more} more (--all-files)" if more > 0 else ""))
|
|
505
|
+
tk = g.get("tests_kept") or {}
|
|
506
|
+
for reason, ps in (tk.get("paths") or {}).items():
|
|
507
|
+
n = (tk.get("by_reason") or {}).get(reason, len(ps))
|
|
508
|
+
show = ps if all_files else ps[:SHOW_PATHS]
|
|
509
|
+
more = n - len(show)
|
|
510
|
+
out.append(f"{indent} {reason}, indexed as test code: " + ", ".join(show)
|
|
511
|
+
+ (f" … +{more} more (--all-files)" if more > 0 else ""))
|
|
512
|
+
return out
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def apply(builder, clf: Classifier) -> dict:
|
|
516
|
+
"""Final pass over the graph: nodes whose file is classified leave the graph with their edges (default), or carry
|
|
517
|
+
`attrs.generated` (included); an included copy gets a COPY_OF edge to the module node of its source file."""
|
|
518
|
+
from .plugin import gc_paused
|
|
519
|
+
by_file: dict[str, Hit | None] = {}
|
|
520
|
+
drop: set[str] = set()
|
|
521
|
+
tagged = 0
|
|
522
|
+
copies = []
|
|
523
|
+
with gc_paused():
|
|
524
|
+
for nid, n in builder.nodes.items():
|
|
525
|
+
f = n.file
|
|
526
|
+
if not f:
|
|
527
|
+
continue
|
|
528
|
+
h = by_file.get(f, 0)
|
|
529
|
+
if h == 0:
|
|
530
|
+
h = by_file[f] = clf.lookup(f)
|
|
531
|
+
if h is None:
|
|
532
|
+
continue
|
|
533
|
+
if clf.include or h.get("test"):
|
|
534
|
+
n.attrs = {**(n.attrs or {}), "generated": dict(h)}
|
|
535
|
+
tagged += 1
|
|
536
|
+
if h.get("copy_of") and n.kind == "module":
|
|
537
|
+
copies.append((nid, n.file, h["copy_of"]))
|
|
538
|
+
else:
|
|
539
|
+
drop.add(nid)
|
|
540
|
+
edges_dropped = 0
|
|
541
|
+
if drop:
|
|
542
|
+
for nid in drop:
|
|
543
|
+
del builder.nodes[nid]
|
|
544
|
+
dead = [k for k, e in builder.edges.items() if e.src in drop or e.dst in drop]
|
|
545
|
+
for k in dead:
|
|
546
|
+
del builder.edges[k]
|
|
547
|
+
edges_dropped = len(dead)
|
|
548
|
+
linked = 0
|
|
549
|
+
if copies:
|
|
550
|
+
mods = {n.file: nid for nid, n in builder.nodes.items() if n.kind == "module" and n.file}
|
|
551
|
+
for nid, f, src in copies:
|
|
552
|
+
dst = mods.get(src)
|
|
553
|
+
if dst and dst != nid:
|
|
554
|
+
builder.add_edge(nid, dst, "COPY_OF", file=f, line=1)
|
|
555
|
+
linked += 1
|
|
556
|
+
with_nodes = sum(1 for h in by_file.values() if h)
|
|
557
|
+
out = {**classified_counts(clf), "mode": "indexed" if clf.include else "excluded"}
|
|
558
|
+
if clf.include:
|
|
559
|
+
out.update(files_with_nodes=with_nodes, nodes_labelled=tagged, copy_of_edges=linked)
|
|
560
|
+
else:
|
|
561
|
+
out.update(files_with_dropped_nodes=with_nodes, nodes_dropped=len(drop), edges_dropped=edges_dropped)
|
|
562
|
+
return out
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def classified_counts(clf: Classifier) -> dict:
|
|
566
|
+
"""The files the classifier labelled (the up-front scan, the same list `cg coverage` shows): the total, per
|
|
567
|
+
language (coverage language keys; `other` for non-source files) and per reason / kind."""
|
|
568
|
+
from ..coverage import UNSUPPORTED, _lang_of_file
|
|
569
|
+
by_lang, by_reason, by_kind = Counter(), Counter(), Counter()
|
|
570
|
+
for rel, h in clf.files.items():
|
|
571
|
+
by_lang[_lang_of_file(rel) or UNSUPPORTED.get(os.path.splitext(rel)[1].lower()) or "other"] += 1
|
|
572
|
+
by_reason[h["reason"]] += 1
|
|
573
|
+
by_kind[h["kind"]] += 1
|
|
574
|
+
return {"files": len(clf.files), "by_language": dict(by_lang.most_common()), "by_reason": dict(by_reason.most_common()),
|
|
575
|
+
"by_kind": dict(by_kind.most_common())}
|