sourcecode 4.5.3__py3-none-any.whl → 4.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of sourcecode might be problematic. Click here for more details.

sourcecode/__init__.py CHANGED
@@ -4,4 +4,4 @@ ASK Engine is the product. ``ask`` is the canonical CLI command; ``sourcecode``
4
4
  the legacy compatibility alias and the Python/PyPI package name. See
5
5
  docs/PRODUCT_IDENTITY.md (normative)."""
6
6
 
7
- __version__ = "4.5.3"
7
+ __version__ = "4.6.0"
sourcecode/archetype.py CHANGED
@@ -86,6 +86,14 @@ _MIN_CONFIDENT_SCORE = 0.5
86
86
  # discriminating. Tuned so a clear library (no app/server/servlet entry) is not confidently
87
87
  # labeled an engine, while runnable systems (which never trigger the gate) are untouched.
88
88
  _LIBRARY_ENGINE_DISCOUNT = 0.55
89
+ # The same shape for the opposite mistake (C3-43): a repository whose primary
90
+ # surface is HTTP owns storage/kernel packages and hub topology *because* it
91
+ # serves requests with them, so that evidence is not discriminating for "engine"
92
+ # either. Scaled by the measured density, so a repository with a marginal surface
93
+ # is barely discounted and one that is nothing but endpoints is discounted fully.
94
+ # Below the library figure on purpose: an application CAN be built around an
95
+ # engine, whereas a non-runnable artifact cannot be an engine product at all.
96
+ _HTTP_ENGINE_DISCOUNT = 0.45
89
97
  # Entry kinds that make a repo a runnable system (an app/server/servlet/daemon) rather
90
98
  # than a consumable library. Used to gate the library de-bias (P0).
91
99
  _RUNNABLE_ENTRY_KINDS = frozenset({
@@ -409,6 +417,42 @@ class ArchetypeClassifier:
409
417
  add("application", "runtime_entry_plus_services",
410
418
  f"bootstrap/server entry + service mass ≈ {cm['service']:.0%}",
411
419
  2.0, 1.0, max(cm["service"], 0.2))
420
+ # C3-43. The strongest MEASURED discriminator this build has for what a
421
+ # repository is — its HTTP surface — was read by `_primary_interface` and
422
+ # ignored here, so a 3 574-endpoint CRUD REST monolith was labelled
423
+ # `engine` on package-name masses and a fan-in Gini, while the legacy
424
+ # classifier it supersedes answered `api` and was right. Same ramp as the
425
+ # interface dimension, from the same features: one authority for the
426
+ # figure, two dimensions reading it.
427
+ if f.endpoint_share is None:
428
+ # Unmeasured is not zero: contribute nothing, and say so.
429
+ add("application", "http_surface_density",
430
+ "endpoint surface not measured — this dimension is scored without it",
431
+ 2.5, 0.0, 0.0)
432
+ else:
433
+ density = f.endpoint_share * 100
434
+ http_strength = (
435
+ max(0.0, min(1.0, (density - 0.5) / 1.5)) if f.endpoint_total else 0.0
436
+ )
437
+ add("application", "http_surface_density",
438
+ f"{f.endpoint_total} endpoints over {f.total_files} files "
439
+ f"({density:.2f}/100 files) — the repository's primary surface is HTTP",
440
+ 2.5, http_strength, 1.0)
441
+ # And the same fact as NAMED negative evidence for `engine`, the way
442
+ # the library de-bias below does it: a codebase whose product is a
443
+ # request surface owns storage and kernel packages *because* it is an
444
+ # application, so that mass is not, alone, evidence of an engine.
445
+ engine_positive_http = sum(max(0.0, e.contribution) for e in cands["engine"])
446
+ if http_strength > 0 and engine_positive_http > 0:
447
+ cands["engine"].append(Evidence(
448
+ "http_surface_discount",
449
+ f"{f.endpoint_total} endpoints over {f.total_files} files: the "
450
+ "primary surface of this repository is HTTP, and the storage/"
451
+ "kernel mass and hub topology an application needs to serve it "
452
+ "are not, alone, evidence of an engine product",
453
+ 1.0, 1.0, 1.0,
454
+ -engine_positive_http * _HTTP_ENGINE_DISCOUNT * http_strength,
455
+ ))
412
456
  # platform: many modules + multiple concept clusters + distribution
413
457
  multi = 1.0 if f.module_count >= 8 else f.module_count / 8
414
458
  add("platform", "many_modules_multi_concern",
@@ -210,7 +210,19 @@ def worktree_dirty(
210
210
  return None
211
211
  if out.returncode != 0:
212
212
  return None
213
- lines = [ln for ln in out.stdout.splitlines() if ln.strip()]
213
+ # C1-27. "Dirty" means the tree *the analysis reads* differs from HEAD. A
214
+ # change under an editor or agent state directory is not one — no analyser
215
+ # opens those files — and counting it reported STALE on a snapshot that
216
+ # described the tree exactly, with `Delta: 0` beside it. One authority answers
217
+ # which paths qualify (`path_filters.analysis_reads`) and its default is
218
+ # inclusive: anything not named there still counts, because a missed change
219
+ # is a false *fresh* and a spurious one only costs a rebuild.
220
+ from sourcecode.path_filters import analysis_reads
221
+
222
+ lines = [
223
+ ln for ln in out.stdout.splitlines()
224
+ if ln.strip() and analysis_reads(_porcelain_path(ln))
225
+ ]
214
226
  if ignore is None:
215
227
  return bool(lines)
216
228
  candidates = [ignore] if isinstance(ignore, (str, Path)) else list(ignore)
sourcecode/cache.py CHANGED
@@ -213,6 +213,14 @@ def _untracked_tree_fingerprint(target: Path) -> str:
213
213
  return f"ff{h.hexdigest()[:14]}"
214
214
 
215
215
 
216
+ def _porcelain_status_path(line: str) -> str:
217
+ """The path a `git status --porcelain` line refers to (destination on renames)."""
218
+ body = line[3:] if len(line) > 3 else ""
219
+ if " -> " in body:
220
+ body = body.split(" -> ", 1)[1]
221
+ return body.strip().strip('"')
222
+
223
+
216
224
  def worktree_signature(repo_root: Path, scope: Optional[Path] = None) -> str:
217
225
  """Hex fingerprint of the **exact tree state an analysis would read**.
218
226
 
@@ -252,12 +260,30 @@ def worktree_signature(repo_root: Path, scope: Optional[Path] = None) -> str:
252
260
  porcelain = st.stdout if st.returncode == 0 else ""
253
261
  except Exception:
254
262
  porcelain = ""
263
+ # C1-27. The signature describes *the tree an analysis would read*, so a line
264
+ # about a file no analyser opens does not belong in it — an editor settings
265
+ # file was invalidating every cached answer and publishing STALE beside
266
+ # `Delta: 0`. One authority decides which paths qualify, and its default is
267
+ # inclusive: anything not named there still invalidates.
268
+ from sourcecode.path_filters import analysis_reads
269
+
270
+ porcelain = "\n".join(
271
+ ln for ln in porcelain.splitlines()
272
+ if ln.strip() and analysis_reads(_porcelain_status_path(ln))
273
+ )
255
274
  if not porcelain.strip():
256
275
  return head # clean — HEAD fully describes the tree
257
276
 
258
277
  try:
278
+ from sourcecode.path_filters import _TOOL_STATE_DIRS
279
+
280
+ # The same exclusion, expressed to git for the half that is not a list of
281
+ # paths. A tracked editor-state file must not enter the signature through
282
+ # the diff after being kept out of the status.
283
+ _diff_spec = list(pathspec) or ["--", "."]
284
+ _diff_spec += [f":(exclude,glob)**/{d}/**" for d in sorted(_TOOL_STATE_DIRS)]
259
285
  df = subprocess.run(
260
- ["git", "-C", str(root), "diff", "HEAD", *pathspec],
286
+ ["git", "-C", str(root), "diff", "HEAD", *_diff_spec],
261
287
  capture_output=True, text=True, timeout=20,
262
288
  )
263
289
  diff_txt = df.stdout if df.returncode == 0 else ""
sourcecode/cli.py CHANGED
@@ -8853,7 +8853,9 @@ def explain_cmd(
8853
8853
  except Exception:
8854
8854
  cir = ContextGraph.build(file_list, target).cir # fallback: never break explain
8855
8855
  model = SpringSemanticModel.build(cir)
8856
- explanation = explain_class(class_name, cir, model).capped(limit)
8856
+ # `root` is what lets the security section read the request-chain rules a
8857
+ # configuration class declares in a DSL rather than in annotations (C3-41).
8858
+ explanation = explain_class(class_name, cir, model, root=target).capped(limit)
8857
8859
  finally:
8858
8860
  _prog.finish()
8859
8861
  if _cc_look is not None:
sourcecode/explain.py CHANGED
@@ -15,6 +15,7 @@ are transitional and tracked for a later phase.
15
15
  from __future__ import annotations
16
16
 
17
17
  from dataclasses import dataclass, field
18
+ from pathlib import Path
18
19
  from typing import TYPE_CHECKING, Optional
19
20
 
20
21
  from sourcecode.caller_metrics import CALLER_METRIC_RECONCILIATION
@@ -397,7 +398,15 @@ def _structural_purpose(
397
398
 
398
399
 
399
400
  def _build_public_methods(class_fqn: str, raw_nodes: list[dict]) -> list[str]:
400
- """Return public method names for class_fqn from raw_ir nodes."""
401
+ """Return public method names for class_fqn from raw_ir nodes.
402
+
403
+ C3-41: a factory member (`symbol_kind: "bean"`) is a method, and it is the
404
+ *whole* API of a configuration class. Excluding the kind reported
405
+ `public_methods: []` for every `@Configuration` in every repository. Its
406
+ Java visibility is not the test either — a factory member is called by the
407
+ container, not by a caller in this repository, so a package-private one is
408
+ still the class's published surface.
409
+ """
401
410
  prefix = class_fqn + "#"
402
411
  methods: list[str] = []
403
412
  for node in raw_nodes:
@@ -405,6 +414,11 @@ def _build_public_methods(class_fqn: str, raw_nodes: list[dict]) -> list[str]:
405
414
  if not fqn.startswith(prefix):
406
415
  continue
407
416
  kind = node.get("symbol_kind") or node.get("type") or ""
417
+ if kind == "bean":
418
+ name = fqn[len(prefix):]
419
+ if name and not name.startswith("<"):
420
+ methods.append(name)
421
+ continue
408
422
  if kind not in ("method", "endpoint", "constructor", ""):
409
423
  continue
410
424
  modifiers: list[str] = node.get("modifiers") or []
@@ -465,10 +479,45 @@ def _build_callers(
465
479
  return [_simple(f) if counts[_simple(f)] == 1 else f for f in fqns]
466
480
 
467
481
 
468
- def _build_deps(class_fqn: str, graph: "ContextGraph") -> list[str]:
469
- """DI injected dependencies, simple names — via the ContextGraph public API."""
470
- deps = graph.injected_dependencies_of(class_fqn)
471
- return sorted({_simple(d) for d in deps})
482
+ def _build_deps(
483
+ class_fqn: str, graph: "ContextGraph", raw_nodes: "Optional[list[dict]]" = None
484
+ ) -> list[str]:
485
+ """What this class depends on, simple names — via the ContextGraph public API.
486
+
487
+ Injected dependencies for a class that is wired; **plus what a factory
488
+ declares**, for a class that does the wiring (C3-41). A configuration class
489
+ injects nothing: it names its collaborators in the signatures of its factory
490
+ members and builds them in their bodies, so reading only the injection graph
491
+ reported `outgoing_deps: []` for a class with eight collaborators per method.
492
+
493
+ Evidence, not naming: the extra sources are read only when the class actually
494
+ declares factory members, and both are atoms the IR already publishes — the
495
+ class-scope type surface and the instantiation facts.
496
+ """
497
+ deps = set(graph.injected_dependencies_of(class_fqn))
498
+ if _factory_members(class_fqn, raw_nodes or []):
499
+ deps.update(
500
+ ref.type for ref in graph.class_type_references_in(class_fqn) if ref.type
501
+ )
502
+ for member in _factory_members(class_fqn, raw_nodes or []):
503
+ deps.update(inst.type for inst in graph.instantiations_in(member) if inst.type)
504
+ return sorted({_simple(d) for d in deps if d})
505
+
506
+
507
+ def _factory_members(class_fqn: str, raw_nodes: list[dict]) -> list[str]:
508
+ """Members of this class that declare a bean rather than behaviour.
509
+
510
+ The IR already classifies them (`symbol_kind: "bean"`); this is the one place
511
+ that reads it, so both the method list and the dependency list agree about
512
+ what a configuration class contains.
513
+ """
514
+ prefix = class_fqn + "#"
515
+ return [
516
+ str(node.get("fqn"))
517
+ for node in raw_nodes
518
+ if str(node.get("fqn") or "").startswith(prefix)
519
+ and (node.get("symbol_kind") or "") == "bean"
520
+ ]
472
521
 
473
522
 
474
523
  def _build_events_published(class_fqn: str, model: "SpringSemanticModel") -> list[str]:
@@ -566,6 +615,44 @@ def _build_security(
566
615
  return result
567
616
 
568
617
 
618
+ def _build_chain_rules(class_fqn: str, cir: "CanonicalRepositoryIR", root) -> list[str]:
619
+ """Access rules this class declares for the request chain.
620
+
621
+ C3-41. A class that configures the chain writes its constraints in a DSL, not
622
+ in annotations, so the annotation reader returned `[]` for the one class whose
623
+ entire job is deciding access — while `posture`, in the same build, read its
624
+ rules from that very file. One authority answers it (`chain_rules`), so the
625
+ two commands cannot describe the same file differently.
626
+ """
627
+ if root is None:
628
+ return []
629
+ from sourcecode.chain_rules import rules_from_source
630
+
631
+ rel = _source_file_of(class_fqn, cir)
632
+ if not rel:
633
+ return []
634
+ try:
635
+ source = (Path(root) / rel).read_text(encoding="utf-8", errors="replace")
636
+ except OSError:
637
+ return []
638
+ out: list[str] = []
639
+ for rule in rules_from_source(source, rel):
640
+ patterns = ", ".join(rule.patterns) if rule.patterns else "**"
641
+ out.append(
642
+ f"request chain: {patterns} → {rule.decision} "
643
+ f"({rel}:{rule.line})"
644
+ )
645
+ return out
646
+
647
+
648
+ def _source_file_of(class_fqn: str, cir: "CanonicalRepositoryIR") -> str:
649
+ """The repo-relative file declaring this class, from the IR's own node."""
650
+ for node in _get_raw_nodes(cir):
651
+ if node.get("fqn") == class_fqn:
652
+ return str(node.get("file") or node.get("source_file") or "")
653
+ return ""
654
+
655
+
569
656
  def _build_endpoints(class_fqn: str, model: "SpringSemanticModel") -> list[str]:
570
657
  """REST endpoints declared on this controller class."""
571
658
  endpoints = model.endpoint_index.endpoints_for(class_fqn)
@@ -585,13 +672,27 @@ def explain_class(
585
672
  class_name: str,
586
673
  cir: "CanonicalRepositoryIR",
587
674
  model: "SpringSemanticModel",
675
+ root: "Optional[Path]" = None,
588
676
  ) -> ClassExplanation:
589
677
  """Build a ClassExplanation for class_name from existing CIR + model.
590
678
 
591
- Never raises — wraps all derivation in try/except.
679
+ Never raises. C3-41: a section that *failed* now says so. Every section was
680
+ wrapped in a bare `except: []`, which made a crash and a measured emptiness
681
+ the same output — and an empty section reads as *"there are none"*.
592
682
  """
593
683
  warnings: list[str] = []
594
684
 
685
+ def _section_of(name: str, build, default):
686
+ """Run one section builder; a failure becomes a warning, not a silence."""
687
+ try:
688
+ return build()
689
+ except Exception as exc: # pragma: no cover — defensive by contract
690
+ warnings.append(
691
+ f"{name} could not be derived ({type(exc).__name__}): this section "
692
+ f"is empty because the derivation failed, not because there is none"
693
+ )
694
+ return default
695
+
595
696
  try:
596
697
  class_fqn, all_matches = _resolve_fqn(class_name, cir)
597
698
  except Exception:
@@ -630,40 +731,32 @@ def explain_class(
630
731
  except Exception:
631
732
  purpose = f"{stereotype} class"
632
733
 
633
- try:
634
- public_methods = _build_public_methods(class_fqn, raw_nodes)
635
- except Exception:
636
- public_methods = []
637
-
638
- try:
639
- incoming_callers = _build_callers(class_fqn, cir, graph)
640
- except Exception:
641
- incoming_callers = []
642
-
643
- try:
644
- outgoing_deps = _build_deps(class_fqn, graph)
645
- except Exception:
646
- outgoing_deps = []
647
-
648
- try:
649
- events_published = _build_events_published(class_fqn, model)
650
- except Exception:
651
- events_published = []
652
-
653
- try:
654
- events_consumed = _build_events_consumed(class_fqn, model)
655
- except Exception:
656
- events_consumed = []
657
-
658
- try:
659
- transactions = _build_transactions(class_fqn, model)
660
- except Exception:
661
- transactions = []
662
-
663
- try:
664
- security_constraints = _build_security(class_fqn, raw_nodes, cir)
665
- except Exception:
666
- security_constraints = []
734
+ public_methods = _section_of(
735
+ "public_methods", lambda: _build_public_methods(class_fqn, raw_nodes), []
736
+ )
737
+ incoming_callers = _section_of(
738
+ "incoming_callers", lambda: _build_callers(class_fqn, cir, graph), []
739
+ )
740
+ outgoing_deps = _section_of(
741
+ "outgoing_deps", lambda: _build_deps(class_fqn, graph, raw_nodes), []
742
+ )
743
+ events_published = _section_of(
744
+ "events_published", lambda: _build_events_published(class_fqn, model), []
745
+ )
746
+ events_consumed = _section_of(
747
+ "events_consumed", lambda: _build_events_consumed(class_fqn, model), []
748
+ )
749
+ transactions = _section_of(
750
+ "transactions", lambda: _build_transactions(class_fqn, model), []
751
+ )
752
+ security_constraints = _section_of(
753
+ "security_constraints", lambda: _build_security(class_fqn, raw_nodes, cir), []
754
+ )
755
+ # A class that configures the request chain declares its constraints in a
756
+ # DSL. Same section, one authority (`chain_rules`) — never a second parser.
757
+ security_constraints = security_constraints + _section_of(
758
+ "security_constraints", lambda: _build_chain_rules(class_fqn, cir, root), []
759
+ )
667
760
 
668
761
  try:
669
762
  rest_endpoints = _build_endpoints(class_fqn, model)
@@ -62,52 +62,62 @@ _JAVA_TEST_ROOTS = (
62
62
  "\\src\\test\\",
63
63
  )
64
64
 
65
+ #: Directories that hold **tool, editor and agent state** — nothing any analyser
66
+ #: in this engine opens, so a change inside one cannot change an answer.
67
+ #:
68
+ #: C1-27. C3-40 made an untracked file count as a change, correctly: an untracked
69
+ #: `.java` file is in the IR. It counted *any* porcelain line, though, so an
70
+ #: editor settings file no analyser reads flipped a snapshot to STALE with
71
+ #: `RIS HEAD == current HEAD` and `Delta: 0`. A freshness signal that fires on a
72
+ #: file the analysis cannot see teaches its reader to ignore the one that matters.
73
+ #:
74
+ #: Deliberately short, and the default is **inclusive**: anything not listed here
75
+ #: is a change. That asymmetry is the point — a missed change is a false *fresh*,
76
+ #: which is the confident-falsehood direction, and a spurious change only costs a
77
+ #: rebuild. `.github` is **not** here: CI descriptors are read as tooling
78
+ #: evidence. Version-control internals are, because git's own directory is never
79
+ #: analysis input.
80
+ _TOOL_STATE_DIRS = frozenset({
81
+ ".git", ".hg", ".svn",
82
+ ".idea", ".vscode", ".fleet", ".eclipse", ".settings",
83
+ ".claude", ".cursor", ".aider", ".continue",
84
+ ".devcontainer", ".husky",
85
+ })
65
86
 
66
- def is_test_path(path: str) -> bool:
67
- """Return True when *path* is part of a test tree, not production code.
68
87
 
69
- Handles:
70
- - Standard Maven/Gradle layout (src/test/java/…)
71
- - Common naming conventions (/tests/, /spec/, /it/)
72
- - Java file name conventions (FooTest.java, TestFoo.java)
73
- - Python conventions (test_foo.py, foo_test.py)
74
- - JS/TS conventions (foo.test.ts, foo.spec.ts)
88
+ def analysis_reads(path: str) -> bool:
89
+ """Whether a change to *path* can change any answer this engine produces.
90
+
91
+ The one authority for that question (ADR-0008 R1). Used by cache freshness so
92
+ "the tree changed" means the tree *the analysis reads* changed, which is what
93
+ a snapshot describes.
75
94
  """
76
- norm = path.replace("\\", "/").lower()
95
+ parts = [p for p in str(path).replace("\\", "/").split("/") if p and p != "."]
96
+ return not any(part in _TOOL_STATE_DIRS for part in parts)
77
97
 
78
- # Maven/Gradle standard test root (fast path)
79
- if "/src/test/" in norm:
80
- return True
81
98
 
82
- # Segment-based check – any directory component is a test segment
83
- parts = norm.split("/")
84
- seen: set[str] = set()
85
- for part in parts[:-1]: # skip filename itself
86
- bare = part.rstrip("/")
87
- if bare in _AMBIGUOUS_TEST_SEGMENTS and seen & _DOC_ROOT_SEGMENTS:
88
- seen.add(bare)
89
- continue # a specification under `docs/`, not a spec suite
90
- if bare in _TEST_SEGMENTS:
91
- return True
92
- seen.add(bare)
93
-
94
- # File-name conventions
95
- name = parts[-1]
96
- if (
97
- name.startswith("test_")
98
- or name.endswith("_test.py")
99
- or name.endswith(".test.ts")
100
- or name.endswith(".test.js")
101
- or name.endswith(".spec.ts")
102
- or name.endswith(".spec.js")
103
- or (name.endswith("test.java") and name != "test.java")
104
- or name.endswith("tests.java")
105
- or (name.startswith("test") and name.endswith(".java") and len(name) > 9
106
- and "/src/main/" not in norm)
107
- ):
108
- return True
99
+ def is_test_path(path: str) -> bool:
100
+ """Return True when *path* is part of a test tree, not production code.
109
101
 
110
- return False
102
+ C1-28: **this is a passthrough**, and it exists only so the six modules that
103
+ import it here keep working. `test_sources.is_test_path` is the authority for
104
+ this fact and has been since 3.2.1 (C1-2) — but this second derivation was
105
+ never retired, so two rules answered one question and disagreed on five path
106
+ shapes, in both directions:
107
+
108
+ src/main/java/**/Test*.java here: True authority: False
109
+ src/main/java/**/*IT.java here: False authority: True
110
+ src/main/java/**/test/X.java here: True authority: False
111
+
112
+ The field met the first as *"`--compact` counts 1 test file that is actually
113
+ in `src/main`"*. Everything this module knew and the authority did not — the
114
+ `test-helpers`/`it`/`integrationtest` directory spellings, the JUnit-3
115
+ `TestFoo` prefix — moved there rather than being dropped, because retiring an
116
+ authority means absorbing what it knew.
117
+ """
118
+ from sourcecode.test_sources import is_test_path as _authority
119
+
120
+ return _authority(path)
111
121
 
112
122
 
113
123
  def is_test_or_fixture_path(path: str) -> bool:
sourcecode/posture.py CHANGED
@@ -45,7 +45,16 @@ POSTURE_SCHEMA = "posture-v1"
45
45
  _POSTURE_LIST_CAP = 200
46
46
 
47
47
 
48
- def _declare_cap(parent: dict, name: str, total: int) -> None:
48
+ #: The same, for the per-candidate lists of `--resolve-environments`. A separate
49
+ #: number because a candidate row is one of many in a single answer, and a
50
+ #: separate *name* because a cap that is not declared with the limit that
51
+ #: produced it cannot be checked (C2-25).
52
+ _ENVIRONMENT_LIST_CAP = 50
53
+
54
+
55
+ def _declare_cap(
56
+ parent: dict, name: str, total: int, limit: int = _POSTURE_LIST_CAP
57
+ ) -> None:
49
58
  """Record what the cut on ``parent[name]`` omitted, or nothing when it did not
50
59
  bite (C2-20). Same shape as `migrate-check`'s `findings_cap`, because two ways
51
60
  to say "this list is not the whole list" is one more than a reader should learn.
@@ -59,7 +68,7 @@ def _declare_cap(parent: dict, name: str, total: int) -> None:
59
68
  "total": total,
60
69
  "shown": shown,
61
70
  "omitted": total - shown,
62
- **cap_effect("display_list", limit=_POSTURE_LIST_CAP, total=total),
71
+ **cap_effect("display_list", limit=limit, total=total),
63
72
  }
64
73
 
65
74
  #: Resolved against the repository's own configuration (see `spring_properties`),
@@ -750,6 +759,7 @@ def access_projection(
750
759
 
751
760
  by_endpoint: dict[str, str] = {}
752
761
  details: dict[str, dict] = {}
762
+ decided_by: dict[str, list[str]] = {}
753
763
  for endpoint in getattr(cir, "endpoints", []) or []:
754
764
  policy = getattr(getattr(endpoint, "security", None), "policy", None) or ""
755
765
  if policy not in _NO_HANDLER_GUARD_POLICIES:
@@ -788,6 +798,14 @@ def access_projection(
788
798
  decision, evidence = next(iter(decisions.items()))
789
799
  by_endpoint[endpoint.id] = decision
790
800
  details[endpoint.id] = evidence
801
+ # C1-25. Which configuration file decided this request. Uncapped and kept
802
+ # beside the decision, because it is the declared scope of anything wrong
803
+ # in that file: a defect in a chain configuration is not "unreachable"
804
+ # because nothing calls the class — its blast radius is every request it
805
+ # decides.
806
+ _decider = str(evidence.get("source_file") or "") if isinstance(evidence, dict) else ""
807
+ if _decider:
808
+ decided_by.setdefault(_decider, []).append(endpoint.id)
791
809
 
792
810
  # A declared-but-undecidable servlet path means the URL the chain matches is
793
811
  # unknown. Where any pattern rule is declared, that unknown decides which rule
@@ -800,6 +818,9 @@ def access_projection(
800
818
  for rule in rules_by_file[source_file]
801
819
  )
802
820
  if not servlet.decided and path_dependent:
821
+ # No chain-derived verdict survives an unknown servlet path, so no file
822
+ # can be said to have decided a request either (C1-25).
823
+ decided_by.clear()
803
824
  for endpoint_id, decision in list(by_endpoint.items()):
804
825
  if decision == "guarded_by_own_annotation":
805
826
  continue
@@ -821,6 +842,13 @@ def access_projection(
821
842
  "inactive_files": sorted(f for f in rules_by_file if states.get(f) == "inactive"),
822
843
  "unresolved_files": undecided_files,
823
844
  "rules": sum(len(rules_by_file[f]) for f in active_files),
845
+ # How much of the request surface each configuration file decides.
846
+ # Uncapped on purpose: it is a per-file figure, not a per-endpoint
847
+ # list, and it is the declared scope of any defect written in that
848
+ # file (C1-25). A count, not a sample, so nothing here needs a cap.
849
+ "decides_endpoints": {
850
+ f: len(ids) for f, ids in sorted(decided_by.items())
851
+ },
824
852
  },
825
853
  "summary": {k: v for k, v in summary.items() if v},
826
854
  }
@@ -900,6 +928,28 @@ def endpoint_access_under(
900
928
  return _posture(root, profiles, properties, cir=cir)[1]
901
929
 
902
930
 
931
+ def access_resolution(
932
+ root: Path,
933
+ profiles: "Optional[set[str]]" = None,
934
+ *,
935
+ cir: "Optional[Any]" = None,
936
+ properties: "Optional[dict[str, str]]" = None,
937
+ ) -> "tuple[dict[str, str], dict[str, int]]":
938
+ """Both halves of one resolution: the per-endpoint decision, and how much of
939
+ the surface each chain configuration file decides.
940
+
941
+ One pass, because both consumers are the same consumer: `ask risk` weighs the
942
+ decision as its access axis (C1-26) **and** needs the second figure to know
943
+ that a defect written in a chain configuration is not unreachable merely
944
+ because nothing calls its class (C1-25). Running the resolution twice to
945
+ answer one question about one profile set is how two authorities are born.
946
+ """
947
+ payload, by_endpoint = _posture(root, profiles, properties, cir=cir)
948
+ access = ((payload.get("endpoints") or {}).get("effective_access") or {})
949
+ decides = (access.get("chains") or {}).get("decides_endpoints") or {}
950
+ return by_endpoint, {str(k): int(v) for k, v in decides.items()}
951
+
952
+
903
953
  def access_verdict(decision: str) -> str:
904
954
  """A profile-resolved decision, in the vocabulary the risk composition weighs.
905
955
 
@@ -956,15 +1006,25 @@ def resolve_environments(
956
1006
  payload, by_endpoint = _posture(root, set(candidate), properties, cir=cir)
957
1007
  open_endpoints = sorted(e for e, d in by_endpoint.items() if d in _OPEN_BY_RULE)
958
1008
  undecided = sorted(e for e, d in by_endpoint.items() if d in _UNDECIDED_ACCESS)
959
- evaluated.append({
1009
+ # C2-25. The list is a sample and the count is the answer; the cut says so
1010
+ # in the same vocabulary every other list in this module uses. Publishing
1011
+ # 50 of 2 635 with no `total`/`omitted`/`direction` was the one place the
1012
+ # product broke a convention it keeps everywhere else — and it was the
1013
+ # list an auditor needs whole.
1014
+ row = {
960
1015
  "profiles": list(candidate),
961
1016
  "open_by_rule": len(open_endpoints),
962
1017
  "access_not_decided": len(undecided),
963
1018
  "endpoints_total": len(by_endpoint),
964
1019
  "access_summary": payload["endpoints"]["effective_access"].get("summary", {}),
965
1020
  "unresolved_beans": len(payload.get("unresolved") or []),
966
- "endpoints_open_by_rule": open_endpoints[:50],
967
- })
1021
+ "endpoints_open_by_rule": open_endpoints[:_ENVIRONMENT_LIST_CAP],
1022
+ }
1023
+ _declare_cap(
1024
+ row, "endpoints_open_by_rule", len(open_endpoints),
1025
+ limit=_ENVIRONMENT_LIST_CAP,
1026
+ )
1027
+ evaluated.append(row)
968
1028
 
969
1029
  worst = max(
970
1030
  evaluated,
@@ -39,6 +39,7 @@ from sourcecode.path_filters import (
39
39
  is_test_path as _is_test_path,
40
40
  is_test_fixture_module_path as _is_test_fixture_module_path,
41
41
  )
42
+ from sourcecode.test_sources import matches_test_naming as _matches_test_naming
42
43
  from sourcecode.security_config import (
43
44
  CustomSecuritySpec,
44
45
  capture_markers as _capture_markers,
@@ -6749,15 +6750,23 @@ def find_java_files(
6749
6750
  # Skip test dirs — use centralised is_test_path (consistent with
6750
6751
  # extract_java_endpoints), guarded against false positives where a Java
6751
6752
  # *package* is named "test" inside a production src/main/ source root.
6752
- if not include_tests and _is_test_path(rel):
6753
- _skip = True
6754
- # Prepend "/" so the check works whether or not rel has a leading slash.
6755
- _rel_sl = "/" + rel
6756
- if "/src/main/" in _rel_sl:
6757
- # is_test_path may fire on a package segment (e.g. com.example.test)
6758
- # rather than a true test module directory. Only skip when the path
6759
- # prefix BEFORE src/main/ is itself a test path (meaning the whole
6760
- # module is a test module, not just a package named "test").
6753
+ if not include_tests:
6754
+ # Two facts, and C1-28 is why they are now written as two.
6755
+ # `is_test_path` answers "is this file a test source"; since the
6756
+ # authority took over it never fires inside a declared main source
6757
+ # root, which is correct everywhere except inside a module that is
6758
+ # itself test infrastructure — a `test-framework` module has no
6759
+ # product for its `src/main` to belong to. There the weaker question
6760
+ # is the right one: does this file look like a test by name or by
6761
+ # package. Both come from the one authority, so this stays a
6762
+ # composition of published facts rather than a third rule. A test
6763
+ # *fixture* module (`*-test-utils`) is deliberately NOT excluded
6764
+ # here: its endpoints are kept as raw evidence and tagged
6765
+ # `scope: test_util` downstream.
6766
+ _rel_sl = "/" + rel # leading slash: the checks below are prefix-safe
6767
+ if _is_test_path(rel):
6768
+ continue
6769
+ if "/src/main/" in _rel_sl and _matches_test_naming(rel):
6761
6770
  _prefix = _rel_sl.split("/src/main/")[0]
6762
6771
  _prefix_parts = [p for p in _prefix.split("/") if p]
6763
6772
  # A module is a test module if is_test_path says so OR if any
@@ -6773,10 +6782,8 @@ def find_java_files(
6773
6782
  for w in p.lower().replace("-", " ").replace("_", " ").split()
6774
6783
  )
6775
6784
  )
6776
- if not _prefix_is_test:
6777
- _skip = False
6778
- if _skip:
6779
- continue
6785
+ if _prefix_is_test:
6786
+ continue
6780
6787
  # Skip vendor/generated/build dirs
6781
6788
  if any(part in _VENDOR_DIRS for part in parts[:-1]):
6782
6789
  continue
sourcecode/risk.py CHANGED
@@ -58,7 +58,14 @@ _SEVERITY_FACTOR = {"critical": 4.0, "high": 3.0, "medium": 2.0, "low": 1.0}
58
58
  #: What reaching HTTP does to a defect. *"TX-001 en un método privado es una nota de
59
59
  #: linter. TX-001 que alcanza cuatro POST de pago es un ticket."* Unreached is not
60
60
  #: zero — the defect is still real, it is simply not exposed.
61
- _REACH_FACTOR = {"none": 0.6, "internal": 1.0, "http": 1.6}
61
+ #:
62
+ #: `request_chain` is the **declared** scope, not a walked one (C1-25): a defect
63
+ #: written in a configuration that decides access for the request surface is not
64
+ #: unreachable because nothing calls its class. Its blast radius is every request
65
+ #: those rules decide, which is why it weighs what an HTTP-reached defect weighs
66
+ #: rather than what an internal one does. The count comes from `posture`, which is
67
+ #: the authority for which file decided which request.
68
+ _REACH_FACTOR = {"none": 0.6, "internal": 1.0, "request_chain": 1.6, "http": 1.6}
62
69
 
63
70
  #: What the access verdict on the reached endpoints does. Reading the WORST verdict
64
71
  #: among them, because one open route is what an attacker needs; and an endpoint the
@@ -153,6 +160,12 @@ class RiskRow:
153
160
  #: reaches more. Both are correct and the difference is the scope, so the scope
154
161
  #: travels with the number instead of in a note the reader has to fetch (C2-23).
155
162
  reach_scope: str = "class"
163
+ #: Requests this defect's own file decides access for, when it is a chain
164
+ #: configuration (C1-25). Zero for every other defect. Kept apart from
165
+ #: `endpoints_reached`, which is a *walked* figure: merging a declared scope
166
+ #: into a reach count would be the same unit collision C2-23 published notes
167
+ #: about.
168
+ declared_scope: int = 0
156
169
  endpoints: list = field(default_factory=list)
157
170
  statement: str = ""
158
171
  blind_axes: list = field(default_factory=list)
@@ -174,8 +187,15 @@ class RiskRow:
174
187
  # number is what makes a risk list impossible to argue with.
175
188
  "factors": self.factors,
176
189
  "endpoints_reached": self.endpoints_reached,
190
+ **({"decides_requests": self.declared_scope} if self.declared_scope else {}),
177
191
  "endpoints_reached_basis": (
178
192
  f"endpoints reachable from this {self.reach_scope}"
193
+ + (
194
+ f"; this defect is written in a chain configuration, whose "
195
+ f"scope is declared rather than walked — see "
196
+ f"`decides_requests` ({self.declared_scope})"
197
+ if self.declared_scope else ""
198
+ )
179
199
  + (
180
200
  "; `ask impact-chain <Class>` answers the wider question — what "
181
201
  "the whole enclosing class reaches — and returns a larger figure "
@@ -410,6 +430,15 @@ def _statement(row: "dict[str, Any]") -> str:
410
430
  )
411
431
  if reach == "http":
412
432
  return f"{row['title']} — reachable over {row['count']} endpoint(s){tail}"
433
+ if reach == "request_chain":
434
+ # C1-25. Never the "nothing reaches it" sentence for the file that decides
435
+ # what is reachable. The scope is declared, not walked, and the sentence
436
+ # says which it is.
437
+ return (
438
+ f"{row['title']} — declared in the configuration that decides access "
439
+ f"for {row['governs']} request(s) in this repository, so its scope is "
440
+ f"that surface and not a call path into this class{tail}"
441
+ )
413
442
  return f"{row['title']} — no HTTP endpoint in this repository reaches it{tail}"
414
443
 
415
444
 
@@ -477,20 +506,49 @@ class RiskComposer:
477
506
  )
478
507
  self._reach_cache: "dict[str, Any]" = {}
479
508
 
480
- # M10's exit criterion: profile context per finding. Without a profile set
481
- # the access axis is the endpoint security surface, which answers for the
482
- # repository as configured. With one, `posture` resolves the conditional
483
- # bean graph and says what *that deployment* leaves reachable — the axis
484
- # nothing else in the market models, and the reason a finding can be
485
- # `protected` under one profile set and open under another.
509
+ # M10's exit criterion: profile context per finding. With a profile set,
510
+ # `posture` resolves the conditional bean graph and says what *that
511
+ # deployment* leaves reachable — the axis nothing else in the market
512
+ # models, and the reason a finding can be `protected` under one profile
513
+ # set and open under another.
514
+ #
515
+ # C1-26: the resolution is read **whether or not a set is named**. It used
516
+ # to be the price of the flag, so the default invocation weighed an
517
+ # endpoint at `unknown` (×1.4) that `posture`, in the same process, over
518
+ # the same CIR, decided was reachable without authentication (×2.0) — the
519
+ # difference between `high` and `critical` on the class this product is
520
+ # bought for. With no set named the resolution is the one Spring itself
521
+ # performs with none given: `default` alone, which CL-8 records as the
522
+ # permissive case rather than a neutral one. It is a **default, never a
523
+ # guarantee** — the process environment outranks every file — and the row
524
+ # says which set it read.
486
525
  self.profiles = set(profiles) if profiles else None
487
- self.profile_access: "dict[str, str]" = {}
488
- if self.profiles is not None:
489
- from sourcecode.posture import endpoint_access_under
526
+ from sourcecode.posture import access_resolution
527
+
528
+ # One pass answers both: the access axis (C1-26) and how much of the
529
+ # request surface each chain configuration file decides (C1-25).
530
+ self.profile_access, self.chain_reach = access_resolution(
531
+ self.root,
532
+ self.profiles if self.profiles is not None else set(),
533
+ cir=self.cir,
534
+ )
490
535
 
491
- self.profile_access = endpoint_access_under(
492
- self.root, self.profiles, cir=self.cir
493
- )
536
+ def chain_scope(self, source_file: str) -> int:
537
+ """How many requests the chain configuration in ``source_file`` decides.
538
+
539
+ 0 when this file declares no readable chain rule, which is every file
540
+ that is not one. Path spellings are normalised because a finding carries
541
+ the path its own emitter produced and `posture` keys on the repository's
542
+ relative POSIX path (C2-18/C2-24's convention).
543
+ """
544
+ if not source_file or not self.chain_reach:
545
+ return 0
546
+ wanted = str(source_file).replace("\\", "/").lstrip("./")
547
+ for candidate, count in self.chain_reach.items():
548
+ normalized = candidate.replace("\\", "/").lstrip("./")
549
+ if normalized == wanted or normalized.endswith(f"/{wanted}") or wanted.endswith(f"/{normalized}"):
550
+ return count
551
+ return 0
494
552
 
495
553
  def reach(self, symbol: str) -> "Any":
496
554
  from sourcecode.spring_impact import run_impact_chain
@@ -532,10 +590,21 @@ class RiskComposer:
532
590
  endpoint.endpoint_id, "coverage_unknown"
533
591
  )))
534
592
 
535
- reach = "http" if endpoints else ("internal" if not blind else "unknown")
593
+ # C1-25. A defect declared in a configuration that decides the request
594
+ # chain has a scope even with no call path into it, and reporting
595
+ # `internal` for it published a sentence that is false in effect: *"no
596
+ # HTTP endpoint in this repository reaches it"*, about the file that
597
+ # decides whether any of them are reachable at all.
598
+ governs = self.chain_scope(source_file)
599
+ if endpoints:
600
+ reach = "http"
601
+ elif governs:
602
+ reach = "request_chain"
603
+ else:
604
+ reach = "internal" if not blind else "unknown"
536
605
  auth = _auth_axis(verdicts)
537
606
  profile_decisions: "list[str]" = []
538
- if self.profiles is not None and endpoints:
607
+ if endpoints:
539
608
  from sourcecode.posture import access_verdict
540
609
 
541
610
  profile_decisions = [
@@ -599,30 +668,48 @@ class RiskComposer:
599
668
  "reachability": {
600
669
  "value": reach,
601
670
  "factor": _REACH_FACTOR.get(reach, 1.0),
602
- "authority": "impact-chain",
671
+ "authority": (
672
+ "posture.chains.decides_endpoints"
673
+ if reach == "request_chain" else "impact-chain"
674
+ ),
675
+ **({
676
+ "governs_endpoints": governs,
677
+ "basis": (
678
+ "declared scope, not a walked one: this file's chain "
679
+ "rules decide access for that many requests, so a "
680
+ "defect in it is not internal because no call path "
681
+ "leads to the class"
682
+ ),
683
+ } if reach == "request_chain" else {}),
603
684
  },
604
685
  "auth_verdict": {
605
686
  "value": auth,
606
687
  "factor": _AUTH_FACTOR.get(auth, 1.0),
607
688
  "authority": (
608
- "security_posture.endpoint_security_surface"
689
+ "security_posture.endpoint_security_surface + "
690
+ "posture.effective_access under "
609
691
  + (
610
- f" + posture.effective_access under "
611
- f"{','.join(sorted(self.profiles))}"
612
- if self.profiles is not None else ""
692
+ ",".join(sorted(self.profiles))
693
+ if self.profiles is not None else "default"
613
694
  )
614
695
  ),
615
696
  "worst_of": sorted(set(verdicts)),
616
- **({
617
- "profile_set": sorted(self.profiles),
618
- "profile_decisions": sorted(set(profile_decisions)),
619
- "basis": (
620
- "the worse of the two authorities: what the repository's "
621
- "own security surface says, and what this profile set "
622
- "resolves the request chain to. `no_rule_matched` is "
623
- "unknown coverage, never open access"
624
- ),
625
- } if self.profiles is not None else {}),
697
+ "profile_set": sorted(self.profiles) if self.profiles is not None else [],
698
+ "profile_set_basis": (
699
+ "named by --profile"
700
+ if self.profiles is not None else
701
+ "no profile set named, so the resolution is the one Spring "
702
+ "performs with none given — `default` alone, which is the "
703
+ "permissive case, not a neutral one. A default, never a "
704
+ "guarantee: the process environment outranks every file"
705
+ ),
706
+ "profile_decisions": sorted(set(profile_decisions)),
707
+ "basis": (
708
+ "the worse of the two authorities: what the repository's "
709
+ "own security surface says, and what this profile set "
710
+ "resolves the request chain to. `no_rule_matched` is "
711
+ "unknown coverage, never open access"
712
+ ),
626
713
  },
627
714
  "write_effect": {
628
715
  "value": write,
@@ -655,11 +742,12 @@ class RiskComposer:
655
742
  },
656
743
  endpoints_reached=len(endpoints),
657
744
  reach_scope="method" if "#" in symbol else "class",
745
+ declared_scope=governs if reach == "request_chain" else 0,
658
746
  endpoints=[e.to_dict() for e in endpoints[:20]],
659
747
  statement=_statement({
660
748
  "title": title, "reach": reach, "auth": auth,
661
749
  "write": write, "count": len(endpoints), "query": query,
662
- "constraints": constraints,
750
+ "constraints": constraints, "governs": governs,
663
751
  }),
664
752
  blind_axes=blind,
665
753
  witness_count=witness_count,
@@ -372,6 +372,39 @@ def _class_of(fqn: str) -> str:
372
372
  return fqn
373
373
 
374
374
 
375
+ #: Attribute the class→symbols index is memoized under, on the CIR it describes.
376
+ #: Not a dataclass field: it is a derived index, it must not enter `cir_hash`,
377
+ #: equality or the serialized form. The CIR is immutable in practice once built,
378
+ #: so one index per CIR is correct for its whole lifetime.
379
+ _SYMBOL_INDEX_ATTR = "_impact_symbols_by_class"
380
+
381
+
382
+ def _symbols_by_class(cir: CanonicalRepositoryIR) -> "dict[str, list[str]]":
383
+ """`{class FQN → [its symbols, in `cir.symbols` order]}`, built once per CIR.
384
+
385
+ C3-42. The seed expansions below need "every symbol of this class" once per
386
+ subtype and once per interface, and they used to answer it by scanning
387
+ `cir.symbols` in full for each one — `len(subtypes) × len(symbols)` string
388
+ splits, inside a query the field measured at **1 187 s against a 516 ms model
389
+ build**. It is also `risk`'s inner loop, once per defect, which is where a
390
+ 45-minute `risk` comes from.
391
+
392
+ The order of each bucket is `cir.symbols` order, which is what the expansions
393
+ appended in before; the answer is identical, only the cost is not.
394
+ """
395
+ index = getattr(cir, _SYMBOL_INDEX_ATTR, None)
396
+ if index is not None:
397
+ return index
398
+ index = {}
399
+ for sym in cir.symbols:
400
+ index.setdefault(_class_of(sym), []).append(sym)
401
+ try:
402
+ setattr(cir, _SYMBOL_INDEX_ATTR, index)
403
+ except Exception: # pragma: no cover — a CIR that cannot carry the memo still works
404
+ pass
405
+ return index
406
+
407
+
375
408
  def _resolve_symbol(
376
409
  symbol: str,
377
410
  cir_symbols: list[str],
@@ -910,8 +943,14 @@ class ImpactOrchestrator:
910
943
  return True
911
944
  return "#" in sym and sym.rsplit("#", 1)[1] in _queried_methods
912
945
 
946
+ # C3-42. The class→symbols index and the seed set are both loop-invariant:
947
+ # `seed_fqns` is only rebound *after* the loop, so the membership test asks
948
+ # about the same set on every symbol. Built once here, they turn an
949
+ # O(subtypes × symbols × seeds) scan into O(subtypes) lookups.
950
+ symbols_by_class = _symbols_by_class(cir)
913
951
  if impl_graph is not None:
914
952
  seed_classes_ch001 = {_class_of(s) for s in seed_fqns}
953
+ seed_set = frozenset(seed_fqns)
915
954
  impl_seeds: list[str] = []
916
955
  for seed_class in sorted(seed_classes_ch001):
917
956
  subtypes = impl_graph.all_subtypes_of(seed_class)
@@ -921,9 +960,8 @@ class ImpactOrchestrator:
921
960
  if impl_class in subtype_classes_added:
922
961
  continue
923
962
  subtype_classes_added.add(impl_class)
924
- for sym in cir.symbols:
925
- if (_class_of(sym) == impl_class and sym not in set(seed_fqns)
926
- and _member_in_scope(sym)):
963
+ for sym in symbols_by_class.get(impl_class, ()):
964
+ if sym not in seed_set and _member_in_scope(sym):
927
965
  impl_seeds.append(sym)
928
966
  if impl_seeds:
929
967
  seed_fqns = list(dict.fromkeys(seed_fqns + impl_seeds))
@@ -943,6 +981,7 @@ class ImpactOrchestrator:
943
981
  # (e.g. RefByUuid) and produces false-positive callers from sibling implementors.
944
982
  if impl_graph is not None:
945
983
  current_seed_classes = {_class_of(s) for s in seed_fqns}
984
+ seed_set = frozenset(seed_fqns)
946
985
  iface_seeds: list[str] = []
947
986
  iface_classes_added: set[str] = set()
948
987
  for seed_class in sorted(original_seed_classes):
@@ -951,9 +990,8 @@ class ImpactOrchestrator:
951
990
  if iface_class in iface_classes_added or iface_class in current_seed_classes:
952
991
  continue
953
992
  iface_classes_added.add(iface_class)
954
- for sym in cir.symbols:
955
- if (_class_of(sym) == iface_class and sym not in set(seed_fqns)
956
- and _member_in_scope(sym)):
993
+ for sym in symbols_by_class.get(iface_class, ()):
994
+ if sym not in seed_set and _member_in_scope(sym):
957
995
  iface_seeds.append(sym)
958
996
  if iface_seeds:
959
997
  seed_fqns = list(dict.fromkeys(seed_fqns + iface_seeds))
@@ -1032,10 +1070,18 @@ class ImpactOrchestrator:
1032
1070
  # The security audit already derives the posture projection; take it from
1033
1071
  # there rather than deriving the same answer a second time (ADR-0008 R1).
1034
1072
  posture_dict: dict = {}
1073
+ # C3-42. This is a **repository-wide** audit, run inside what the payload
1074
+ # then publishes as `query_time_ms` — so a reader comparing that figure
1075
+ # against `model_build_time_ms` is comparing a model build against a model
1076
+ # build plus a full audit of every file. The share is measured and named
1077
+ # rather than hidden: the cost is real, and which part of it is the walk is
1078
+ # the first thing anyone reporting a slow query needs to know.
1079
+ _audit_t0 = time.monotonic()
1035
1080
  if prebuilt_findings is not None:
1036
1081
  all_findings = prebuilt_findings
1037
1082
  else:
1038
1083
  all_findings, posture_dict = _run_audit_for_chain(cir, model, root)
1084
+ audit_ms = round((time.monotonic() - _audit_t0) * 1000, 2)
1039
1085
 
1040
1086
  impact_findings_raw = _filter_findings(
1041
1087
  all_findings, seed_fqns, direct_callers_raw, indirect_callers_raw,
@@ -1405,6 +1451,20 @@ class ImpactOrchestrator:
1405
1451
  "risk_score": risk_score,
1406
1452
  "model_build_time_ms": model.build_time_ms,
1407
1453
  "query_time_ms": elapsed_ms,
1454
+ # What that number is made of. A field report of
1455
+ # `query_time_ms: 1 187 063` beside `model_build_time_ms: 516`
1456
+ # reads as a 2 300× graph walk; the walk is the difference between
1457
+ # these two figures, and on every repository measured here the
1458
+ # repository-wide audit is the larger part (C3-42).
1459
+ "audit_time_ms": audit_ms,
1460
+ "query_time_basis": (
1461
+ "wall time of the whole call, which includes a repository-wide "
1462
+ "TX + security audit (`audit_time_ms`) whose findings are then "
1463
+ "filtered to this chain. The graph walk itself is "
1464
+ "`query_time_ms - audit_time_ms`. `model_build_time_ms` covers "
1465
+ "the semantic model only, so the two are not comparable as they "
1466
+ "stand"
1467
+ ),
1408
1468
  "blind_spots": blind_spots,
1409
1469
  "reconciliation": reconciliation_findings,
1410
1470
  "external_supertypes": external_supertypes,
@@ -48,9 +48,14 @@ _TEST_ROOT_SEGMENTS: tuple[tuple[str, ...], ...] = (
48
48
  )
49
49
 
50
50
  # Repository-level test directories, for ecosystems with no src/main split.
51
- _TEST_DIR_NAMES: frozenset[str] = frozenset(
52
- {"test", "tests", "spec", "specs", "__tests__"}
53
- )
51
+ # The `test-helpers`/`it`/`integrationtest` spellings arrived with C1-28, from
52
+ # the second module that used to answer this question: retiring an authority
53
+ # means absorbing what it knew, not dropping it.
54
+ _TEST_DIR_NAMES: frozenset[str] = frozenset({
55
+ "test", "tests", "spec", "specs", "__tests__",
56
+ "test-helpers", "test_helpers", "testfixtures",
57
+ "it", "integrationtest", "integrationtests",
58
+ })
54
59
 
55
60
  #: `spec/` is a test suite in the RSpec/Jasmine sense — and the opposite under a
56
61
  #: documentation root, where a *specification* lives. Found by this project's own suite
@@ -73,6 +78,10 @@ _TEST_NAME_PATTERNS: tuple[re.Pattern[str], ...] = tuple(
73
78
  r"[^/]*_test\.go$",
74
79
  r"[^/]*\.(spec|test)\.(js|jsx|ts|tsx|mjs|cjs)$",
75
80
  r"[^/]*(Test|Tests|Spec|IT)\.(java|kt|scala)$",
81
+ # The JUnit-3 prefix spelling, absorbed from the module C1-28 retired.
82
+ # It is a name, so the main-root rule disarms it exactly as it disarms
83
+ # the suffix one.
84
+ r"(?:^|/)Test[A-Z][^/]*\.(java|kt|scala)$",
76
85
  r"[^/]*_test\.rb$",
77
86
  r"[^/]*_spec\.rb$",
78
87
  r"[^/]*_test\.dart$",
@@ -122,10 +131,44 @@ def is_test_path(path: str) -> bool:
122
131
 
123
132
  The single derivation of this fact. A file under a declared test source root
124
133
  is a test whatever it is named; a file named by convention is a test wherever
125
- it sits; a `test` package inside a main source root is neither.
134
+ it sits **except inside a declared main source root**; a `test` package
135
+ inside a main source root is neither.
136
+
137
+ C1-28 added the exception, and it is the same rule the directory half has
138
+ always applied, in the same direction: a build tool that separates test
139
+ sources says so with a directory, and that statement outranks a name — in
140
+ both directions. `src/main/java/**/OrderIT.java` is production code that a
141
+ convention happens to match, and counting it inflates a test population in
142
+ the one direction a coverage figure must never be wrong about. The field met
143
+ it as *"`--compact` counts 1 test file that is actually in `src/main`"*, with
144
+ the verdict (0.0 %, critical) right and the numerator wrong.
126
145
  """
127
146
  if declared_test_root(path) is not None:
128
147
  return True
148
+ if _under_main_root(_parts(path)):
149
+ return False
150
+ normalized = _norm(path)
151
+ return any(p.search(normalized) for p in _TEST_NAME_PATTERNS)
152
+
153
+
154
+ def matches_test_naming(path: str) -> bool:
155
+ """Whether ``path`` *looks* like a test by convention alone — a test-shaped
156
+ directory segment or file name — with no source-root reasoning at all.
157
+
158
+ Published because one caller genuinely needs the weaker question. Inside a
159
+ module that is itself test infrastructure (`test-framework/`, `testsuite/`),
160
+ a file in an `it` package or named `TestResource` is a test even though it
161
+ sits under `src/main`: the module has no product to belong to. Everywhere
162
+ else the declared root outranks the name and `is_test_path` is the question
163
+ to ask — this one, used alone, is the C1-2 defect.
164
+ """
165
+ parts = _parts(path)
166
+ for i, part in enumerate(parts[:-1]):
167
+ if part not in _TEST_DIR_NAMES:
168
+ continue
169
+ if part in _AMBIGUOUS_TEST_DIR_NAMES and (set(parts[:i]) & _DOC_ROOT_NAMES):
170
+ continue
171
+ return True
129
172
  normalized = _norm(path)
130
173
  return any(p.search(normalized) for p in _TEST_NAME_PATTERNS)
131
174
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sourcecode
3
- Version: 4.5.3
3
+ Version: 4.6.0
4
4
  Summary: Persistent structural context and ultra-fast repeated analysis for AI coding agents
5
5
  License-File: LICENSE
6
6
  Keywords: agents,ai,codebase,context,developer-tools,llm
@@ -42,7 +42,7 @@ Description-Content-Type: text/markdown
42
42
 
43
43
  **Context · Impact · Migration · Architecture · Review — everything from one structural model.**
44
44
 
45
- ![Version](https://img.shields.io/badge/version-4.5.3-blue)
45
+ ![Version](https://img.shields.io/badge/version-4.6.0-blue)
46
46
  ![Python](https://img.shields.io/badge/python-3.9%2B-green)
47
47
 
48
48
  > **ASK Engine** is the product. The CLI command is **`ask`**. The legacy **`sourcecode`**
@@ -97,7 +97,7 @@ brew tap haroundominique/sourcecode && brew install sourcecode
97
97
  # pip / pipx
98
98
  pipx install sourcecode # or: pip install sourcecode
99
99
 
100
- ask version # ask 4.5.3 — and, on a build that has aged,
100
+ ask version # ask 4.6.0 — and, on a build that has aged,
101
101
  # how many releases have probably shipped since
102
102
  ```
103
103
 
@@ -1,13 +1,13 @@
1
- sourcecode/__init__.py,sha256=TB3NC3ZIhSq8CPZFlNi9Y6qRAqfWAwRIj4a5XRpDjx0,308
1
+ sourcecode/__init__.py,sha256=4dXthmwMtG3-Py64d-gE1ZDK9uCes0_GCsrNgSGBt3I,308
2
2
  sourcecode/adaptive_scanner.py,sha256=yJBKjNpkY6bpueYJ2YnRezen3sYZDecEt7WaaNWdqug,9466
3
- sourcecode/archetype.py,sha256=HBGTTaS-bVHS6pdacPUMKcMxklkDEKnZMgz48Kc3yec,37630
3
+ sourcecode/archetype.py,sha256=CZvRLpkHot_D8D3JFQVorr-EHDJyh0BbS7RnSTqBigM,40499
4
4
  sourcecode/architectural_baseline.py,sha256=agDSwGEakdkgLLh5xnd3hiD_TC6pRujWt4klnuYF1wo,18689
5
5
  sourcecode/architectural_delta.py,sha256=E6MkyjWl-1ZgR4KSKnI785gUAfc8nq-lK-jd_rupQBU,12496
6
6
  sourcecode/architecture_analyzer.py,sha256=GFc4ek-s1IHWM7pl-0L32WahZ93AmDgrAMcOuBKA5Dk,61463
7
7
  sourcecode/architecture_summary.py,sha256=BVVRHd952cjRhjHnR6CPrvKgaa-tdM16l-pBi1yDCPs,32395
8
8
  sourcecode/ast_extractor.py,sha256=aXJjZ7XjAdxa99hWNm4SyzPqx4gNWiTbUYjNWGNl-Vc,51213
9
- sourcecode/baseline_autocapture.py,sha256=xxfXT9Nu4uiDbWMZRawcCXsA4fjgEjZW2qI173dq3GE,14962
10
- sourcecode/cache.py,sha256=Es9ZLqGRoheMv2bOD1azaFKqU2zJKJVudHhjQKcIxy8,40128
9
+ sourcecode/baseline_autocapture.py,sha256=tmMLexQSeaQRRbqxAd7walvGQqnF0mdpxQmkL40rlhI,15623
10
+ sourcecode/cache.py,sha256=9XJUNSGzqM6DNWy4erUO-OcZY9qkrTCCcOJ-OANVrw8,41409
11
11
  sourcecode/cache_model.py,sha256=hUGWNKBKLV76mGveFffOzr9fqFD1ZydJVakseRsg14s,17401
12
12
  sourcecode/call_surface.py,sha256=fiqYfHooxN1fX9oQoysq1LS3LoZcobhyUNjAGEZKpwk,4148
13
13
  sourcecode/caller_metrics.py,sha256=--sFGDnIog_YGu9xZHMcB91F5xZakfaAKvy15xU54hg,7904
@@ -17,7 +17,7 @@ sourcecode/chain_rules.py,sha256=Bi6UHfgd-GxWswmnHRcPz5jdbAuqka3Zkz_P-MTvqhw,127
17
17
  sourcecode/change_plan.py,sha256=kFjjp16XYbupgkv1CPkfqo39_SiZPMRQ8OOfC-Vy9eg,7929
18
18
  sourcecode/cir_graphs.py,sha256=9G0HHj1kw2325IDyzo2OpX73BNswEckecf4MZUXB4JM,12078
19
19
  sourcecode/classifier.py,sha256=JBzPwSSrDG-tUHAbcKB678HRbjLpD-ohzbzzO62mgpo,20114
20
- sourcecode/cli.py,sha256=J9pEXkcMWFHg0FXW6CxWa69zvP2DHHOXlgliKXNc4Dc,492534
20
+ sourcecode/cli.py,sha256=iTTJJyP4yugJ37xOA-G7W0WslC31d7fL4If-Gx4mIpM,492713
21
21
  sourcecode/client_calls.py,sha256=daRTgbXNUOfkzGXJpVb6A737R_Thhka8vw1_YgxCaLg,13548
22
22
  sourcecode/code_notes_analyzer.py,sha256=EJemNCNc9Dn-1RZYu-aNbK0ELzmsyC4s6FdHi3XyNEI,9392
23
23
  sourcecode/compare.py,sha256=xq3zsqAOAw4AWkoD9khb9xDP_O3KvwqO9k-pf8sbi3g,10951
@@ -48,7 +48,7 @@ sourcecode/envelope.py,sha256=fpF_8znvPqGXKZb0UPYcCYjm27yO7Pr-WsdXNE-tEoY,8911
48
48
  sourcecode/environment_resolution.py,sha256=2cEsF741-Cb6PR9Hiv57BpOTWl4gEnHMQsiRzcNcjJM,18872
49
49
  sourcecode/error_schema.py,sha256=uwosfNaSujtYm11_732Hu92z5ITV040fQDaIyefSvR4,1683
50
50
  sourcecode/evidence_provider.py,sha256=GSSL44JEaouO5AHks2sB3d1YvC9xIKIld1yBYxZpXxo,4277
51
- sourcecode/explain.py,sha256=8i1gQL8qNUO6NuUmzgO4vYdGGBn5_lp-1XqI9LPyx2k,25271
51
+ sourcecode/explain.py,sha256=yqxvKiLF0XMMg11PdgqMFrsim5SXuEKM1U2cLvCLspM,29935
52
52
  sourcecode/file_chunker.py,sha256=3vkM3mDQ5eE_yTPvUgjyjpGFBIjkW6_mrBmIbrylnA8,16444
53
53
  sourcecode/file_classifier.py,sha256=pJCeN9KqWpAwKMCgGP4KDsBjuWMeo4zlbj5fv3hk9dA,15587
54
54
  sourcecode/filter_surface.py,sha256=HeV1qU0y_UO95KNTy3Xb551zXvv7_ns2gyxCS5uC3fs,16418
@@ -73,10 +73,10 @@ sourcecode/openrewrite_recipe.py,sha256=4dyY5twRB6Xew-G9Zy5FafNZwlMU1S7DnRa2FqPk
73
73
  sourcecode/output_budget.py,sha256=__DQrIg7MGsYrd0_S3lyd3AG0a0jWNGr55v5Gg9kK0U,12347
74
74
  sourcecode/parse_cache.py,sha256=SnHOhNTvAqHm_PXImPVBuj7NEoOQRn8dfIZHEqOFENs,7565
75
75
  sourcecode/path_admission.py,sha256=OGNSoluhVh8YYOBZBnACUrkmH4Tp_yxjGLifkcsxQpo,5838
76
- sourcecode/path_filters.py,sha256=9yk8fA29sZSPyvI8UjF6fIOY37Hd3hs8txUiNXsQGwg,11182
76
+ sourcecode/path_filters.py,sha256=LmTYq735orssKTIuxTX1mE1FHAXF7dVEJEse35xE1RA,12371
77
77
  sourcecode/perf.py,sha256=GAcEoouPIlPMCQIcHNToxK6K3WdIR-lj9aFg4prOYJI,9743
78
78
  sourcecode/pipe_contract.py,sha256=PML0Er5d8uDyec4OrdUXuyqHbGPUYndjeaKUjkFz2u4,8369
79
- sourcecode/posture.py,sha256=x1LY98XSjZzG4xThZYJYK3e5BTYDmJhyg5eqJ5KquNk,72999
79
+ sourcecode/posture.py,sha256=nmvlJOBwnwV5aAOmO7GMfU35hNgArCQFSuESsYzcS8s,76101
80
80
  sourcecode/pr_comment_renderer.py,sha256=239PmJdf95av_ZW236C7_tvq_ahECeuF9AZxrheJ4OQ,15573
81
81
  sourcecode/pr_impact.py,sha256=eaFgcPMDsjaF8VZvcitFWCPCabMTi-khtGdxpK0wXTQ,22461
82
82
  sourcecode/prepare_context.py,sha256=B8TiBQbazxUOz6D37Dt9zqbOtD5b2S_fR3MB15P9DrI,237206
@@ -91,9 +91,9 @@ sourcecode/relevance_scorer.py,sha256=0AgEt4KrV73nioMqBgjhGjtY7L2C7L7cSyKtj3IKcr
91
91
  sourcecode/remedies.py,sha256=8lgvKkW1WuBG-QfTnA0uBhEKnwWGSvgSMWCfTjzYhEY,6516
92
92
  sourcecode/rename_refactor.py,sha256=h6dNFlB9aZ_3q6heeHBkgXQeXaT03nvPSsYH6P8qxFg,12965
93
93
  sourcecode/repo_classifier.py,sha256=FG1vaWKdWXsWdl-S8hjVMiTqcwgaRXkDyvK4rPcOGtQ,22681
94
- sourcecode/repository_ir.py,sha256=uuJsPmtgSEi4MKVKJEiZl8bpvU1QFU2Oslxn96p5xi8,356849
94
+ sourcecode/repository_ir.py,sha256=yrQd8yB9NO0M2SLkI0bQ5ty1RjTldQOhVbr3IoNkXJk,357437
95
95
  sourcecode/ris.py,sha256=SjjDNMcc9Zr9us9G8ikBVkoPXKK3BpM-dBodSGMh_g4,24783
96
- sourcecode/risk.py,sha256=cgkHu4fjAyLs9xSytegdv4b8ua8RqRJEQ5rHbQKgzfo,34851
96
+ sourcecode/risk.py,sha256=xYzXMINAkA1lGdlTiBpK9ODhor3c_UrNtL33P88mGJs,39850
97
97
  sourcecode/rule_catalog.py,sha256=buU6qZI1j1EjqAWOIzeDhZ9kO_6Tp_Ma-dmexNWkMe0,4324
98
98
  sourcecode/runtime_classifier.py,sha256=uTAD6BDCiBLUZEDRfqk718kM4RTT_vAbfkcOI2_Xx58,18432
99
99
  sourcecode/sarif.py,sha256=_3ggbtV0O2iK5zeUiJSS36KxpPRVAtGy0hzXAFBmww8,25590
@@ -111,7 +111,7 @@ sourcecode/servlet_surface.py,sha256=kwIBS6aT-DJHu8mqzvhlYrhaqAq_005SDAeVZ7yx8Y0
111
111
  sourcecode/source_text.py,sha256=Q5v5kAq2QFZ-HcaK5UXp_xZTbiLiFW5p09v7r_hQHnU,4878
112
112
  sourcecode/spring_event_topology.py,sha256=5_ON_21Le5zbG-1GRc5GLIi5HJfy_QjcXLVPC5WeUGQ,18055
113
113
  sourcecode/spring_findings.py,sha256=qgWz4LLL5TFe3eG-mTttlrDTx-6ps3EOuLA-S2HvQCM,16945
114
- sourcecode/spring_impact.py,sha256=KkThpPmBbBJk_1H8f0cVWOuz6vBN-yX5qcTQ3Sjmjnk,74797
114
+ sourcecode/spring_impact.py,sha256=SW7gyjnzf1f9dLH296s1-EmxgZvitl1MjHAnTQlUfmQ,78145
115
115
  sourcecode/spring_model.py,sha256=zOAgFmrRbG4a6KLm1TJl55aWMyPNsz3OS3FSczqPG6A,16594
116
116
  sourcecode/spring_profiles.py,sha256=-kwrCK0O-MRjrCp6SA1d31t--o4Tg87iqPVrTwxoIUs,19605
117
117
  sourcecode/spring_properties.py,sha256=kPTk5qAJxdbHc1hclhHAP0O-hCIzBn8PTh3lzl-I4a4,8400
@@ -121,7 +121,7 @@ sourcecode/spring_tx_analyzer.py,sha256=_wqjRktqdPS6PiXXxtTkCD1t6BaE2uhqx6ioYY-U
121
121
  sourcecode/summarizer.py,sha256=0aD4x3vgPngqBCEBKGuES1J2Vk5f7mqCm_ZWErwm3js,27025
122
122
  sourcecode/target_admission.py,sha256=wFZ4pzlxhiF6Q6s2lEAZEzcCfj1Y6xNvujjt8MdO0Qo,7154
123
123
  sourcecode/test_gap_ranking.py,sha256=hl-tTyQUXZGJycG1b6npbLeE_DaVHaOsvPwG9qqC5y8,15711
124
- sourcecode/test_sources.py,sha256=cMGVNLbYZZ2yt2PVcySK4scbjKpCLDAKhIViyWN_pRM,6986
124
+ sourcecode/test_sources.py,sha256=6qXI47rVt_DFEbziFUyU-_KIUIGt8YQ5Uf4-sdiAhKU,9277
125
125
  sourcecode/token_estimate.py,sha256=G-ivzIJfVq5TLQN04dC0WK4LZJjWVYL54Ebcxe4fJ7M,9268
126
126
  sourcecode/tree_utils.py,sha256=8GAkIfQAsvtEudIeW1l4ooH_oRtrWR8cpJQJsEa_Pfw,2093
127
127
  sourcecode/type_usage_surface.py,sha256=51IrKRQoIoRnlsiDjHnqpJBn2rc6E59aRhgS0HTzAF0,4428
@@ -190,8 +190,8 @@ sourcecode/telemetry/consent.py,sha256=pQdl-QeLl6Gcibn0eWHSKZrm-HYSsjpVqOnjrgFp8
190
190
  sourcecode/telemetry/events.py,sha256=4_yeO58U-Cwc1Qb27VB0_EjhmroY0k91n3_VGxeALB8,2776
191
191
  sourcecode/telemetry/filters.py,sha256=RzxauTz8HliO4BllQnXEXc7zTeqdCZi5MgqGEDuW7OQ,6570
192
192
  sourcecode/telemetry/transport.py,sha256=4gGHsq0WeY9VywEZXA3vUxykfiYnw9uuqfjAAec7F8o,1681
193
- sourcecode-4.5.3.dist-info/METADATA,sha256=md2ShdxOBK6RppqSdpqL8YjqIDoo3PRAHMZxb96YT64,34161
194
- sourcecode-4.5.3.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
195
- sourcecode-4.5.3.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
196
- sourcecode-4.5.3.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
197
- sourcecode-4.5.3.dist-info/RECORD,,
193
+ sourcecode-4.6.0.dist-info/METADATA,sha256=6XkofNasppQ6luCtFPesmVOf4wVVd-JWSACEyVw85r4,34161
194
+ sourcecode-4.6.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
195
+ sourcecode-4.6.0.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
196
+ sourcecode-4.6.0.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
197
+ sourcecode-4.6.0.dist-info/RECORD,,