sourcecode 3.2.0__py3-none-any.whl → 3.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of sourcecode might be problematic. Click here for more details.

sourcecode/cli.py CHANGED
@@ -149,6 +149,88 @@ def _check_pipeline_coherence(sm: "SourceMap") -> list[str]: # type: ignore[nam
149
149
 
150
150
  return issues
151
151
 
152
+ # ---------------------------------------------------------------------------
153
+ # Command tiers — what we promise about a command, published where it is read
154
+ # ---------------------------------------------------------------------------
155
+
156
+ #: The support commitment behind every command — one authority for it, read by
157
+ #: `--help`, by the README command table and by the user guide.
158
+ #:
159
+ #: A tier is a **stability** promise, never a value ranking. `posture` is the most
160
+ #: differentiated capability in the product *and* it is experimental: its shape can
161
+ #: still change under a minor, so nobody should gate a pipeline on it yet. Those are
162
+ #: two facts, and collapsing them onto one axis is how a surface starts lying — the
163
+ #: first panel of `--help` still says "start here" for the commands worth learning
164
+ #: first, and this table says what each command's output is worth relying on.
165
+ #:
166
+ #: This is not the pricing tier: Free/Pro is a separate axis (docs/PRODUCT_TIERS.md)
167
+ #: and gates repository size, never capability.
168
+ #:
169
+ #: Assigning the tiers is a later milestone's work (commands moving between them,
170
+ #: `parked` disappearing from the default help). Naming them is what ships here, so
171
+ #: the vocabulary exists before it is used against a command.
172
+ COMMAND_TIERS: "tuple[tuple[str, str, tuple[str, ...]], ...]" = (
173
+ ("core", "contract stable within a major — safe to gate CI on", (
174
+ "endpoints", "spring-audit", "migrate-check",
175
+ "impact", "impact-chain", "pr-impact", "verify",
176
+ )),
177
+ ("supported", "maintained; fields are added, never removed without a major", (
178
+ "verify-edit", "review-pr", "plan", "compare", "delta", "contract-diff",
179
+ "fix-bug", "rename-class", "prepare-context", "onboard", "explain",
180
+ "export", "repo-ir", "validation", "modernize", "chunk-file",
181
+ "cold-start", "baseline", "activate", "config", "schema", "version",
182
+ "cache", "auth", "mcp", "telemetry",
183
+ )),
184
+ ("experimental", "shape may change in a minor — do not gate CI on it", (
185
+ "posture", "archetype", "retrieve",
186
+ )),
187
+ ("parked", "kept working, no longer developed", ()),
188
+ )
189
+
190
+ #: Tiers whose membership is short enough to name in `--help`. `supported` is the
191
+ #: remainder by construction, and printing twenty-six names would bury the two lists
192
+ #: a reader acts on.
193
+ _TIERS_NAMED_IN_HELP = ("core", "experimental", "parked")
194
+
195
+
196
+ def command_tier(name: str) -> "str | None":
197
+ """The tier of a registered command, or None when it is in no tier.
198
+
199
+ None is a defect, not a state: the battery fails on it. It is returned rather
200
+ than defaulted so an unassigned command can never be silently published as
201
+ `supported`.
202
+ """
203
+ for tier, _promise, members in COMMAND_TIERS:
204
+ if name in members:
205
+ return tier
206
+ return None
207
+
208
+
209
+ def _tier_help_block() -> str:
210
+ """The tier table as `--help` prints it — generated from `COMMAND_TIERS`, because
211
+ a hand-written catalogue beside a generated one drifts (it is what left
212
+ `endpoints` unlisted while it shipped)."""
213
+ import textwrap
214
+
215
+ lines = ["[bold]Command tiers[/bold] [dim](a stability promise, not a value ranking):[/dim]"]
216
+ indent = " " * 4
217
+ for tier, promise, members in COMMAND_TIERS:
218
+ if tier not in _TIERS_NAMED_IN_HELP:
219
+ listed = "every other command in the panels below"
220
+ elif members:
221
+ listed = " · ".join(members)
222
+ else:
223
+ listed = "none today"
224
+ lines.append(f" [bold]{tier:<14}[/bold]{promise}")
225
+ # Pre-wrapped so the renderer never breaks a command name at its hyphen.
226
+ wrapped = textwrap.fill(
227
+ listed, width=70, initial_indent=indent, subsequent_indent=indent,
228
+ break_on_hyphens=False, break_long_words=False,
229
+ )
230
+ lines.append(f"[dim]{wrapped}[/dim]")
231
+ return "\n".join(lines)
232
+
233
+
152
234
  def _build_help_text() -> str:
153
235
  """Build --help text dynamically based on current license state."""
154
236
  try:
@@ -170,10 +252,30 @@ Cache warms on first scan; later calls reuse pre-built context instead of rescan
170
252
  Scan and warm time scale with repo size — small repos in seconds, large repos (thousands
171
253
  of files) in minutes. Semantic analysis itself is sub-second; repo indexing dominates.
172
254
 
173
- [bold]Primary usage:[/bold]
174
- ask --compact high-signal summary (~2,500–4,000 tokens)
175
- ask --compact --git-context include git hotspots and uncommitted files
176
- ask --agent full structured JSON for AI agents
255
+ [bold]Start here — Java/Spring analysis:[/bold]
256
+ posture . --diff dev:prod [dim]# effective access, two profile sets (exp.)[/dim]
257
+ endpoints . [dim]# endpoints + effective path + policy[/dim]
258
+ spring-audit . [dim]# TX anomalies + security surface[/dim]
259
+ migrate-check . --compact [dim]# Boot 2→3: located blockers + effort[/dim]
260
+
261
+ [bold]Agent context:[/bold]
262
+ ask --compact [dim]# high-signal summary (~2,500–4,000 tokens)[/dim]
263
+ ask --compact --git-context [dim]# + git hotspots and uncommitted files[/dim]
264
+ ask --agent [dim]# full structured JSON for AI agents[/dim]
265
+
266
+ [bold]Change and risk:[/bold]
267
+ impact-chain <Class> . [dim]# blast radius w/ TX + security per hop[/dim]
268
+ impact <Class> . [dim]# reverse deps → endpoints reached[/dim]
269
+ pr-impact . --since main [dim]# same, scoped to a diff; gating codes[/dim]
270
+ verify . [dim]# contract gate, baseline-relative[/dim]
271
+ verify-edit . [dim]# did working-tree edits change behaviour?[/dim]
272
+ [dim]modernize · explain <Class> · validation · export · repo-ir[/dim]
273
+
274
+ [dim]Spring commands take the INTERFACE, not the Impl — callers inject it.[/dim]
275
+
276
+ [dim]Every command is grouped in the panels below; full reference in docs/USER_GUIDE.md[/dim]
277
+
278
+ {_tier_help_block()}
177
279
 
178
280
  [bold]Auth commands:[/bold]
179
281
  auth status [dim]# show current plan and auth state[/dim]
@@ -181,14 +283,17 @@ of files) in minutes. Semantic analysis itself is sub-second; repo indexing domi
181
283
 
182
284
  [bold]Cache commands:[/bold]
183
285
  cache status [dim]# cache size, hit keys, last-warmed timestamp[/dim]
184
- cache warm [dim]# pre-build cache ahead of an agent session[/dim]
286
+ cache warm [dim]# pre-build structural layers + compact view
287
+ # (--agent warms the agent view too; --full,
288
+ # --env-map and raised --depth are NOT warmed)[/dim]
185
289
  cache clear [dim]# clear all cached results for this repo[/dim]
186
290
 
187
291
  [bold]Examples:[/bold]
292
+ ask posture . --diff default:prod -o posture.json
293
+ ask endpoints . -f json | jq '.endpoints | map(select(.method=="POST"))'
294
+ ask spring-audit . --min-severity high -o audit.json
188
295
  ask my-project --compact
189
296
  ask . --compact --git-context --copy
190
- ask . --changed-only --git-context
191
- ask prepare-context onboard my-project
192
297
  ask prepare-context delta . --since main
193
298
 
194
299
  [bold]Subcommands:[/bold]
@@ -808,7 +913,7 @@ try:
808
913
  except Exception:
809
914
  pass
810
915
 
811
- telemetry_app = typer.Typer(help="Manage anonymous telemetry (on by default; opt-out).", rich_markup_mode="rich")
916
+ telemetry_app = typer.Typer(help="Manage anonymous telemetry (off by default; opt-in).", rich_markup_mode="rich")
812
917
  app.add_typer(telemetry_app, name="telemetry")
813
918
 
814
919
  mcp_app = typer.Typer(help="MCP integration: setup, status, serve, remove.", rich_markup_mode="rich")
@@ -836,8 +941,9 @@ app.add_typer(retrieve_app, name="retrieve")
836
941
  def _maybe_show_telemetry_notice() -> None:
837
942
  """Show first-run telemetry notice once, on interactive TTYs only.
838
943
 
839
- Telemetry is on by default (opt-out). We inform rather than ask, then
840
- mark the notice as shown so it appears only once.
944
+ Telemetry is off by default (opt-in). The notice is an invitation, not a
945
+ disclosure: nothing has been collected when it appears. Marked as shown so it
946
+ appears only once.
841
947
  """
842
948
  try:
843
949
  from sourcecode.telemetry.config import has_been_asked, mark_asked
@@ -1678,12 +1784,21 @@ def main(
1678
1784
  f"ex={_excl_key},depth={effective_depth}"
1679
1785
  )
1680
1786
  _core_h = _hashlib.sha256(_core_flags_str.encode()).hexdigest()[:8]
1681
- if _git_sha and _git_root_str:
1682
- _core_key = f"{_git_sha}-{_core_h}"
1787
+ # Freshness comes from ONE authority (cache.worktree_signature): the exact
1788
+ # tree state an analysis would read, not the committed HEAD. Keyed on HEAD,
1789
+ # this cache answered for a tree it had not read — an uncommitted pom.xml
1790
+ # gaining a dependency was served the pre-edit analysis with is_stale:false.
1791
+ # Clean tree → the sha, so the ordinary repeat run still hits.
1792
+ _tree_sig = _cache_mod.worktree_signature(
1793
+ Path(_git_root_str) if _git_root_str else target,
1794
+ scope=target,
1795
+ )
1796
+ if _tree_sig:
1797
+ _core_key = f"{_tree_sig}-{_core_h}"
1683
1798
  else:
1684
- # No git history (untracked/no-commit repo) — stable synthetic key
1685
- # scoped per repo path via cache_dir(); invalidated by --no-cache or cache clear.
1686
- _core_key = f"nogit-{_core_h}"
1799
+ # The tree cannot be described (no git, unreadable) — a stable synthetic
1800
+ # key would answer forever from the first run, so skip the cache instead.
1801
+ _core_key = ""
1687
1802
 
1688
1803
  # ── View flags: output presentation only (no re-analysis needed) ──
1689
1804
  _view_flags_str = (
@@ -1698,7 +1813,7 @@ def main(
1698
1813
 
1699
1814
  # ── Lookup ──────────────────────────────────────────────────────
1700
1815
  # Step 1: try L1 to obtain the core_hash needed for L2 key
1701
- _l1_result = _cache_mod.read_core(target, _core_key)
1816
+ _l1_result = _cache_mod.read_core(target, _core_key) if _core_key else None
1702
1817
 
1703
1818
  # Additive overlays (--env-map / --git-context) miss L1 because they sit in
1704
1819
  # the core key, yet neither changes the semantic core — env walks config
@@ -1714,8 +1829,8 @@ def main(
1714
1829
  # we inject exactly the overlays we flipped to land the hit.
1715
1830
  _l1_needs_env_inject = False
1716
1831
  _l1_needs_git_inject = False
1717
- if _l1_result is None and (env_map or git_context):
1718
- _sha_prefix = _git_sha if _git_sha else "nogit"
1832
+ if _l1_result is None and _core_key and (env_map or git_context):
1833
+ _sha_prefix = _tree_sig
1719
1834
  _flippable = []
1720
1835
  if git_context:
1721
1836
  _flippable.append("gc") # inject git is cheap + additive
@@ -3271,10 +3386,11 @@ def prepare_context_cmd(
3271
3386
  from dataclasses import asdict
3272
3387
  import time as _time
3273
3388
 
3274
- # Task-level cache: keyed on (task, git_head, symptom) so warm calls complete in <1s.
3389
+ # Task-level cache: keyed on (task, tree state, symptom) so warm calls complete in <1s.
3390
+ # The tree state comes from the same authority as every other layer, so an
3391
+ # uncommitted edit invalidates here exactly as it does for the root command.
3275
3392
  # Skip for diff-dependent tasks (delta, review-pr), fast mode, and llm_prompt
3276
3393
  # (those embed per-call content that must not be served from cache).
3277
- import subprocess as _pctx_sub
3278
3394
  import hashlib as _pctx_hash
3279
3395
  from sourcecode import cache as _pctx_cache
3280
3396
  _pctx_git_sha = ""
@@ -3282,15 +3398,19 @@ def prepare_context_cmd(
3282
3398
  _pctx_cacheable = task not in ("delta", "review-pr") and not fast and not llm_prompt
3283
3399
  if _pctx_cacheable:
3284
3400
  try:
3285
- _sha_r2 = _pctx_sub.run(
3286
- ["git", "-C", str(target), "rev-parse", "--short", "HEAD"],
3287
- capture_output=True, text=True, timeout=3,
3401
+ _pctx_git_sha = _pctx_cache.worktree_signature(
3402
+ _resolve_repo_root(target), scope=target
3288
3403
  )
3289
- _pctx_git_sha = _sha_r2.stdout.strip()
3290
3404
  except Exception:
3291
3405
  pass
3292
3406
  if _pctx_git_sha:
3293
- _sym_h = _pctx_hash.sha256((symptom or "").encode()).hexdigest()[:8]
3407
+ # Every option that changes the answer belongs in the key. --all and
3408
+ # --include-config did not, and went unnoticed only because the key
3409
+ # required a git sha: in a non-git tree nothing was cached at all, so
3410
+ # the collision could not surface.
3411
+ _sym_h = _pctx_hash.sha256(
3412
+ f"sym={symptom or ''};all={all_gaps};cfg={include_config}".encode()
3413
+ ).hexdigest()[:8]
3294
3414
  _pctx_cache_key = f"pctx-{task}-{_pctx_git_sha}-{_sym_h}-{format or 'json'}"
3295
3415
  _cached_pctx = _pctx_cache.read(target, _pctx_cache_key)
3296
3416
  if _cached_pctx is not None:
@@ -3722,16 +3842,22 @@ def prepare_context_cmd(
3722
3842
  @telemetry_app.command("status")
3723
3843
  def telemetry_status() -> None:
3724
3844
  """Show current telemetry setting."""
3725
- from sourcecode.telemetry.config import config_file_path, has_been_asked, is_enabled
3845
+ from sourcecode.telemetry.config import config_file_path, is_enabled, stored_choice
3726
3846
  enabled = is_enabled()
3727
- asked = has_been_asked()
3847
+ choice = stored_choice()
3728
3848
  status = "enabled" if enabled else "disabled"
3729
- typer.echo(f"Telemetry: {status} (on by default; opt-out)")
3730
- if not asked:
3731
- typer.echo(" (first-run notice not yet shown — will show on next run)")
3849
+ typer.echo(f"Telemetry: {status} (off by default; opt-in)")
3850
+ # "off because you said so" and "off because nobody asked you" are different
3851
+ # answers, and a buyer auditing this needs to be told which one they have.
3852
+ if choice is None:
3853
+ typer.echo(" No choice recorded — nothing has been collected or sent.")
3854
+ else:
3855
+ typer.echo(f" Your recorded choice: {'enabled' if choice else 'disabled'}.")
3732
3856
  typer.echo(f" Config: {config_file_path()}")
3733
- typer.echo(" Disable: ask telemetry disable")
3734
- typer.echo(" Or set env var: SOURCECODE_TELEMETRY=0 (or DO_NOT_TRACK=1)")
3857
+ if enabled:
3858
+ typer.echo(" Disable: ask telemetry disable (or SOURCECODE_TELEMETRY=0, DO_NOT_TRACK=1)")
3859
+ else:
3860
+ typer.echo(" Enable: ask telemetry enable (or SOURCECODE_TELEMETRY=1)")
3735
3861
 
3736
3862
 
3737
3863
  @telemetry_app.command("enable")
@@ -3753,7 +3879,7 @@ def telemetry_disable() -> None:
3753
3879
  from sourcecode.telemetry.config import set_enabled
3754
3880
  set_enabled(False)
3755
3881
  typer.echo("Telemetry disabled. No data will be collected or sent.")
3756
- typer.echo("Telemetry is on by default; this opt-out is remembered.")
3882
+ typer.echo("Telemetry is off by default; this choice is recorded so the notice stops asking.")
3757
3883
  typer.echo("Re-enable at any time: ask telemetry enable")
3758
3884
 
3759
3885
 
@@ -4332,6 +4458,10 @@ def endpoints_cmd(
4332
4458
  if controller:
4333
4459
  _ctrl_lower = controller.lower()
4334
4460
  endpoints_list = [e for e in endpoints_list if _ctrl_lower in e.get("controller", "").lower()]
4461
+ # A filter changes the population a count is about; `--limit` does not — it
4462
+ # cuts the rendering. Measuring `total` after the cut published the display
4463
+ # size as the measurement (ADR-0008 R5), so the two are separated here.
4464
+ _selected = endpoints_list
4335
4465
  if limit is not None and limit > 0:
4336
4466
  endpoints_list = endpoints_list[:limit]
4337
4467
  if path_prefix or controller or limit is not None:
@@ -4341,11 +4471,12 @@ def endpoints_cmd(
4341
4471
  _no_sec_before = data.get("no_security_signal")
4342
4472
  _undoc_before = data.get("undocumented")
4343
4473
  _no_sec_after = sum(
4344
- 1 for e in endpoints_list
4474
+ 1 for e in _selected
4345
4475
  if e.get("security", {}).get("policy") == "none_detected"
4346
4476
  )
4347
4477
  data["endpoints"] = endpoints_list
4348
- data["total"] = len(endpoints_list)
4478
+ data["total"] = len(_selected)
4479
+ data["shown"] = len(endpoints_list)
4349
4480
  data["no_security_signal"] = _no_sec_after
4350
4481
  data["undocumented"] = _no_sec_after
4351
4482
  data["_filter"] = {
@@ -4355,6 +4486,10 @@ def endpoints_cmd(
4355
4486
  "total_before_filter": _total_before,
4356
4487
  "no_security_signal_before_filter": _no_sec_before,
4357
4488
  "undocumented_before_filter": _undoc_before,
4489
+ "note": (
4490
+ "`total` counts the endpoints the filters selected; `shown` is how "
4491
+ "many of them this document lists, which `--limit` cuts."
4492
+ ),
4358
4493
  }
4359
4494
 
4360
4495
  if by_controller:
@@ -5818,10 +5953,14 @@ def spring_audit_cmd(
5818
5953
  SEC-003 @Transactional on @Controller/@RestController (TX in wrong layer)
5819
5954
 
5820
5955
  \b
5821
- CI/CD usage:
5956
+ CI/CD usage — this gate is ABSOLUTE (any finding fails, including debt that
5957
+ was already there). For a baseline-relative gate that fails only on findings
5958
+ a change INTRODUCES, use `ask verify --fail-on new` (or `ask pr-impact
5959
+ --fail-on`); `ask baseline capture/diff` tracks the debt over time.
5822
5960
  ask spring-audit . --ci # exit 1 on any finding
5823
5961
  ask spring-audit . --ci --min-severity high # exit 1 only on high/critical
5824
5962
  ask spring-audit . --ci --format github-comment # Markdown PR comment + exit 1
5963
+ ask verify . --fail-on new # exit 1 only on NEW findings
5825
5964
 
5826
5965
  \b
5827
5966
  Examples:
@@ -7330,138 +7469,24 @@ def fix_bug_cmd(
7330
7469
  )
7331
7470
 
7332
7471
 
7333
- # Method signatures that are almost always FRAMEWORK ENTRY POINTS, not dead code —
7334
- # invoked by a dispatcher (reflection / XML / SPI), invisible to a static Java
7335
- # call-graph. Generalizes beyond any one framework.
7336
- _DYNAMIC_ENTRY_SIGNATURE_RE = __import__("re").compile(
7337
- r"\(\s*DispatchContext\b" # OFBiz Service Engine service
7338
- r"|HttpServletRequest\s+\w+\s*,\s*HttpServletResponse" # OFBiz event / servlet handler
7339
- r"|@(?:Scheduled|PostConstruct|PreDestroy|EventListener|Bean|"
7340
- r"RequestMapping|GetMapping|PostMapping|Path|GET|POST|Provider|"
7341
- r"ApplicationScoped|Singleton)\b", # annotation-dispatched entry
7342
- )
7343
- # Config file extensions where a framework wires classes by name (XML/props/yaml).
7344
- _CONFIG_REF_EXTS: frozenset = frozenset({".xml", ".properties", ".yml", ".yaml", ".groovy"})
7345
- _CONFIG_SCAN_MAX_FILES: int = 12000
7346
- _CONFIG_SCAN_MAX_BYTES: int = 256 * 1024
7347
-
7348
-
7349
- def _partition_static_unreferenced(nodes: list[dict], root: Path) -> tuple[list[dict], list[dict]]:
7350
- """Split zero-degree classes into (truly_unreferenced, framework_dispatched).
7351
-
7352
- A class with no static callers is NOT necessarily dead: frameworks invoke
7353
- classes via reflection, XML/SPI config, or annotations that a static call-graph
7354
- cannot see (e.g. Apache OFBiz Service Engine services, JAX-RS resources,
7355
- ServiceLoader providers, scheduled beans). We exclude a candidate when EITHER:
7356
- 1. its source declares a dynamic-entry method signature, OR
7357
- 2. its simple name / FQN is referenced from a non-Java config file.
7358
- Whatever survives is reported as *statically_unreferenced* — never a confident
7359
- "dead zone".
7360
- """
7361
- import os
7362
- if not nodes:
7363
- return [], []
7364
- by_simple: dict[str, list[dict]] = {}
7365
- for n in nodes:
7366
- simple = (n.get("fqn") or "").rsplit(".", 1)[-1]
7367
- if simple:
7368
- by_simple.setdefault(simple, []).append(n)
7369
-
7370
- dispatched_fqns: set[str] = set()
7371
-
7372
- # 1. Source-signature allowlist (bounded — candidate set is small).
7373
- for n in nodes:
7374
- src = n.get("source_file")
7375
- if not src:
7376
- continue
7377
- try:
7378
- txt = (root / src).read_text(encoding="utf-8", errors="replace")
7379
- except OSError:
7380
- continue
7381
- if _DYNAMIC_ENTRY_SIGNATURE_RE.search(txt):
7382
- dispatched_fqns.add(n["fqn"])
7383
-
7384
- # 2. Config-reference scan — find candidate names wired from XML/props/yaml.
7385
- unresolved_simple = {s for s, ns in by_simple.items()
7386
- if any(x["fqn"] not in dispatched_fqns for x in ns)}
7387
- if unresolved_simple:
7388
- files_scanned = 0
7389
- for dirpath, dirnames, filenames in os.walk(root):
7390
- dirnames[:] = [d for d in dirnames
7391
- if d not in {".git", "build", "out", "target", "node_modules", ".gradle"}]
7392
- for fname in filenames:
7393
- ext = os.path.splitext(fname)[1].lower()
7394
- if ext not in _CONFIG_REF_EXTS:
7395
- continue
7396
- if files_scanned >= _CONFIG_SCAN_MAX_FILES or not unresolved_simple:
7397
- break
7398
- fpath = os.path.join(dirpath, fname)
7399
- try:
7400
- with open(fpath, "r", encoding="utf-8", errors="replace") as fh:
7401
- text = fh.read(_CONFIG_SCAN_MAX_BYTES)
7402
- except OSError:
7403
- continue
7404
- files_scanned += 1
7405
- for simple in list(unresolved_simple):
7406
- if simple in text:
7407
- for x in by_simple.get(simple, []):
7408
- dispatched_fqns.add(x["fqn"])
7409
- unresolved_simple.discard(simple)
7410
- if files_scanned >= _CONFIG_SCAN_MAX_FILES or not unresolved_simple:
7411
- break
7472
+ def _partition_static_unreferenced(
7473
+ nodes: list[dict], root: Path
7474
+ ) -> tuple[list[dict], list[dict]]:
7475
+ """Split callerless types into (no dispatch signal found, framework-dispatched).
7412
7476
 
7413
- # 3. Nested-type qualified-reference scan (SC-2 residual). The static call-graph
7414
- # does not emit an edge for a nested-type member access in ordinary method-body
7415
- # code — e.g. `PropertyType.AdminGroupPresentation.NAME` credits the enclosing
7416
- # type, not the nested holder — so a referenced nested class can look
7417
- # zero-degree. When a candidate is a nested type (its second-to-last FQN segment
7418
- # is an UpperCamel enclosing type) and its qualified `Outer.Nested` form appears
7419
- # in some OTHER Java source, it is statically referenced: neither dead nor
7420
- # framework-dispatched. Structural, name-agnostic (pattern derived from the
7421
- # candidate's own FQN).
7422
- import re as _re_nested
7423
- statically_referenced: set[str] = set()
7424
- nested_patterns: dict[str, "tuple"] = {}
7425
- for n in nodes:
7426
- if n["fqn"] in dispatched_fqns:
7427
- continue
7428
- parts = (n.get("fqn") or "").split(".")
7429
- if len(parts) >= 2 and parts[-2][:1].isupper():
7430
- outer, nested = parts[-2], parts[-1]
7431
- nested_patterns[n["fqn"]] = (
7432
- _re_nested.compile(r"\b" + _re_nested.escape(outer) + r"\." + _re_nested.escape(nested) + r"\b"),
7433
- n.get("source_file") or "",
7434
- )
7435
- if nested_patterns:
7436
- files_scanned = 0
7437
- for dirpath, dirnames, filenames in os.walk(root):
7438
- dirnames[:] = [d for d in dirnames
7439
- if d not in {".git", "build", "out", "target", "node_modules", ".gradle"}]
7440
- for fname in filenames:
7441
- if not fname.endswith(".java"):
7442
- continue
7443
- if files_scanned >= _CONFIG_SCAN_MAX_FILES or len(statically_referenced) == len(nested_patterns):
7444
- break
7445
- fpath = os.path.join(dirpath, fname)
7446
- rel = os.path.relpath(fpath, root)
7447
- try:
7448
- with open(fpath, "r", encoding="utf-8", errors="replace") as fh:
7449
- text = fh.read(_CONFIG_SCAN_MAX_BYTES)
7450
- except OSError:
7451
- continue
7452
- files_scanned += 1
7453
- for fqn, (pat, own) in nested_patterns.items():
7454
- if fqn in statically_referenced or rel == own:
7455
- continue
7456
- if pat.search(text):
7457
- statically_referenced.add(fqn)
7458
- if files_scanned >= _CONFIG_SCAN_MAX_FILES or len(statically_referenced) == len(nested_patterns):
7459
- break
7477
+ Kept as the shape `modernize` renders; the fact itself is derived once, in
7478
+ `reference_facts` — the authority that also states the third answer this pair
7479
+ cannot express, `unknown`.
7480
+ """
7481
+ from sourcecode.reference_facts import (
7482
+ NO_STATIC_CALLERS, UNKNOWN_DISPATCH, analyze_type_references,
7483
+ )
7460
7484
 
7461
- unreferenced = [n for n in nodes
7462
- if n["fqn"] not in dispatched_fqns and n["fqn"] not in statically_referenced]
7463
- dispatched = [n for n in nodes if n["fqn"] in dispatched_fqns]
7464
- return unreferenced, dispatched
7485
+ facts = analyze_type_references(nodes, root)
7486
+ by_fqn = {str(n.get("fqn")): n for n in nodes}
7487
+ def _nodes_of(status: str) -> list[dict]:
7488
+ return [by_fqn[e.fqn] for e in facts.of_status(status) if e.fqn in by_fqn]
7489
+ return _nodes_of(NO_STATIC_CALLERS), _nodes_of(UNKNOWN_DISPATCH)
7465
7490
 
7466
7491
 
7467
7492
  @app.command("modernize")
@@ -7479,7 +7504,7 @@ def modernize_cmd(
7479
7504
  help="Copy output to clipboard after a successful run.",
7480
7505
  ),
7481
7506
  ) -> None:
7482
- """[Pro*] Modernization planning: coupling, dead zones, risky modules, refactor candidates.
7507
+ """[Pro*] Modernization planning: coupling, callerless types, risky modules, refactor candidates.
7483
7508
 
7484
7509
  Note: [Pro*] label is reserved for a future licensing gate. This command currently
7485
7510
  runs without authentication. Behavior may change in a future version.
@@ -7489,7 +7514,7 @@ def modernize_cmd(
7489
7514
 
7490
7515
  Analyzes the repo for:
7491
7516
  - High-coupling modules (high in-degree + out-degree nodes)
7492
- - Dead zones (isolated symbols with no callers)
7517
+ - Types with no static caller, split from those a framework may dispatch (unknown)
7493
7518
  - Risk hotspots (high fan-in + security annotations + transaction boundaries)
7494
7519
  - Cross-module dependency tangles
7495
7520
  - Subsystem summary with member counts
@@ -7559,27 +7584,29 @@ def modernize_cmd(
7559
7584
  if _src and _src in _file_churn:
7560
7585
  _fqn_churn[_n["fqn"]] = _file_churn[_src]
7561
7586
 
7562
- # High-coupling nodes: high in_degree (many dependents = risky to change)
7563
- coupling_nodes = sorted(
7587
+ # High-coupling nodes: high in_degree (many dependents = risky to change).
7588
+ # The measurement is the full set; the list below is the display cut, and the
7589
+ # summary counts the first, never the second (ADR-0008 R5).
7590
+ _coupling_all = sorted(
7564
7591
  [n for n in graph_nodes if n.get("in_degree", 0) >= 3],
7565
7592
  key=lambda n: (-n.get("in_degree", 0), n.get("fqn", "")),
7566
- )[:20]
7567
-
7568
- # Statically-unreferenced zones: classes with zero in-degree AND zero out-degree
7569
- # in the Java call-graph. These are NOT necessarily dead — framework dispatch
7570
- # (reflection / XML / SPI / annotations) is invisible to a static graph — so we
7571
- # partition out framework-dispatched entry points before reporting, and never
7572
- # call the survivors "dead". (Defect 5: OFBiz Service-Engine services and event
7573
- # handlers were false-positive "dead zones".)
7574
- _zero_degree = sorted(
7575
- [n for n in graph_nodes
7576
- if n.get("in_degree", 0) == 0 and n.get("out_degree", 0) == 0
7577
- and n.get("type") in ("class", "interface")],
7578
- key=lambda n: n.get("fqn", ""),
7579
7593
  )
7580
- dead_zones, framework_dispatched = _partition_static_unreferenced(_zero_degree, root)
7581
- dead_zones = dead_zones[:20]
7582
- framework_dispatched = framework_dispatched[:20]
7594
+ coupling_nodes = _coupling_all[:20]
7595
+
7596
+ # Which types nothing calls — one authority (`reference_facts`), three answers.
7597
+ # A static call-graph cannot see reflection, XML/SPI wiring or annotation
7598
+ # dispatch, so "no caller found" and "not called" are different statements: the
7599
+ # first is published here, the second never is. (M7 seam 6 / defect C3-8: this
7600
+ # surface examined only types with no edges in EITHER direction while calling
7601
+ # the result "zero static callers", and published `0` where the answer was
7602
+ # unknown.)
7603
+ from sourcecode.reference_facts import (
7604
+ NO_STATIC_CALLERS as _NO_CALLERS,
7605
+ UNKNOWN_DISPATCH as _UNKNOWN_DISPATCH,
7606
+ analyze_type_references as _analyze_type_references,
7607
+ )
7608
+
7609
+ _reference_facts = _analyze_type_references(graph_nodes, root)
7583
7610
 
7584
7611
  # Hotspot candidates: high in-degree service/repository/controller nodes,
7585
7612
  # ranked by composite score (in_degree × 2 + git_churn) for volatility signal.
@@ -7682,11 +7709,17 @@ def modernize_cmd(
7682
7709
  _cross_module_tangles = _cross_module_tangles[:15]
7683
7710
 
7684
7711
  _summary = {
7685
- "total_classes": len([n for n in graph_nodes if n.get("type") in ("class", "interface")]),
7712
+ # Same population the reference partition is measured over, so the three
7713
+ # statuses below sum to exactly this number.
7714
+ "total_classes": _reference_facts.classes_examined,
7686
7715
  "total_subsystems": len(subsystems),
7687
- "high_coupling_nodes": len(coupling_nodes),
7688
- "statically_unreferenced": len(dead_zones),
7689
- "framework_dispatched": len(framework_dispatched),
7716
+ "high_coupling_nodes": len(_coupling_all),
7717
+ "high_coupling_nodes_shown": len(coupling_nodes),
7718
+ # From the measurement, not from the lists below — those are truncated for
7719
+ # display, and a count taken off them is a different number (ADR-0008 R5).
7720
+ "statically_unreferenced": _reference_facts.count(_NO_CALLERS),
7721
+ "framework_dispatched": _reference_facts.count(_UNKNOWN_DISPATCH),
7722
+ "reference_status": _reference_facts.to_dict(),
7690
7723
  }
7691
7724
  # BUG #6 (v1.68.0): `member_count` counts ALL graph members in the subsystem —
7692
7725
  # classes, methods AND fields — so it runs ~5x higher than the class count and
@@ -7725,7 +7758,7 @@ def modernize_cmd(
7725
7758
  "tier_note": (
7726
7759
  "This repository exceeds the free-tier size limit. "
7727
7760
  "Upgrade to Pro for full analysis on enterprise-scale monoliths: "
7728
- "dead zones, dependency tangles, refactor candidates ranked by git "
7761
+ "callerless types, dependency tangles, refactor candidates ranked by git "
7729
7762
  "churn, and complete coupling graphs."
7730
7763
  ),
7731
7764
  "summary": _summary,
@@ -7753,19 +7786,36 @@ def modernize_cmd(
7753
7786
  # three share one reconciliation note so they can never contradict.
7754
7787
  "high_coupling_nodes_note": CALLER_METRIC_RECONCILIATION,
7755
7788
  "statically_unreferenced": [
7756
- {"fqn": n["fqn"], "type": n.get("type", ""), "role": n.get("role", "other")}
7757
- for n in dead_zones
7789
+ {
7790
+ "fqn": e.fqn, "type": e.type, "role": e.role,
7791
+ # Why this type is in this list, so the claim can be checked
7792
+ # rather than trusted.
7793
+ "basis": e.basis,
7794
+ }
7795
+ for e in _reference_facts.of_status(_NO_CALLERS)[:20]
7758
7796
  ],
7759
7797
  "statically_unreferenced_note": (
7760
- "Zero static callers in the Java call-graph. NOT confirmed dead: verify "
7761
- "no framework dispatch (reflection, XML/SPI config, annotations) before "
7762
- "removing. Classes detected as framework-dispatched are listed separately "
7763
- "under framework_dispatched and excluded from this list."
7798
+ "No incoming edge in the Java call-graph and no framework-dispatch "
7799
+ "signal found. This is absence of evidence, NOT confirmed dead code: "
7800
+ "a static graph cannot see reflection, XML/SPI wiring or annotation "
7801
+ "dispatch, and this scan reads only the repository's own sources and "
7802
+ "configuration. Types that DO carry a dispatch signal are reported "
7803
+ "separately as framework_dispatched — status "
7804
+ "`unknown_framework_dispatch`, because whether they run is not "
7805
+ "decidable here. Counts live in summary.reference_status; this list "
7806
+ f"shows at most 20 of {_reference_facts.count(_NO_CALLERS)}."
7764
7807
  ),
7765
7808
  "framework_dispatched": [
7766
- {"fqn": n["fqn"], "type": n.get("type", ""), "role": n.get("role", "other")}
7767
- for n in framework_dispatched
7809
+ {"fqn": e.fqn, "type": e.type, "role": e.role, "basis": e.basis}
7810
+ for e in _reference_facts.of_status(_UNKNOWN_DISPATCH)[:20]
7768
7811
  ],
7812
+ "framework_dispatched_note": (
7813
+ "No static caller, but a framework may invoke them (published "
7814
+ "annotation, entry-point signature, or the name wired from a "
7815
+ "configuration file). Status is `unknown`, never 'alive' and never "
7816
+ "'dead'. This list shows at most 20 of "
7817
+ f"{_reference_facts.count(_UNKNOWN_DISPATCH)}."
7818
+ ),
7769
7819
  "subsystem_summary": _subsystem_summary,
7770
7820
  "subsystem_summary_note": _subsystem_summary_note,
7771
7821
  "cross_module_tangles": _cross_module_tangles,
@@ -7787,8 +7837,10 @@ def modernize_cmd(
7787
7837
  if hotspots else
7788
7838
  "high_coupling_nodes shows the most-referenced classes — start there. "
7789
7839
  )
7790
- + "statically_unreferenced lists classes with no Java callers — review "
7791
- + "for framework dispatch (XML/reflection/SPI) before removing. "
7840
+ + "statically_unreferenced lists types with no Java caller and no "
7841
+ + "dispatch signal found — a starting point for review, not a "
7842
+ + "removal list; confirm at runtime before deleting. "
7843
+ + _reference_facts.statement + ". "
7792
7844
  + "In cross_module_tangles, coupling_type=cyclic entries are the "
7793
7845
  + "real tangles to decompose first; directional entries are normal "
7794
7846
  + "layering ranked by coupling strength."
@@ -8138,6 +8190,9 @@ def version_cmd() -> None:
8138
8190
  # contract does.
8139
8191
  "envelope_version": ENVELOPE_VERSION,
8140
8192
  "published_schemas": available_schemas(),
8193
+ # Same list `ask schema` prints, so the two surfaces cannot disagree
8194
+ # about what this release publishes.
8195
+ "registries": ["facts-v1"],
8141
8196
  }, ensure_ascii=False))
8142
8197
 
8143
8198
 
@@ -8162,13 +8217,33 @@ def schema_cmd(
8162
8217
  Examples:
8163
8218
  ask schema # list published schemas
8164
8219
  ask schema envelope-v1 # print the response-envelope schema
8220
+ ask schema facts-v1 # print the fact registry (one fact, one authority)
8165
8221
  """
8166
8222
  from sourcecode.envelope import available_schemas, load_schema
8223
+ from sourcecode.facts import load_registry
8167
8224
 
8225
+ # The fact registry is a contract too: which facts have a single authority,
8226
+ # who derives each, and which modules may emit it (ADR-0008 R11). Published
8227
+ # through the command that already answers "what shape does this release
8228
+ # produce?" rather than a command of its own.
8168
8229
  names = available_schemas()
8230
+ if name == "facts-v1":
8231
+ _emit_command_output(
8232
+ json.dumps(load_registry(), indent=2, ensure_ascii=False),
8233
+ output_path,
8234
+ False,
8235
+ stamp_envelope=False,
8236
+ )
8237
+ return
8169
8238
  if name is None:
8170
8239
  _emit_command_output(
8171
- json.dumps({"schemas": names}, indent=2, ensure_ascii=False),
8240
+ # Schemas describe output shapes; the fact registry describes which
8241
+ # answers have a single authority. Both are contracts this release
8242
+ # carries, listed apart because they are not the same kind of thing.
8243
+ json.dumps(
8244
+ {"schemas": names, "registries": ["facts-v1"]},
8245
+ indent=2, ensure_ascii=False,
8246
+ ),
8172
8247
  output_path,
8173
8248
  False,
8174
8249
  )
@@ -8202,7 +8277,7 @@ def config_cmd() -> None:
8202
8277
  from sourcecode.telemetry.config import config_file_path, is_enabled
8203
8278
  typer.echo(f"ask {__version__}")
8204
8279
  typer.echo(f"Config: {config_file_path()}")
8205
- typer.echo(f"Telemetry: {'enabled' if is_enabled() else 'disabled'}")
8280
+ typer.echo(f"Telemetry: {'enabled' if is_enabled() else 'disabled'} (off by default; opt-in)")
8206
8281
  typer.echo("")
8207
8282
  typer.echo("Manage telemetry:")
8208
8283
  typer.echo(" ask telemetry enable")
@@ -9330,17 +9405,32 @@ def cache_warm_cmd(
9330
9405
  ) -> None:
9331
9406
  """Pre-populate the cache by running a fresh analysis.
9332
9407
 
9333
- Runs a full analysis to populate L1/L2 caches and rebuild the RIS
9334
- (Repository Intelligence Snapshot). Useful after a merge/pull in CI.
9408
+ Runs a full analysis to populate the snapshot cache, rebuild the RIS and build
9409
+ the shared Canonical IR. Useful after a merge/pull in CI.
9410
+
9411
+ \b
9412
+ It is not a general warm, and it says so when it finishes: it warms the
9413
+ compact view (add --agent for the agent view), and analysis flags that change
9414
+ what is analysed — --env-map, --depth N, --exclude — rescan anyway. Run
9415
+ `ask cache model` for what a warm gives each command.
9416
+
9417
+ \b
9418
+ In CI without a persisted cache directory, every pipeline pays the cold cost
9419
+ this command reports. Cache ~/.sourcecode/cache and ~/.sourcecode between jobs,
9420
+ or budget the cold run explicitly.
9335
9421
  """
9336
9422
  import shutil as _shutil
9337
9423
  import subprocess as _sub
9424
+ import time as _warm_time
9338
9425
  # Warm exactly the path given. Resolving up to the enclosing git root warmed
9339
9426
  # (and reported on) the whole monorepo when asked for one module — every other
9340
9427
  # command scopes to the argument, so `cache warm ./service-a` populated a cache
9341
9428
  # for a different target than `ask ./service-a` reads.
9342
9429
  target = Path(path).resolve()
9343
9430
  _git_root = _resolve_repo_root(Path(path))
9431
+ # Cost is reported for the whole warm, the CIR build included — that is what a
9432
+ # pipeline pays, and quoting only the analysis half would understate it (C4-5).
9433
+ _warm_t0 = _warm_time.monotonic()
9344
9434
  typer.echo(f"Warming cache for {target} …", err=True)
9345
9435
  if _git_root != target:
9346
9436
  typer.echo(
@@ -9375,6 +9465,58 @@ def cache_warm_cmd(
9375
9465
  err=True,
9376
9466
  )
9377
9467
 
9468
+ # A warm that reports only what it built lets the reader assume it built
9469
+ # everything — which is how "el warm no es un warm general" became a field
9470
+ # finding (C4-3) rather than a documented limit. State both halves, and state
9471
+ # the cost, because in CI it is paid per pipeline (C4-5).
9472
+ _warm_elapsed = _warm_time.monotonic() - _warm_t0
9473
+ _warmed_view = "compact + agent views" if (compact and agent) else (
9474
+ "agent view" if agent else "compact view"
9475
+ )
9476
+ typer.echo(
9477
+ f"Warmed in {_warm_elapsed:.0f}s: {_warmed_view}, RIS, shared CIR, parse cache.",
9478
+ err=True,
9479
+ )
9480
+ typer.echo(
9481
+ "NOT warmed: prepare-context task answers (refactor / fix-bug / generate-tests / "
9482
+ "delta / review-pr), and any run whose analysis flags differ (--env-map, --depth N, "
9483
+ "--exclude) — those rescan. Measured: `endpoints` and `migrate-check` gain nothing "
9484
+ "from a warm. `ask cache model` says what a warm gives each command.",
9485
+ err=True,
9486
+ )
9487
+ typer.echo(
9488
+ f"In CI without a persisted ~/.sourcecode, every pipeline pays this {_warm_elapsed:.0f}s again.",
9489
+ err=True,
9490
+ )
9491
+
9492
+
9493
+ @cache_app.command("model")
9494
+ def cache_model_cmd(
9495
+ json_output: bool = typer.Option(False, "--json", help="Output as JSON."),
9496
+ markdown: bool = typer.Option(False, "--markdown", help="Output the tables published in the user guide."),
9497
+ ) -> None:
9498
+ """What each cache layer stores, what invalidates it, and what a warm helps.
9499
+
9500
+ \b
9501
+ Answers, per command, the question a warm raises: will this be fast next time?
9502
+ the answer — a warm stores what this command returns
9503
+ the shared work — a warm removes the Java parse / the shared IR; the command
9504
+ still computes its own answer
9505
+ nothing — a warm does not touch it
9506
+
9507
+ One rule covers invalidation: every layer keys on the exact tree state, so any
9508
+ change to the analysed files invalidates it, committed or not.
9509
+ """
9510
+ from sourcecode import cache_model as _cmodel
9511
+
9512
+ if json_output:
9513
+ import json as _j
9514
+ typer.echo(_j.dumps(_cmodel.as_dict(), indent=2, ensure_ascii=False))
9515
+ elif markdown:
9516
+ typer.echo(_cmodel.render_markdown())
9517
+ else:
9518
+ typer.echo(_cmodel.render_text())
9519
+
9378
9520
 
9379
9521
  @cache_app.command("freshness")
9380
9522
  def cache_freshness_cmd(
@@ -9516,6 +9658,78 @@ def _stderr_is_interactive() -> bool:
9516
9658
  return False
9517
9659
 
9518
9660
 
9661
+ #: How `--help` groups and orders the commands — one authority for it, applied
9662
+ #: below to the registered commands themselves.
9663
+ #:
9664
+ #: `--help` is the first thing a new user reads, and it was ordered by the file's
9665
+ #: registration order: a reader met `prepare-context` and `repo-ir` before
9666
+ #: `posture` or `endpoints`, so the surface introduced itself with its weakest
9667
+ #: 20 %. Field evaluation #3 scored discoverability 3/10 and found `endpoints` —
9668
+ #: the command it valued most — by accident, inside another command's JSON.
9669
+ #:
9670
+ #: Panels are ordered by what a reader should meet first, and inside a panel by
9671
+ #: what answers the most common question. Nothing is hidden: every command is
9672
+ #: still listed, and the battery fails if one is missing from this table.
9673
+ HELP_PANELS: "tuple[tuple[str, tuple[str, ...]], ...]" = (
9674
+ ("Java/Spring analysis — start here", (
9675
+ "posture", "endpoints", "spring-audit", "migrate-check",
9676
+ )),
9677
+ ("Change and risk", (
9678
+ "impact-chain", "impact", "pr-impact", "verify", "verify-edit",
9679
+ "review-pr", "plan", "compare", "delta", "contract-diff", "fix-bug",
9680
+ "rename-class",
9681
+ )),
9682
+ ("Context for AI agents", (
9683
+ "prepare-context", "onboard", "explain", "export", "repo-ir",
9684
+ "validation", "modernize", "chunk-file", "cold-start",
9685
+ )),
9686
+ # Inside a panel, Typer renders plain commands before command groups, so the
9687
+ # groups (`cache`, `auth`, …) are listed last here to keep this table and the
9688
+ # rendered help in the same order.
9689
+ ("Setup and inspection", (
9690
+ "activate", "config", "schema", "version",
9691
+ "cache", "auth", "mcp", "telemetry", "baseline",
9692
+ )),
9693
+ ("Experimental — shape may change", (
9694
+ "archetype", "retrieve",
9695
+ )),
9696
+ )
9697
+
9698
+ #: Where a command lands when it is not in the table. Visible on purpose: an
9699
+ #: unlisted command must look unfinished, not disappear.
9700
+ _UNPANELLED = "Other commands"
9701
+
9702
+
9703
+ def _apply_help_panels() -> None:
9704
+ """Group and order the registered commands per `HELP_PANELS`.
9705
+
9706
+ Typer renders panels in the order it first meets them and commands in
9707
+ registration order, so the grouping is applied to the registered objects
9708
+ rather than to thirty decorator call sites — one place to read, one place a
9709
+ new command has to be added to.
9710
+ """
9711
+ rank: dict[str, tuple[int, int]] = {}
9712
+ panel_of: dict[str, str] = {}
9713
+ for panel_index, (panel, names) in enumerate(HELP_PANELS):
9714
+ for name_index, name in enumerate(names):
9715
+ rank[name] = (panel_index, name_index)
9716
+ panel_of[name] = panel
9717
+
9718
+ def _sort_key(entry: Any) -> "tuple[int, int, str]":
9719
+ name = str(getattr(entry, "name", "") or "")
9720
+ position = rank.get(name, (len(HELP_PANELS), 0))
9721
+ return (position[0], position[1], name)
9722
+
9723
+ for registry in (app.registered_commands, app.registered_groups):
9724
+ for entry in registry:
9725
+ name = str(getattr(entry, "name", "") or "")
9726
+ entry.rich_help_panel = panel_of.get(name, _UNPANELLED)
9727
+ registry.sort(key=_sort_key)
9728
+
9729
+
9730
+ _apply_help_panels()
9731
+
9732
+
9519
9733
  def _force_utf8_streams() -> None:
9520
9734
  """Force UTF-8 on stdout AND stderr so Unicode characters (em-dash, arrows, box
9521
9735
  drawing) survive on Windows where the default console codec is cp1252 (BUG-1).