sourcecode 3.8.0__py3-none-any.whl → 4.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of sourcecode might be problematic. Click here for more details.

sourcecode/__init__.py CHANGED
@@ -4,4 +4,4 @@ ASK Engine is the product. ``ask`` is the canonical CLI command; ``sourcecode``
4
4
  the legacy compatibility alias and the Python/PyPI package name. See
5
5
  docs/PRODUCT_IDENTITY.md (normative)."""
6
6
 
7
- __version__ = "3.8.0"
7
+ __version__ = "4.0.1"
sourcecode/cache.py CHANGED
@@ -354,6 +354,7 @@ def status(repo_root: Path) -> dict[str, Any]:
354
354
  "cores": 0, "snapshots": 0, "views": 0, "cas_blobs": 0,
355
355
  "total_size_bytes": 0, "total_size_mb": 0.0,
356
356
  "current_git_head": current_head,
357
+ "stores": _store_breakdown(repo_root, 0),
357
358
  **ris_fields,
358
359
  }
359
360
  cores = list(cache_d.glob("core-*.json.gz"))
@@ -371,10 +372,57 @@ def status(repo_root: Path) -> dict[str, Any]:
371
372
  "total_size_bytes": total_bytes,
372
373
  "total_size_mb": round(total_bytes / (1024 * 1024), 2),
373
374
  "current_git_head": current_head,
375
+ # C3-36: `CAS blobs: 0`, `Total size: 0.1 MB` immediately after an 89-second
376
+ # warm of 3 342 files. Every figure above was true and described one store
377
+ # of three — the warm's output mostly lands in the shared CIR and the parse
378
+ # cache, which this command never counted. A status that reports a third of
379
+ # the state reads as a warm that did nothing.
380
+ "stores": _store_breakdown(repo_root, total_bytes),
374
381
  **ris_fields,
375
382
  }
376
383
 
377
384
 
385
+ def _store_breakdown(repo_root: Path, core_bytes: int) -> "dict[str, Any]":
386
+ """What a warm actually populated, by store. Best-effort per store: a store
387
+ that cannot be inspected is reported as unavailable, never as empty."""
388
+ out: "dict[str, Any]" = {
389
+ "core": {
390
+ "cache_dir": str(cache_dir(repo_root)),
391
+ "bytes": core_bytes,
392
+ "scope": "this repository",
393
+ "holds": "core snapshots, rendered views and their CAS blobs",
394
+ },
395
+ }
396
+ try:
397
+ from sourcecode.context_cache import ContextCache # noqa: PLC0415
398
+
399
+ ctx = ContextCache.for_repo(repo_root).stats()
400
+ out["shared_cir"] = {
401
+ "cache_dir": ctx["cache_dir"],
402
+ "entries": ctx["contexts"],
403
+ "bytes": ctx["bytes_stored"],
404
+ "scope": "this repository",
405
+ "holds": "the shared Canonical IR that explain/impact/posture reuse",
406
+ }
407
+ except Exception:
408
+ out["shared_cir"] = {"available": False}
409
+ try:
410
+ from sourcecode import parse_cache as _pc # noqa: PLC0415
411
+
412
+ out["parse"] = {
413
+ **_pc.store_stats(),
414
+ "holds": "per-file parses, content-addressed across every repository",
415
+ }
416
+ except Exception:
417
+ out["parse"] = {"available": False}
418
+ out["note"] = (
419
+ "`cache warm` populates all of these; the core store alone is a third of "
420
+ "the answer, and reading it as the whole is how a completed warm looks like "
421
+ "an empty cache."
422
+ )
423
+ return out
424
+
425
+
378
426
  def clear(repo_root: Path, *, clear_ris: bool = False) -> int:
379
427
  """Delete cache files for *repo_root*. Returns the number of files removed.
380
428
 
sourcecode/cache_model.py CHANGED
@@ -131,6 +131,13 @@ COMMANDS: tuple[CommandCache, ...] = (
131
131
  "builds — the parse it used to repeat for itself. `--diff` compares two profile "
132
132
  "sets over that one IR, so the second side costs the resolution only.",
133
133
  "10.1 s → 1.6 s"),
134
+ CommandCache("risk", ("cir", "parse"), "shared", False,
135
+ "Composes what the audit, impact-chain and the posture already answer, so it "
136
+ "pays each of their costs once over the shared CIR a warm builds — one parse "
137
+ "for the whole composition, and the reachability query is cached per symbol "
138
+ "within the run.",
139
+ "not measured on the battery yet — the composition is bounded by the "
140
+ "`spring-audit` + `impact-chain` costs listed here, not by new analysis"),
134
141
  CommandCache("endpoints", ("ris", "parse"), "shared", False,
135
142
  "Recomputes the endpoint surface on every run, over a parse a warm has already "
136
143
  "paid for. Until 3.7.0 the extractor parsed every file itself instead of reading "
sourcecode/cli.py CHANGED
@@ -183,7 +183,7 @@ COMMAND_TIERS: "tuple[tuple[str, str, tuple[str, ...]], ...]" = (
183
183
  "cache", "auth", "mcp", "telemetry",
184
184
  )),
185
185
  ("experimental", "shape may change in a minor — do not gate CI on it", (
186
- "posture", "archetype",
186
+ "risk", "posture", "archetype",
187
187
  )),
188
188
  # `retrieve` publishes 15+ intents whose answers the other commands already
189
189
  # give better: measured on the battery, `security-surface` merely re-states
@@ -297,6 +297,27 @@ def _group_help_block() -> str:
297
297
  return "\n".join(lines)
298
298
 
299
299
 
300
+ def _remedy_help_block() -> str:
301
+ """The front page names the answers the field asked for as missing features.
302
+
303
+ C4-8 was closed in the payload: a limitation now carries the invocation that
304
+ resolves it. But the evaluator who wrote `ask posture --provenance` and `ask
305
+ contracts init` by hand was reading `--help` first, and neither shipped flag
306
+ was on it. The block is rendered from `remedies.REMEDIES`, so the pointer and
307
+ the capability cannot drift — `tests/test_remedies.py` already refuses a
308
+ remedy whose option the CLI does not accept.
309
+ """
310
+ from sourcecode.remedies import headlines
311
+
312
+ lines = ["[bold]Questions the repository alone does not answer:[/bold]"]
313
+ for invocation, headline in headlines():
314
+ # Bare invocation, like every other block on this page: the header panels
315
+ # print without the `ask ` prefix (C3-16).
316
+ bare = invocation.removeprefix("ask ")
317
+ lines.append(f" {bare:<33}[dim]# {headline}[/dim]")
318
+ return "\n".join(lines)
319
+
320
+
300
321
  def _build_help_text() -> str:
301
322
  """Build --help text dynamically based on current license state."""
302
323
  try:
@@ -323,6 +344,9 @@ of files) in minutes. Semantic analysis itself is sub-second; repo indexing domi
323
344
  endpoints . [dim]# endpoints + effective path + policy[/dim]
324
345
  spring-audit . [dim]# TX anomalies + security surface[/dim]
325
346
  migrate-check . --compact [dim]# Boot 2→3: located blockers + effort[/dim]
347
+ risk . [dim]# defects ranked by reach × access (exp.)[/dim]
348
+
349
+ {_remedy_help_block()}
326
350
 
327
351
  [bold]Agent context:[/bold]
328
352
  ask --compact [dim]# high-signal summary (~2,500–4,000 tokens)[/dim]
@@ -5475,11 +5499,21 @@ def validation_cmd(
5475
5499
  # The payload carries the same note, so a pipeline loses nothing.
5476
5500
  if _note:
5477
5501
  _notice(f"Note: {_note}")
5502
+ # C2-15: the console said "0 body endpoints, 1179 gaps" over a payload carrying
5503
+ # `body_endpoints_in_code: 1201` — it printed the *declared-constraint* count
5504
+ # under the *code-surface* name, so the one line most readers see stated the
5505
+ # opposite of the answer. Both axes are named, in their own units.
5506
+ _declared = _summary.get("endpoints_with_body", 0)
5507
+ _in_code = _summary.get("body_endpoints_in_code")
5508
+ _bodies = (
5509
+ f"{_in_code} body endpoints in code, {_declared} with a declared constraint "
5510
+ f"surface" if _in_code is not None
5511
+ else f"{_declared} routes with a declared constraint surface"
5512
+ )
5478
5513
  _emit_command_output(
5479
5514
  output, output_path, copy,
5480
5515
  success_msg=f"Validation surface written to {output_path} "
5481
- f"({_summary.get('endpoints_with_body', 0)} body endpoints, "
5482
- f"{_summary.get('gaps', 0)} gaps)",
5516
+ f"({_bodies}, {_summary.get('gaps', 0)} gaps)",
5483
5517
  )
5484
5518
 
5485
5519
  from sourcecode.mcp_nudge import nudge_mcp_if_needed as _nudge
@@ -6308,11 +6342,13 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6308
6342
  gc = (result.security_posture or {}).get("gate_coverage")
6309
6343
  if not gc:
6310
6344
  return []
6311
- not_covered = gc.get("endpoints_not_carrying_gate", gc.get("not_carrying_gate", 0))
6345
+ not_covered = gc.get("endpoints_not_carrying_gate", 0)
6312
6346
  gates = ", ".join(f"`{g}`" for g in gc.get("gate_annotations", [])) or "the detected gate"
6313
6347
  # One unit, named: these are endpoints, not handler methods. The two counts differ
6314
6348
  # (one method can serve several mappings) and must never be summed together.
6315
- total = gc.get("endpoints_total", gc.get("total_controller_handlers", 0))
6349
+ # C2-14: read only the explicit keys — falling back to the retired aliases is how
6350
+ # a name that lies about its unit survives its own removal.
6351
+ total = gc.get("endpoints_total", 0)
6316
6352
  lines: list[str] = ["", "---", ""]
6317
6353
  if not_covered == 0:
6318
6354
  _gated = gc.get("endpoints_carrying_gate", total)
@@ -6326,7 +6362,6 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6326
6362
  lines.append(f"✅ **Gate coverage** — all {total} endpoints carry {gates}.")
6327
6363
  return lines
6328
6364
 
6329
- covered = gc.get("possibly_filter_covered", 0)
6330
6365
  _standard = gc.get("endpoints_standard_guarded", 0)
6331
6366
  lines.append(
6332
6367
  f"🔓 **Gate coverage** — {not_covered} of {total} endpoints do not carry "
@@ -6338,9 +6373,18 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6338
6373
  f"standard guard, {not_covered} on neither — the three sum to {total}._"
6339
6374
  )
6340
6375
  if gc.get("reconstructed_filter_patterns"):
6376
+ # C1-20: the split, not the total. "N match a filter pattern" is the sentence
6377
+ # a reader turns into "N are covered", and in the field the filters doing the
6378
+ # matching were CORS, headers and logging.
6379
+ _auth = gc.get("filter_covered_authenticating", 0)
6380
+ _gap = gc.get("filter_gap_non_authenticating", 0)
6381
+ _unknown = gc.get("filter_authentication_unknown", 0)
6341
6382
  lines.append(
6342
- f"_{covered} of those match a reconstructed servlet filter pattern "
6343
- f"(possibly filter-covered); {gc.get('no_matching_filter_pattern', 0)} match none._"
6383
+ f"_{_auth} are covered by a servlet filter that authenticates; {_gap} match "
6384
+ f"only filters that do not check the caller (CORS, headers, logging and the "
6385
+ f"like); {_unknown} match a filter whose implementation is not in this "
6386
+ f"repository; {gc.get('no_matching_filter_pattern', 0)} match no pattern at "
6387
+ f"all. Filter-chain order is not reconstructed._"
6344
6388
  )
6345
6389
  lines += ["", "<details>", "<summary>Handlers without the gate</summary>", ""]
6346
6390
  lines += [
@@ -6418,19 +6462,11 @@ def spring_audit_cmd(
6418
6462
  help=_NO_CACHE_SUBCOMMAND_HELP,
6419
6463
  ),
6420
6464
  ) -> None:
6421
- """Spring semantic audit: TX anomalies (TX-001..006) + security surface (SEC-001..003).
6465
+ """Spring semantic audit: {{RULE_SUMMARY}}.
6422
6466
 
6423
6467
  \b
6424
6468
  Detects:
6425
- TX-001 @Transactional on private/final method (CGLIB proxy bypass)
6426
- TX-002 REQUIRES_NEW nested in REQUIRED call chain
6427
- TX-003 readOnly=true boundary propagating to write operation
6428
- TX-004 NOT_SUPPORTED/NEVER within active TX chain
6429
- TX-005 Exception swallowing inside @Transactional
6430
- TX-006 Self-invocation of @Transactional sibling (proxy bypass)
6431
- SEC-001 Unsecured endpoint in annotation_based security model
6432
- SEC-002 CVE-2025-41248: @PreAuthorize on inherited method from generic supertype
6433
- SEC-003 @Transactional on @Controller/@RestController (TX in wrong layer)
6469
+ {{RULE_LIST}}
6434
6470
 
6435
6471
  \b
6436
6472
  Findings and defects are two counts, both published. A rule fires per
@@ -6462,10 +6498,7 @@ def spring_audit_cmd(
6462
6498
  \b
6463
6499
  What it does not look at (the whole list is in the payload's `non_coverage`,
6464
6500
  and in the README):
6465
- NC-001 whether request input reaches a sink — no dataflow or taint
6466
- NC-002 secrets outside Java and Spring configuration
6467
- NC-003 filter-chain ORDER — a custom filter's presence is structural only
6468
- NC-004 known vulnerabilities in dependencies — no CVE matching
6501
+ {{NON_COVERAGE}}
6469
6502
  Silence about those is not evidence that there is nothing to find.
6470
6503
  """
6471
6504
  import json as _json
@@ -6567,6 +6600,30 @@ def spring_audit_cmd(
6567
6600
  raise typer.Exit(code=1)
6568
6601
 
6569
6602
 
6603
+ def _render_audit_help(doc: str) -> str:
6604
+ """Fill the rule and non-coverage blocks from the tables that own them.
6605
+
6606
+ The rules this command detects were written out by hand in its own help, and
6607
+ stayed at `SEC-001..003` for two releases after CL-9 shipped three more — the
6608
+ C4-1 shape, where a curated list beside a generated one drifts and the
6609
+ curated one is what the reader gets. Both blocks are rendered now, so adding
6610
+ a rule updates the help by construction.
6611
+ """
6612
+ from sourcecode import non_coverage, rule_catalog
6613
+
6614
+ nc_lines = [
6615
+ f" {line}" for line in non_coverage.render_help_lines("security_surface", width=68)
6616
+ ]
6617
+ return (
6618
+ doc.replace("{{RULE_SUMMARY}}", rule_catalog.summary_line())
6619
+ .replace("{{RULE_LIST}}", rule_catalog.render_help_block())
6620
+ .replace("{{NON_COVERAGE}}", "\n".join(nc_lines))
6621
+ )
6622
+
6623
+
6624
+ spring_audit_cmd.__doc__ = _render_audit_help(spring_audit_cmd.__doc__ or "")
6625
+
6626
+
6570
6627
  # ── verify-edit: in-loop semantic diff gate ───────────────────────────────────
6571
6628
 
6572
6629
 
@@ -6840,6 +6897,74 @@ def verify_cmd(
6840
6897
  raise typer.Exit(code=report.exit_code)
6841
6898
 
6842
6899
 
6900
+ @app.command("risk")
6901
+ def risk_cmd(
6902
+ path: Path = typer.Argument(
6903
+ Path("."),
6904
+ help="Repository path (default: current directory).",
6905
+ ),
6906
+ limit: int = typer.Option(
6907
+ 50, "--limit", help="How many composed risks to publish (highest first)."
6908
+ ),
6909
+ min_band: str = typer.Option(
6910
+ "low",
6911
+ "--min-band",
6912
+ help="Floor on the composed band: critical | high | medium | low.",
6913
+ ),
6914
+ output_path: Optional[Path] = typer.Option(
6915
+ None, "--output", "-o", help="Write the report to a file instead of stdout."
6916
+ ),
6917
+ format: str = typer.Option("json", "--format", "-f", help="Output format: json or yaml."),
6918
+ ) -> None:
6919
+ """[EXPERIMENTAL] What each defect actually costs, once reach and access are in it.
6920
+
6921
+ \b
6922
+ The other commands answer one axis each, correctly, and leave the composition
6923
+ to the reader. In the field, one class was `medium` in `spring-audit`,
6924
+ `medium/5.0` in `impact-chain`, `coverage_unknown → permit_all` in `posture`,
6925
+ and executed a stored procedure that mutates the database. Composed, that is
6926
+ unauthenticated write access under the release build; separately, it was two
6927
+ commands saying "medium".
6928
+
6929
+ \b
6930
+ severity_effective = defect_severity × reachability × auth_verdict × write_effect
6931
+
6932
+ \b
6933
+ No new analysis: every factor is read from the command that already publishes
6934
+ it, and every row publishes its four factors with the authority each came
6935
+ from, so a reader can disagree with one and keep the rest. An axis that could
6936
+ not be measured is `unknown`, multiplies by 1.0, and is named in `blind_axes`.
6937
+
6938
+ \b
6939
+ Examples:
6940
+ ask risk .
6941
+ ask risk . --min-band high
6942
+ ask risk . --limit 10 -o risk.json
6943
+ """
6944
+ from sourcecode.risk import build_risk
6945
+
6946
+ path = _admit_path(path)
6947
+ if min_band not in ("critical", "high", "medium", "low"):
6948
+ _emit_error_json(
6949
+ INVALID_INPUT_CODE,
6950
+ f"--min-band expects critical | high | medium | low (got {min_band!r}).",
6951
+ hint="Example: --min-band high",
6952
+ expected="critical|high|medium|low",
6953
+ )
6954
+ raise typer.Exit(code=1)
6955
+
6956
+ data = build_risk(path, limit=limit, min_band=min_band)
6957
+ _emit_command_output(
6958
+ _serialize_dict(data, format),
6959
+ output_path,
6960
+ False,
6961
+ success_msg=(
6962
+ f"risk written to {output_path} ({data['shown']} of "
6963
+ f"{data['total_defects']} defects composed)"
6964
+ ),
6965
+ )
6966
+
6967
+
6843
6968
  @app.command("posture")
6844
6969
  def posture_cmd(
6845
6970
  path: Path = typer.Argument(
@@ -9969,6 +10094,20 @@ def cache_status_cmd(
9969
10094
  typer.echo(f"Views: {stats['views']}")
9970
10095
  typer.echo(f"CAS blobs: {stats['cas_blobs']}")
9971
10096
  typer.echo(f"Total size: {stats['total_size_mb']} MB")
10097
+ # C3-36: the three lines above describe ONE of the three stores a warm
10098
+ # fills. Printed alone after an 89 s warm they read as "nothing was
10099
+ # cached", which is the opposite of what happened.
10100
+ _stores = stats.get("stores") or {}
10101
+ for _label, _key in (("Shared CIR", "shared_cir"), ("Parse cache", "parse")):
10102
+ _store = _stores.get(_key) or {}
10103
+ if not _store or _store.get("available") is False:
10104
+ typer.echo(f"{_label + ':':<13}unavailable")
10105
+ continue
10106
+ _mb = round(_store.get("bytes", 0) / (1024 * 1024), 2)
10107
+ _scope = " (shared across repositories)" if _store.get("scope") == "shared" else ""
10108
+ typer.echo(
10109
+ f"{_label + ':':<13}{_store.get('entries', 0)} entries, {_mb} MB{_scope}"
10110
+ )
9972
10111
  # RIS section
9973
10112
  if stats.get("ris_exists"):
9974
10113
  _stale_tag = " [STALE]" if stats.get("ris_is_stale") else ""
@@ -10326,7 +10465,7 @@ HELP_PANELS: "tuple[tuple[str, tuple[str, ...]], ...]" = (
10326
10465
  "cache", "auth", "mcp", "telemetry", "baseline",
10327
10466
  )),
10328
10467
  ("Experimental — shape may change", (
10329
- "archetype", "retrieve",
10468
+ "risk", "archetype", "retrieve",
10330
10469
  )),
10331
10470
  )
10332
10471
 
@@ -343,9 +343,14 @@ class ConfidenceAnalyzer:
343
343
  gaps.append(AnalysisGap(
344
344
  area="testing",
345
345
  reason=(
346
+ # C1-19: "Java files" here has always meant the non-test ones
347
+ # — the denominator of a test ratio cannot include the tests.
348
+ # Unqualified, it read as a third file count contradicting
349
+ # `migrate-check.java_files_scanned` (which counts them all).
346
350
  f"Backend test coverage critical: {len(_java_tests)} test files "
347
- f"for {len(_java_prod)} Java files "
348
- f"({_ratio:.1%}) — {_java_test_facts.basis}"
351
+ f"for {len(_java_prod)} non-test Java files "
352
+ f"(of {len(_java_all)} Java files, {_ratio:.1%}) — "
353
+ f"{_java_test_facts.basis}"
349
354
  ),
350
355
  impact="high",
351
356
  ))
@@ -93,6 +93,25 @@ def assign_defect_ids(findings: "Iterable[SpringFinding]") -> None:
93
93
  finding.defect_id = make_defect_id(category, kind, symbol)
94
94
 
95
95
 
96
+ def _witness_sites(witnesses: "list[SpringFinding]") -> "dict[str, Any]":
97
+ """The distinct places this defect was observed, when the witnesses record one."""
98
+ sites: "list[dict[str, Any]]" = []
99
+ seen: "set[tuple[str, Any]]" = set()
100
+ for finding in witnesses:
101
+ site = (finding.evidence or {}).get("call_site")
102
+ if not isinstance(site, dict):
103
+ continue
104
+ key = (str(site.get("source_file") or ""), site.get("line"))
105
+ if key in seen:
106
+ continue
107
+ seen.add(key)
108
+ sites.append(site)
109
+ if not sites:
110
+ return {}
111
+ sites.sort(key=lambda s: (str(s.get("source_file") or ""), s.get("line") or 0))
112
+ return {"witness_sites": sites, "distinct_site_count": len(sites)}
113
+
114
+
96
115
  def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
97
116
  """One row per defect, most severe first, each naming its witnesses.
98
117
 
@@ -137,6 +156,11 @@ def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
137
156
  "rule_ids": sorted({f.pattern_id for f in witnesses}),
138
157
  "witness_count": len(witnesses),
139
158
  "witnesses": [f.id for f in witnesses],
159
+ # C2-16: witnesses of one defect are all located at the symbol that
160
+ # carries the remedy, so several of them print the same line and read
161
+ # as duplicates. Where each was actually observed is listed here, and
162
+ # a count of distinct sites so "N witnesses" can be checked against it.
163
+ **_witness_sites(witnesses),
140
164
  })
141
165
  rows.sort(key=lambda r: (SEVERITY_RANK.get(r["severity"], 9), r["symbol"]))
142
166
  return rows
@@ -209,7 +209,7 @@ def collect_signals(root: Path) -> list[Signal]:
209
209
  if _PROPERTY_KEY not in text and _ENV_KEY not in text:
210
210
  continue
211
211
  try:
212
- rel = str(path.relative_to(root))
212
+ rel = path.relative_to(root).as_posix()
213
213
  except ValueError:
214
214
  rel = str(path)
215
215
  lines = text.splitlines()