sourcecode 4.16.0__py3-none-any.whl → 4.18.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of sourcecode might be problematic. Click here for more details.
- sourcecode/__init__.py +1 -1
- sourcecode/audit_report.py +48 -6
- sourcecode/cli.py +173 -35
- sourcecode/context_cache.py +94 -7
- sourcecode/data_exposure.py +5 -1
- sourcecode/data_labels.py +10 -3
- sourcecode/declarations.py +81 -0
- sourcecode/identity_fallback.py +6 -12
- sourcecode/mcp/orchestrator.py +3 -1
- sourcecode/non_coverage.py +86 -0
- sourcecode/partial_contract.py +81 -0
- sourcecode/phased_run.py +14 -0
- sourcecode/posture.py +20 -14
- sourcecode/ris.py +4 -1
- sourcecode/risk.py +164 -34
- sourcecode/rule_pass.py +11 -1
- sourcecode/security_config.py +6 -5
- sourcecode/security_config_scan.py +4 -8
- sourcecode/security_posture.py +43 -3
- sourcecode/source_text.py +65 -0
- sourcecode/spring_findings.py +22 -0
- sourcecode/spring_model.py +22 -1
- sourcecode/spring_security_audit.py +118 -10
- sourcecode/spring_tx_analyzer.py +5 -1
- {sourcecode-4.16.0.dist-info → sourcecode-4.18.0.dist-info}/METADATA +3 -3
- {sourcecode-4.16.0.dist-info → sourcecode-4.18.0.dist-info}/RECORD +29 -27
- {sourcecode-4.16.0.dist-info → sourcecode-4.18.0.dist-info}/WHEEL +0 -0
- {sourcecode-4.16.0.dist-info → sourcecode-4.18.0.dist-info}/entry_points.txt +0 -0
- {sourcecode-4.16.0.dist-info → sourcecode-4.18.0.dist-info}/licenses/LICENSE +0 -0
sourcecode/__init__.py
CHANGED
sourcecode/audit_report.py
CHANGED
|
@@ -12,9 +12,10 @@ import hmac
|
|
|
12
12
|
import json
|
|
13
13
|
from datetime import datetime, timezone
|
|
14
14
|
from pathlib import Path
|
|
15
|
-
from typing import Any, Optional
|
|
15
|
+
from typing import Any, Callable, Optional
|
|
16
16
|
|
|
17
17
|
from sourcecode import __version__
|
|
18
|
+
from sourcecode.partial_contract import floor_counts, read_count
|
|
18
19
|
|
|
19
20
|
AUDIT_REPORT_SCHEMA = "audit-report-v1"
|
|
20
21
|
|
|
@@ -40,14 +41,30 @@ def build_audit_report(
|
|
|
40
41
|
sign_key: Optional[bytes] = None,
|
|
41
42
|
risk_limit: int = 10,
|
|
42
43
|
risk_payload: Optional[dict[str, Any]] = None,
|
|
44
|
+
progress: "Optional[Callable[..., Any]]" = None,
|
|
45
|
+
checkpoint: "Optional[Callable[[str, dict], None]]" = None,
|
|
46
|
+
partial_status: "Optional[Callable[..., dict]]" = None,
|
|
43
47
|
) -> dict[str, Any]:
|
|
44
|
-
"""Build the buyer-readable audit bundle from existing command payloads.
|
|
48
|
+
"""Build the buyer-readable audit bundle from existing command payloads.
|
|
49
|
+
|
|
50
|
+
C3-89. This command composes the whole of `risk`, and for four releases it did
|
|
51
|
+
so with none of the three arguments above — no observer carrying the deadline,
|
|
52
|
+
no checkpoint, no `partial_status` — while `cli` built no `PhasedRun` for it
|
|
53
|
+
either. The operator's budget therefore did not exist here at all: the field
|
|
54
|
+
measured **2 265 s under `ASK_MAX_ANALYSIS_SECONDS=300`** (7,55×), CPU-bound in
|
|
55
|
+
one worker, with no `partial` and no answer. C3-87 bound the second consumer of
|
|
56
|
+
that fact and moved the composition of the `_partial` keys into
|
|
57
|
+
`PhasedRun.status` so the third would get it by construction; this is the third.
|
|
58
|
+
"""
|
|
45
59
|
from sourcecode.posture import build_posture
|
|
46
60
|
|
|
47
61
|
root = Path(root).resolve()
|
|
48
62
|
if risk_payload is None:
|
|
49
63
|
from sourcecode.risk import build_risk
|
|
50
|
-
risk = build_risk(
|
|
64
|
+
risk = build_risk(
|
|
65
|
+
root, limit=risk_limit, min_band="low", profiles=profiles,
|
|
66
|
+
progress=progress, checkpoint=checkpoint, partial_status=partial_status,
|
|
67
|
+
)
|
|
51
68
|
risk_source = "computed"
|
|
52
69
|
else:
|
|
53
70
|
risk = risk_payload
|
|
@@ -65,8 +82,10 @@ def build_audit_report(
|
|
|
65
82
|
"summary": {
|
|
66
83
|
"risk_model": risk.get("model"),
|
|
67
84
|
"risk_source": risk_source,
|
|
68
|
-
|
|
69
|
-
|
|
85
|
+
# F-AV: read under either spelling, and republished under the one
|
|
86
|
+
# the bundle's own `summary.partial` already qualifies.
|
|
87
|
+
"total_defects": read_count(risk, "total_defects"),
|
|
88
|
+
"total_findings": read_count(risk, "total_findings"),
|
|
70
89
|
"risk_bands": risk.get("by_band", {}),
|
|
71
90
|
"top_risks_shown": len(top_risks),
|
|
72
91
|
"endpoint_access": access.get("summary", {}),
|
|
@@ -85,6 +104,28 @@ def build_audit_report(
|
|
|
85
104
|
"non_coverage": (risk.get("non_coverage") or {}).get("items", []),
|
|
86
105
|
},
|
|
87
106
|
}
|
|
107
|
+
# C3-89. A bundle whose evidence was cut says so **before** it is signed, and
|
|
108
|
+
# says it where a reader of the summary sees it — a signature over a truncated
|
|
109
|
+
# payload that does not declare the truncation is the worst artifact this
|
|
110
|
+
# product can produce. The keys come from `risk` itself, whichever way it
|
|
111
|
+
# arrived: a `--from-risk` file that was cut carries its own `_partial`, and
|
|
112
|
+
# packaging it silently would launder a floor into a verdict (C2-31's rule).
|
|
113
|
+
if risk.get("partial"):
|
|
114
|
+
unsigned["summary"]["partial"] = True
|
|
115
|
+
floor_counts(unsigned["summary"], True) # F-AV, on the bundle too
|
|
116
|
+
unsigned["summary"]["counts_are_floor"] = True
|
|
117
|
+
unsigned["summary"]["counts_basis"] = (
|
|
118
|
+
"The risk evidence in this bundle was stopped by the analysis budget. "
|
|
119
|
+
"`total_defects` and `risk_bands` count what the families that ran "
|
|
120
|
+
"measured, and are a floor over the repository."
|
|
121
|
+
)
|
|
122
|
+
_risk_partial = risk.get("_partial")
|
|
123
|
+
if isinstance(_risk_partial, dict):
|
|
124
|
+
unsigned["_partial"] = dict(_risk_partial)
|
|
125
|
+
unsigned["_partial"]["command"] = "audit-report"
|
|
126
|
+
unsigned["_partial"]["cut_in"] = "risk"
|
|
127
|
+
if risk_source == "from-risk":
|
|
128
|
+
unsigned["_partial"]["cut_in"] = "the `--from-risk` payload"
|
|
88
129
|
from sourcecode.provenance import build_evidence_manifest
|
|
89
130
|
unsigned["evidence_manifest"] = build_evidence_manifest(
|
|
90
131
|
unsigned,
|
|
@@ -117,7 +158,8 @@ def render_markdown(report: dict[str, Any]) -> str:
|
|
|
117
158
|
f"- Repository: `{repo}`",
|
|
118
159
|
f"- ASK version: `{(report.get('tool') or {}).get('version', '')}`",
|
|
119
160
|
f"- Profile set: `{profile}`",
|
|
120
|
-
f"- Total defects: `{summary
|
|
161
|
+
f"- Total defects: `{read_count(summary, 'total_defects')}`"
|
|
162
|
+
+ (" **(floor — this run was cut short)**" if summary.get("partial") else ""),
|
|
121
163
|
f"- Risk bands: `{summary.get('risk_bands')}`",
|
|
122
164
|
f"- Endpoint access: `{summary.get('endpoint_access')}`",
|
|
123
165
|
"",
|
sourcecode/cli.py
CHANGED
|
@@ -24,6 +24,8 @@ from sourcecode.output_encoding import json_ensure_ascii as _json_ensure_ascii
|
|
|
24
24
|
from sourcecode.output_encoding import set_ascii_fallback
|
|
25
25
|
from sourcecode.phased_run import WHY_BUDGET as _WHY_BUDGET
|
|
26
26
|
from sourcecode.phased_run import PhasedRun
|
|
27
|
+
from sourcecode.partial_contract import PARTIAL_EXIT_CODE as _PARTIAL_EXIT
|
|
28
|
+
from sourcecode.partial_contract import read_count as _read_count
|
|
27
29
|
from sourcecode import perf
|
|
28
30
|
from sourcecode.caller_metrics import (
|
|
29
31
|
CALLER_METRIC_RECONCILIATION,
|
|
@@ -238,6 +240,16 @@ def _writes_help_block() -> str:
|
|
|
238
240
|
lines.append(f" {name.ljust(width)} {what}")
|
|
239
241
|
lines.append("")
|
|
240
242
|
lines.append(" --no-write (ASK_READONLY=1) refuses all of it and says so on stderr.")
|
|
243
|
+
# F-AX: the other half of an audit that may not write — where the repository's
|
|
244
|
+
# own declaration is allowed to live.
|
|
245
|
+
lines.append(
|
|
246
|
+
" --config <path> (ASK_CONFIG) reads the repository's declaration from "
|
|
247
|
+
"outside it,"
|
|
248
|
+
)
|
|
249
|
+
lines.append(
|
|
250
|
+
" so labels and custom security annotations can be "
|
|
251
|
+
"declared without writing."
|
|
252
|
+
)
|
|
241
253
|
return "\n".join(lines)
|
|
242
254
|
|
|
243
255
|
|
|
@@ -746,7 +758,7 @@ def _preprocess_args(args: list[str]) -> list[str]:
|
|
|
746
758
|
|
|
747
759
|
def _preprocess_argv() -> None:
|
|
748
760
|
"""Apply _preprocess_args to sys.argv in-place (used by main_entry)."""
|
|
749
|
-
modified = _preprocess_args(_apply_no_write(sys.argv[1:]))
|
|
761
|
+
modified = _preprocess_args(_apply_external_config(_apply_no_write(sys.argv[1:])))
|
|
750
762
|
sys.argv = sys.argv[:1] + modified
|
|
751
763
|
|
|
752
764
|
|
|
@@ -2045,6 +2057,57 @@ NO_WRITE_OPTION_HELP = (
|
|
|
2045
2057
|
NO_WRITE_FLAG = "--no-write"
|
|
2046
2058
|
|
|
2047
2059
|
|
|
2060
|
+
#: The one help string for `--config` (F-AX).
|
|
2061
|
+
CONFIG_OPTION_HELP = (
|
|
2062
|
+
"Read this repository's declaration (data labels, custom security "
|
|
2063
|
+
"annotations) from <path> instead of <repo>/sourcecode.config.json — so an "
|
|
2064
|
+
"auditor can declare without writing into somebody else's tree. Env: "
|
|
2065
|
+
"ASK_CONFIG."
|
|
2066
|
+
)
|
|
2067
|
+
|
|
2068
|
+
#: Handled before the parser, for `--no-write`'s reason: a declaration honoured
|
|
2069
|
+
#: by some commands and not others is worse than none, because the reader cannot
|
|
2070
|
+
#: tell which answer used it.
|
|
2071
|
+
CONFIG_FLAG = "--config"
|
|
2072
|
+
|
|
2073
|
+
|
|
2074
|
+
def _apply_external_config(argv: "list[str]") -> "list[str]":
|
|
2075
|
+
"""Consume `--config <path>`, putting the declaration in force. Never raises.
|
|
2076
|
+
|
|
2077
|
+
F-AX. Under `ASK_READONLY=1` the only way to declare anything used to be a
|
|
2078
|
+
file inside the analysed tree, which left `data-exposure`, the `Verified`
|
|
2079
|
+
upgrade for a custom gate, and everything downstream of them unreachable by
|
|
2080
|
+
construction for the exact user this mode exists for — the auditor of
|
|
2081
|
+
somebody else's repository.
|
|
2082
|
+
"""
|
|
2083
|
+
if CONFIG_FLAG not in argv:
|
|
2084
|
+
return argv
|
|
2085
|
+
out: "list[str]" = []
|
|
2086
|
+
index = 0
|
|
2087
|
+
while index < len(argv):
|
|
2088
|
+
token = argv[index]
|
|
2089
|
+
if token != CONFIG_FLAG:
|
|
2090
|
+
out.append(token)
|
|
2091
|
+
index += 1
|
|
2092
|
+
continue
|
|
2093
|
+
value = argv[index + 1] if index + 1 < len(argv) else ""
|
|
2094
|
+
if not value or value.startswith("-"):
|
|
2095
|
+
print(
|
|
2096
|
+
f"error: {CONFIG_FLAG} needs a path to a declaration file.\n"
|
|
2097
|
+
f" use: ask {CONFIG_FLAG} ./declarations.json <command> …",
|
|
2098
|
+
file=sys.stderr,
|
|
2099
|
+
)
|
|
2100
|
+
raise SystemExit(2)
|
|
2101
|
+
try:
|
|
2102
|
+
from sourcecode import declarations
|
|
2103
|
+
|
|
2104
|
+
declarations.set_config_path(value)
|
|
2105
|
+
except Exception:
|
|
2106
|
+
pass
|
|
2107
|
+
index += 2
|
|
2108
|
+
return out
|
|
2109
|
+
|
|
2110
|
+
|
|
2048
2111
|
def _apply_no_write(argv: "list[str]") -> "list[str]":
|
|
2049
2112
|
"""Consume `--no-write` from *argv*, putting the mode in force. Never raises."""
|
|
2050
2113
|
if NO_WRITE_FLAG not in argv:
|
|
@@ -8222,14 +8285,20 @@ def spring_audit_cmd(
|
|
|
8222
8285
|
)
|
|
8223
8286
|
_stopped_early: Optional[str] = None
|
|
8224
8287
|
|
|
8288
|
+
from sourcecode import context_cache as _ctxcache
|
|
8289
|
+
|
|
8225
8290
|
phase = f"auditing {len(file_list)} Java files"
|
|
8226
8291
|
with _expensive_analysis_scope("spring-audit", target, phase):
|
|
8227
8292
|
_prog = Progress()
|
|
8228
8293
|
_prog.start(phase)
|
|
8229
8294
|
try:
|
|
8230
|
-
|
|
8231
|
-
|
|
8232
|
-
|
|
8295
|
+
# The parse is ~89 % of this command's wall clock (measured: 8,54 s of
|
|
8296
|
+
# 9,53 s on a 1 303-file repository) and it is the same parse the other
|
|
8297
|
+
# knowledge commands share, so it is fetched rather than rebuilt — which
|
|
8298
|
+
# is what makes a warm worth anything here (cold == warm before C3-94).
|
|
8299
|
+
cir = _ctxcache.shared_cir(
|
|
8300
|
+
target, file_list, progress=_work_sink(_prog)
|
|
8301
|
+
)
|
|
8233
8302
|
# C3-85: the stretch between the last `linking` tick and the first rule
|
|
8234
8303
|
# tick reports nothing, and what stays on the line is `n/n` — a counter
|
|
8235
8304
|
# that finished, reading as a stage still running. Measured on keycloak
|
|
@@ -8395,8 +8464,23 @@ def spring_audit_cmd(
|
|
|
8395
8464
|
else:
|
|
8396
8465
|
output = _serialize_dict(data, format)
|
|
8397
8466
|
|
|
8398
|
-
|
|
8467
|
+
# F-AV: our own readers ask for the count under either spelling — the break
|
|
8468
|
+
# is meant for a consumer that never asked whether the run finished.
|
|
8469
|
+
_total = _read_count(combined.summary, "total_findings") or 0
|
|
8399
8470
|
_partial_msg = " — PARTIAL, budget exhausted" if _stopped_early else ""
|
|
8471
|
+
# CL-19: the payload carries `stack_fit`, and the person who most needs it is
|
|
8472
|
+
# the one who will read the counts and stop. Said once, on stderr, so piped
|
|
8473
|
+
# output stays byte-identical.
|
|
8474
|
+
if not combined.spring_detected:
|
|
8475
|
+
_notice(
|
|
8476
|
+
"[ask] stack fit: no Spring detected in this scope. The access, "
|
|
8477
|
+
"transaction and "
|
|
8478
|
+
"Boot-readiness axes model Spring semantics and are NOT measuring "
|
|
8479
|
+
"this repository — read their answers as `unknown`, never as `none`. "
|
|
8480
|
+
"The structural axes (call graph, endpoint census, coupling, JDK "
|
|
8481
|
+
"inventory) answer here as they do anywhere. See `stack_fit` in the "
|
|
8482
|
+
"payload."
|
|
8483
|
+
)
|
|
8400
8484
|
_emit_command_output(
|
|
8401
8485
|
output, output_path, copy,
|
|
8402
8486
|
success_msg=(
|
|
@@ -8411,8 +8495,10 @@ def spring_audit_cmd(
|
|
|
8411
8495
|
|
|
8412
8496
|
if ci and _stopped_early:
|
|
8413
8497
|
# A gate on incomplete evidence must not read as a pass — the same rule
|
|
8414
|
-
# `pr-impact` holds for UNKNOWN.
|
|
8415
|
-
|
|
8498
|
+
# `pr-impact` holds for UNKNOWN. F-AV: and it must not read as a *failure*
|
|
8499
|
+
# either, because "the gate found something" and "the gate did not finish"
|
|
8500
|
+
# call for different pipeline decisions. 75 is neither 0, 1 nor 2.
|
|
8501
|
+
raise typer.Exit(code=_PARTIAL_EXIT)
|
|
8416
8502
|
if ci and combined.findings:
|
|
8417
8503
|
raise typer.Exit(code=1)
|
|
8418
8504
|
|
|
@@ -8962,7 +9048,7 @@ def risk_cmd(
|
|
|
8962
9048
|
copy,
|
|
8963
9049
|
success_msg=(
|
|
8964
9050
|
f"risk written to {output_path} ({data['shown']} of "
|
|
8965
|
-
f"{data
|
|
9051
|
+
f"{_read_count(data, 'total_defects')} defects composed)"
|
|
8966
9052
|
# C3-87: the terminal says it too. A reader who never opens the payload
|
|
8967
9053
|
# must not take a truncated ranking for the repository's risk.
|
|
8968
9054
|
+ (" — PARTIAL, budget exhausted" if data.get("partial") else "")
|
|
@@ -9256,26 +9342,58 @@ def audit_report_cmd(
|
|
|
9256
9342
|
)
|
|
9257
9343
|
raise typer.Exit(code=1)
|
|
9258
9344
|
|
|
9259
|
-
|
|
9260
|
-
|
|
9261
|
-
|
|
9262
|
-
|
|
9263
|
-
|
|
9264
|
-
|
|
9265
|
-
|
|
9266
|
-
|
|
9267
|
-
|
|
9345
|
+
# C3-89: this command composes the whole of `risk`, and it did so with no
|
|
9346
|
+
# deadline of any kind — no `PhasedRun`, no `stop_when` on the observers, no
|
|
9347
|
+
# checkpoint. The field measured 2 265 s under a 300 s budget, CPU-bound in one
|
|
9348
|
+
# worker, killed by hand with nothing to show. `risk` was bound in 4.16.0
|
|
9349
|
+
# (C3-87) and the composition of the `_partial` keys moved into
|
|
9350
|
+
# `PhasedRun.status` precisely so the next consumer would inherit it; this is
|
|
9351
|
+
# that consumer. The phases are `risk`'s own (it is the analysis being run)
|
|
9352
|
+
# plus this command's packaging step.
|
|
9353
|
+
_phase = "packaging audit evidence"
|
|
9354
|
+
_budget = _analysis_budget("audit-report")
|
|
9355
|
+
_run = PhasedRun(
|
|
9356
|
+
command="audit-report",
|
|
9357
|
+
root=path,
|
|
9358
|
+
phases=["audit", "compose", "bundle"],
|
|
9359
|
+
output_path=output_path,
|
|
9360
|
+
budget_seconds=_budget.get("configured_max_seconds"),
|
|
9361
|
+
budget_source=_budget.get("configured_source"),
|
|
9362
|
+
writer=_safe_write_file,
|
|
9363
|
+
)
|
|
9364
|
+
with _expensive_analysis_scope("audit-report", path, _phase):
|
|
9365
|
+
_prog = Progress()
|
|
9366
|
+
_prog.start(_phase)
|
|
9367
|
+
try:
|
|
9368
|
+
data = build_audit_report(
|
|
9369
|
+
path, profiles=_profile_set(profile), sign_key=key_bytes,
|
|
9370
|
+
risk_payload=risk_payload,
|
|
9371
|
+
progress=lambda stage, unit="rule families": _rule_pass_progress(
|
|
9372
|
+
_prog, stage, unit, stop_when=_run.exhausted
|
|
9373
|
+
),
|
|
9374
|
+
checkpoint=_run.checkpoint,
|
|
9375
|
+
partial_status=lambda **cut: _run.status(_WHY_BUDGET, **cut),
|
|
9376
|
+
)
|
|
9377
|
+
finally:
|
|
9378
|
+
_prog.stop()
|
|
9268
9379
|
output = render_markdown(data) if fmt == "markdown" else _serialize_dict(data, fmt)
|
|
9380
|
+
_summary = data.get("summary") if isinstance(data.get("summary"), dict) else {}
|
|
9269
9381
|
_emit_command_output(
|
|
9270
9382
|
output,
|
|
9271
9383
|
output_path,
|
|
9272
9384
|
copy,
|
|
9273
|
-
success_msg=f"audit report written to {output_path}"
|
|
9385
|
+
success_msg=f"audit report written to {output_path}"
|
|
9386
|
+
# The terminal says it too, for the same reason `risk` does: a signed
|
|
9387
|
+
# bundle that was cut must not be read as the repository's audit.
|
|
9388
|
+
+ (" — PARTIAL, budget exhausted" if _summary.get("partial") else ""),
|
|
9274
9389
|
# A signed artifact cannot be mutated after signing. The report carries
|
|
9275
9390
|
# tool/version metadata itself, and `signature.payload_sha256` is over
|
|
9276
9391
|
# the exact JSON payload the user receives.
|
|
9277
9392
|
stamp_envelope=False,
|
|
9278
9393
|
)
|
|
9394
|
+
# The complete bundle is written; the checkpoint that stood in for it must not
|
|
9395
|
+
# outlive it (C3-77's invariant, as `risk` applies it).
|
|
9396
|
+
_run.discard_checkpoint()
|
|
9279
9397
|
|
|
9280
9398
|
|
|
9281
9399
|
@app.command("migrate-recipe")
|
|
@@ -9854,7 +9972,7 @@ def _migration_blast_radius(
|
|
|
9854
9972
|
from sourcecode.spring_model import SpringSemanticModel
|
|
9855
9973
|
|
|
9856
9974
|
try:
|
|
9857
|
-
cir
|
|
9975
|
+
cir = _ctxcache.shared_cir(target, file_list)
|
|
9858
9976
|
except Exception:
|
|
9859
9977
|
cir = ContextGraph.build(file_list, target).cir
|
|
9860
9978
|
model = SpringSemanticModel.build(cir)
|
|
@@ -10408,10 +10526,7 @@ def impact_chain_cmd(
|
|
|
10408
10526
|
# a fresh build, exactly as explain does, so impact-chain never breaks.
|
|
10409
10527
|
from sourcecode import context_cache as _ctxcache
|
|
10410
10528
|
try:
|
|
10411
|
-
cir
|
|
10412
|
-
_resolve_repo_root(target), target, file_list,
|
|
10413
|
-
progress=_work_sink(_prog),
|
|
10414
|
-
)
|
|
10529
|
+
cir = _ctxcache.shared_cir(target, file_list, progress=_work_sink(_prog))
|
|
10415
10530
|
except Exception:
|
|
10416
10531
|
cir = ContextGraph.build(file_list, target, progress=_work_sink(_prog)).cir
|
|
10417
10532
|
_model = SpringSemanticModel.build(cir)
|
|
@@ -10737,11 +10852,13 @@ def explain_cmd(
|
|
|
10737
10852
|
)
|
|
10738
10853
|
raise typer.Exit(code=1)
|
|
10739
10854
|
|
|
10740
|
-
# AI Context Cache — operates at the *knowledge* level: it caches the
|
|
10741
|
-
#
|
|
10742
|
-
#
|
|
10743
|
-
# shared CIR
|
|
10744
|
-
# Provider-agnostic and best-effort: any fault degrades to a fresh build.
|
|
10855
|
+
# AI Context Cache — operates at the *knowledge* level: it caches the reusable
|
|
10856
|
+
# Canonical IR (the expensive Java parse), keyed by knowledge state and by the
|
|
10857
|
+
# analysed set, not by command. explain then derives its answer cheaply from
|
|
10858
|
+
# that shared CIR, and any other command over the same files reuses it for
|
|
10859
|
+
# free. Provider-agnostic and best-effort: any fault degrades to a fresh build.
|
|
10860
|
+
# This call site keeps the lower-level entry point because it reports the
|
|
10861
|
+
# lookup (`_cc_look`); `shared_cir` is the same door without the receipt.
|
|
10745
10862
|
_prog = Progress()
|
|
10746
10863
|
_prog.start(f"explaining {class_name} ({len(file_list)} files)")
|
|
10747
10864
|
_cc_look = None
|
|
@@ -11782,7 +11899,15 @@ def schema_cmd(
|
|
|
11782
11899
|
# through the command that already answers "what shape does this release
|
|
11783
11900
|
# produce?" rather than a command of its own.
|
|
11784
11901
|
names = available_schemas()
|
|
11785
|
-
|
|
11902
|
+
# C4-22: one authority for *what this argument accepts*. The registry names
|
|
11903
|
+
# itself (`registry_version`), so the listing, the dispatch below and the
|
|
11904
|
+
# rejection all read the same value — the field's audit was told
|
|
11905
|
+
# `available: ["envelope-v1"]` by a rejection that knew only half of what the
|
|
11906
|
+
# very next line accepts, and concluded that was the whole published surface.
|
|
11907
|
+
registries = [str(load_registry().get("registry_version") or "")]
|
|
11908
|
+
registries = [r for r in registries if r]
|
|
11909
|
+
accepted = list(names) + registries
|
|
11910
|
+
if name in registries:
|
|
11786
11911
|
_emit_command_output(
|
|
11787
11912
|
json.dumps(load_registry(), indent=2, ensure_ascii=False),
|
|
11788
11913
|
output_path,
|
|
@@ -11797,7 +11922,7 @@ def schema_cmd(
|
|
|
11797
11922
|
# answers have a single authority. Both are contracts this release
|
|
11798
11923
|
# carries, listed apart because they are not the same kind of thing.
|
|
11799
11924
|
json.dumps(
|
|
11800
|
-
{"schemas": names, "registries":
|
|
11925
|
+
{"schemas": names, "registries": registries},
|
|
11801
11926
|
indent=2, ensure_ascii=False,
|
|
11802
11927
|
),
|
|
11803
11928
|
output_path,
|
|
@@ -11810,10 +11935,14 @@ def schema_cmd(
|
|
|
11810
11935
|
except FileNotFoundError:
|
|
11811
11936
|
_emit_error_json(
|
|
11812
11937
|
INVALID_INPUT_CODE,
|
|
11813
|
-
f"
|
|
11814
|
-
available=
|
|
11815
|
-
hint=
|
|
11816
|
-
|
|
11938
|
+
f"'{name}' is not a published schema or registry.",
|
|
11939
|
+
available=accepted,
|
|
11940
|
+
hint=(
|
|
11941
|
+
"Run `ask schema` to list them: "
|
|
11942
|
+
+ ", ".join(accepted)
|
|
11943
|
+
+ "."
|
|
11944
|
+
),
|
|
11945
|
+
expected="One of: " + " | ".join(accepted),
|
|
11817
11946
|
)
|
|
11818
11947
|
raise typer.Exit(code=1)
|
|
11819
11948
|
# A schema document describes the envelope; stamping one into it would put
|
|
@@ -11871,8 +12000,17 @@ def config_cmd(
|
|
|
11871
12000
|
_answer.say(f"Telemetry: {'enabled' if is_enabled() else 'disabled'} (off by default; opt-in)")
|
|
11872
12001
|
_answer.say("")
|
|
11873
12002
|
|
|
11874
|
-
|
|
12003
|
+
# F-AX: the declaration may sit outside the analysed tree, and *which* file
|
|
12004
|
+
# answered is the first thing this command exists to say.
|
|
12005
|
+
from sourcecode.declarations import config_path as _config_path, is_external
|
|
12006
|
+
|
|
12007
|
+
_declaration_path = _config_path(_repo) or (_repo / CONFIG_FILENAME)
|
|
11875
12008
|
_answer.say(f"Repository: {_repo}")
|
|
12009
|
+
if is_external():
|
|
12010
|
+
_answer.say(
|
|
12011
|
+
"Declared with --config / ASK_CONFIG — read from outside the "
|
|
12012
|
+
"repository, which is unchanged by this run."
|
|
12013
|
+
)
|
|
11876
12014
|
if not _declaration_path.is_file():
|
|
11877
12015
|
_answer.say(f"Declaration: none at {_declaration_path}")
|
|
11878
12016
|
_answer.say(
|
sourcecode/context_cache.py
CHANGED
|
@@ -651,6 +651,33 @@ class KnowledgeLookup:
|
|
|
651
651
|
return f"Context cache: MISS (build {self.build_ms:.0f}ms)"
|
|
652
652
|
|
|
653
653
|
|
|
654
|
+
def analysed_set_signature(root: Path, file_list: list[str]) -> str:
|
|
655
|
+
"""Fingerprint of *what was analysed*: the scope root and the exact file set.
|
|
656
|
+
|
|
657
|
+
The shared CIR is keyed on knowledge state, and until C3-94 "knowledge state"
|
|
658
|
+
was read as the worktree alone — which is only half of it. A run bounded to a
|
|
659
|
+
subdirectory parses a *subset* of the same worktree, and the resulting CIR is
|
|
660
|
+
a different artifact answering a different question. Keyed on the worktree
|
|
661
|
+
alone, both artifacts collided on one entry: measured on a 861-file repository,
|
|
662
|
+
``explain WebUtil <repo>/web`` stored a 42-file / 382-symbol CIR under the
|
|
663
|
+
repo-wide key, and the next repo-wide command read it and answered about 3 % of
|
|
664
|
+
the repository — silently, because the reader passes its own ``file_paths`` to
|
|
665
|
+
``ir_dict_to_canonical`` and the reconstructed CIR then reports 861 files
|
|
666
|
+
beside 382 symbols.
|
|
667
|
+
|
|
668
|
+
So the analysed set is stated, not hoped for. The scope path and the file list
|
|
669
|
+
are both in it: two roots can yield the same file count, and a caller that
|
|
670
|
+
filters its list (tests in or out) is asking about a different population than
|
|
671
|
+
one that does not.
|
|
672
|
+
"""
|
|
673
|
+
payload = "".join([
|
|
674
|
+
str(root.resolve()),
|
|
675
|
+
str(len(file_list)),
|
|
676
|
+
hashlib.sha256("\n".join(sorted(file_list)).encode("utf-8")).hexdigest(),
|
|
677
|
+
])
|
|
678
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16]
|
|
679
|
+
|
|
680
|
+
|
|
654
681
|
def get_or_build_cir(
|
|
655
682
|
repo_root: Path,
|
|
656
683
|
root: Path,
|
|
@@ -662,11 +689,19 @@ def get_or_build_cir(
|
|
|
662
689
|
"""Return ``(CanonicalRepositoryIR, KnowledgeLookup)``, caching the CIR.
|
|
663
690
|
|
|
664
691
|
This is the knowledge-level entry point. It caches the **reusable** raw IR
|
|
665
|
-
(the expensive Java parse) keyed
|
|
666
|
-
so every command that
|
|
667
|
-
CIR is rebuilt cheaply from the cached raw IR via
|
|
668
|
-
skipping the parse entirely. Best-effort: any cache
|
|
669
|
-
fresh build.
|
|
692
|
+
(the expensive Java parse) keyed by knowledge state and by the analysed set —
|
|
693
|
+
not by command — so every command that parses the same files shares one cache
|
|
694
|
+
entry. On a hit the full CIR is rebuilt cheaply from the cached raw IR via
|
|
695
|
+
``ir_dict_to_canonical``, skipping the parse entirely. Best-effort: any cache
|
|
696
|
+
fault falls back to a fresh build.
|
|
697
|
+
|
|
698
|
+
A **repo-wide** run (``root`` is ``repo_root``) owns the scope-free entry —
|
|
699
|
+
the one :func:`peek_cir` reads without knowing any file list, which is why
|
|
700
|
+
that key must stay free of one. Every **bounded** run keys on its own analysed
|
|
701
|
+
set instead, so it can neither read nor overwrite the repo-wide answer. Either
|
|
702
|
+
way the entry carries the analysed signature it was built from and a reader
|
|
703
|
+
whose set does not match rebuilds: the key prevents the collision, the check
|
|
704
|
+
makes the prevention verifiable rather than assumed (C3-94).
|
|
670
705
|
|
|
671
706
|
`progress` is the (done, total, stage) sink the caller's heartbeat reads
|
|
672
707
|
(C3-56). It reaches only the *building* paths, which is the honest place for
|
|
@@ -684,14 +719,23 @@ def get_or_build_cir(
|
|
|
684
719
|
cir = build_canonical_ir(file_list, root, since=since, progress=progress)
|
|
685
720
|
return cir, KnowledgeLookup(hit=False, enabled=False)
|
|
686
721
|
|
|
687
|
-
|
|
722
|
+
analysed = analysed_set_signature(root, file_list)
|
|
723
|
+
options: dict[str, Any] = {"since": since} if since else None
|
|
724
|
+
if root.resolve() != repo_root.resolve():
|
|
725
|
+
# A bounded run: its own entry, never the repo-wide one.
|
|
726
|
+
options = dict(options or {})
|
|
727
|
+
options["analysed"] = analysed
|
|
728
|
+
key = cache.knowledge_key(SCOPE_JAVA_CIR, options=options)
|
|
688
729
|
|
|
689
730
|
t0 = time.perf_counter()
|
|
690
731
|
cached = cache.get(key)
|
|
691
732
|
lookup_ms = (time.perf_counter() - t0) * 1000
|
|
692
733
|
if cached is not None:
|
|
693
734
|
raw = cached.payload.get("raw_ir")
|
|
694
|
-
|
|
735
|
+
stored = (cached.metadata or {}).get("analysed_set")
|
|
736
|
+
# An entry written before C3-94 carries no signature; it cannot be shown to
|
|
737
|
+
# describe this analysed set, so it is not served as if it did.
|
|
738
|
+
if isinstance(raw, dict) and stored == analysed:
|
|
695
739
|
try:
|
|
696
740
|
cir = ir_dict_to_canonical(raw, file_paths=file_list)
|
|
697
741
|
if not validate_canonical_ir(cir):
|
|
@@ -708,6 +752,8 @@ def get_or_build_cir(
|
|
|
708
752
|
"scope": SCOPE_JAVA_CIR,
|
|
709
753
|
"cir_hash": cir.cir_hash,
|
|
710
754
|
"schema_version": cir.schema_version,
|
|
755
|
+
"analysed_set": analysed,
|
|
756
|
+
"analysed_root": str(root.resolve()),
|
|
711
757
|
"file_count": len(cir.files),
|
|
712
758
|
"symbol_count": len(cir.symbols),
|
|
713
759
|
},
|
|
@@ -725,6 +771,41 @@ def get_or_build_cir(
|
|
|
725
771
|
return cir, KnowledgeLookup(hit=False, enabled=True, build_ms=build_ms)
|
|
726
772
|
|
|
727
773
|
|
|
774
|
+
def repo_root_of(root: Path) -> Path:
|
|
775
|
+
"""The enclosing git root, or *root* itself outside a repository."""
|
|
776
|
+
candidate = root.resolve()
|
|
777
|
+
while True:
|
|
778
|
+
if (candidate / ".git").exists():
|
|
779
|
+
return candidate
|
|
780
|
+
if candidate.parent == candidate:
|
|
781
|
+
return root.resolve()
|
|
782
|
+
candidate = candidate.parent
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def shared_cir(
|
|
786
|
+
root: Path,
|
|
787
|
+
file_list: list[str],
|
|
788
|
+
*,
|
|
789
|
+
since: Optional[str] = None,
|
|
790
|
+
progress: "Optional[Callable[[int, int, str], None]]" = None,
|
|
791
|
+
) -> Any:
|
|
792
|
+
"""The Java CIR for *root*, from the shared knowledge cache.
|
|
793
|
+
|
|
794
|
+
The one door for a command that needs the repository parse and does not care
|
|
795
|
+
where it comes from: it resolves the enclosing repository itself, so a caller
|
|
796
|
+
holds no opinion about cache scope — the reason each call-site that built its
|
|
797
|
+
own CIR was one more place for the scope rule to be got wrong (C3-94).
|
|
798
|
+
|
|
799
|
+
Best-effort by construction: `get_or_build_cir` degrades to a fresh build on
|
|
800
|
+
any cache fault, so the answer never depends on the cache being there.
|
|
801
|
+
"""
|
|
802
|
+
cir, _lookup = get_or_build_cir(
|
|
803
|
+
repo_root_of(root), Path(root).resolve(), file_list,
|
|
804
|
+
since=since, progress=progress,
|
|
805
|
+
)
|
|
806
|
+
return cir
|
|
807
|
+
|
|
808
|
+
|
|
728
809
|
def peek_cir(
|
|
729
810
|
repo_root: Path,
|
|
730
811
|
*,
|
|
@@ -755,6 +836,12 @@ def peek_cir(
|
|
|
755
836
|
raw = cached.payload.get("raw_ir")
|
|
756
837
|
if not isinstance(raw, dict):
|
|
757
838
|
return None
|
|
839
|
+
# The caller is promised the *repo-wide* CIR, so the entry has to say it is one.
|
|
840
|
+
# C3-94 makes the key unreachable to bounded runs; this re-reads the fact from
|
|
841
|
+
# the entry rather than inferring it from the key, and treats an entry written
|
|
842
|
+
# before that guarantee (no recorded root) as a miss.
|
|
843
|
+
if (cached.metadata or {}).get("analysed_root") != str(repo_root.resolve()):
|
|
844
|
+
return None
|
|
758
845
|
try:
|
|
759
846
|
cir = ir_dict_to_canonical(raw, file_paths=None)
|
|
760
847
|
# validate_canonical_ir returns a LIST of problems — empty means valid.
|
sourcecode/data_exposure.py
CHANGED
|
@@ -40,6 +40,7 @@ from pathlib import Path
|
|
|
40
40
|
from typing import TYPE_CHECKING, Any, Optional
|
|
41
41
|
|
|
42
42
|
from sourcecode.data_labels import DataLabel, LabelDeclaration, load_labels
|
|
43
|
+
from sourcecode.declarations import source_note as _declaration_source
|
|
43
44
|
|
|
44
45
|
if TYPE_CHECKING: # pragma: no cover - typing only
|
|
45
46
|
from sourcecode.canonical_ir import CanonicalRepositoryIR
|
|
@@ -139,7 +140,10 @@ def build_data_exposure(
|
|
|
139
140
|
|
|
140
141
|
payload: dict = {
|
|
141
142
|
"schema_version": SCHEMA_VERSION,
|
|
142
|
-
|
|
143
|
+
# F-AX: which file answered. A declaration read from outside the analysed
|
|
144
|
+
# tree is the auditor's case, and a payload that does not say so leaves
|
|
145
|
+
# the reader unable to reproduce the answer.
|
|
146
|
+
"declaration": {**decl.to_dict(), "source": _declaration_source()},
|
|
143
147
|
"profiles": sorted(profiles) if profiles else None,
|
|
144
148
|
}
|
|
145
149
|
|
sourcecode/data_labels.py
CHANGED
|
@@ -39,7 +39,9 @@ from pathlib import Path
|
|
|
39
39
|
from typing import Optional
|
|
40
40
|
|
|
41
41
|
#: The same file the custom security annotations are declared in.
|
|
42
|
-
|
|
42
|
+
#: F-AX: the name lives in `declarations`, which also decides *where* it is
|
|
43
|
+
#: looked for. Re-exported so existing importers keep working.
|
|
44
|
+
from sourcecode.declarations import CONFIG_FILENAME # noqa: E402,F401
|
|
43
45
|
|
|
44
46
|
#: The key under which labels are declared.
|
|
45
47
|
CONFIG_KEY = "dataLabels"
|
|
@@ -108,9 +110,14 @@ def load_labels(root: Optional[Path]) -> LabelDeclaration:
|
|
|
108
110
|
Never raises. An absent file is not a problem — it is a repository that has
|
|
109
111
|
declared nothing, and the caller says so in the words of the question.
|
|
110
112
|
"""
|
|
111
|
-
|
|
113
|
+
# F-AX: the path is resolved by one authority, so `--config <external>` is
|
|
114
|
+
# honoured identically here and in `security_config` — a declaration that
|
|
115
|
+
# applied to one loader and not the other would be worse than none.
|
|
116
|
+
from sourcecode.declarations import config_path
|
|
117
|
+
|
|
118
|
+
cfg_path = config_path(root)
|
|
119
|
+
if cfg_path is None:
|
|
112
120
|
return LabelDeclaration()
|
|
113
|
-
cfg_path = Path(root) / CONFIG_FILENAME
|
|
114
121
|
try:
|
|
115
122
|
if not cfg_path.is_file():
|
|
116
123
|
return LabelDeclaration(config_file=None)
|