sourcecode 5.0.0__py3-none-any.whl → 5.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of sourcecode might be problematic. Click here for more details.
- sourcecode/__init__.py +1 -1
- sourcecode/cache_model.py +156 -32
- sourcecode/execution_plan.py +44 -6
- sourcecode/posture.py +13 -26
- sourcecode/repository_ir.py +21 -31
- sourcecode/security_chain.py +148 -0
- sourcecode/security_config_scan.py +54 -4
- sourcecode/source_text.py +18 -0
- {sourcecode-5.0.0.dist-info → sourcecode-5.0.1.dist-info}/METADATA +3 -3
- {sourcecode-5.0.0.dist-info → sourcecode-5.0.1.dist-info}/RECORD +13 -12
- {sourcecode-5.0.0.dist-info → sourcecode-5.0.1.dist-info}/WHEEL +0 -0
- {sourcecode-5.0.0.dist-info → sourcecode-5.0.1.dist-info}/entry_points.txt +0 -0
- {sourcecode-5.0.0.dist-info → sourcecode-5.0.1.dist-info}/licenses/LICENSE +0 -0
sourcecode/__init__.py
CHANGED
sourcecode/cache_model.py
CHANGED
|
@@ -87,7 +87,28 @@ class CommandCache:
|
|
|
87
87
|
#: releases (spring-audit, 408 s in #13 against 2 351 s in #16) is itself the
|
|
88
88
|
#: measurement C3-76 exists for.
|
|
89
89
|
field_seconds: "Optional[float]" = None
|
|
90
|
+
#: **C3-100.** `field_blocked` is a *measurement outcome*, not a property of the
|
|
91
|
+
#: command: it says "this run did not finish on that build, at that size", and
|
|
92
|
+
#: like every other measurement here it belongs to the build named in
|
|
93
|
+
#: `field_measured_version` and expires with it. It was read as a property for
|
|
94
|
+
#: three releases — `repo-ir` and `verify-edit` carried a 4.10.4 outcome while
|
|
95
|
+
#: eval #23 ran them in 9,8 s and 76,6 s — which is worse than a stale number,
|
|
96
|
+
#: because a stale number is a measurement and *"the field could not run this"*
|
|
97
|
+
#: is an incapacity the product does not have. A blocked row without a build is
|
|
98
|
+
#: rejected by the battery for the same reason a figure without one is.
|
|
90
99
|
field_blocked: bool = False
|
|
100
|
+
#: The cache state the field figure was taken in. Eval #22 measured a warm
|
|
101
|
+
#: machine; eval #23 purged every cache first (`cache clear --all`,
|
|
102
|
+
#: `context-clear`, `rm -rf parse-cache-v1`) and then ran the whole catalogue
|
|
103
|
+
#: in one round, so its first figures are cold and its later ones are neither —
|
|
104
|
+
#: the shared layers filled as the round went on. Published because *72,3 s
|
|
105
|
+
#: cold* and *72,3 s warm* are two different facts about the same command, and
|
|
106
|
+
#: the row that says which is the only thing standing between the two:
|
|
107
|
+
#: ``warm`` — the caches were warmed before the run
|
|
108
|
+
#: ``cold`` — every cache was purged before the run
|
|
109
|
+
#: ``mixed`` — inside a round that began purged: earlier commands had already
|
|
110
|
+
#: filled some of the shared layers this one reads
|
|
111
|
+
field_cache_state: str = "warm"
|
|
91
112
|
#: **C3-97.** The build the field figure was taken on. An anchor without one is
|
|
92
113
|
#: a measurement that can outlive its build and keep deciding: 4.18.0 deleted
|
|
93
114
|
#: two catastrophic-backtracking patterns (C3-95, C3-96) and every figure taken
|
|
@@ -137,21 +158,20 @@ REFERENCE_MEASURED_VERSION = "3.2.2"
|
|
|
137
158
|
#: What can be said honestly is the class, the two measured anchors, **the build
|
|
138
159
|
#: each was measured on**, and how to run it.
|
|
139
160
|
#:
|
|
140
|
-
#: History of
|
|
161
|
+
#: History of the anchor, kept because it is the measurement C3-76 exists for:
|
|
141
162
|
#: 408 s (#13) → 2 351 s (#16, C3-72) → 8,4 s (#22, C3-97). The middle figure was
|
|
142
163
|
#: quoted as an operational anchor for four releases after the build that produced
|
|
143
164
|
#: it, and told operators to detach commands that finish in seconds.
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
165
|
+
#:
|
|
166
|
+
#: ⚠ `FIELD_ANCHOR`, `FIELD_ANCHOR_SECONDS` and `FIELD_ANCHOR_MEASURED_VERSION`
|
|
167
|
+
#: are **derived from the `spring-audit` row**, below `COMMANDS`. They used to be
|
|
168
|
+
#: written out here as well, which made the anchor two facts kept in step by hand:
|
|
169
|
+
#: C3-97 refreshed both, and C3-100 is what it costs when the next refresh reaches
|
|
170
|
+
#: only one of the copies. One authority per fact — the table — and the headline
|
|
171
|
+
#: reads itself off it.
|
|
172
|
+
#:
|
|
173
|
+
#: The size of the field repository every `field_seconds` below was taken on.
|
|
148
174
|
FIELD_ANCHOR_JAVA_FILES = 3342
|
|
149
|
-
FIELD_ANCHOR_SECONDS = 8.4
|
|
150
|
-
|
|
151
|
-
#: The build the field anchors above were taken on. Rows carrying a different
|
|
152
|
-
#: `field_measured_version` are published with that build's name attached rather
|
|
153
|
-
#: than as figures for this one (C3-97).
|
|
154
|
-
FIELD_ANCHOR_MEASURED_VERSION = "4.18.0"
|
|
155
175
|
|
|
156
176
|
|
|
157
177
|
#: Every layer keys on `cache.worktree_signature` — the exact tree state (C1-9).
|
|
@@ -217,9 +237,12 @@ COMMANDS: tuple[CommandCache, ...] = (
|
|
|
217
237
|
CommandCache("ask (root)", ("snapshot", "ris", "parse"), "answer", True,
|
|
218
238
|
"`--compact` is what a warm stores by default; `--agent` needs `cache warm --agent`. "
|
|
219
239
|
"`--env-map`, `--depth N` and `--exclude` change the *analysis*, so they miss the "
|
|
220
|
-
"warmed core and rescan —
|
|
240
|
+
"warmed core and rescan — the 171 s the field measured after a 103 s warm on 4.10.4. "
|
|
241
|
+
"Eval #23 timed the same invocation at 72,3 s from a purged cache, and then paid "
|
|
242
|
+
"71,7 s again for `--agent` on the state it had just analysed (C3-103).",
|
|
221
243
|
"--compact 13.3 s cold → 0.3 s warm (cold re-measured on 3.7.0: was 19.3 s, C3-6); "
|
|
222
|
-
"--agent --full --env-map --depth 20 34.7 s → 33.9 s (no gain)", analysis_class="repo-wide", cold_seconds=13.3, warm_seconds=0.3,
|
|
244
|
+
"--agent --full --env-map --depth 20 34.7 s → 33.9 s (no gain)", analysis_class="repo-wide", cold_seconds=13.3, warm_seconds=0.3,
|
|
245
|
+
field_seconds=72.3, field_cache_state="cold", field_measured_version="5.0.0"),
|
|
223
246
|
CommandCache("posture", ("cir", "parse"), "shared", False,
|
|
224
247
|
"Resolves the conditional bean graph on every run, over the shared CIR a warm "
|
|
225
248
|
"builds — the parse it used to repeat for itself. `--diff` compares two profile "
|
|
@@ -284,9 +307,11 @@ COMMANDS: tuple[CommandCache, ...] = (
|
|
|
284
307
|
CommandCache("verify", (), "none", False, "Runs the contracts against a fresh reading.", analysis_class="core"),
|
|
285
308
|
CommandCache("verify-edit", ("parse",), "shared", True,
|
|
286
309
|
"Built for the edit loop: the parse cache is what keeps an unchanged file out of the "
|
|
287
|
-
"next run. Its own second run is faster again."
|
|
310
|
+
"next run. Its own second run is faster again. At field size that loop is not short "
|
|
311
|
+
"yet: eval #23 measured 76,6 s on a tree with no edits, because the HEAD side is "
|
|
312
|
+
"still built in a throwaway worktree instead of reusing the shared CIR (C3-102).",
|
|
288
313
|
"14.6 s → 9.6 s → 5.6 s on repeat", analysis_class="repo-wide", cold_seconds=14.6, warm_seconds=9.6,
|
|
289
|
-
|
|
314
|
+
field_seconds=76.6, field_cache_state="mixed", field_measured_version="5.0.0"),
|
|
290
315
|
CommandCache("review-pr", ("cir",), "none", False,
|
|
291
316
|
"Diff-dependent, and it reuses the CIR only if one exists. On a small diff, loading "
|
|
292
317
|
"the warmed CIR costs more than the work it saves.",
|
|
@@ -318,8 +343,12 @@ COMMANDS: tuple[CommandCache, ...] = (
|
|
|
318
343
|
"9.8 s → 1.6 s", analysis_class="core", cold_seconds=9.8, warm_seconds=1.6,
|
|
319
344
|
field_seconds=4.4, field_measured_version="4.18.0"),
|
|
320
345
|
CommandCache("export", ("parse",), "shared", False, "", "8.8 s → 3.7 s", analysis_class="repo-wide", cold_seconds=8.8, warm_seconds=3.7),
|
|
321
|
-
CommandCache("repo-ir", ("parse",), "shared", False,
|
|
322
|
-
|
|
346
|
+
CommandCache("repo-ir", ("parse",), "shared", False,
|
|
347
|
+
"Carried as *did not finish* from 4.10.4 until eval #23 ran it at field size in "
|
|
348
|
+
"9,8 s — a run that did not finish once is not a command that cannot finish "
|
|
349
|
+
"(C3-100, and the same correction `modernize` needed).",
|
|
350
|
+
"5.1 s → 2.9 s", analysis_class="repo-wide", cold_seconds=5.1, warm_seconds=2.9,
|
|
351
|
+
field_seconds=9.8, field_cache_state="mixed", field_measured_version="5.0.0"),
|
|
323
352
|
CommandCache("validation", ("parse",), "shared", False, "", "11.6 s → 6.4 s", analysis_class="repo-wide", cold_seconds=11.6, warm_seconds=6.4,
|
|
324
353
|
field_seconds=17.4, field_measured_version="4.18.0"),
|
|
325
354
|
CommandCache("modernize", ("parse",), "shared", False,
|
|
@@ -333,8 +362,11 @@ COMMANDS: tuple[CommandCache, ...] = (
|
|
|
333
362
|
"`no_ris` instead of a snapshot.",
|
|
334
363
|
"0.2 s either way", analysis_class="core", cold_seconds=0.2, warm_seconds=0.2),
|
|
335
364
|
CommandCache("trend", (), "none", False, "Reads stored baseline artifacts from disk; analyses no source, so no cache layer applies. Same command as `baseline trend`.", analysis_class="none"),
|
|
336
|
-
CommandCache("baseline", ("parse",), "shared", False,
|
|
337
|
-
"capture
|
|
365
|
+
CommandCache("baseline", ("parse",), "shared", False,
|
|
366
|
+
"`capture`/`diff`/`trend` over architectural metrics. The field figure is `capture`, "
|
|
367
|
+
"the subcommand that analyses; `trend` reads stored artifacts and has its own row.",
|
|
368
|
+
"capture 8.7 s → 3.6 s", analysis_class="repo-wide", cold_seconds=8.7, warm_seconds=3.6,
|
|
369
|
+
field_seconds=2.2, field_cache_state="mixed", field_measured_version="5.0.0"),
|
|
338
370
|
CommandCache("retrieve", ("cir", "parse"), "shared", False,
|
|
339
371
|
"Every query builds or reuses the shared CIR a warm builds.", analysis_class="core"),
|
|
340
372
|
CommandCache("archetype", ("parse",), "shared", False, "", "9.3 s → 5.6 s", analysis_class="repo-wide", cold_seconds=9.3, warm_seconds=5.6),
|
|
@@ -355,6 +387,73 @@ _WARM_LABEL = {
|
|
|
355
387
|
"none": "nothing",
|
|
356
388
|
}
|
|
357
389
|
|
|
390
|
+
#: The cache states a field figure may declare. A row outside this set is a
|
|
391
|
+
#: measurement whose conditions nobody can read (C3-100).
|
|
392
|
+
FIELD_CACHE_STATES = ("warm", "cold", "mixed")
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _thousands(n: int) -> str:
|
|
396
|
+
return f"{n:,}".replace(",", " ")
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def _row(command: str) -> "CommandCache":
|
|
400
|
+
for entry in COMMANDS:
|
|
401
|
+
if entry.command == command:
|
|
402
|
+
return entry
|
|
403
|
+
raise KeyError(command)
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
#: The headline anchor, read off the table rather than written twice (C3-100).
|
|
407
|
+
#: `spring-audit` is the command the whole series measured — 408 s (#13), 2 351 s
|
|
408
|
+
#: (#16), 8,4 s (#22) — so it is the row the front page quotes, and quoting it
|
|
409
|
+
#: means reading it, not copying it.
|
|
410
|
+
_FIELD_ANCHOR_ROW = _row("spring-audit")
|
|
411
|
+
FIELD_ANCHOR_SECONDS = _FIELD_ANCHOR_ROW.field_seconds
|
|
412
|
+
FIELD_ANCHOR_MEASURED_VERSION = _FIELD_ANCHOR_ROW.field_measured_version
|
|
413
|
+
FIELD_ANCHOR = (
|
|
414
|
+
f"field evaluation #22: spring-audit on {_thousands(FIELD_ANCHOR_JAVA_FILES)} "
|
|
415
|
+
f"Java files took {FIELD_ANCHOR_SECONDS:g} s (Windows, pipx, "
|
|
416
|
+
f"{_FIELD_ANCHOR_ROW.field_cache_state} cache) against "
|
|
417
|
+
f"{_FIELD_ANCHOR_ROW.cold_seconds:g} s on the "
|
|
418
|
+
f"{_thousands(REFERENCE_JAVA_FILES)}-file reference"
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _version_key(version: "Optional[str]") -> tuple:
|
|
423
|
+
"""A comparable key for a release name, for ordering measurements only.
|
|
424
|
+
|
|
425
|
+
Not a distance — `reference_currency` refuses to invent one and this does not
|
|
426
|
+
either. It answers exactly one question: is this row's build older than the
|
|
427
|
+
newest build anything in the table was measured on (C3-100)? Anything
|
|
428
|
+
unparseable sorts oldest, which is the safe direction: it gets labelled.
|
|
429
|
+
"""
|
|
430
|
+
if not version:
|
|
431
|
+
return ()
|
|
432
|
+
parts: list[int] = []
|
|
433
|
+
for chunk in version.split("."):
|
|
434
|
+
digits = "".join(c for c in chunk if c.isdigit())
|
|
435
|
+
if not digits:
|
|
436
|
+
break
|
|
437
|
+
parts.append(int(digits))
|
|
438
|
+
return tuple(parts)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def newest_field_measurement_version() -> "Optional[str]":
|
|
442
|
+
"""The most recent build any field figure in the table was taken on.
|
|
443
|
+
|
|
444
|
+
The reference point for *"this row is older than the table it sits in"* — the
|
|
445
|
+
state C3-100 lived in for three releases, where a refresh reached five rows
|
|
446
|
+
and left four behind on a build four releases older, with nothing anywhere
|
|
447
|
+
saying so.
|
|
448
|
+
"""
|
|
449
|
+
versions = [
|
|
450
|
+
row.field_measured_version
|
|
451
|
+
for row in COMMANDS
|
|
452
|
+
if (row.field_seconds is not None or row.field_blocked)
|
|
453
|
+
and row.field_measured_version
|
|
454
|
+
]
|
|
455
|
+
return max(versions, key=_version_key) if versions else None
|
|
456
|
+
|
|
358
457
|
|
|
359
458
|
def layer(layer_id: str) -> Layer:
|
|
360
459
|
for entry in LAYERS:
|
|
@@ -459,10 +558,6 @@ def cost_sentence() -> str:
|
|
|
459
558
|
)
|
|
460
559
|
|
|
461
560
|
|
|
462
|
-
def _thousands(n: int) -> str:
|
|
463
|
-
return f"{n:,}".replace(",", " ")
|
|
464
|
-
|
|
465
|
-
|
|
466
561
|
def field_currency() -> dict:
|
|
467
562
|
"""The same statement as `reference_currency`, for the field anchors (C3-97).
|
|
468
563
|
|
|
@@ -473,11 +568,32 @@ def field_currency() -> dict:
|
|
|
473
568
|
from sourcecode import __version__
|
|
474
569
|
|
|
475
570
|
current = __version__
|
|
571
|
+
measured = [
|
|
572
|
+
row for row in COMMANDS
|
|
573
|
+
if row.field_seconds is not None or row.field_blocked
|
|
574
|
+
]
|
|
476
575
|
stale = sorted(
|
|
477
576
|
row.command
|
|
478
|
-
for row in
|
|
479
|
-
if
|
|
480
|
-
|
|
577
|
+
for row in measured
|
|
578
|
+
if row.field_measured_version not in (None, current)
|
|
579
|
+
)
|
|
580
|
+
# C3-100: the other staleness, and the one that hid for three releases — not
|
|
581
|
+
# "older than the build you are running" (every row is, the day after a
|
|
582
|
+
# release) but "older than the newest measurement in this same table", which
|
|
583
|
+
# is a refresh that reached some rows and not others. Named here so the gap is
|
|
584
|
+
# readable without diffing the rows by hand.
|
|
585
|
+
newest = newest_field_measurement_version()
|
|
586
|
+
behind = sorted(
|
|
587
|
+
row.command
|
|
588
|
+
for row in measured
|
|
589
|
+
if _version_key(row.field_measured_version) < _version_key(newest)
|
|
590
|
+
)
|
|
591
|
+
lagging = (
|
|
592
|
+
f" {len(behind)} of the {len(measured)} field figures were taken before "
|
|
593
|
+
f"the newest one in this table ({newest}): "
|
|
594
|
+
f"{', '.join(behind)}."
|
|
595
|
+
if behind
|
|
596
|
+
else ""
|
|
481
597
|
)
|
|
482
598
|
return {
|
|
483
599
|
"anchor": FIELD_ANCHOR,
|
|
@@ -485,12 +601,17 @@ def field_currency() -> dict:
|
|
|
485
601
|
"running_version": current,
|
|
486
602
|
"current": FIELD_ANCHOR_MEASURED_VERSION == current,
|
|
487
603
|
"rows_measured_on_an_earlier_build": stale,
|
|
604
|
+
"newest_field_measurement_version": newest,
|
|
605
|
+
"rows_behind_the_newest_field_measurement": behind,
|
|
488
606
|
"statement": (
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
607
|
+
(
|
|
608
|
+
f"The field anchors were taken on {FIELD_ANCHOR_MEASURED_VERSION}; you "
|
|
609
|
+
f"are running {current}. Each row states the build its field figure "
|
|
610
|
+
f"belongs to, and no figure is projected onto this one."
|
|
611
|
+
if FIELD_ANCHOR_MEASURED_VERSION != current
|
|
612
|
+
else f"The field anchors were taken on {current}, the build you are running."
|
|
613
|
+
)
|
|
614
|
+
+ lagging
|
|
494
615
|
),
|
|
495
616
|
}
|
|
496
617
|
|
|
@@ -590,6 +711,7 @@ def as_dict(here: "Optional[Conditioning]" = None) -> dict:
|
|
|
590
711
|
"seconds": cmd.field_seconds,
|
|
591
712
|
"did_not_finish": cmd.field_blocked,
|
|
592
713
|
"java_files": FIELD_ANCHOR_JAVA_FILES,
|
|
714
|
+
"cache_state": cmd.field_cache_state,
|
|
593
715
|
"measured_on_version": cmd.field_measured_version,
|
|
594
716
|
"note": field_anchor_note(cmd),
|
|
595
717
|
}
|
|
@@ -713,13 +835,15 @@ def render_text(here: "Optional[Conditioning]" = None) -> str:
|
|
|
713
835
|
lines.append(f" {' ' * width} measured: {cmd.measured}")
|
|
714
836
|
if cmd.field_seconds is not None or cmd.field_blocked:
|
|
715
837
|
# C3-97: never the seconds without the build they were taken on.
|
|
838
|
+
# C3-100: nor without the cache state they were taken in.
|
|
716
839
|
note = field_anchor_note(cmd)
|
|
717
840
|
shown = (
|
|
718
841
|
"did not finish in an interactive session"
|
|
719
842
|
if cmd.field_seconds is None
|
|
720
843
|
else f"{cmd.field_seconds:g} s"
|
|
721
844
|
)
|
|
722
|
-
|
|
845
|
+
qualifiers = [f"{cmd.field_cache_state} cache"] + ([note] if note else [])
|
|
846
|
+
suffix = f" ({', '.join(qualifiers)})"
|
|
723
847
|
lines.append(
|
|
724
848
|
f" {' ' * width} field: {shown} at "
|
|
725
849
|
f"{_thousands(FIELD_ANCHOR_JAVA_FILES)} Java files{suffix}"
|
sourcecode/execution_plan.py
CHANGED
|
@@ -204,7 +204,16 @@ def cost_row(command: str):
|
|
|
204
204
|
`retrieve blast-radius <sym> .` — because that is what the front page and the
|
|
205
205
|
remedies hold. Flags and arguments are dropped; what is left is the command.
|
|
206
206
|
"""
|
|
207
|
-
|
|
207
|
+
named = (command or "").strip()
|
|
208
|
+
if named in _BY_COMMAND:
|
|
209
|
+
# A caller may also ask by the name the table itself gives a row, which is
|
|
210
|
+
# how `cache model` builds its `here:` lines. The root analysis is filed as
|
|
211
|
+
# "ask (root)" — a name the tokenizer below strips to `(root)` and then
|
|
212
|
+
# fails to find, so the root row's `here:` line was the only one published
|
|
213
|
+
# from its class instead of from its measurement (C3-100: a cell nothing
|
|
214
|
+
# reads is the cell that goes stale).
|
|
215
|
+
return _BY_COMMAND[named]
|
|
216
|
+
tokens = [t for t in named.split() if not t.startswith("-")]
|
|
208
217
|
if tokens and tokens[0] in ("ask", "sourcecode"):
|
|
209
218
|
tokens = tokens[1:]
|
|
210
219
|
if not tokens:
|
|
@@ -259,16 +268,22 @@ def _reference(row) -> Optional[str]:
|
|
|
259
268
|
f"{row.cold_seconds:g} s cold{warm} at {REFERENCE_JAVA_FILES} files"
|
|
260
269
|
)
|
|
261
270
|
note = field_anchor_note(row)
|
|
262
|
-
|
|
271
|
+
# C3-100: the cache state travels with the figure for the same reason the
|
|
272
|
+
# build does — a root analysis is 72,3 s from a purged cache and 0,3 s from a
|
|
273
|
+
# warm one, so a figure that does not say which is not a measurement anybody
|
|
274
|
+
# can act on.
|
|
275
|
+
qualified = ", ".join(
|
|
276
|
+
["field", f"{row.field_cache_state} cache"] + ([note] if note else [])
|
|
277
|
+
)
|
|
263
278
|
if row.field_seconds is not None:
|
|
264
279
|
parts.append(
|
|
265
280
|
f"{row.field_seconds:g} s at {_files(FIELD_ANCHOR_JAVA_FILES)} files "
|
|
266
|
-
f"(
|
|
281
|
+
f"({qualified})"
|
|
267
282
|
)
|
|
268
283
|
elif row.field_blocked:
|
|
269
284
|
parts.append(
|
|
270
285
|
f"did not finish in an interactive session at "
|
|
271
|
-
f"{_files(FIELD_ANCHOR_JAVA_FILES)} files (
|
|
286
|
+
f"{_files(FIELD_ANCHOR_JAVA_FILES)} files ({qualified})"
|
|
272
287
|
)
|
|
273
288
|
return " · ".join(parts) or None
|
|
274
289
|
|
|
@@ -310,11 +325,27 @@ def verdict(command: str, scope: Scope) -> Verdict:
|
|
|
310
325
|
# critical finding — under "too big for this session" because its class
|
|
311
326
|
# is repo-wide would repeat C4-19 in the other direction.
|
|
312
327
|
if row is not None and row.field_blocked:
|
|
328
|
+
# C3-100: an outcome is a measurement, so it is dated like one. "It did
|
|
329
|
+
# not finish" is the strongest claim this table makes about a command,
|
|
330
|
+
# and it was published undated while the field was running two of those
|
|
331
|
+
# commands in 9,8 s and 76,6 s. It still decides — an outcome at this
|
|
332
|
+
# size outranks a class, as C3-97 established for seconds — but it says
|
|
333
|
+
# which build it happened on and that nothing has re-measured it since.
|
|
334
|
+
note = field_anchor_note(row)
|
|
335
|
+
dated = ""
|
|
336
|
+
if note:
|
|
337
|
+
dated = (
|
|
338
|
+
f" on {row.field_measured_version}"
|
|
339
|
+
if row.field_measured_version
|
|
340
|
+
else " on a build the row does not record"
|
|
341
|
+
)
|
|
342
|
+
since = ", and nothing has re-measured it since" if note else ""
|
|
313
343
|
return Verdict(
|
|
314
344
|
command, DETACH,
|
|
315
345
|
(
|
|
316
346
|
f"measured at this size and it did not finish in an interactive "
|
|
317
|
-
f"session ({_files(FIELD_ANCHOR_JAVA_FILES)} Java files,
|
|
347
|
+
f"session{dated} ({_files(FIELD_ANCHOR_JAVA_FILES)} Java files, "
|
|
348
|
+
f"{row.field_cache_state} cache){since} — "
|
|
318
349
|
f"run it with `--output` and `--detach`, or nightly"
|
|
319
350
|
),
|
|
320
351
|
cls, reference,
|
|
@@ -325,7 +356,14 @@ def verdict(command: str, scope: Scope) -> Verdict:
|
|
|
325
356
|
# a measurement at this size and the class is not — but it is never
|
|
326
357
|
# printed as though it described the build in hand.
|
|
327
358
|
note = field_anchor_note(row)
|
|
328
|
-
|
|
359
|
+
# C3-100: and the cache state it was taken in. `ask --compact` is the
|
|
360
|
+
# row this matters most on — 72,3 s from a purged cache against 0,3 s
|
|
361
|
+
# warm on the reference — and the honest thing to publish is the
|
|
362
|
+
# measurement's conditions, not a warm figure nobody took at this size.
|
|
363
|
+
qualifiers = ", ".join(
|
|
364
|
+
[f"{row.field_cache_state} cache"] + ([note] if note else [])
|
|
365
|
+
)
|
|
366
|
+
dated = f" ({qualifiers})"
|
|
329
367
|
if row.field_seconds <= FOREGROUND_SECONDS:
|
|
330
368
|
return Verdict(
|
|
331
369
|
command, RUNS_NOW,
|
sourcecode/posture.py
CHANGED
|
@@ -33,6 +33,9 @@ from pathlib import Path
|
|
|
33
33
|
from typing import TYPE_CHECKING, Any, Optional
|
|
34
34
|
|
|
35
35
|
from sourcecode.remedies import remedy
|
|
36
|
+
from sourcecode.security_chain import CHAIN_PARTICIPANT_TYPES as _CHAIN_PARTICIPANT_TYPES
|
|
37
|
+
from sourcecode.security_chain import simple_name as _chain_simple_name
|
|
38
|
+
from sourcecode.security_chain import types_from_edges as _chain_types_from_edges
|
|
36
39
|
|
|
37
40
|
if TYPE_CHECKING:
|
|
38
41
|
from sourcecode.canonical_ir import CanonicalRepositoryIR
|
|
@@ -478,20 +481,17 @@ _SECURITY_CONFIG_ANNOTATIONS = frozenset({
|
|
|
478
481
|
"@EnableWebSecurity", "@EnableMethodSecurity", "@EnableGlobalMethodSecurity",
|
|
479
482
|
"@EnableReactiveMethodSecurity", "@EnableWebFluxSecurity",
|
|
480
483
|
})
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
#: Edge types that say "this type IS one of those" — the inheritance half of the
|
|
488
|
-
#: request-chain SPI test.
|
|
489
|
-
_SUPERTYPE_EDGES = frozenset({"extends", "implements"})
|
|
484
|
+
#: C1-43: the request-chain SPI vocabulary is `security_chain`'s, not a second
|
|
485
|
+
#: copy here. This command asks the *participant* question — a filter takes part
|
|
486
|
+
#: in configuring the chain without declaring one — while the endpoint surface
|
|
487
|
+
#: asks the narrower declaration question. Same vocabulary, same member rule, two
|
|
488
|
+
#: named subsets.
|
|
489
|
+
_SECURITY_CONFIG_SUPERTYPES = _CHAIN_PARTICIPANT_TYPES
|
|
490
490
|
|
|
491
491
|
|
|
492
492
|
def _simple_name(type_ref: str) -> str:
|
|
493
493
|
"""`a.b.Chain<T>` → `Chain`. Generic arguments are not part of the type's name."""
|
|
494
|
-
return
|
|
494
|
+
return _chain_simple_name(type_ref)
|
|
495
495
|
|
|
496
496
|
|
|
497
497
|
def _security_config_types(cir: "CanonicalRepositoryIR") -> "frozenset[str]":
|
|
@@ -520,25 +520,12 @@ def _security_config_types(cir: "CanonicalRepositoryIR") -> "frozenset[str]":
|
|
|
520
520
|
cached = getattr(cir, "_posture_security_types", None)
|
|
521
521
|
if cached is not None:
|
|
522
522
|
return cached
|
|
523
|
-
relevant: set[str] = set()
|
|
524
523
|
raw = getattr(cir, "_raw_ir", None) or {}
|
|
525
524
|
graph = raw.get("graph") if isinstance(raw, dict) else None
|
|
526
525
|
edges = (graph or {}).get("edges") if isinstance(graph, dict) else None
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
kind = edge.get("type")
|
|
531
|
-
if kind not in _SUPERTYPE_EDGES and kind != "returns":
|
|
532
|
-
continue
|
|
533
|
-
if _simple_name(edge.get("to") or "") not in _SECURITY_CONFIG_SUPERTYPES:
|
|
534
|
-
continue
|
|
535
|
-
source = str(edge.get("from") or "")
|
|
536
|
-
if not source:
|
|
537
|
-
continue
|
|
538
|
-
# A member returning the SPI type makes its *declaring class* the
|
|
539
|
-
# configuration; a type extending it is the configuration itself.
|
|
540
|
-
relevant.add(source.partition("#")[0] if kind == "returns" else source)
|
|
541
|
-
result = frozenset(relevant)
|
|
526
|
+
# C1-43: the scan itself lives in `security_chain`, so the endpoint surface's
|
|
527
|
+
# narrower reading of the same edges cannot drift from this one.
|
|
528
|
+
result = _chain_types_from_edges(edges or [], targets=_SECURITY_CONFIG_SUPERTYPES)
|
|
542
529
|
try:
|
|
543
530
|
object.__setattr__(cir, "_posture_security_types", result)
|
|
544
531
|
except Exception:
|
sourcecode/repository_ir.py
CHANGED
|
@@ -348,15 +348,19 @@ _SECURITY_MARKER_ANNOTATIONS: frozenset[str] = frozenset({
|
|
|
348
348
|
"@PermitAll", "@DenyAll", "@Authenticated",
|
|
349
349
|
})
|
|
350
350
|
|
|
351
|
-
#
|
|
352
|
-
#
|
|
353
|
-
#
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
351
|
+
# A centralized security filter chain: when the repository declares one,
|
|
352
|
+
# per-endpoint no_security_signal is expected and does NOT mean endpoints are
|
|
353
|
+
# unprotected.
|
|
354
|
+
#: C1-43: "does this repository declare a request filter chain?" is one fact with
|
|
355
|
+
#: one reading, and it lives in `security_chain` — which is also where the reason
|
|
356
|
+
#: `@EnableMethodSecurity` is *not* in it is written down. This module used to
|
|
357
|
+
#: answer it from annotations and `extends WebSecurityConfigurerAdapter` alone,
|
|
358
|
+
#: i.e. Spring Security 5 vocabulary, while `posture` had already learned that a
|
|
359
|
+
#: chain can be a `@Bean SecurityFilterChain` and nothing else. The consequence
|
|
360
|
+
#: was not a missing signal: it was SEC-001 telling the reader that a Security 6
|
|
361
|
+
#: repository has no centralized filter, and naming a route the chain admits on
|
|
362
|
+
#: purpose as reachable without authentication.
|
|
363
|
+
from sourcecode.security_chain import declares_request_chain as _declares_request_chain
|
|
360
364
|
|
|
361
365
|
# Programmatic security: method-call patterns that indicate runtime auth enforcement.
|
|
362
366
|
# Requires method-call or field-access context — bare class name mentions (imports,
|
|
@@ -4586,16 +4590,10 @@ def _assemble(
|
|
|
4586
4590
|
# Stored here so CIR projections (project_endpoint_surface) can read it without
|
|
4587
4591
|
# re-parsing symbols.
|
|
4588
4592
|
_class_syms_asm = [s for s in sorted_syms if s.type in ("class", "interface")]
|
|
4589
|
-
_filter_based_asm = (
|
|
4590
|
-
|
|
4591
|
-
|
|
4592
|
-
|
|
4593
|
-
for ann in sym.annotations
|
|
4594
|
-
)
|
|
4595
|
-
or any(
|
|
4596
|
-
_extends_map.get(sym.symbol, "") == "WebSecurityConfigurerAdapter"
|
|
4597
|
-
for sym in _class_syms_asm
|
|
4598
|
-
)
|
|
4593
|
+
_filter_based_asm = _declares_request_chain(
|
|
4594
|
+
annotations=(a for sym in _class_syms_asm for a in sym.annotations),
|
|
4595
|
+
supertypes=(_extends_map.get(sym.symbol, "") for sym in _class_syms_asm),
|
|
4596
|
+
return_types=(s.return_type for s in sorted_syms if s.type == "method"),
|
|
4599
4597
|
)
|
|
4600
4598
|
# A repo whose authorization runs through its OWN annotation is not an
|
|
4601
4599
|
# unsecured repo. The `endpoints` command applied this predicate to its own
|
|
@@ -6487,18 +6485,10 @@ def extract_java_endpoints(root: Path) -> "dict[str, Any]":
|
|
|
6487
6485
|
# When present, high no_security_signal is expected — security is enforced by
|
|
6488
6486
|
# the filter chain, not per-endpoint annotations.
|
|
6489
6487
|
_class_syms = [s for s in all_symbols if s.type in ("class", "interface")]
|
|
6490
|
-
_filter_based = (
|
|
6491
|
-
|
|
6492
|
-
|
|
6493
|
-
|
|
6494
|
-
for sym in _class_syms
|
|
6495
|
-
for ann in sym.annotations
|
|
6496
|
-
)
|
|
6497
|
-
# Class extends WebSecurityConfigurerAdapter (pre-Spring 5.7 style)
|
|
6498
|
-
or any(
|
|
6499
|
-
extends_map.get(sym.symbol, "") == "WebSecurityConfigurerAdapter"
|
|
6500
|
-
for sym in _class_syms
|
|
6501
|
-
)
|
|
6488
|
+
_filter_based = _declares_request_chain(
|
|
6489
|
+
annotations=(a for sym in _class_syms for a in sym.annotations),
|
|
6490
|
+
supertypes=(extends_map.get(sym.symbol, "") for sym in _class_syms),
|
|
6491
|
+
return_types=(s.return_type for s in all_symbols if s.type == "method"),
|
|
6502
6492
|
)
|
|
6503
6493
|
# A repo whose authorization runs through its OWN annotation is not an
|
|
6504
6494
|
# unsecured repo. The structural gate predicate (coverage + specificity over
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""security_chain.py — the Spring Security request filter chain, read once.
|
|
2
|
+
|
|
3
|
+
C1-43. *"Does this repository declare a request filter chain?"* had two readings
|
|
4
|
+
that disagreed, and the weaker one decided whether a `high`-severity finding
|
|
5
|
+
family ran at all.
|
|
6
|
+
|
|
7
|
+
`posture` learned the modern shape and recorded why in its own docstring: the
|
|
8
|
+
chain used to be read off annotations only, so **a repository that wires its
|
|
9
|
+
chain with a `@Bean SecurityFilterChain` and no `@EnableWebSecurity` reported
|
|
10
|
+
`security.active: 0`** — a false zero on the axis that command exists for. The
|
|
11
|
+
endpoint surface never learned it. It still tested `@EnableWebSecurity` and
|
|
12
|
+
`extends WebSecurityConfigurerAdapter`, which is Spring Security 5 vocabulary,
|
|
13
|
+
so a Spring Security 6 application — where the chain is a bean and the
|
|
14
|
+
annotation is optional — came out `annotation_based`, and SEC-001 then told the
|
|
15
|
+
reader, per unannotated handler, that *"there is no centralized filter — any
|
|
16
|
+
caller can reach this endpoint without authentication"*.
|
|
17
|
+
|
|
18
|
+
Field evaluation #24 measured what that costs: on a greenfield Boot 3 service,
|
|
19
|
+
the one SEC-001 finding was `POST /api/v1/auth/login`, a route the chain admits
|
|
20
|
+
on purpose (`PUBLIC_ENDPOINTS` … `permitAll()`), sitting under an
|
|
21
|
+
`anyRequest().authenticated()` that guards the other 24. The detection was
|
|
22
|
+
right; the binding was missing.
|
|
23
|
+
|
|
24
|
+
**Two questions, deliberately named apart, because they have different answers
|
|
25
|
+
and only one of them may silence a finding.**
|
|
26
|
+
|
|
27
|
+
``declares_request_chain`` — *does the repository declare a chain at all?* A
|
|
28
|
+
declaration is the chain itself: the enabling annotation, the pre-5.7 adapter,
|
|
29
|
+
or a member that produces a ``SecurityFilterChain``. This is the premise
|
|
30
|
+
SEC-001 rests on, so it is deliberately narrow: a filter is **not** a
|
|
31
|
+
declaration. Widening it here would silence the rule on repositories that have
|
|
32
|
+
no centralized authorization, which is the failure mode P1-A already records.
|
|
33
|
+
|
|
34
|
+
``chain_participant_types`` — *which types take part in configuring the chain?*
|
|
35
|
+
Broader: a ``OncePerRequestFilter`` participates without declaring anything.
|
|
36
|
+
`posture` asks this one, to show which beans a profile switch moves.
|
|
37
|
+
|
|
38
|
+
Both read the same vocabulary and the same member rule from this file, so the
|
|
39
|
+
two answers can differ in scope but never in mechanism.
|
|
40
|
+
"""
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
from typing import Iterable
|
|
44
|
+
|
|
45
|
+
#: The annotation that turns on the request filter chain. **One entry, on
|
|
46
|
+
#: purpose.** `@EnableMethodSecurity` / `@EnableGlobalMethodSecurity` enable
|
|
47
|
+
#: per-method annotation security (`@PreAuthorize`, `@Secured`) and configure no
|
|
48
|
+
#: chain; counting them here would suppress SEC-001 for every unannotated
|
|
49
|
+
#: endpoint in exactly the repositories the rule exists for.
|
|
50
|
+
CHAIN_DECLARATION_ANNOTATIONS: "frozenset[str]" = frozenset({
|
|
51
|
+
"@EnableWebSecurity",
|
|
52
|
+
})
|
|
53
|
+
|
|
54
|
+
#: Types whose presence *is* the chain. `WebSecurityConfigurerAdapter` is the
|
|
55
|
+
#: pre-5.7 form (extended); `SecurityFilterChain` is the current one (returned
|
|
56
|
+
#: from a `@Bean` method). Either one means the request path has a centralized
|
|
57
|
+
#: authorization configuration.
|
|
58
|
+
CHAIN_DECLARATION_TYPES: "frozenset[str]" = frozenset({
|
|
59
|
+
"WebSecurityConfigurerAdapter",
|
|
60
|
+
"SecurityFilterChain",
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
#: Everything that participates in configuring the chain, which includes
|
|
64
|
+
#: declaring it. A filter or an interceptor changes what the chain does without
|
|
65
|
+
#: declaring that there is one — true for `posture`'s question, false for
|
|
66
|
+
#: SEC-001's premise.
|
|
67
|
+
CHAIN_PARTICIPANT_TYPES: "frozenset[str]" = CHAIN_DECLARATION_TYPES | frozenset({
|
|
68
|
+
"AbstractSecurityInterceptor",
|
|
69
|
+
"OncePerRequestFilter",
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
#: Edge kinds that say "this type IS one of those" — the inheritance half of the
|
|
73
|
+
#: test. The other half is `returns`, handled separately because it attributes to
|
|
74
|
+
#: the declaring class rather than to the member.
|
|
75
|
+
SUPERTYPE_EDGES: "frozenset[str]" = frozenset({"extends", "implements"})
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def simple_name(type_ref: str) -> str:
|
|
79
|
+
"""``a.b.Chain<T>`` → ``Chain``. Generic arguments are not part of a name.
|
|
80
|
+
|
|
81
|
+
Array and whitespace noise is stripped too, because the same vocabulary is
|
|
82
|
+
matched against a declared return type (`SecurityFilterChain`, written by
|
|
83
|
+
hand) and against a resolved edge target (`org.springframework…`, produced
|
|
84
|
+
by the resolver), and a rule that only worked on one of them would be the
|
|
85
|
+
drift this module exists to prevent.
|
|
86
|
+
"""
|
|
87
|
+
return str(type_ref).partition("<")[0].replace("[]", "").strip().rsplit(".", 1)[-1].strip()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _matches(type_ref: str, targets: "frozenset[str]") -> bool:
|
|
91
|
+
return simple_name(type_ref) in targets
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def declares_request_chain(
|
|
95
|
+
*,
|
|
96
|
+
annotations: "Iterable[str]" = (),
|
|
97
|
+
supertypes: "Iterable[str]" = (),
|
|
98
|
+
return_types: "Iterable[str]" = (),
|
|
99
|
+
) -> bool:
|
|
100
|
+
"""True when this repository declares a Spring Security request filter chain.
|
|
101
|
+
|
|
102
|
+
Three evidence shapes, because the two callers hold the evidence in different
|
|
103
|
+
forms and neither should have to reshape it into the other's: annotations
|
|
104
|
+
anywhere in the repository, supertypes of its classes, and the declared
|
|
105
|
+
return types of its members. The rule is one rule; only the input differs.
|
|
106
|
+
|
|
107
|
+
A member returning the chain type is a declaration of the chain — that is the
|
|
108
|
+
`@Bean SecurityFilterChain` shape, and the reason this function exists.
|
|
109
|
+
"""
|
|
110
|
+
for annotation in annotations:
|
|
111
|
+
if str(annotation).split("(", 1)[0].strip() in CHAIN_DECLARATION_ANNOTATIONS:
|
|
112
|
+
return True
|
|
113
|
+
for supertype in supertypes:
|
|
114
|
+
if supertype and _matches(supertype, CHAIN_DECLARATION_TYPES):
|
|
115
|
+
return True
|
|
116
|
+
for return_type in return_types:
|
|
117
|
+
if return_type and _matches(return_type, CHAIN_DECLARATION_TYPES):
|
|
118
|
+
return True
|
|
119
|
+
return False
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def types_from_edges(
|
|
123
|
+
edges: "Iterable[dict]",
|
|
124
|
+
*,
|
|
125
|
+
targets: "frozenset[str]",
|
|
126
|
+
) -> "frozenset[str]":
|
|
127
|
+
"""Types related to *targets* by inheritance or by a member that returns one.
|
|
128
|
+
|
|
129
|
+
Reads the IR's own edge list — the vocabulary of a node is fqn/annotations/
|
|
130
|
+
signature/role, so inheritance and return types are `extends`, `implements`
|
|
131
|
+
and `returns` **edges** and nothing else. A member returning the SPI type
|
|
132
|
+
makes its *declaring class* the configuration; a type extending it is the
|
|
133
|
+
configuration itself.
|
|
134
|
+
"""
|
|
135
|
+
relevant: "set[str]" = set()
|
|
136
|
+
for edge in edges or ():
|
|
137
|
+
if not isinstance(edge, dict):
|
|
138
|
+
continue
|
|
139
|
+
kind = edge.get("type")
|
|
140
|
+
if kind not in SUPERTYPE_EDGES and kind != "returns":
|
|
141
|
+
continue
|
|
142
|
+
if not _matches(edge.get("to") or "", targets):
|
|
143
|
+
continue
|
|
144
|
+
source = str(edge.get("from") or "")
|
|
145
|
+
if not source:
|
|
146
|
+
continue
|
|
147
|
+
relevant.add(source.partition("#")[0] if kind == "returns" else source)
|
|
148
|
+
return frozenset(relevant)
|
|
@@ -467,26 +467,73 @@ def _comment_excerpt(text: str) -> str:
|
|
|
467
467
|
return cleaned[:120] + ("..." if len(cleaned) > 120 else "")
|
|
468
468
|
|
|
469
469
|
|
|
470
|
+
#: How far past a comment the declaration it heads is looked for. A declaration
|
|
471
|
+
#: head is annotations + modifiers + signature; 600 characters covers a generously
|
|
472
|
+
#: annotated method and stops well short of the next member.
|
|
473
|
+
_DECLARATION_HEAD_WINDOW = 600
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _control_is_live_below(live_source: str, end: int, pattern: "re.Pattern") -> bool:
|
|
477
|
+
"""True when the same control is live in the declaration this comment heads.
|
|
478
|
+
|
|
479
|
+
C3-101, the second discriminator. A comment that *names* a control which the
|
|
480
|
+
very next declaration *carries* is documenting it, not disabling it — the
|
|
481
|
+
shape is `// authorization is @PreAuthorize below` followed by the live
|
|
482
|
+
annotation. The window ends at the first `{` or `;` (the end of a declaration
|
|
483
|
+
head), so a control commented out on one member is never excused by a live
|
|
484
|
+
one on the next.
|
|
485
|
+
|
|
486
|
+
Read on comment-blanked source, so a second comment inside the window cannot
|
|
487
|
+
be mistaken for live code.
|
|
488
|
+
"""
|
|
489
|
+
window = live_source[end:end + _DECLARATION_HEAD_WINDOW]
|
|
490
|
+
stop = min(
|
|
491
|
+
(pos for pos in (window.find("{"), window.find(";")) if pos != -1),
|
|
492
|
+
default=len(window),
|
|
493
|
+
)
|
|
494
|
+
return bool(pattern.search(window[:stop]))
|
|
495
|
+
|
|
496
|
+
|
|
470
497
|
def _scan_commented_controls(
|
|
471
498
|
source: str,
|
|
472
499
|
rel: str,
|
|
473
500
|
*,
|
|
474
501
|
xml: bool = False,
|
|
502
|
+
live_source: "Optional[str]" = None,
|
|
475
503
|
) -> "list[SecurityConfigObservation]":
|
|
476
|
-
"""DEAD-001 — security controls visible only inside comments.
|
|
477
|
-
|
|
504
|
+
"""DEAD-001 — security controls visible only inside comments.
|
|
505
|
+
|
|
506
|
+
C3-101: *only inside comments* is the claim, and it is not what a comment span
|
|
507
|
+
alone establishes. Two discriminators separate a switched-off control from a
|
|
508
|
+
documented one, both lexical and both decidable from the source already read:
|
|
509
|
+
a Javadoc block is documentation by construction (`source_text.is_javadoc`),
|
|
510
|
+
and a control the adjacent declaration carries live is being described rather
|
|
511
|
+
than disabled. Neither was applied, and on a well-documented repository the
|
|
512
|
+
result was 8 `high` findings and 0 true positives — the rule fired *because*
|
|
513
|
+
the repository documents its authorization decisions.
|
|
514
|
+
"""
|
|
515
|
+
from sourcecode.source_text import commented_spans, is_javadoc
|
|
478
516
|
|
|
479
517
|
rules = _COMMENTED_XML_CONTROLS if xml else _COMMENTED_JAVA_CONTROLS
|
|
480
518
|
hints = _COMMENTED_XML_HINTS if xml else _COMMENTED_JAVA_HINTS
|
|
481
519
|
if not any(hint in source for hint in hints):
|
|
482
520
|
return []
|
|
521
|
+
if live_source is None and not xml:
|
|
522
|
+
from sourcecode.source_text import blank_java_comments
|
|
523
|
+
|
|
524
|
+
live_source = blank_java_comments(source)
|
|
483
525
|
out: "list[SecurityConfigObservation]" = []
|
|
484
526
|
seen: set[tuple[str, int]] = set()
|
|
485
527
|
for start, end in commented_spans(source, xml=xml):
|
|
528
|
+
# Javadoc is addressed to a reader; a disabled control is not written in it.
|
|
529
|
+
if not xml and is_javadoc(source, start):
|
|
530
|
+
continue
|
|
486
531
|
text = source[start:end]
|
|
487
532
|
for pattern, symbol, detail in rules:
|
|
488
533
|
if not pattern.search(text):
|
|
489
534
|
continue
|
|
535
|
+
if not xml and _control_is_live_below(live_source or "", end, pattern):
|
|
536
|
+
continue
|
|
490
537
|
line = _line_of(source, start)
|
|
491
538
|
key = (symbol, line)
|
|
492
539
|
if key in seen:
|
|
@@ -623,8 +670,11 @@ def scan_security_configuration(
|
|
|
623
670
|
source = (root / rel).read_text(encoding="utf-8", errors="replace")
|
|
624
671
|
except OSError:
|
|
625
672
|
continue
|
|
626
|
-
|
|
627
|
-
|
|
673
|
+
# One blanking per file: the live view is what `_scan_java` reads and what
|
|
674
|
+
# C3-101's second discriminator tests the declaration head against.
|
|
675
|
+
_live = blank_java_comments(source)
|
|
676
|
+
out.extend(_scan_commented_controls(source, rel, xml=False, live_source=_live))
|
|
677
|
+
out.extend(_scan_java(_live, rel))
|
|
628
678
|
|
|
629
679
|
# C3-64 — one pruned walk for all three populations. This used to be five
|
|
630
680
|
# `rglob` calls (one per config suffix, one for descriptors, one for
|
sourcecode/source_text.py
CHANGED
|
@@ -165,3 +165,21 @@ def commented_spans(source: str, *, xml: bool) -> "list[tuple[int, int]]":
|
|
|
165
165
|
if xml or match.lastgroup in ("line", "block"):
|
|
166
166
|
spans.append((match.start(), match.end()))
|
|
167
167
|
return spans
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def is_javadoc(source: str, start: int) -> bool:
|
|
171
|
+
"""True when the comment beginning at *start* is a Javadoc block.
|
|
172
|
+
|
|
173
|
+
C3-101: a comment span says *what is not compiled*, and DEAD-001 reads it as
|
|
174
|
+
*what was switched off* — two different things, and the difference is
|
|
175
|
+
lexical. Code is disabled with `//` or `/* */`; `/** */` is the documentation
|
|
176
|
+
form, addressed to a reader and processed by a doc tool. Nobody disables an
|
|
177
|
+
annotation by moving it into a Javadoc block, and a repository that documents
|
|
178
|
+
the authorization decision it makes writes the annotation's name there on
|
|
179
|
+
purpose — which is how eight `high` findings were produced by good
|
|
180
|
+
documentation, with the live annotations sixty lines below in the same file.
|
|
181
|
+
|
|
182
|
+
`/**/` is an empty block comment, not a Javadoc opener, and is excluded the
|
|
183
|
+
way `javadoc` itself excludes it.
|
|
184
|
+
"""
|
|
185
|
+
return source.startswith("/**", start) and not source.startswith("/**/", start)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sourcecode
|
|
3
|
-
Version: 5.0.
|
|
3
|
+
Version: 5.0.1
|
|
4
4
|
Summary: Persistent structural context and ultra-fast repeated analysis for AI coding agents
|
|
5
5
|
License-File: LICENSE
|
|
6
6
|
Keywords: agents,ai,codebase,context,developer-tools,llm
|
|
@@ -42,7 +42,7 @@ Description-Content-Type: text/markdown
|
|
|
42
42
|
|
|
43
43
|
**Context · Impact · Migration · Architecture · Review — everything from one structural model.**
|
|
44
44
|
|
|
45
|
-

|
|
46
46
|

|
|
47
47
|
|
|
48
48
|
> **ASK Engine** is the product. The CLI command is **`ask`**. The legacy **`sourcecode`**
|
|
@@ -126,7 +126,7 @@ brew tap haroundominique/sourcecode && brew install sourcecode
|
|
|
126
126
|
# pip / pipx
|
|
127
127
|
pipx install sourcecode # or: pip install sourcecode
|
|
128
128
|
|
|
129
|
-
ask version # ask 5.0.
|
|
129
|
+
ask version # ask 5.0.1 — and, on a build that has aged,
|
|
130
130
|
# how many releases have probably shipped since
|
|
131
131
|
```
|
|
132
132
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
sourcecode/__init__.py,sha256=
|
|
1
|
+
sourcecode/__init__.py,sha256=Vi4TJSP5-Nuef7bTyGWp6icmexh3SfDW_L2-OJMHyWk,308
|
|
2
2
|
sourcecode/adaptive_scanner.py,sha256=yJBKjNpkY6bpueYJ2YnRezen3sYZDecEt7WaaNWdqug,9466
|
|
3
3
|
sourcecode/archetype.py,sha256=CZvRLpkHot_D8D3JFQVorr-EHDJyh0BbS7RnSTqBigM,40499
|
|
4
4
|
sourcecode/architectural_baseline.py,sha256=4GiMVBLJVHvRKSWQGf2ZVurI1-R0qk5pmaevgvGSgVA,26325
|
|
@@ -9,7 +9,7 @@ sourcecode/ast_extractor.py,sha256=aXJjZ7XjAdxa99hWNm4SyzPqx4gNWiTbUYjNWGNl-Vc,5
|
|
|
9
9
|
sourcecode/audit_report.py,sha256=CLvvlQH2oA6Qj4UADUWhNwSsp1OcyzwXXey5DN_IiWI,8660
|
|
10
10
|
sourcecode/baseline_autocapture.py,sha256=tmMLexQSeaQRRbqxAd7walvGQqnF0mdpxQmkL40rlhI,15623
|
|
11
11
|
sourcecode/cache.py,sha256=CK_J8NqyaTNZ57KQT-R4puqI8yYlzG5WL7uLFRbzods,41727
|
|
12
|
-
sourcecode/cache_model.py,sha256=
|
|
12
|
+
sourcecode/cache_model.py,sha256=8xaRd_9zS6XMGPWZ7E6FHSimAG-pZQKNGQBi1jBFLhY,47724
|
|
13
13
|
sourcecode/call_surface.py,sha256=fiqYfHooxN1fX9oQoysq1LS3LoZcobhyUNjAGEZKpwk,4148
|
|
14
14
|
sourcecode/caller_metrics.py,sha256=--sFGDnIog_YGu9xZHMcB91F5xZakfaAKvy15xU54hg,7904
|
|
15
15
|
sourcecode/caller_reach.py,sha256=RRF49tv4-QswraJ2vsZ89VAMOj7BmYXlAdXr-tLP_CA,9483
|
|
@@ -52,7 +52,7 @@ sourcecode/envelope.py,sha256=fpF_8znvPqGXKZb0UPYcCYjm27yO7Pr-WsdXNE-tEoY,8911
|
|
|
52
52
|
sourcecode/environment_resolution.py,sha256=bfhkM0RGSyLwzFvfi-YhHjU1cy2ufA_WIkXBVeHs9y8,22130
|
|
53
53
|
sourcecode/error_schema.py,sha256=uwosfNaSujtYm11_732Hu92z5ITV040fQDaIyefSvR4,1683
|
|
54
54
|
sourcecode/evidence_provider.py,sha256=GSSL44JEaouO5AHks2sB3d1YvC9xIKIld1yBYxZpXxo,4277
|
|
55
|
-
sourcecode/execution_plan.py,sha256=
|
|
55
|
+
sourcecode/execution_plan.py,sha256=c_pYnk-JFY7QNuk_Wc1F0z6tBJswUeu_j5i3QQtakGg,20390
|
|
56
56
|
sourcecode/explain.py,sha256=yqxvKiLF0XMMg11PdgqMFrsim5SXuEKM1U2cLvCLspM,29935
|
|
57
57
|
sourcecode/file_chunker.py,sha256=3vkM3mDQ5eE_yTPvUgjyjpGFBIjkW6_mrBmIbrylnA8,16444
|
|
58
58
|
sourcecode/file_classifier.py,sha256=pJCeN9KqWpAwKMCgGP4KDsBjuWMeo4zlbj5fv3hk9dA,15587
|
|
@@ -85,7 +85,7 @@ sourcecode/path_filters.py,sha256=LmTYq735orssKTIuxTX1mE1FHAXF7dVEJEse35xE1RA,12
|
|
|
85
85
|
sourcecode/perf.py,sha256=ZCPxDIg6ww6PXfLwcqNTH99lc1UgkYUZI3MywFgRO90,25991
|
|
86
86
|
sourcecode/phased_run.py,sha256=cFd0Lxfc8XL-XkCVCCzcjd9bZbtgkJkGZAHfmJ6dBJ4,16542
|
|
87
87
|
sourcecode/pipe_contract.py,sha256=PML0Er5d8uDyec4OrdUXuyqHbGPUYndjeaKUjkFz2u4,8369
|
|
88
|
-
sourcecode/posture.py,sha256=
|
|
88
|
+
sourcecode/posture.py,sha256=7d3zM4ExrSXVKrWNTlXQ0fUQNqDdl1LOuKwRVaXdRss,76574
|
|
89
89
|
sourcecode/pr_comment_renderer.py,sha256=239PmJdf95av_ZW236C7_tvq_ahECeuF9AZxrheJ4OQ,15573
|
|
90
90
|
sourcecode/pr_impact.py,sha256=KyVBvHCCbgByKA_wqTbO1dSkfi05GEudIAuH4CyDPK4,27395
|
|
91
91
|
sourcecode/prepare_context.py,sha256=ntB2MSScHtYRKRm9SansA0FrklMnXZmE9_yRXqXk0qc,239264
|
|
@@ -102,7 +102,7 @@ sourcecode/relevance_scorer.py,sha256=0AgEt4KrV73nioMqBgjhGjtY7L2C7L7cSyKtj3IKcr
|
|
|
102
102
|
sourcecode/remedies.py,sha256=9eAuP9ndw78YvuUtOi2VvoeNnVCMh1q0uxfnPIkvx7c,7838
|
|
103
103
|
sourcecode/rename_refactor.py,sha256=h6dNFlB9aZ_3q6heeHBkgXQeXaT03nvPSsYH6P8qxFg,12965
|
|
104
104
|
sourcecode/repo_classifier.py,sha256=FG1vaWKdWXsWdl-S8hjVMiTqcwgaRXkDyvK4rPcOGtQ,22681
|
|
105
|
-
sourcecode/repository_ir.py,sha256=
|
|
105
|
+
sourcecode/repository_ir.py,sha256=R2LjUHUxdOQdx4lGCCgKLY-9K7ldc7JAuW7MpEFJ124,368882
|
|
106
106
|
sourcecode/ris.py,sha256=nc0d6boOiREFqXO4eLHHaYcnqZj4pT_AG43-Xz9ejkc,24862
|
|
107
107
|
sourcecode/risk.py,sha256=QSm_F-3tGncQNMcdKGgr5bVONM8614J8n8qmP9x4FDU,73200
|
|
108
108
|
sourcecode/rule_catalog.py,sha256=pTkgkQZ1u7atB3fkQJ9BPWo8lNiamvEuRN7MtRgle0c,5263
|
|
@@ -112,8 +112,9 @@ sourcecode/runtime_classifier.py,sha256=uTAD6BDCiBLUZEDRfqk718kM4RTT_vAbfkcOI2_X
|
|
|
112
112
|
sourcecode/sarif.py,sha256=3hQEegUxIZbojFdY59oB-yKsK-rnafHdTNmYsEP2--A,25984
|
|
113
113
|
sourcecode/scanner.py,sha256=z3CV0rcGunu0Y8mpNgp07wI7nxT0pxw1BkXRRtI0Rpo,9609
|
|
114
114
|
sourcecode/schema.py,sha256=aHNXDf8LGyUC8ZDE_VS9kiskC2-Oswhi_WnpdGy6HDw,24897
|
|
115
|
+
sourcecode/security_chain.py,sha256=1YZVeFLwy65kntMBhMNvz1cUpOoKCmvEqYICgxs66Gw,6823
|
|
115
116
|
sourcecode/security_config.py,sha256=_tdyI939NeOlgwQWNCXPfsqaB1fTyNROizYqTOdvWGQ,3718
|
|
116
|
-
sourcecode/security_config_scan.py,sha256=
|
|
117
|
+
sourcecode/security_config_scan.py,sha256=CttiIYVhBlPCuLNYMWTEtgbZmwXGYm2GYiFo9ehD1Go,32487
|
|
117
118
|
sourcecode/security_posture.py,sha256=FiEGood01y5gKhuYyL6NK4VT-WtCUMJwHOSR5xTfnPI,54063
|
|
118
119
|
sourcecode/semantic_analyzer.py,sha256=bpgdC6m0_ftVtRf3rSdwhbhWjnZnGxRXaZVcfe4BbcQ,95414
|
|
119
120
|
sourcecode/semantic_impact_engine.py,sha256=t09IirGC3JjQDy33JZd1_WKzQVKXkoNl3-XEUr5kjis,20563
|
|
@@ -121,7 +122,7 @@ sourcecode/semantic_integration_engine.py,sha256=7a0WqAInOv39f0Yr_94TYo_JP_8QpeI
|
|
|
121
122
|
sourcecode/semantic_services.py,sha256=nbUuPv-F01USTt_9CHT8iy_ucCIw3fz4W3Aquea_pd4,10782
|
|
122
123
|
sourcecode/serializer.py,sha256=KiUUHMQNWOOJye95bZQKsRKFkVeVK0Vb2Y7IocQjVbk,140262
|
|
123
124
|
sourcecode/servlet_surface.py,sha256=aOeRne07man8DwHsbog4FTOyF3sYV-oP_N1e4gt3Uvk,9115
|
|
124
|
-
sourcecode/source_text.py,sha256=
|
|
125
|
+
sourcecode/source_text.py,sha256=R52gmSljz7L6YVPbixv5W-LOZ5eKUVvSUP3S-QH_IWc,9047
|
|
125
126
|
sourcecode/spring_event_topology.py,sha256=5_ON_21Le5zbG-1GRc5GLIi5HJfy_QjcXLVPC5WeUGQ,18055
|
|
126
127
|
sourcecode/spring_findings.py,sha256=-0EZMFpZxPRmK2FiR4d7Mk4O73c4wkga-Dqs1-fBSUA,22895
|
|
127
128
|
sourcecode/spring_impact.py,sha256=GW77k7KMn13QmzP5520BbVcXSCYlTFeD1bFibhbmzJM,83907
|
|
@@ -205,8 +206,8 @@ sourcecode/telemetry/consent.py,sha256=pQdl-QeLl6Gcibn0eWHSKZrm-HYSsjpVqOnjrgFp8
|
|
|
205
206
|
sourcecode/telemetry/events.py,sha256=4_yeO58U-Cwc1Qb27VB0_EjhmroY0k91n3_VGxeALB8,2776
|
|
206
207
|
sourcecode/telemetry/filters.py,sha256=RzxauTz8HliO4BllQnXEXc7zTeqdCZi5MgqGEDuW7OQ,6570
|
|
207
208
|
sourcecode/telemetry/transport.py,sha256=4gGHsq0WeY9VywEZXA3vUxykfiYnw9uuqfjAAec7F8o,1681
|
|
208
|
-
sourcecode-5.0.
|
|
209
|
-
sourcecode-5.0.
|
|
210
|
-
sourcecode-5.0.
|
|
211
|
-
sourcecode-5.0.
|
|
212
|
-
sourcecode-5.0.
|
|
209
|
+
sourcecode-5.0.1.dist-info/METADATA,sha256=wKxYzyIT5zVAkDjb12M3Wzj2yyLMr-uklRY8AmRlHGI,42410
|
|
210
|
+
sourcecode-5.0.1.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
|
|
211
|
+
sourcecode-5.0.1.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
|
|
212
|
+
sourcecode-5.0.1.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
|
|
213
|
+
sourcecode-5.0.1.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|