sourcecode 4.10.7__py3-none-any.whl → 4.12.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of sourcecode might be problematic. Click here for more details.

sourcecode/__init__.py CHANGED
@@ -4,4 +4,4 @@ ASK Engine is the product. ``ask`` is the canonical CLI command; ``sourcecode``
4
4
  the legacy compatibility alias and the Python/PyPI package name. See
5
5
  docs/PRODUCT_IDENTITY.md (normative)."""
6
6
 
7
- __version__ = "4.10.7"
7
+ __version__ = "4.12.0"
@@ -30,6 +30,7 @@ the perf baselines do.
30
30
  from __future__ import annotations
31
31
 
32
32
  import json
33
+ import os
33
34
  import subprocess
34
35
  from datetime import datetime, timezone
35
36
  from pathlib import Path
@@ -365,21 +366,40 @@ def _baseline_filename(baseline: dict) -> str:
365
366
  #: a 14 587-entry structural inventory of the code. A self-confined
366
367
  #: `.ask/.gitignore` is the honest form: it never touches the user's own
367
368
  #: `.gitignore`, and it stops applying the moment they delete the directory.
369
+ #: C3-74 corrected the last line. It used to read *"Delete this file to version
370
+ #: them deliberately"*, and deleting it is the one action that would commit a
371
+ #: baseline carrying `env.hostname_hash`, `cpu` and `cores` — an invitation to do
372
+ #: the thing the file exists to prevent. It now states the consequence instead.
368
373
  _ASK_DIR_GITIGNORE = (
369
374
  "# Written by `ask` the first time it created this directory.\n"
370
375
  "# These are local analysis artefacts: a structural inventory of your code\n"
371
376
  "# and, if you asked for it, the machine that measured it. Nothing here\n"
372
- "# belongs in a commit. Delete this file to version them deliberately.\n"
377
+ "# belongs in a commit.\n"
378
+ "#\n"
379
+ "# Removing this file makes everything here committable, including any\n"
380
+ "# recorded machine fingerprint (`--record-env`: hostname hash, CPU, cores).\n"
381
+ "# Version these deliberately only if you intend to publish that.\n"
373
382
  "*\n"
374
383
  )
375
384
 
376
385
 
377
- def protect_ask_dir(out_dir: Path) -> None:
386
+ def protect_ask_dir(out_dir: Path, *, announce: bool = True) -> None:
378
387
  """Give a freshly created `.ask/` its own `.gitignore` (C3-68).
379
388
 
380
389
  Best-effort and never overwriting: a user who edited it decided something,
381
390
  and a tool that re-imposes its default on every run is worse than one that
382
391
  never wrote it.
392
+
393
+ **Call this where the write happens, never where the path is computed.** This
394
+ function *creates* the directory in order to protect it, so calling it
395
+ speculatively is a write — which is how `migrate-check` came to leave a
396
+ `.ask/.gitignore` inside a repository it had only read (C3-74).
397
+
398
+ `announce` puts one line on stderr the first time the directory is created,
399
+ with its path. A side effect inside somebody else's repository is exactly what
400
+ I-8 says must be declared, and this one hides from the check an auditor makes:
401
+ `git status --porcelain` is clean because the ignore file ignores itself, and
402
+ only `--ignored` reveals `!! <scope>/.ask/`.
383
403
  """
384
404
  try:
385
405
  ask_dir = Path(out_dir)
@@ -391,12 +411,35 @@ def protect_ask_dir(out_dir: Path) -> None:
391
411
  return
392
412
  marker = ask_dir / ".gitignore"
393
413
  if not marker.exists():
414
+ created = not ask_dir.exists()
394
415
  ask_dir.mkdir(parents=True, exist_ok=True)
395
416
  marker.write_text(_ASK_DIR_GITIGNORE, encoding="utf-8")
417
+ if announce and created:
418
+ _announce_ask_dir(ask_dir)
396
419
  except OSError:
397
420
  return
398
421
 
399
422
 
423
+ def _announce_ask_dir(ask_dir: Path) -> None:
424
+ """One line, on stderr, naming what was created and how to move it (C3-74)."""
425
+ import sys
426
+
427
+ try:
428
+ # No flag is named here on purpose: the commands that write under `.ask/`
429
+ # spell the override differently (`baseline capture --dir`,
430
+ # `migrate-check --history-dir`), and a message that names the wrong one
431
+ # is the class of defect this ledger is about.
432
+ sys.stderr.write(
433
+ f"[ask] created {ask_dir}{os.sep} for this run's local artefacts — "
434
+ f"git-ignored by its own .gitignore, so `git status` will not show it "
435
+ f"(`git status --ignored` will). The command's directory option takes a "
436
+ f"path outside the repository if you want them elsewhere.\n"
437
+ )
438
+ sys.stderr.flush()
439
+ except Exception:
440
+ pass
441
+
442
+
400
443
  def write_baseline(baseline: dict, out_dir: Path) -> Path:
401
444
  """Write `baseline` as deterministic JSON under `out_dir`; return the file path."""
402
445
  out_dir.mkdir(parents=True, exist_ok=True)
sourcecode/cache_model.py CHANGED
@@ -75,10 +75,17 @@ class CommandCache:
75
75
  warm_seconds: "Optional[float]" = None
76
76
  #: The other measured point: this command on the field repository
77
77
  #: (3 342 Java files, Windows 11 / PowerShell 5.1 / pipx, cold), from field
78
- #: evaluations #13 and #14. Seconds where a run finished; `field_blocked`
78
+ #: evaluations #13, #14 and #16. Seconds where a run finished; `field_blocked`
79
79
  #: where it did not — the harness promoted the process and the session died,
80
80
  #: which is a measurement of a different kind and is never written as a
81
81
  #: duration. Rows with neither were not attempted there.
82
+ #:
83
+ #: C3-72: where two evaluations of the same command at this size disagree, the
84
+ #: figure here is the **most recent** one, because it describes the build a
85
+ #: reader is about to run — and the previous figure is kept in the comment
86
+ #: beside it rather than dropped, since a cost that moved by 5,8× between two
87
+ #: releases (spring-audit, 408 s in #13 against 2 351 s in #16) is itself the
88
+ #: measurement C3-76 exists for.
82
89
  field_seconds: "Optional[float]" = None
83
90
  field_blocked: bool = False
84
91
 
@@ -93,19 +100,26 @@ REFERENCE_REPOSITORY = (
93
100
  REFERENCE_JAVA_FILES = 2000
94
101
 
95
102
  #: The other end of the measured range, and the reason this module publishes a
96
- #: *class* rather than a projected duration (C4-19). Field evaluation #13, on
103
+ #: *class* rather than a projected duration (C4-19). Field evaluation #16, on
97
104
  #: Windows 11 / PowerShell 5.1 / pipx / Python 3.10: `spring-audit` on a
98
- #: 3 342-file repository ran **408 s** — 46× the 8.8 s the same command takes on
99
- #: the 2 000-file reference. Cost does not scale with file count in any way this
100
- #: product has measured, so a projected "~15 s on your repository" would be a
101
- #: confident falsehood of exactly the kind the ledger exists to prevent. What can
102
- #: be said honestly is the class, the two measured anchors, and how to run it.
105
+ #: 3 342-file repository ran **2 351 s** — 267× the 8.8 s the same command takes
106
+ #: on the 2 000-file reference, for a repository 1,7× the size. Cost does not
107
+ #: scale with file count in any way this product has measured, so a projected
108
+ #: "~15 s on your repository" would be a confident falsehood of exactly the kind
109
+ #: the ledger exists to prevent. What can be said honestly is the class, the two
110
+ #: measured anchors, and how to run it.
111
+ #:
112
+ #: C3-72 replaced evaluation #13's 408 s with #16's 2 351 s here: the same command
113
+ #: on the same repository at the same commit, measured three releases later
114
+ #: (2 557 s in 4.10.6, 2 398 s in 4.10.7, 2 351 s in 4.11.0). The older figure was
115
+ #: the one a budget message quoted as "~600s" for three releases while the
116
+ #: measurement stood four times higher.
103
117
  FIELD_ANCHOR = (
104
- "field evaluation #13: spring-audit on 3 342 Java files took 408 s "
105
- "(Windows, pipx, cold) against 8.8 s on the 2 000-file reference"
118
+ "field evaluation #16: spring-audit on 3 342 Java files took 2 351 s "
119
+ "(Windows, pipx, warm cache) against 8.8 s on the 2 000-file reference"
106
120
  )
107
121
  FIELD_ANCHOR_JAVA_FILES = 3342
108
- FIELD_ANCHOR_SECONDS = 408.0
122
+ FIELD_ANCHOR_SECONDS = 2351.4
109
123
 
110
124
 
111
125
  #: Every layer keys on `cache.worktree_signature` — the exact tree state (C1-9).
@@ -176,14 +190,15 @@ COMMANDS: tuple[CommandCache, ...] = (
176
190
  "Resolves the conditional bean graph on every run, over the shared CIR a warm "
177
191
  "builds — the parse it used to repeat for itself. `--diff` compares two profile "
178
192
  "sets over that one IR, so the second side costs the resolution only.",
179
- "10.1 s → 1.6 s", analysis_class="repo-wide", cold_seconds=10.1, warm_seconds=1.6, field_blocked=True),
193
+ "10.1 s → 1.6 s", analysis_class="repo-wide", cold_seconds=10.1, warm_seconds=1.6, field_seconds=38.0),
180
194
  CommandCache("risk", ("cir", "parse"), "shared", False,
181
195
  "Composes what the audit, impact-chain and the posture already answer, so it "
182
196
  "pays each of their costs once over the shared CIR a warm builds — one parse "
183
197
  "for the whole composition, and the reachability query is cached per symbol "
184
198
  "within the run.",
185
199
  "not measured on the battery yet — the composition is bounded by the "
186
- "`spring-audit` + `impact-chain` costs listed here, not by new analysis", analysis_class="deep"),
200
+ "`spring-audit` + `impact-chain` costs listed here, not by new analysis",
201
+ analysis_class="deep", field_seconds=3232.2),
187
202
  CommandCache("enrich", ("cir", "parse"), "shared", False,
188
203
  "Runs the same composition as `risk` over the repository, then joins a SARIF "
189
204
  "log to it. Reading the log is negligible; everything a warm helps with is the "
@@ -213,14 +228,14 @@ COMMANDS: tuple[CommandCache, ...] = (
213
228
  "Recomputes the endpoint surface on every run, over a parse a warm has already "
214
229
  "paid for. Until 3.7.0 the extractor parsed every file itself instead of reading "
215
230
  "the shared parse cache, and a warm measurably bought it nothing (C3-6).",
216
- "3.3 s → 1.4 s (re-measured on 3.7.0; was 2.8 s → 2.9 s)", analysis_class="repo-wide", cold_seconds=3.3, warm_seconds=1.4, field_seconds=4.2),
231
+ "3.3 s → 1.4 s (re-measured on 3.7.0; was 2.8 s → 2.9 s)", analysis_class="repo-wide", cold_seconds=3.3, warm_seconds=1.4, field_seconds=5.0),
217
232
  CommandCache("spring-audit", ("ris", "parse"), "shared", False,
218
233
  "Recomputes every run, but over a parse a warm has already paid for.",
219
- "8.8 s → 3.7 s", analysis_class="repo-wide", cold_seconds=8.8, warm_seconds=3.7, field_seconds=408.0),
234
+ "8.8 s → 3.7 s", analysis_class="repo-wide", cold_seconds=8.8, warm_seconds=3.7, field_seconds=2351.4),
220
235
  CommandCache("migrate-check", ("cir",), "none", False,
221
236
  "Computes its own inventory and shares nothing a warm builds. Only `--blast-radius` "
222
237
  "reuses the shared CIR.",
223
- "4.8 s → 4.8 s", analysis_class="repo-wide", cold_seconds=4.8, warm_seconds=4.8, field_blocked=True),
238
+ "4.8 s → 4.8 s", analysis_class="repo-wide", cold_seconds=4.8, warm_seconds=4.8, field_seconds=26.7),
224
239
  CommandCache("impact-chain", ("cir", "parse"), "shared", False,
225
240
  "The CIR is the expensive half — this is where a warm pays most.",
226
241
  "9.9 s → 1.7 s", analysis_class="core", cold_seconds=9.9, warm_seconds=1.7),
@@ -301,9 +316,59 @@ def layer(layer_id: str) -> Layer:
301
316
  raise KeyError(layer_id)
302
317
 
303
318
 
304
- def as_dict() -> dict:
319
+ @dataclass(frozen=True)
320
+ class Conditioning:
321
+ """The reader's repository, so the figures below are read against it (C3-60).
322
+
323
+ `verify-edit` is published here as *14,6 s → 9,6 s → 5,6 s* and described as
324
+ *"built for the edit loop"*; on the field repository the same command had not
325
+ answered at 495 s. Both are true, and printing only the first is what made
326
+ the second read as a defect in the reader's setup rather than as the scale
327
+ the figure was never measured at. The numbers do not change — what changes is
328
+ that they arrive with the repository they belong to, and with what is known
329
+ about the one in hand.
330
+
331
+ Built by the caller (`cli`) from `execution_plan`, which owns the scope
332
+ measurement; this module stays free of that dependency because
333
+ `execution_plan` already reads *it*.
334
+ """
335
+
336
+ scope: str # "3 342 Java files, cache cold"
337
+ transfers: "Optional[bool]" # do the reference figures apply here?
338
+ per_command: "dict[str, str]" # command → how it can be run here
339
+
340
+ def statement(self) -> str:
341
+ """One sentence about whether the figures below transfer to this scope."""
342
+ if self.transfers is None:
343
+ return (
344
+ f"This repository ({self.scope}) was not measured, so nothing below "
345
+ f"is claimed about it: the figures are the reference repository's."
346
+ )
347
+ if self.transfers:
348
+ return (
349
+ f"This repository ({self.scope}) is within the size the figures "
350
+ f"below were measured at ({REFERENCE_JAVA_FILES} Java files)."
351
+ )
352
+ return (
353
+ f"This repository ({self.scope}) is larger than the repository the "
354
+ f"figures below were measured on ({REFERENCE_JAVA_FILES} Java files), "
355
+ f"and cost has not been observed to scale with file count: "
356
+ f"{FIELD_ANCHOR}. Read each figure as the reference's, not as yours; "
357
+ f"the `here:` lines say how each command can be run on this one."
358
+ )
359
+
360
+
361
+ def as_dict(here: "Optional[Conditioning]" = None) -> dict:
305
362
  """The whole model as data — what `ask cache model --json` emits."""
363
+ conditioned: dict = {}
364
+ if here is not None:
365
+ conditioned = {
366
+ "scope_measured": here.scope,
367
+ "reference_figures_transfer": here.transfers,
368
+ "conditioning": here.statement(),
369
+ }
306
370
  return {
371
+ **conditioned,
307
372
  "invalidation": (
308
373
  "Every layer keys on the exact tree state: any change to the analysed files "
309
374
  "invalidates it, committed or not. A clean tree keys on the commit, so a repeat "
@@ -334,6 +399,13 @@ def as_dict() -> dict:
334
399
  "note": cmd.note,
335
400
  "measured": cmd.measured or None,
336
401
  "analysis_class": cmd.analysis_class,
402
+ # How this command can be run on the repository in hand, when one
403
+ # was measured. Absent rather than guessed otherwise (C3-60).
404
+ **(
405
+ {"here": here.per_command[cmd.command]}
406
+ if here is not None and cmd.command in here.per_command
407
+ else {}
408
+ ),
337
409
  "reference_seconds": (
338
410
  None
339
411
  if cmd.cold_seconds is None
@@ -366,7 +438,7 @@ def render_markdown() -> str:
366
438
  return "\n".join(out)
367
439
 
368
440
 
369
- def render_summary() -> str:
441
+ def render_summary(here: "Optional[Conditioning]" = None) -> str:
370
442
  """The model as `ask cache model` prints it by default (C3-63).
371
443
 
372
444
  One line per layer and one line per command — every command named (nothing
@@ -376,6 +448,9 @@ def render_summary() -> str:
376
448
  own selling argument is token economy. That detail is still one flag away.
377
449
  """
378
450
  lines: list[str] = []
451
+ if here is not None:
452
+ lines.append(here.statement())
453
+ lines.append("")
379
454
  lines.append(
380
455
  "Invalidation: every layer keys on the exact tree state — any change to the "
381
456
  "analysed files invalidates it, committed or not."
@@ -396,6 +471,8 @@ def render_summary() -> str:
396
471
  f" {cmd.command.ljust(cmd_width)} {_WARM_LABEL[cmd.warm].ljust(warm_width)} "
397
472
  f"({repeat}) [{layers}]"
398
473
  )
474
+ if here is not None and cmd.command in here.per_command:
475
+ lines.append(f" {' ' * cmd_width} here: {here.per_command[cmd.command]}")
399
476
  lines.append("")
400
477
  lines.append(
401
478
  "Full detail (measured timings, per-command notes): ask cache model --full"
@@ -404,11 +481,14 @@ def render_summary() -> str:
404
481
  return "\n".join(lines)
405
482
 
406
483
 
407
- def render_text() -> str:
484
+ def render_text(here: "Optional[Conditioning]" = None) -> str:
408
485
  """The model in full: every layer and command with its measured timing and note.
409
486
 
410
487
  `ask cache model --full`. See `render_summary` for the default."""
411
488
  lines: list[str] = []
489
+ if here is not None:
490
+ lines.append(here.statement())
491
+ lines.append("")
412
492
  lines.append("Invalidation — one rule for every layer")
413
493
  lines.append(" Every layer keys on the exact tree state. Any change to the analysed files")
414
494
  lines.append(" invalidates it, committed or not, tracked or not. A clean tree keys on the")
@@ -432,4 +512,6 @@ def render_text() -> str:
432
512
  lines.append(f" {' ' * width} measured: {cmd.measured}")
433
513
  if cmd.note:
434
514
  lines.append(f" {' ' * width} {cmd.note}")
515
+ if here is not None and cmd.command in here.per_command:
516
+ lines.append(f" {' ' * width} here: {here.per_command[cmd.command]}")
435
517
  return "\n".join(lines)
@@ -19,7 +19,7 @@ import json
19
19
  import re
20
20
  from dataclasses import dataclass, field
21
21
  from pathlib import Path
22
- from typing import Any, Optional
22
+ from typing import Any, Callable, Optional
23
23
 
24
24
  from sourcecode.cir_graphs import ImplementationGraph, InjectionGraph
25
25
  from sourcecode.repository_ir import (
@@ -526,6 +526,7 @@ def build_canonical_ir(
526
526
  root: Path,
527
527
  *,
528
528
  since: Optional[str] = None,
529
+ progress: "Optional[Callable[[int, int, str], None]]" = None,
529
530
  ) -> CanonicalRepositoryIR:
530
531
  """Build CanonicalRepositoryIR from Java files.
531
532
 
@@ -537,8 +538,13 @@ def build_canonical_ir(
537
538
  file_paths: Relative paths to Java files (from find_java_files).
538
539
  root: Absolute repo root.
539
540
  since: Git ref for symbol diff (e.g. "HEAD~1", "main").
541
+ progress: Optional (done, total, stage) sink passed straight through to
542
+ the builder — the only place that knows how much of the work
543
+ is left (C3-56).
540
544
  """
541
- ir = build_repo_ir(file_paths, root, since=since, emit_body_facts=True)
545
+ ir = build_repo_ir(
546
+ file_paths, root, since=since, emit_body_facts=True, progress=progress
547
+ )
542
548
  return ir_dict_to_canonical(ir, file_paths=file_paths)
543
549
 
544
550