sourcecode 5.0.1__py3-none-any.whl → 5.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sourcecode/__init__.py CHANGED
@@ -4,4 +4,4 @@ ASK Engine is the product. ``ask`` is the canonical CLI command; ``sourcecode``
4
4
  the legacy compatibility alias and the Python/PyPI package name. See
5
5
  docs/PRODUCT_IDENTITY.md (normative)."""
6
6
 
7
- __version__ = "5.0.1"
7
+ __version__ = "5.1.0"
sourcecode/cache_model.py CHANGED
@@ -616,6 +616,103 @@ def field_currency() -> dict:
616
616
  }
617
617
 
618
618
 
619
+ def gate_anchor(command: str) -> "Optional[dict]":
620
+ """What the release gate measured for *command*, or ``None`` (C3-109).
621
+
622
+ The third figure in this module, and the only one this repository can refresh
623
+ by itself. `REFERENCE_*` is a 3.2.2 measurement on a 2 000-file tree;
624
+ `field_seconds` is the field's, at 3 342 files, on their machine and their
625
+ build. Both age, and neither can be re-measured from here — which is exactly
626
+ how B17 reopened one release after it was closed: the labelling worked and
627
+ the number still expired.
628
+
629
+ The gate measures a 2 985-file repository under a declared host class every
630
+ release and commits its cells, so this figure carries the build that produced
631
+ it and can be brought current by running the gate. It never overrides a field
632
+ anchor — a measurement at field scale outranks one at gate scale — it is
633
+ published *beside* it, so an anchor from an older build is no longer the only
634
+ number a reader has.
635
+ """
636
+ try:
637
+ from sourcecode.gate_anchors import (
638
+ GATE_ANCHORS,
639
+ GATE_HOST_CLASS,
640
+ GATE_JAVA_FILES,
641
+ GATE_MEASURED_VERSION,
642
+ GATE_REPOSITORY,
643
+ )
644
+ except Exception:
645
+ return None
646
+ cell = GATE_ANCHORS.get(command)
647
+ if not cell:
648
+ return None
649
+ return {
650
+ "seconds": cell["seconds"],
651
+ "cache_state": cell["mode"],
652
+ "runs": cell["runs"],
653
+ "measured_on_version": GATE_MEASURED_VERSION,
654
+ "repository": GATE_REPOSITORY,
655
+ "java_files": GATE_JAVA_FILES,
656
+ "host_class": GATE_HOST_CLASS,
657
+ }
658
+
659
+
660
+ def gate_currency() -> dict:
661
+ """Whether the in-house gate figures describe the build in hand (C3-109).
662
+
663
+ Same shape and same purpose as `field_currency`, for the anchor set that can
664
+ actually be refreshed. When this says `current: false`, the remedy is a
665
+ command rather than a hope: re-run the gate on the tag and commit the cells.
666
+ """
667
+ from sourcecode import __version__
668
+
669
+ current = __version__
670
+ try:
671
+ from sourcecode.gate_anchors import (
672
+ GATE_ANCHORS,
673
+ GATE_CAPTURED_AT,
674
+ GATE_HOST_CLASS,
675
+ GATE_JAVA_FILES,
676
+ GATE_MEASURED_VERSION,
677
+ GATE_REPOSITORY,
678
+ )
679
+ except Exception:
680
+ return {
681
+ "available": False,
682
+ "running_version": current,
683
+ "statement": (
684
+ "This build carries no gate measurement of its own. The gate "
685
+ "publishes one per release (docs/perf/REGRESSION-GATE.md); until "
686
+ "it runs, the field anchors are the only measured figures here."
687
+ ),
688
+ }
689
+ return {
690
+ "available": True,
691
+ "measured_on_version": GATE_MEASURED_VERSION,
692
+ "running_version": current,
693
+ "current": GATE_MEASURED_VERSION == current,
694
+ "captured_at": GATE_CAPTURED_AT,
695
+ "repository": GATE_REPOSITORY,
696
+ "java_files": GATE_JAVA_FILES,
697
+ "host_class": GATE_HOST_CLASS,
698
+ "commands": sorted(GATE_ANCHORS),
699
+ "statement": (
700
+ f"The release gate measured {len(GATE_ANCHORS)} commands on "
701
+ f"{GATE_REPOSITORY} ({GATE_JAVA_FILES} Java files, {GATE_HOST_CLASS}) "
702
+ + (
703
+ f"on {current}, the build you are running."
704
+ if GATE_MEASURED_VERSION == current
705
+ else (
706
+ f"on {GATE_MEASURED_VERSION}; you are running {current}, which "
707
+ f"the gate has not measured. Re-run it on this build "
708
+ f"(docs/perf/REGRESSION-GATE.md) — unlike the field anchors, "
709
+ f"this is a figure this project can refresh."
710
+ )
711
+ )
712
+ ),
713
+ }
714
+
715
+
619
716
  def reference_currency() -> dict:
620
717
  """How old the reference figures are, in releases the reader can name (F-AP).
621
718
 
@@ -680,6 +777,11 @@ def as_dict(here: "Optional[Conditioning]" = None) -> dict:
680
777
  # measured 2 351 s and 8,4 s on two of our own builds (C4-19, C3-97).
681
778
  "field_anchor": FIELD_ANCHOR,
682
779
  "field_currency": field_currency(),
780
+ # C3-109: and the figure this project can refresh by itself. A field
781
+ # anchor belongs to the build the field ran; this one belongs to the
782
+ # build the gate ran, which is a release away at worst instead of
783
+ # unreachable. Published beside, never instead.
784
+ "gate_currency": gate_currency(),
683
785
  "commands": [
684
786
  {
685
787
  "command": cmd.command,
@@ -716,6 +818,12 @@ def as_dict(here: "Optional[Conditioning]" = None) -> dict:
716
818
  "note": field_anchor_note(cmd),
717
819
  }
718
820
  ),
821
+ # C3-109: the in-house measurement of this command, on the build
822
+ # the gate last ran. Absent for commands the gate does not cover —
823
+ # its population is declared in `sourcecode.perf`, with the
824
+ # exclusions and their reasons, and an axis nobody measured is
825
+ # never a number here.
826
+ "gate_measurement": gate_anchor(cmd.command),
719
827
  }
720
828
  for cmd in COMMANDS
721
829
  ],
@@ -827,6 +935,7 @@ def render_text(here: "Optional[Conditioning]" = None) -> str:
827
935
  lines.append(f" Timings: {REFERENCE_REPOSITORY}, each command measured in isolation.")
828
936
  lines.append(f" {reference_currency()['statement']}")
829
937
  lines.append(f" {field_currency()['statement']}")
938
+ lines.append(f" {gate_currency()['statement']}")
830
939
  width = max(len(c.command) for c in COMMANDS)
831
940
  for cmd in COMMANDS:
832
941
  repeat = "repeat cached" if cmd.repeat else "recomputes"
@@ -848,6 +957,17 @@ def render_text(here: "Optional[Conditioning]" = None) -> str:
848
957
  f" {' ' * width} field: {shown} at "
849
958
  f"{_thousands(FIELD_ANCHOR_JAVA_FILES)} Java files{suffix}"
850
959
  )
960
+ _gate = gate_anchor(cmd.command)
961
+ if _gate:
962
+ # C3-109: the in-house figure, beside the field's rather than instead
963
+ # of it — one is at field scale on another build, one is on a build
964
+ # this project can bring current.
965
+ lines.append(
966
+ f" {' ' * width} gate: {_gate['seconds']:g} s at "
967
+ f"{_thousands(_gate['java_files'])} Java files "
968
+ f"({_gate['cache_state']} cache, on {_gate['measured_on_version']}, "
969
+ f"{_gate['host_class']})"
970
+ )
851
971
  if cmd.note:
852
972
  lines.append(f" {' ' * width} {cmd.note}")
853
973
  if here is not None and cmd.command in here.per_command:
sourcecode/cli.py CHANGED
@@ -934,6 +934,67 @@ def _emit_error_json(error: str, message: str, **context: object) -> None:
934
934
  sys.stderr.flush()
935
935
 
936
936
 
937
+ #: Exit code for a confirmation that could not be read. Argument-validation
938
+ #: convention (the same 2 an invalid `--format` returns): nothing was done, and
939
+ #: the input is what has to change.
940
+ _UNANSWERABLE_PROMPT_EXIT = 2
941
+
942
+
943
+ def _stdin_is_interactive() -> bool:
944
+ """True when a human could answer a prompt on stdin.
945
+
946
+ The counterpart of `_stderr_is_interactive`, and one seam rather than three
947
+ call sites, because the two states this gate distinguishes are not the two
948
+ states the world has: stdin can also *claim* to be a terminal and answer
949
+ nothing, which is C3-106 and is handled by `_confirm` rather than here.
950
+ """
951
+ try:
952
+ return bool(sys.stdin.isatty())
953
+ except Exception:
954
+ return False
955
+
956
+
957
+ def _confirm(question: str, *, command: str, default: bool = False, err: bool = False) -> bool:
958
+ """Ask a yes/no question, and diagnose a prompt that cannot be answered.
959
+
960
+ C3-106. v1.45.0 guarded the case it knew about — `sys.stdin.isatty()` false,
961
+ so never block on a pipe — and the field found the case it does not cover: a
962
+ PowerShell 5.1 session driven by a harness where `isatty()` answers **true**
963
+ and the read hits end-of-input immediately. `click.confirm` turns that into
964
+ `Abort`, nothing catches it, and the operator gets a Python traceback under a
965
+ half-printed prompt — from the first command anyone runs when following our
966
+ own advice to clear the cache between versions.
967
+
968
+ An unanswerable prompt is malformed input, and this product answers malformed
969
+ input with a diagnosed envelope everywhere else. So does this: what was asked,
970
+ why no answer arrived, and the flag that makes the question unnecessary.
971
+
972
+ `click` collapses end-of-input and interrupt into the same `Abort`, so the
973
+ message names the outcome (*no answer*) rather than a cause it cannot
974
+ distinguish. A plain "no" is not this: it returns `False` and the caller
975
+ decides.
976
+ """
977
+ import click as _click
978
+
979
+ try:
980
+ return bool(_click.confirm(question, default=default, err=err))
981
+ except (EOFError, OSError, _click.Abort):
982
+ _emit_error_json(
983
+ INVALID_INPUT_CODE,
984
+ f"{command} needs confirmation and no answer could be read from stdin",
985
+ hint=f"Pass --yes to skip the prompt: `{command} --yes`.",
986
+ expected="--yes when running non-interactively",
987
+ reason="unanswerable_prompt",
988
+ detail=(
989
+ "the prompt was written and the read ended without an answer "
990
+ "(end of input, or an interrupt). A terminal that reports itself "
991
+ "interactive and returns EOF — a harness-driven shell, a "
992
+ "redirected stdin — reaches here."
993
+ ),
994
+ )
995
+ raise typer.Exit(code=_UNANSWERABLE_PROMPT_EXIT)
996
+
997
+
937
998
  def _notice(message: str) -> None:
938
999
  """Say something to the person, and nothing to the pipeline.
939
1000
 
@@ -2478,14 +2539,24 @@ _FREE_TIER_NODE_CAP: int = 50 # graph/semantic node cap — applies only to lar
2478
2539
  _JAVA_MIN_SCAN_DEPTH: int = 12 # Maven src/main/java/<pkg>/<module>/File depth floor
2479
2540
  _LARGE_REPO_ADVISORY_FILES: int = 3000 # cold-scan advisory threshold (files); honest expectation-setting, not a speed claim
2480
2541
 
2481
- #: What one Java file costs in serialized `repo-ir` output, measured across four
2542
+ #: What one Java file costs in serialized `repo-ir` output, measured across five
2482
2543
  #: repositories (C3-18): spring-petclinic 30 files/318 KB (10.8 KB/file),
2483
2544
  #: jobrunr 581/9.5 MB (16.8), openmrs-core 866/21 MB (25.2), eureka 299/8.0 MB
2484
- #: (27.3). Kept as a band, never collapsed to a midpoint: the 2.6x spread is
2485
- #: real — it tracks how densely the code cross-references — and a point estimate
2486
- #: would assert a precision the sample does not support (the C1-12 rule).
2545
+ #: (27.3), and the field repository 3 342/101 254 852 B (**30.3**, eval #23).
2546
+ #: Kept as a band, never collapsed to a midpoint: the 2.8x spread is real — it
2547
+ #: tracks how densely the code cross-references — and a point estimate would
2548
+ #: assert a precision the sample does not support (the C1-12 rule).
2549
+ #:
2550
+ #: C3-105: the fifth sample is the one that opened the row. The band was 10.8-27.9
2551
+ #: and the largest repository anybody had run this on landed **above** it, so the
2552
+ #: one number offered for planning excluded the measurement a reader would take.
2553
+ #: Same class as C3-97 — a published figure a later measurement falsifies — and
2554
+ #: the remedy is the same: admit the sample, and say how many there are.
2487
2555
  _IR_BYTES_PER_FILE_LOW: int = 10_800
2488
- _IR_BYTES_PER_FILE_HIGH: int = 27_900
2556
+ _IR_BYTES_PER_FILE_HIGH: int = 30_300
2557
+ #: How many repositories the band is drawn from. Published with the band, because
2558
+ #: a range without its sample size reads as a bound rather than an observation.
2559
+ _IR_BYTES_PER_FILE_SAMPLES: int = 5
2489
2560
  _IR_SIZE_WARN_MB: int = 10 # advise above this; below it the write is not worth a sentence
2490
2561
 
2491
2562
  #: What `--no-cache` means on a subcommand (C3-5). The old text — "this command
@@ -2516,12 +2587,19 @@ def _ir_size_advisory_message(file_count: int) -> "Optional[str]":
2516
2587
  file already on disk — so it could tell a user what they had just spent,
2517
2588
  never what they were about to.
2518
2589
 
2519
- The figure is a band, not a point. Sampled across four repositories the IR
2520
- costs 10.8-27.9 KB per Java file, and that 2.6x spread is a real property of
2590
+ The figure is a band, not a point. Sampled across five repositories the IR
2591
+ costs 10.8-30.3 KB per Java file, and that 2.8x spread is a real property of
2521
2592
  the input (how densely the code cross-references), not sampling noise.
2522
2593
  Publishing its midpoint as one number would assert a precision the sample
2523
2594
  does not support — the same defect C1-12 records for migration effort — so
2524
- the band is published with the word "estimate" on it.
2595
+ the band is published with the word "estimate" and its sample size on it.
2596
+
2597
+ C3-105: the band's own direction is declared, because the sample that sets
2598
+ its high end (the field repository, 101 254 852 B) counted **every** Java
2599
+ file in the repository, tests included, while this estimate multiplies the
2600
+ files this run will actually scan. Per scanned file the true rate is
2601
+ therefore higher, so the top of the band under-states — stated rather than
2602
+ left for a reader to discover the way C3-105 was discovered.
2525
2603
 
2526
2604
  Pure and TTY-agnostic: the caller owns the terminal gate, so this never
2527
2605
  reaches stdout or a pipeline (C3-26).
@@ -2532,7 +2610,10 @@ def _ir_size_advisory_message(file_count: int) -> "Optional[str]":
2532
2610
  return None
2533
2611
  return (
2534
2612
  f"[repo-ir] {file_count} files — estimated output "
2535
- f"{est_low_mb:.0f}-{est_high_mb:.0f}MB (estimate, not a measurement). "
2613
+ f"{est_low_mb:.0f}-{est_high_mb:.0f}MB (estimate from "
2614
+ f"{_IR_BYTES_PER_FILE_SAMPLES} repositories, not a measurement; the top "
2615
+ "of the band under-states, since its sample counted tests among its "
2616
+ "files and this run does not). "
2536
2617
  "To spend less: --summary-only, --max-nodes N --max-edges N, or --gzip."
2537
2618
  )
2538
2619
 
@@ -3360,6 +3441,43 @@ def main(
3360
3441
  _l1_needs_env_inject = "em" in _subset
3361
3442
  break
3362
3443
 
3444
+ # C3-103: and the same reasoning in the other direction, which was never
3445
+ # tried. The candidate set above turns requested overlays OFF to find a
3446
+ # poorer base and re-attaches them; a base that is *richer* than the
3447
+ # request is reusable too, because an overlay only ever attaches a block —
3448
+ # the same additivity the inject path already depends on — so projecting
3449
+ # the richer core down is a drop, not a rescan. Measured in the field:
3450
+ # `--compact --git-context --env-map` 72,3 s cold, then `--agent` 71,7 s
3451
+ # with the RIS already built, because `--agent` asks for `gc=False` and
3452
+ # the only core on disk had `gc=True`. 144 s for two views of one state.
3453
+ _l1_needs_env_drop = False
3454
+ _l1_needs_git_drop = False
3455
+ if _l1_result is None and _core_key and not (env_map and git_context):
3456
+ _sha_prefix = _tree_sig
3457
+ _addable = []
3458
+ if not git_context:
3459
+ _addable.append("gc")
3460
+ if not env_map:
3461
+ _addable.append("em")
3462
+ import itertools as _it2
3463
+ _up_subsets = [
3464
+ s for n in range(1, len(_addable) + 1)
3465
+ for s in _it2.combinations(_addable, n)
3466
+ ]
3467
+ for _subset in _up_subsets:
3468
+ _bf = _core_flags_str
3469
+ if "gc" in _subset:
3470
+ _bf = _bf.replace(",gc=False,", ",gc=True,")
3471
+ if "em" in _subset:
3472
+ _bf = _bf.replace(",em=False,", ",em=True,")
3473
+ _bk = f"{_sha_prefix}-{_hashlib.sha256(_bf.encode()).hexdigest()[:8]}"
3474
+ _br = _cache_mod.read_core(target, _bk)
3475
+ if _br is not None:
3476
+ _l1_result = _br
3477
+ _l1_needs_git_drop = "gc" in _subset
3478
+ _l1_needs_env_drop = "em" in _subset
3479
+ break
3480
+
3363
3481
  if _l1_result is not None:
3364
3482
  _core_dict_l1, _core_hash = _l1_result
3365
3483
  _view_key = f"{_core_hash}-{_view_h}"
@@ -3373,6 +3491,14 @@ def main(
3373
3491
  f"{_view_key}-ov:gc{int(_l1_needs_git_inject)}"
3374
3492
  f"em{int(_l1_needs_env_inject)}"
3375
3493
  )
3494
+ # C3-103: a projected-down view shares the richer base's core hash
3495
+ # with the base's own view, and it is a different answer — one block
3496
+ # shorter. Its own suffix, for the reason the inject path has one.
3497
+ if _l1_needs_git_drop or _l1_needs_env_drop:
3498
+ _view_key = (
3499
+ f"{_view_key}-pd:gc{int(_l1_needs_git_drop)}"
3500
+ f"em{int(_l1_needs_env_drop)}"
3501
+ )
3376
3502
 
3377
3503
  # Step 2: try L2 (exact view match).
3378
3504
  # Skip L2 for --changed-only: the stored view is a previous
@@ -3443,6 +3569,17 @@ def main(
3443
3569
  _rebuilt["git_context"] = _gc_block
3444
3570
  except Exception:
3445
3571
  pass # git inject failed — continue without git data
3572
+ # C3-103: drop the overlays the richer base carried and this
3573
+ # request did not ask for. The view is what the flags decide,
3574
+ # so a block the caller did not request must not arrive just
3575
+ # because a previous run paid for it.
3576
+ if _rebuilt is not None and (_l1_needs_git_drop or _l1_needs_env_drop):
3577
+ _rebuilt = dict(_rebuilt)
3578
+ if _l1_needs_git_drop:
3579
+ _rebuilt.pop("git_context", None)
3580
+ if _l1_needs_env_drop:
3581
+ _rebuilt.pop("env_summary", None)
3582
+ _rebuilt.pop("env_map", None)
3446
3583
  if _rebuilt is not None:
3447
3584
  # Apply redaction
3448
3585
  if not no_redact:
@@ -6948,19 +7085,34 @@ def _two_states(
6948
7085
  root = repo_root(target)
6949
7086
  base_sha = resolve_ref(root, base_ref)
6950
7087
  head_sha = resolve_ref(root, head_ref)
7088
+ _identical = bool(base_sha) and base_sha == head_sha
7089
+ block = {
7090
+ "basis": "git_refs",
7091
+ "repository": str(root),
7092
+ "base": {"ref": base_ref, "commit": base_sha},
7093
+ "head": {"ref": head_ref, "commit": head_sha},
7094
+ "note": (
7095
+ "Both states were read with `git archive`: nothing was "
7096
+ "written to the repository, no worktree was created and "
7097
+ "the working tree and index were not read."
7098
+ ),
7099
+ }
7100
+ _record_state_identity(
7101
+ block,
7102
+ _identical,
7103
+ f"--base-ref and --head-ref resolve to the same commit ({base_sha})",
7104
+ )
7105
+ if _identical:
7106
+ # C3-107, the first half: two refs that name one commit are one
7107
+ # tree, so it is read out once. The second `git archive` copied
7108
+ # the same bytes to a second directory for a build that is going
7109
+ # to be reused rather than repeated.
7110
+ with materialize_ref(target, base_ref) as tree:
7111
+ yield tree, tree, block
7112
+ return
6951
7113
  with materialize_ref(target, base_ref) as base_tree, \
6952
7114
  materialize_ref(target, head_ref) as head_tree:
6953
- yield base_tree, head_tree, {
6954
- "basis": "git_refs",
6955
- "repository": str(root),
6956
- "base": {"ref": base_ref, "commit": base_sha},
6957
- "head": {"ref": head_ref, "commit": head_sha},
6958
- "note": (
6959
- "Both states were read with `git archive`: nothing was "
6960
- "written to the repository, no worktree was created and "
6961
- "the working tree and index were not read."
6962
- ),
6963
- }
7115
+ yield base_tree, head_tree, block
6964
7116
  except RefError as exc:
6965
7117
  _emit_error_json(exc.code, exc.message, ref=exc.ref, hint=usage)
6966
7118
  raise typer.Exit(code=1)
@@ -6980,11 +7132,53 @@ def _two_states(
6980
7132
  usage=usage,
6981
7133
  extra_hint=f"Refused as the {label} checkout.",
6982
7134
  )
6983
- yield base.resolve(), head.resolve(), {
7135
+ block = {
6984
7136
  "basis": "checkouts",
6985
7137
  "base": {"path": str(base.resolve())},
6986
7138
  "head": {"path": str(head.resolve())},
6987
7139
  }
7140
+ _record_state_identity(block, *_checkouts_are_one_state(base, head))
7141
+ yield base.resolve(), head.resolve(), block
7142
+
7143
+
7144
+ def _checkouts_are_one_state(base: Path, head: Path) -> "tuple[bool, Optional[str]]":
7145
+ """Are these two checkouts provably the same tree state?
7146
+
7147
+ `cache.worktree_signature` is the authority this product already keys every
7148
+ cached answer on — *"any change to the analysed files invalidates, committed
7149
+ or not"* — so the question needs no second definition of "the same tree".
7150
+
7151
+ Identity is claimed only when a signature was computed on both sides: a tree
7152
+ that cannot be described answers `""`, and two unknowns are not a match. The
7153
+ error direction is deliberate — a missed identity costs a second build, a
7154
+ false one would answer for a tree nobody read.
7155
+ """
7156
+ from sourcecode import cache as _cache
7157
+
7158
+ try:
7159
+ base_sig = _cache.worktree_signature(Path(base).resolve())
7160
+ head_sig = _cache.worktree_signature(Path(head).resolve())
7161
+ except Exception:
7162
+ return False, None
7163
+ if base_sig and base_sig == head_sig:
7164
+ return True, f"both checkouts carry the same tree signature ({base_sig})"
7165
+ return False, None
7166
+
7167
+
7168
+ def _record_state_identity(block: dict, identical: bool, basis: "Optional[str]") -> None:
7169
+ """Publish the identity verdict beside the two states it is about.
7170
+
7171
+ C3-107: a reader who sees one analysis where the command names two states is
7172
+ owed the reason, and a consumer gating on `delta` is owed the fact that the
7173
+ comparison was degenerate. Always present, so its absence is never the answer.
7174
+ """
7175
+ block["identical_states"] = bool(identical)
7176
+ if identical:
7177
+ block["identical_states_basis"] = basis
7178
+ block["identical_states_effect"] = (
7179
+ "One tree state, so it was read and analysed once and compared with "
7180
+ "itself. The result is a diff of a real analysis, not an assumed-empty one."
7181
+ )
6988
7182
 
6989
7183
 
6990
7184
  @app.command("delta")
@@ -7045,10 +7239,16 @@ def delta_cmd(
7045
7239
  with _two_states(
7046
7240
  base, head, base_ref, head_ref, repo, usage="ask delta <base> <head>",
7047
7241
  ) as (_base_root, _head_root, _comparison):
7242
+ # C3-107: `_two_states` has already decided this, from the commit on the
7243
+ # ref path and from the tree signature on the checkout path.
7244
+ _same = bool(_comparison.get("identical_states"))
7048
7245
  _prog.start("building base IR")
7049
7246
  _base_cir = ContextGraph.build_from_root(_base_root).cir
7050
- _prog.update("building head IR")
7051
- _head_cir = ContextGraph.build_from_root(_head_root).cir
7247
+ if _same:
7248
+ _head_cir = _base_cir
7249
+ else:
7250
+ _prog.update("building head IR")
7251
+ _head_cir = ContextGraph.build_from_root(_head_root).cir
7052
7252
  _prog.update("diffing snapshots")
7053
7253
  data = architectural_delta(_base_cir, _head_cir)
7054
7254
  data["comparison"] = _comparison
@@ -7136,17 +7336,24 @@ def contract_diff_cmd(
7136
7336
  with _two_states(
7137
7337
  base, head, base_ref, head_ref, repo, usage="ask contract-diff <base> <head>",
7138
7338
  ) as (_base_root, _head_root, _comparison):
7339
+ # C3-107: `_two_states` has already decided this, from the commit on the
7340
+ # ref path and from the tree signature on the checkout path.
7341
+ _same = bool(_comparison.get("identical_states"))
7139
7342
  _prog.start("building base IR")
7140
7343
  _base_cir = ContextGraph.build_from_root(_base_root).cir
7141
- _prog.update("building head IR")
7142
- _head_cir = ContextGraph.build_from_root(_head_root).cir
7344
+ if _same:
7345
+ _head_cir = _base_cir
7346
+ else:
7347
+ _prog.update("building head IR")
7348
+ _head_cir = ContextGraph.build_from_root(_head_root).cir
7143
7349
  # D2-b: proven request-constraint loosenings fill the non_breaking bucket. The
7144
7350
  # validation surface needs the repo roots (it discovers specs/DTOs), not just the
7145
7351
  # CIR — build it per side and diff the constraint contracts.
7146
7352
  _prog.update("projecting validation constraints")
7353
+ _base_surface = build_validation_surface(_base_root)
7147
7354
  _cfindings = _constraint_findings(
7148
- build_validation_surface(_base_root),
7149
- build_validation_surface(_head_root),
7355
+ _base_surface,
7356
+ _base_surface if _same else build_validation_surface(_head_root),
7150
7357
  )
7151
7358
  _prog.update("diffing contracts")
7152
7359
  data = _contract_diff(_base_cir, _head_cir, constraint_findings=_cfindings)
@@ -12907,8 +13114,7 @@ def mcp_init(
12907
13114
  typer.echo("")
12908
13115
 
12909
13116
  if not yes:
12910
- confirmed = typer.confirm("Proceed?", default=False)
12911
- if not confirmed:
13117
+ if not _confirm("Proceed?", command="ask mcp init"):
12912
13118
  typer.echo("Aborted.")
12913
13119
  raise typer.Exit(code=0)
12914
13120
  typer.echo("")
@@ -13174,8 +13380,7 @@ def mcp_remove(
13174
13380
  typer.echo("")
13175
13381
 
13176
13382
  if not yes:
13177
- confirmed = typer.confirm("Proceed?", default=False)
13178
- if not confirmed:
13383
+ if not _confirm("Proceed?", command="ask mcp remove"):
13179
13384
  typer.echo("Aborted.")
13180
13385
  raise typer.Exit(code=0)
13181
13386
  typer.echo("")
@@ -13384,11 +13589,22 @@ def cache_clear_cmd(
13384
13589
  yes: bool = typer.Option(False, "--yes", "-y", help="Skip confirmation prompt."),
13385
13590
  include_ris: bool = typer.Option(False, "--include-ris", hidden=True, help="Alias for --all. Preserved for backward compatibility."),
13386
13591
  all_: bool = typer.Option(False, "--all", help="Also delete the RIS snapshot (ris.json.gz). By default, RIS is preserved across clears."),
13592
+ global_: bool = typer.Option(
13593
+ False, "--global",
13594
+ help="Also empty the shared parse store (every repository's per-file parses).",
13595
+ ),
13387
13596
  ) -> None:
13388
13597
  """Delete cached snapshots for a repository.
13389
13598
 
13599
+ \b
13390
13600
  By default, RIS (ris.json.gz) is preserved — it is the persistent structural
13391
13601
  index used for cold-start bootstrapping. Use --all to also clear it.
13602
+
13603
+ \b
13604
+ --global additionally empties the shared parse store that `cache status`
13605
+ reports. That store is content-addressed **across repositories**, so it is
13606
+ not this repository's to clear selectively: the flag says so by being a
13607
+ separate one, and the line it prints names the whole store.
13392
13608
  """
13393
13609
  from sourcecode import cache as _cm
13394
13610
  # Clear exactly the path given, for the same reason `cache warm` warms it:
@@ -13403,9 +13619,14 @@ def cache_clear_cmd(
13403
13619
  # The interactive prompt is only meaningful on a TTY; elsewhere it would
13404
13620
  # hang indefinitely waiting for input. Treat non-interactive as confirmed
13405
13621
  # (clear is idempotent cleanup; RIS is preserved unless --all is passed).
13406
- if sys.stdin.isatty():
13407
- import click as _click
13408
- _click.confirm(f"Delete all cache files for {target}{_ris_note}?", abort=True, err=True)
13622
+ if _stdin_is_interactive():
13623
+ if not _confirm(
13624
+ f"Delete all cache files for {target}{_ris_note}?",
13625
+ command="ask cache clear",
13626
+ err=True,
13627
+ ):
13628
+ typer.echo("Aborted.", err=True)
13629
+ raise typer.Exit(code=1)
13409
13630
  else:
13410
13631
  typer.echo(
13411
13632
  f"Non-interactive: clearing cache for {target}{_ris_note}. "
@@ -13414,6 +13635,34 @@ def cache_clear_cmd(
13414
13635
  )
13415
13636
  removed = _cm.clear(target, clear_ris=_clear_ris)
13416
13637
  typer.echo(f"Removed {removed} file(s).")
13638
+ # C3-104: a layer `cache status` publishes and `cache clear` cannot empty is
13639
+ # a fact with no authority over it. The eviction is here; the notice is here
13640
+ # too, because the operator who purges between versions is following our own
13641
+ # advice and should not have to find `rm -rf` in the documentation.
13642
+ if global_:
13643
+ from sourcecode import parse_cache as _pc
13644
+
13645
+ store = _pc.clear_store()
13646
+ typer.echo(
13647
+ f"Removed {store['entries']} shared parse entrie(s), "
13648
+ f"{store['bytes'] / (1024 * 1024):.1f} MB "
13649
+ "(shared across every repository on this machine)."
13650
+ )
13651
+ else:
13652
+ try:
13653
+ from sourcecode import parse_cache as _pc
13654
+
13655
+ _store = _pc.store_stats()
13656
+ if int(_store.get("entries", 0) or 0):
13657
+ _notice(
13658
+ f"[ask] the shared parse store still holds "
13659
+ f"{_store['entries']} entries "
13660
+ f"({int(_store.get('bytes', 0) or 0) / (1024 * 1024):.1f} MB) — "
13661
+ "it is content-addressed across every repository. "
13662
+ "`ask cache clear --global` empties it."
13663
+ )
13664
+ except Exception:
13665
+ pass
13417
13666
 
13418
13667
 
13419
13668
  def _warm_shared_cir(target: Path):
@@ -13794,9 +14043,14 @@ def cache_context_clear_cmd(
13794
14043
  """Delete the AI Context Cache for a repository (structured context + stats)."""
13795
14044
  from sourcecode import context_cache as _ctxcache
13796
14045
  target = _resolve_repo_root(Path(path))
13797
- if not yes and sys.stdin.isatty():
13798
- import click as _click
13799
- _click.confirm(f"Delete AI context cache for {target}?", abort=True, err=True)
14046
+ if not yes and _stdin_is_interactive():
14047
+ if not _confirm(
14048
+ f"Delete AI context cache for {target}?",
14049
+ command="ask cache context-clear",
14050
+ err=True,
14051
+ ):
14052
+ typer.echo("Aborted.", err=True)
14053
+ raise typer.Exit(code=1)
13800
14054
  removed = _ctxcache.clear_for_repo(target)
13801
14055
  typer.echo(f"Removed {removed} context file(s).")
13802
14056
 
@@ -0,0 +1,42 @@
1
+ """gate_anchors.py — what the release gate measured, on the build that ran it.
2
+
3
+ **Generated by `scripts/sync_gate_anchors.py` from `docs/perf/baselines/gate-latest/`.
4
+ Do not edit by hand.** The baseline is the authority; this module is how a build
5
+ carries it.
6
+
7
+ C3-109. `cache_model`'s field anchors are the field's own measurements, on their
8
+ repository and their machine — the reason they are trustworthy is the reason they
9
+ cannot be refreshed here, and B17 reopened one release after closing because a
10
+ refreshed anchor expires the moment the next release ships. This is the figure
11
+ this project *can* re-measure every release: the gate's own population on a
12
+ 2 985-file repository, under a declared host class, with the build that produced
13
+ it recorded beside it.
14
+
15
+ It does not replace a field anchor and never overrides one. It is the second
16
+ figure a reader gets — one measured at field scale on another build, one measured
17
+ on this build on a repository we control — and the pair is honest in a way
18
+ neither is alone.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ #: The build the gate cells below were captured on.
23
+ GATE_MEASURED_VERSION = '5.0.0'
24
+ #: When, so a reader can tell a stale capture from a missing one.
25
+ GATE_CAPTURED_AT = '2026-08-11T07:49:57Z'
26
+ #: The repository the gate measures, and its size axis.
27
+ GATE_REPOSITORY = 'BroadleafCommerce'
28
+ GATE_JAVA_FILES = 2985
29
+ #: The host class the runs are comparable within. Cross-machine numbers
30
+ #: are recorded, never diffed (ADR-0006).
31
+ GATE_HOST_CLASS = 'github-ubuntu-latest'
32
+
33
+ #: command → measured wall time (p50 seconds), the cache mode it was
34
+ #: taken in, and how many runs the median is over.
35
+ GATE_ANCHORS: "dict[str, dict]" = {
36
+ 'ask': {"seconds": 0.46, "mode": 'warm', "runs": 5},
37
+ 'endpoints': {"seconds": 2.07, "mode": 'warm', "runs": 5},
38
+ 'migrate-check': {"seconds": 8.72, "mode": 'warm', "runs": 5},
39
+ 'posture': {"seconds": 2.89, "mode": 'warm', "runs": 5},
40
+ 'spring-audit': {"seconds": 3.51, "mode": 'warm', "runs": 5},
41
+ 'validation': {"seconds": 9.22, "mode": 'warm', "runs": 5},
42
+ }
sourcecode/parse_cache.py CHANGED
@@ -204,6 +204,49 @@ def purge_stale_generations() -> int:
204
204
  return removed
205
205
 
206
206
 
207
+ def clear_store() -> "dict[str, int]":
208
+ """Empty the shared parse store — every generation, every entry (C3-104).
209
+
210
+ `cache status` publishes this layer and, until this existed, nothing could
211
+ evict it: `cache.clear()` deletes the per-repository core, view and snapshot
212
+ files, and the only surface here was `purge_stale_generations()`, which
213
+ deliberately keeps the current build's. The field purging for a from-scratch
214
+ audit had to be told `rm -rf ~/.sourcecode/parse-cache-v1` (6 684 files /
215
+ 142,5 MB) — a path they learned from our documentation rather than from the
216
+ CLI, in the one workflow we actively recommend.
217
+
218
+ Global on purpose: the store is content-addressed **across repositories**, so
219
+ an entry cannot be attributed to one of them. Deleting "this repository's
220
+ entries" would mean deleting parses other repositories share, which is why
221
+ the surface is a separate flag and says what it clears.
222
+
223
+ Returns the entries and bytes removed. Best-effort like everything here: an
224
+ entry that cannot be removed is still a correct cache, and a missing entry is
225
+ only ever a miss.
226
+ """
227
+ root = _cache_root()
228
+ entries = 0
229
+ total = 0
230
+ if not root.exists():
231
+ return {"entries": 0, "bytes": 0}
232
+ for path in root.rglob("*"):
233
+ if path.is_file():
234
+ try:
235
+ total += path.stat().st_size
236
+ except OSError:
237
+ pass
238
+ for child in list(root.iterdir()):
239
+ if child.is_dir():
240
+ entries += _rmtree(child)
241
+ else:
242
+ try:
243
+ child.unlink()
244
+ entries += 1
245
+ except OSError:
246
+ continue
247
+ return {"entries": entries, "bytes": total}
248
+
249
+
207
250
  def enforce_budget(*, force: bool = False) -> "dict[str, object]":
208
251
  """Evict least-recently-used entries until the store fits its budget (F-AR).
209
252
 
sourcecode/perf.py CHANGED
@@ -421,12 +421,31 @@ RELEASE_GATE_REPO = "broadleaf"
421
421
  RELEASE_GATE_REPO_URL = "https://github.com/BroadleafCommerce/BroadleafCommerce"
422
422
  RELEASE_GATE_REPO_SCALE = "~3 000 Java files"
423
423
 
424
- #: Warm, because that is the state a returning user is in and the state every
425
- #: field regression was measured in. Cold is dominated by the parse cache being
426
- #: empty, which each release invalidates by construction (C3-70) — gating on it
427
- #: would measure the cache, not the code.
424
+ #: Warm, because that is the state a returning user is in — and it is the mode
425
+ #: the gate *decides* on.
428
426
  RELEASE_GATE_MODE = "warm"
429
427
 
428
+ #: **C3-108.** What the gate *measures*. Warm was the whole population for five
429
+ #: releases, on the argument that "cold is dominated by the parse cache being
430
+ #: empty, which each release invalidates by construction (C3-70), so gating on it
431
+ #: would measure the cache, not the code". Half of that is right and the half that
432
+ #: is wrong cost four rounds: an empty parse cache means the run pays the parser,
433
+ #: and the parser is code — C3-95 and C3-96 were both found by measuring exactly
434
+ #: that. Meanwhile every field report since #23 purges every layer first, so the
435
+ #: regime the reports live in is the one regime nothing here measured, and four
436
+ #: consecutive in-house A/Bs failed to reproduce a field regression.
437
+ #:
438
+ #: Cold cells are therefore **captured and published, and do not decide.** A cold
439
+ #: median on an ephemeral runner carries the noise of a cold page cache as well as
440
+ #: a cold parse store, which is not a base a release may be blocked on; but a cold
441
+ #: cell that doubles between two releases is the evidence this project did not
442
+ #: have, and the next report of a cold regression can be answered with a
443
+ #: measurement instead of a shrug.
444
+ RELEASE_GATE_MEASURED_MODES: tuple[str, ...] = ("warm", "cold")
445
+
446
+ #: The modes a cell may fail the release on. One, deliberately: see above.
447
+ RELEASE_GATE_DECIDING_MODES: tuple[str, ...] = ("warm",)
448
+
430
449
  #: ADR-0006 §5 floor for a cell that may be compared at all.
431
450
  RELEASE_GATE_MIN_RUNS = 5
432
451
 
@@ -485,21 +504,36 @@ def gate_report(
485
504
 
486
505
  ``require_min_runs`` fails a cell whose sample is below the ADR-0006 floor,
487
506
  for the same reason: a 1-run cell is not a measurement the gate may bless.
507
+
508
+ **C3-108: a cell may be measured without deciding.** Every cell is compared
509
+ and published; only cells whose mode is in `RELEASE_GATE_DECIDING_MODES` can
510
+ turn the verdict red. Cold cells are measured because the field measures cold
511
+ and nothing here did; they do not block a release because a cold median on an
512
+ ephemeral runner carries the noise of a cold page cache as well as a cold
513
+ parse store. Each cell says which it is (`decides`), so "the gate is green" is
514
+ never mistaken for "nothing moved".
488
515
  """
489
516
  cells: list[dict[str, object]] = []
490
517
  regressions: list[str] = []
491
518
  missing: list[str] = []
492
519
  under_sampled: list[str] = []
520
+ observed: list[str] = []
521
+
522
+ def _decides(cell: "Mapping[str, object] | None") -> bool:
523
+ mode = str((cell or {}).get("mode") or "")
524
+ return not mode or mode in RELEASE_GATE_DECIDING_MODES
493
525
 
494
526
  for cid in sorted(set(baseline_cells) | set(current_cells)):
495
527
  base = baseline_cells.get(cid)
496
528
  head = current_cells.get(cid)
497
529
  if head is None:
498
- missing.append(cid)
530
+ decides = _decides(base)
531
+ (missing if decides else observed).append(cid)
499
532
  cells.append({
500
533
  "cell": cid,
501
534
  "status": "missing",
502
- "passed": False,
535
+ "passed": not decides,
536
+ "decides": decides,
503
537
  "reasons": ["measured in the baseline, not measured in this run"],
504
538
  })
505
539
  continue
@@ -508,17 +542,20 @@ def gate_report(
508
542
  "cell": cid,
509
543
  "status": "new",
510
544
  "passed": True,
545
+ "decides": _decides(head),
511
546
  "wall_ms": head.get("wall_ms"),
512
547
  "reasons": ["no baseline cell — recorded, nothing to compare"],
513
548
  })
514
549
  continue
515
550
  runs = head.get("runs")
516
551
  if isinstance(runs, int) and runs < require_min_runs:
517
- under_sampled.append(cid)
552
+ decides = _decides(head)
553
+ (under_sampled if decides else observed).append(cid)
518
554
  cells.append({
519
555
  "cell": cid,
520
556
  "status": "under-sampled",
521
- "passed": False,
557
+ "passed": not decides,
558
+ "decides": decides,
522
559
  "runs": runs,
523
560
  "required_runs": require_min_runs,
524
561
  "reasons": [
@@ -532,10 +569,20 @@ def gate_report(
532
569
  max_regression=max_regression,
533
570
  allow_incomparable=allow_incomparable,
534
571
  )
535
- entry = {"cell": cid, **gate}
572
+ decides = _decides(head)
573
+ entry = {"cell": cid, **gate, "decides": decides}
574
+ if not gate.get("passed") and not decides:
575
+ # Measured, published, and explicitly not a blocker — the distinction
576
+ # C3-108 needs, because the alternative to publishing a cold cell that
577
+ # nobody may block on is not publishing it at all.
578
+ entry["passed"] = True
579
+ entry["reasons"] = list(gate.get("reasons") or []) + [
580
+ f"mode `{head.get('mode')}` is measured, not gated — see "
581
+ f"RELEASE_GATE_DECIDING_MODES"
582
+ ]
536
583
  cells.append(entry)
537
584
  if not gate.get("passed"):
538
- regressions.append(cid)
585
+ (regressions if decides else observed).append(cid)
539
586
 
540
587
  passed = not (regressions or missing or under_sampled)
541
588
  return {
@@ -543,11 +590,17 @@ def gate_report(
543
590
  "threshold": max_regression,
544
591
  "gate_repository": RELEASE_GATE_REPO,
545
592
  "gate_mode": RELEASE_GATE_MODE,
593
+ "measured_modes": list(RELEASE_GATE_MEASURED_MODES),
594
+ "deciding_modes": list(RELEASE_GATE_DECIDING_MODES),
546
595
  "passed": passed,
547
596
  "cells_compared": len(cells),
548
597
  "regressions": regressions,
549
598
  "missing_cells": missing,
550
599
  "under_sampled_cells": under_sampled,
600
+ # C3-108: cells that moved (or went missing, or were under-sampled) in a
601
+ # mode the gate measures but does not decide on. Never empty by policy —
602
+ # it is empty when nothing moved there, which is a fact worth reading.
603
+ "observed_not_gated": observed,
551
604
  "excluded_commands": [
552
605
  {"command": cmd, "reason": why} for cmd, why in RELEASE_GATE_EXCLUSIONS
553
606
  ],
@@ -558,7 +611,9 @@ def gate_report(
558
611
  f"was not measured at all, or when a cell's sample is under "
559
612
  f"{require_min_runs} runs. Commands outside the gated population are "
560
613
  f"listed in `excluded_commands` — a green gate is a claim about the "
561
- f"cells it names and about no others."
614
+ f"cells it names and about no others. Modes in `measured_modes` but "
615
+ f"not in `deciding_modes` are compared and published in "
616
+ f"`observed_not_gated`; they never turn the verdict red (C3-108)."
562
617
  ),
563
618
  }
564
619
 
sourcecode/verify_edit.py CHANGED
@@ -217,6 +217,37 @@ def _build_head_cir_via_worktree(root: Path) -> Any:
217
217
  shutil.rmtree(tmp, ignore_errors=True)
218
218
 
219
219
 
220
+ def _clean_tree_cir(root: Path) -> tuple[Any, bool]:
221
+ """The CIR of a tree git reports identical to HEAD — from the shared entry.
222
+
223
+ C3-102. C3-32 closed the second half of this: on a clean tree the *working*
224
+ CIR is the HEAD CIR, so it is not built twice. What was left is the first
225
+ half — the HEAD side still checked out a throwaway detached worktree and
226
+ parsed it, which on the field's repository is 76,6 s to answer *"you have
227
+ not edited anything"*, and 176,6 s in the round after.
228
+
229
+ When git says the tree does not differ from HEAD, the HEAD tree **is** the
230
+ worktree: copying it to a temp directory copies bytes that are already on
231
+ disk, and the shared repository-wide CIR for exactly this state is what
232
+ every other command on this repository has already built or will build.
233
+ `context_cache` keys that entry on `worktree_signature`, which on a clean
234
+ tree *is* the HEAD sha — the two lanes were separate for a reason that only
235
+ holds while the tree is dirty.
236
+
237
+ The shortcut is reached only through `git_answered` (C3-32's guard): a failed
238
+ `git diff` also returns an empty change set, and mistaking that for a clean
239
+ tree would turn an unknown into a confident *"nothing changed"*.
240
+ """
241
+ from sourcecode import context_cache as _ctx
242
+ from sourcecode.repository_ir import find_java_files
243
+
244
+ repo_root = _ctx.repo_root_of(root)
245
+ cached = _ctx.peek_cir(repo_root)
246
+ if cached is not None:
247
+ return cached, True
248
+ return _ctx.shared_cir(root, find_java_files(root)), False
249
+
250
+
220
251
  def _head_cir(root: Path, head_sha: str) -> tuple[Any, bool]:
221
252
  """Return ``(cir_head, cache_hit)``, caching the HEAD CIR by **HEAD sha**.
222
253
 
@@ -249,21 +280,41 @@ def _head_cir(root: Path, head_sha: str) -> tuple[Any, bool]:
249
280
  pass # corrupt payload — rebuild
250
281
 
251
282
  cir = _build_head_cir_via_worktree(root)
252
- if cache.enabled:
253
- try:
254
- cache.put(
255
- key,
256
- CachedContext(
257
- metadata={"scope": SCOPE_JAVA_CIR, "head_sha": head_sha, "cir_hash": cir.cir_hash},
258
- graph_metrics={"symbol_count": len(cir.symbols), "endpoint_count": len(cir.endpoints)},
259
- payload={"raw_ir": cir._raw_ir},
260
- ),
261
- )
262
- except Exception:
263
- pass # caching is best-effort; correctness never depends on it
283
+ _store_head_cir(root, head_sha, cir)
264
284
  return cir, False
265
285
 
266
286
 
287
+ def _store_head_cir(root: Path, head_sha: str, cir: Any) -> None:
288
+ """Write *cir* into the HEAD-sha lane. Best-effort: correctness never depends on it.
289
+
290
+ Called from both lanes on purpose. A clean-tree run reads the shared entry
291
+ (C3-102) and would otherwise leave the edit lane cold, so the first edit
292
+ after it would pay the worktree checkout this row exists to remove — the
293
+ bytes are identical, and writing them costs one serialisation.
294
+ """
295
+ from sourcecode.context_cache import (
296
+ SCOPE_JAVA_CIR,
297
+ CachedContext,
298
+ ContextCache,
299
+ _cache_dir,
300
+ )
301
+
302
+ cache = ContextCache(_cache_dir(root.resolve()), head_sha)
303
+ if not cache.enabled:
304
+ return
305
+ try:
306
+ cache.put(
307
+ cache.knowledge_key(SCOPE_JAVA_CIR),
308
+ CachedContext(
309
+ metadata={"scope": SCOPE_JAVA_CIR, "head_sha": head_sha, "cir_hash": cir.cir_hash},
310
+ graph_metrics={"symbol_count": len(cir.symbols), "endpoint_count": len(cir.endpoints)},
311
+ payload={"raw_ir": cir._raw_ir},
312
+ ),
313
+ )
314
+ except Exception:
315
+ pass
316
+
317
+
267
318
  # ── entry point ─────────────────────────────────────────────────────────────────
268
319
  def head_vs_working(path: Path) -> HeadVsWorking:
269
320
  """Produce the ``(CIR_HEAD, CIR_working, changed_files)`` diff substrate.
@@ -284,10 +335,6 @@ def head_vs_working(path: Path) -> HeadVsWorking:
284
335
  changed_build = tuple(f for f in all_changed if _is_build_config_path(f))
285
336
  timings["changed_files_ms"] = (time.perf_counter() - t0) * 1000.0
286
337
 
287
- t1 = time.perf_counter()
288
- cir_head, hit = _head_cir(root, head_sha)
289
- timings["head_cir_ms"] = (time.perf_counter() - t1) * 1000.0
290
-
291
338
  # C3-32: on a tree git reports as unmodified, the working CIR is the HEAD CIR
292
339
  # by construction — building it again was 94 s of dead cost on the field's
293
340
  # repository, paid *after* the change set was already known to be empty. The
@@ -295,6 +342,20 @@ def head_vs_working(path: Path) -> HeadVsWorking:
295
342
  # `git diff` also yields an empty list, and mistaking that for a clean tree
296
343
  # would turn an unknown into a confident "nothing changed".
297
344
  reused = git_answered and not all_changed
345
+
346
+ # C3-102: and the same fact decides the HEAD side, which is why it is read
347
+ # before the build rather than after it. A clean tree needs no worktree
348
+ # checkout: the tree on disk is HEAD, and its CIR is the shared one.
349
+ t1 = time.perf_counter()
350
+ if reused:
351
+ cir_head, hit = _clean_tree_cir(root)
352
+ if not hit:
353
+ # Keep the edit lane warm: the next run is the one with an edit in it.
354
+ _store_head_cir(root, head_sha, cir_head)
355
+ else:
356
+ cir_head, hit = _head_cir(root, head_sha)
357
+ timings["head_cir_ms"] = (time.perf_counter() - t1) * 1000.0
358
+
298
359
  t2 = time.perf_counter()
299
360
  cir_working = cir_head if reused else _build_cir(root)
300
361
  timings["working_cir_ms"] = (time.perf_counter() - t2) * 1000.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sourcecode
3
- Version: 5.0.1
3
+ Version: 5.1.0
4
4
  Summary: Persistent structural context and ultra-fast repeated analysis for AI coding agents
5
5
  License-File: LICENSE
6
6
  Keywords: agents,ai,codebase,context,developer-tools,llm
@@ -42,7 +42,7 @@ Description-Content-Type: text/markdown
42
42
 
43
43
  **Context · Impact · Migration · Architecture · Review — everything from one structural model.**
44
44
 
45
- ![Version](https://img.shields.io/badge/version-5.0.1-blue)
45
+ ![Version](https://img.shields.io/badge/version-5.1.0-blue)
46
46
  ![Python](https://img.shields.io/badge/python-3.9%2B-green)
47
47
 
48
48
  > **ASK Engine** is the product. The CLI command is **`ask`**. The legacy **`sourcecode`**
@@ -126,7 +126,7 @@ brew tap haroundominique/sourcecode && brew install sourcecode
126
126
  # pip / pipx
127
127
  pipx install sourcecode # or: pip install sourcecode
128
128
 
129
- ask version # ask 5.0.1 — and, on a build that has aged,
129
+ ask version # ask 5.1.0 — and, on a build that has aged,
130
130
  # how many releases have probably shipped since
131
131
  ```
132
132
 
@@ -1,4 +1,4 @@
1
- sourcecode/__init__.py,sha256=Vi4TJSP5-Nuef7bTyGWp6icmexh3SfDW_L2-OJMHyWk,308
1
+ sourcecode/__init__.py,sha256=AQosRL48ISdUoMCE4htHdgvYGVNDHGx2oIMtrSEFoew,308
2
2
  sourcecode/adaptive_scanner.py,sha256=yJBKjNpkY6bpueYJ2YnRezen3sYZDecEt7WaaNWdqug,9466
3
3
  sourcecode/archetype.py,sha256=CZvRLpkHot_D8D3JFQVorr-EHDJyh0BbS7RnSTqBigM,40499
4
4
  sourcecode/architectural_baseline.py,sha256=4GiMVBLJVHvRKSWQGf2ZVurI1-R0qk5pmaevgvGSgVA,26325
@@ -9,7 +9,7 @@ sourcecode/ast_extractor.py,sha256=aXJjZ7XjAdxa99hWNm4SyzPqx4gNWiTbUYjNWGNl-Vc,5
9
9
  sourcecode/audit_report.py,sha256=CLvvlQH2oA6Qj4UADUWhNwSsp1OcyzwXXey5DN_IiWI,8660
10
10
  sourcecode/baseline_autocapture.py,sha256=tmMLexQSeaQRRbqxAd7walvGQqnF0mdpxQmkL40rlhI,15623
11
11
  sourcecode/cache.py,sha256=CK_J8NqyaTNZ57KQT-R4puqI8yYlzG5WL7uLFRbzods,41727
12
- sourcecode/cache_model.py,sha256=8xaRd_9zS6XMGPWZ7E6FHSimAG-pZQKNGQBi1jBFLhY,47724
12
+ sourcecode/cache_model.py,sha256=w-pmiVRxzNO6BOqYwEQ6fPNXEPDhz0hSmJruX2byNAY,52868
13
13
  sourcecode/call_surface.py,sha256=fiqYfHooxN1fX9oQoysq1LS3LoZcobhyUNjAGEZKpwk,4148
14
14
  sourcecode/caller_metrics.py,sha256=--sFGDnIog_YGu9xZHMcB91F5xZakfaAKvy15xU54hg,7904
15
15
  sourcecode/caller_reach.py,sha256=RRF49tv4-QswraJ2vsZ89VAMOj7BmYXlAdXr-tLP_CA,9483
@@ -18,7 +18,7 @@ sourcecode/chain_rules.py,sha256=Bi6UHfgd-GxWswmnHRcPz5jdbAuqka3Zkz_P-MTvqhw,127
18
18
  sourcecode/change_plan.py,sha256=aX1mp2XnlDN-8R2BcTiu8HlkDwQ_4fN1ZLckj0VMXHM,9945
19
19
  sourcecode/cir_graphs.py,sha256=9G0HHj1kw2325IDyzo2OpX73BNswEckecf4MZUXB4JM,12078
20
20
  sourcecode/classifier.py,sha256=JBzPwSSrDG-tUHAbcKB678HRbjLpD-ohzbzzO62mgpo,20114
21
- sourcecode/cli.py,sha256=5sTk3jJ1Xh-JlnLiwN9bH0y3puiYejENXfl5iN0efrU,604917
21
+ sourcecode/cli.py,sha256=KEYjMKuDdlsrmhfAa3c4MKKPVGywZtPYNNWQriQGrnc,617524
22
22
  sourcecode/client_calls.py,sha256=daRTgbXNUOfkzGXJpVb6A737R_Thhka8vw1_YgxCaLg,13548
23
23
  sourcecode/code_notes_analyzer.py,sha256=EJemNCNc9Dn-1RZYu-aNbK0ELzmsyC4s6FdHi3XyNEI,9392
24
24
  sourcecode/compare.py,sha256=2GdDy0qkDSgmkkpgvoCHitoLj3mt30EeFgJ8PGEXJng,14712
@@ -59,6 +59,7 @@ sourcecode/file_classifier.py,sha256=pJCeN9KqWpAwKMCgGP4KDsBjuWMeo4zlbj5fv3hk9dA
59
59
  sourcecode/filter_surface.py,sha256=HeV1qU0y_UO95KNTy3Xb551zXvv7_ns2gyxCS5uC3fs,16418
60
60
  sourcecode/format_contract.py,sha256=e6GvU88TpV1QBq_BSZ6ZMpCq0gu07q7-zW7AjWb-xnI,6041
61
61
  sourcecode/fqn_utils.py,sha256=XLU7zDkNBXz_RZkIUNfpPmp1nekWtqP-fxV92tDV1vg,2158
62
+ sourcecode/gate_anchors.py,sha256=26v8wUtzOk1K1Un5uXdUxAu20mqSPX3jo82zuM05ByE,2066
62
63
  sourcecode/git_analyzer.py,sha256=irvEwddbd5_tbA_eYUZebf4mH7_nFhpjwaWAV1f7jUY,15755
63
64
  sourcecode/git_checkout.py,sha256=gCnkMwHpU6yLyc10FoGzE8W9-tI8KyYxKTgpbWPjayo,6009
64
65
  sourcecode/graph_analyzer.py,sha256=lp0eB1PWC20BYF-GpPhAyegRpKrUKgOmXZIcZSIX_Ks,65777
@@ -78,11 +79,11 @@ sourcecode/openrewrite_recipe.py,sha256=4dyY5twRB6Xew-G9Zy5FafNZwlMU1S7DnRa2FqPk
78
79
  sourcecode/output_budget.py,sha256=GxoFnbwk0dvqScgmMRPZHj7m-zRIoHLwUKplrH0u89s,13872
79
80
  sourcecode/output_encoding.py,sha256=OxiZQtr7uWBDUtblNQviNHsTH1Xo9AZEpsegPlQWKFk,2139
80
81
  sourcecode/parallel.py,sha256=T9N-I8rtr43eXx2ZmKcmj2e62v6wsEKE9Y7zGNZ62TU,10224
81
- sourcecode/parse_cache.py,sha256=fhXnoaOgVWwp_8q0ngJ76UZ09qJ6uxDd0jaX8jdezvo,20826
82
+ sourcecode/parse_cache.py,sha256=TpIoSeNUba4Aoaz5o5SmgAc0Tz02OKq1Du4Hw7HyJfg,22560
82
83
  sourcecode/partial_contract.py,sha256=x_N_0bF3mpNNjXd7JFB1zr6LSIroo2zN5GCz14GXvXQ,3510
83
84
  sourcecode/path_admission.py,sha256=OGNSoluhVh8YYOBZBnACUrkmH4Tp_yxjGLifkcsxQpo,5838
84
85
  sourcecode/path_filters.py,sha256=LmTYq735orssKTIuxTX1mE1FHAXF7dVEJEse35xE1RA,12371
85
- sourcecode/perf.py,sha256=ZCPxDIg6ww6PXfLwcqNTH99lc1UgkYUZI3MywFgRO90,25991
86
+ sourcecode/perf.py,sha256=l5YW_BBKMRZb88yjzI9FnXpr21pgmQ7LPt1MZBYhf3k,29345
86
87
  sourcecode/phased_run.py,sha256=cFd0Lxfc8XL-XkCVCCzcjd9bZbtgkJkGZAHfmJ6dBJ4,16542
87
88
  sourcecode/pipe_contract.py,sha256=PML0Er5d8uDyec4OrdUXuyqHbGPUYndjeaKUjkFz2u4,8369
88
89
  sourcecode/posture.py,sha256=7d3zM4ExrSXVKrWNTlXQ0fUQNqDdl1LOuKwRVaXdRss,76574
@@ -142,7 +143,7 @@ sourcecode/tree_utils.py,sha256=8GAkIfQAsvtEudIeW1l4ooH_oRtrWR8cpJQJsEa_Pfw,2093
142
143
  sourcecode/type_usage_surface.py,sha256=51IrKRQoIoRnlsiDjHnqpJBn2rc6E59aRhgS0HTzAF0,4428
143
144
  sourcecode/validation_inference.py,sha256=-oWJqE6PqkqcZbCFDJeWCQHgI9Dbuv5ggJWD3LE_2IU,20817
144
145
  sourcecode/validation_surface.py,sha256=jYL-hkDjaaRKkAt7ZUcbxm3Dydl8o6KVRhTkuPXKPVc,31460
145
- sourcecode/verify_edit.py,sha256=O6nhDN3DLwr2RniMdWq7tSeVUkzlbfQfyBz5K_-o4SM,37608
146
+ sourcecode/verify_edit.py,sha256=djBCZcDMEM71UFhls_J4oImYBu7CUeQOHPKzWrT9c9U,40288
146
147
  sourcecode/verify_repo.py,sha256=aHlNIUfoVVUxTXTNtSFryP4dOFBOvCfdyC31fFRXU6s,15346
147
148
  sourcecode/verify_rules.py,sha256=YxG5JkDXBVpwbU1_tIKVtU8_eaF3zjpsq05tjpN4Z7Y,19401
148
149
  sourcecode/version_check.py,sha256=CHp6ZxTIfo8kyHPCBgJA1uFC0xQCoXMuuOfrW8QTL8o,4942
@@ -206,8 +207,8 @@ sourcecode/telemetry/consent.py,sha256=pQdl-QeLl6Gcibn0eWHSKZrm-HYSsjpVqOnjrgFp8
206
207
  sourcecode/telemetry/events.py,sha256=4_yeO58U-Cwc1Qb27VB0_EjhmroY0k91n3_VGxeALB8,2776
207
208
  sourcecode/telemetry/filters.py,sha256=RzxauTz8HliO4BllQnXEXc7zTeqdCZi5MgqGEDuW7OQ,6570
208
209
  sourcecode/telemetry/transport.py,sha256=4gGHsq0WeY9VywEZXA3vUxykfiYnw9uuqfjAAec7F8o,1681
209
- sourcecode-5.0.1.dist-info/METADATA,sha256=wKxYzyIT5zVAkDjb12M3Wzj2yyLMr-uklRY8AmRlHGI,42410
210
- sourcecode-5.0.1.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
211
- sourcecode-5.0.1.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
212
- sourcecode-5.0.1.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
213
- sourcecode-5.0.1.dist-info/RECORD,,
210
+ sourcecode-5.1.0.dist-info/METADATA,sha256=-w-kPbI2gq9YUOyXPsS8f-zgKHUVJBsbC_u-iXcgAZ0,42410
211
+ sourcecode-5.1.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
212
+ sourcecode-5.1.0.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
213
+ sourcecode-5.1.0.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
214
+ sourcecode-5.1.0.dist-info/RECORD,,