sourcecode 2.6.5__py3-none-any.whl → 2.6.10__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sourcecode/__init__.py +1 -1
- sourcecode/cli.py +91 -11
- sourcecode/file_classifier.py +6 -2
- sourcecode/mcp/onboarding/applier.py +29 -15
- sourcecode/mcp/onboarding/detector.py +38 -0
- sourcecode/mcp/onboarding/planner.py +2 -2
- sourcecode/mcp/server.py +45 -6
- sourcecode/mcp_nudge.py +3 -2
- sourcecode/readiness_timeline.py +178 -0
- sourcecode/telemetry/__init__.py +2 -0
- sourcecode/telemetry/consent.py +2 -1
- sourcecode/telemetry/events.py +3 -0
- sourcecode/telemetry/filters.py +20 -1
- sourcecode/token_estimate.py +213 -0
- {sourcecode-2.6.5.dist-info → sourcecode-2.6.10.dist-info}/METADATA +1 -1
- {sourcecode-2.6.5.dist-info → sourcecode-2.6.10.dist-info}/RECORD +19 -17
- {sourcecode-2.6.5.dist-info → sourcecode-2.6.10.dist-info}/WHEEL +0 -0
- {sourcecode-2.6.5.dist-info → sourcecode-2.6.10.dist-info}/entry_points.txt +0 -0
- {sourcecode-2.6.5.dist-info → sourcecode-2.6.10.dist-info}/licenses/LICENSE +0 -0
sourcecode/__init__.py
CHANGED
sourcecode/cli.py
CHANGED
|
@@ -1851,6 +1851,10 @@ def main(
|
|
|
1851
1851
|
)
|
|
1852
1852
|
except Exception:
|
|
1853
1853
|
pass # stale value better than crash
|
|
1854
|
+
# C1: token economy on the agent-facing views (cache-hit path).
|
|
1855
|
+
if format == "json" and (compact or agent):
|
|
1856
|
+
from sourcecode.token_estimate import inject_token_economy as _inj_te
|
|
1857
|
+
_cache_hit_content = _inj_te(_cache_hit_content, target)
|
|
1854
1858
|
_emit_command_output(_cache_hit_content, output, copy)
|
|
1855
1859
|
return
|
|
1856
1860
|
|
|
@@ -2699,7 +2703,14 @@ def main(
|
|
|
2699
2703
|
}, indent=2, ensure_ascii=False)
|
|
2700
2704
|
except Exception:
|
|
2701
2705
|
pass
|
|
2702
|
-
|
|
2706
|
+
# C1: token economy on the agent-facing views (fresh path). Emit-only —
|
|
2707
|
+
# `content` stays clean so the cached L2 view never embeds a stale block
|
|
2708
|
+
# (the cache-hit path recomputes it against the working tree).
|
|
2709
|
+
_emit_content = content
|
|
2710
|
+
if format == "json" and (compact or agent):
|
|
2711
|
+
from sourcecode.token_estimate import inject_token_economy as _inj_te
|
|
2712
|
+
_emit_content = _inj_te(content, target)
|
|
2713
|
+
_emit_command_output(_emit_content, output, copy if not _pipeline_error else False)
|
|
2703
2714
|
|
|
2704
2715
|
# Persist to two-layer cache (git SHA unchanged → re-use on next run).
|
|
2705
2716
|
#
|
|
@@ -5873,6 +5884,23 @@ def migrate_check_cmd(
|
|
|
5873
5884
|
False, "--no-cache",
|
|
5874
5885
|
help="Accepted for compatibility; this command always reads fresh source (no snapshot cache). No-op.",
|
|
5875
5886
|
),
|
|
5887
|
+
snapshot: bool = typer.Option(
|
|
5888
|
+
False, "--snapshot",
|
|
5889
|
+
help="Persist this readiness result as a timeline snapshot under --history-dir.",
|
|
5890
|
+
),
|
|
5891
|
+
trend: bool = typer.Option(
|
|
5892
|
+
False, "--trend",
|
|
5893
|
+
help="Report the readiness trend across stored snapshots (days-remaining over time) "
|
|
5894
|
+
"instead of scanning. Reads --history-dir.",
|
|
5895
|
+
),
|
|
5896
|
+
history_dir: Optional[Path] = typer.Option(
|
|
5897
|
+
None, "--history-dir",
|
|
5898
|
+
help="Directory of readiness snapshots (default: <repo>/.ask/readiness-history).",
|
|
5899
|
+
),
|
|
5900
|
+
ref: Optional[str] = typer.Option(
|
|
5901
|
+
None, "--ref",
|
|
5902
|
+
help="Label for a --snapshot capture (e.g. a version or sprint tag).",
|
|
5903
|
+
),
|
|
5876
5904
|
) -> None:
|
|
5877
5905
|
"""Spring Boot 2→3 migration readiness: detect javax→jakarta namespace blockers.
|
|
5878
5906
|
|
|
@@ -5904,6 +5932,16 @@ def migrate_check_cmd(
|
|
|
5904
5932
|
ask migrate-check /path/to/repo --format text
|
|
5905
5933
|
ask migrate-check . --min-severity high
|
|
5906
5934
|
ask migrate-check . --output migration.json
|
|
5935
|
+
ask migrate-check . --snapshot --ref sprint-12 persist a readiness point
|
|
5936
|
+
ask migrate-check . --trend days-remaining over time
|
|
5937
|
+
|
|
5938
|
+
\b
|
|
5939
|
+
Readiness time series:
|
|
5940
|
+
--snapshot persists this run's budgetable figures (readiness_score, effort
|
|
5941
|
+
days, per-dimension scores, blocking_count) under --history-dir (default
|
|
5942
|
+
<repo>/.ask/readiness-history). --trend reads that series and reports
|
|
5943
|
+
first→last movement — no improving/degrading label, and readiness_score is
|
|
5944
|
+
flagged not-comparable when the applicable dimension set changed.
|
|
5907
5945
|
"""
|
|
5908
5946
|
from sourcecode.repository_ir import find_java_files
|
|
5909
5947
|
from sourcecode.migrate_check import run_migrate_check
|
|
@@ -5930,6 +5968,29 @@ def migrate_check_cmd(
|
|
|
5930
5968
|
)
|
|
5931
5969
|
raise typer.Exit(code=1)
|
|
5932
5970
|
|
|
5971
|
+
_history_dir = history_dir.resolve() if history_dir else (target / ".ask" / "readiness-history")
|
|
5972
|
+
|
|
5973
|
+
# --trend reads the stored series and reports movement over time; it does not scan.
|
|
5974
|
+
if trend:
|
|
5975
|
+
from sourcecode.readiness_timeline import build_readiness_trend, load_snapshots_dir
|
|
5976
|
+
|
|
5977
|
+
_snaps = load_snapshots_dir(_history_dir) if _history_dir.exists() else []
|
|
5978
|
+
if not _snaps:
|
|
5979
|
+
_emit_error_json(
|
|
5980
|
+
INVALID_INPUT_CODE,
|
|
5981
|
+
f"No readiness snapshots found in '{_history_dir}'.",
|
|
5982
|
+
path=str(_history_dir),
|
|
5983
|
+
hint="Capture points first: ask migrate-check <repo> --snapshot",
|
|
5984
|
+
expected="A directory holding readiness-snapshot-v1 artifacts.",
|
|
5985
|
+
)
|
|
5986
|
+
raise typer.Exit(code=1)
|
|
5987
|
+
_trend = build_readiness_trend(_snaps)
|
|
5988
|
+
_emit_command_output(
|
|
5989
|
+
_serialize_dict(_trend, "json"), output_path, copy,
|
|
5990
|
+
success_msg=f"Readiness trend over {_trend['count']} snapshot(s) → {_history_dir}",
|
|
5991
|
+
)
|
|
5992
|
+
return
|
|
5993
|
+
|
|
5933
5994
|
_file_limitations: list[str] = []
|
|
5934
5995
|
file_list = find_java_files(target, limitations=_file_limitations)
|
|
5935
5996
|
_prog = Progress()
|
|
@@ -5950,6 +6011,14 @@ def migrate_check_cmd(
|
|
|
5950
6011
|
payload = report.to_compact_dict() if compact else report.to_dict()
|
|
5951
6012
|
output = _serialize_dict(payload, "json")
|
|
5952
6013
|
|
|
6014
|
+
_snapshot_note = ""
|
|
6015
|
+
if snapshot:
|
|
6016
|
+
from sourcecode.readiness_timeline import build_snapshot, write_snapshot
|
|
6017
|
+
|
|
6018
|
+
_snap = build_snapshot(report.to_dict(), ref=ref)
|
|
6019
|
+
_snap_path = write_snapshot(_snap, _history_dir)
|
|
6020
|
+
_snapshot_note = f"; snapshot → {_snap_path}"
|
|
6021
|
+
|
|
5953
6022
|
_total = report.summary.get("total_findings", 0)
|
|
5954
6023
|
_emit_command_output(
|
|
5955
6024
|
output, output_path, copy,
|
|
@@ -5957,6 +6026,7 @@ def migrate_check_cmd(
|
|
|
5957
6026
|
f"Migration check written to {output_path} "
|
|
5958
6027
|
f"(score: {report.readiness_score if report.readiness_score is not None else 'N/A'}"
|
|
5959
6028
|
f"{'/100' if report.readiness_score is not None else ''}, {_total} findings)"
|
|
6029
|
+
f"{_snapshot_note}"
|
|
5960
6030
|
),
|
|
5961
6031
|
)
|
|
5962
6032
|
|
|
@@ -7965,11 +8035,15 @@ def cold_start_cmd(
|
|
|
7965
8035
|
result["endpoints"] = result["endpoints"][:30]
|
|
7966
8036
|
result["_meta"] = {**(result.get("_meta") or {}), "compact_mode": True,
|
|
7967
8037
|
"full_available": "ask cold-start (without --compact)"}
|
|
8038
|
+
from sourcecode.token_estimate import estimate_tokens as _est_tokens, token_economy as _tok_econ
|
|
7968
8039
|
_out = _json.dumps(result, indent=2, ensure_ascii=False)
|
|
7969
8040
|
_size = len(_out.encode("utf-8"))
|
|
7970
|
-
_tokens =
|
|
8041
|
+
_tokens = _est_tokens(_out)
|
|
7971
8042
|
_out_with_meta = _json.loads(_out)
|
|
7972
|
-
_out_with_meta.setdefault("_meta", {})
|
|
8043
|
+
_meta_cs = _out_with_meta.setdefault("_meta", {})
|
|
8044
|
+
_meta_cs["estimated_tokens"] = _tokens
|
|
8045
|
+
# C1: what this response costs vs. what reading the same files raw would.
|
|
8046
|
+
_meta_cs["token_economy"] = _tok_econ(_out, _out_with_meta, target)
|
|
7973
8047
|
_out = _json.dumps(_out_with_meta, indent=2, ensure_ascii=False)
|
|
7974
8048
|
if not compact and _size > 400_000:
|
|
7975
8049
|
sys.stderr.write(
|
|
@@ -8086,7 +8160,9 @@ def mcp_init(
|
|
|
8086
8160
|
typer.echo("No MCP clients found on this system.")
|
|
8087
8161
|
typer.echo("")
|
|
8088
8162
|
typer.echo("Manual setup — add to your MCP client config:")
|
|
8089
|
-
typer.echo(' "
|
|
8163
|
+
typer.echo(' "ask": {"command": "ask", "args": ["mcp", "serve"]}')
|
|
8164
|
+
typer.echo(' (VS Code keys these under "servers" and wants "type": "stdio";')
|
|
8165
|
+
typer.echo(' other clients use "mcpServers".)')
|
|
8090
8166
|
raise typer.Exit(code=0)
|
|
8091
8167
|
|
|
8092
8168
|
# Show detection results
|
|
@@ -8138,7 +8214,9 @@ def mcp_init(
|
|
|
8138
8214
|
if a.client.config_path.exists():
|
|
8139
8215
|
bak = backup.create(a.client.config_path)
|
|
8140
8216
|
typer.echo(f" ✓ Backup {bak}")
|
|
8141
|
-
updated = applier.apply_entry(
|
|
8217
|
+
updated = applier.apply_entry(
|
|
8218
|
+
config, a.client.servers_key, a.client.entry_extra
|
|
8219
|
+
)
|
|
8142
8220
|
applier.write_config(a.client.config_path, updated)
|
|
8143
8221
|
if not applier.validate(a.client.config_path):
|
|
8144
8222
|
errors.append(f"{a.client.name}: JSON validation failed after write")
|
|
@@ -8222,16 +8300,18 @@ def mcp_status() -> None:
|
|
|
8222
8300
|
typer.echo(f" Fix: ask mcp init --target {client.slug}")
|
|
8223
8301
|
continue
|
|
8224
8302
|
config = applier.read_config(client.config_path)
|
|
8225
|
-
if applier.is_installed(config):
|
|
8303
|
+
if applier.is_installed(config, client.servers_key):
|
|
8226
8304
|
typer.echo(f" {client.name:<20} ✓ configured {client.config_path}")
|
|
8227
8305
|
# FIX-P0-5: inspect registered command for external-server drift.
|
|
8228
|
-
|
|
8306
|
+
# Reads through the client's own servers key and entry name, so drift is
|
|
8307
|
+
# still detected for clients that key their servers differently.
|
|
8308
|
+
_registered = applier.registered_entry(config, client.servers_key)
|
|
8229
8309
|
_reg_cmd = _registered.get("command", "")
|
|
8230
8310
|
_reg_args = _registered.get("args", [])
|
|
8231
8311
|
# Built-in form: command=sourcecode args=[mcp, serve] (or just the binary)
|
|
8232
8312
|
_is_builtin = (
|
|
8233
|
-
_reg_cmd
|
|
8234
|
-
or (not _reg_args and _reg_cmd.endswith("/sourcecode"))
|
|
8313
|
+
_reg_cmd in ("ask", "sourcecode")
|
|
8314
|
+
or (not _reg_args and _reg_cmd.endswith(("/ask", "/sourcecode")))
|
|
8235
8315
|
or (_reg_args and _reg_args[:2] == ["mcp", "serve"])
|
|
8236
8316
|
)
|
|
8237
8317
|
if _is_builtin:
|
|
@@ -8292,7 +8372,7 @@ def mcp_status() -> None:
|
|
|
8292
8372
|
for _c in clients:
|
|
8293
8373
|
if _c.app_installed:
|
|
8294
8374
|
_cfg = applier.read_config(_c.config_path)
|
|
8295
|
-
if applier.is_installed(_cfg):
|
|
8375
|
+
if applier.is_installed(_cfg, _c.servers_key):
|
|
8296
8376
|
_configured_clients.add(_c.slug)
|
|
8297
8377
|
|
|
8298
8378
|
# Stage 3: Process liveness — is the client app currently running?
|
|
@@ -8372,7 +8452,7 @@ def mcp_remove(
|
|
|
8372
8452
|
bak = backup.create(a.client.config_path)
|
|
8373
8453
|
typer.echo(f" ✓ Backup {bak}")
|
|
8374
8454
|
config = applier.read_config(a.client.config_path)
|
|
8375
|
-
updated = applier.remove_entry(config)
|
|
8455
|
+
updated = applier.remove_entry(config, a.client.servers_key)
|
|
8376
8456
|
applier.write_config(a.client.config_path, updated)
|
|
8377
8457
|
if not applier.validate(a.client.config_path):
|
|
8378
8458
|
errors.append(f"{a.client.name}: JSON validation failed — restoring backup")
|
sourcecode/file_classifier.py
CHANGED
|
@@ -300,14 +300,18 @@ class FileClassifier:
|
|
|
300
300
|
found = frozenset(m.group(1) for m in _JAVA_ANNOTATION_RE.finditer(content))
|
|
301
301
|
if not found:
|
|
302
302
|
return None
|
|
303
|
+
# Sorted, not set order: this list is emitted as `evidence` in the agent view,
|
|
304
|
+
# and frozenset iteration order varies with PYTHONHASHSEED — the same repo
|
|
305
|
+
# analysed twice produced the same evidence in a different order.
|
|
306
|
+
evidence = sorted(found)
|
|
303
307
|
for required_annotations, category, relevance, why in _JAVA_STEREOTYPE_RULES:
|
|
304
308
|
# For @Data DTO: must have @Data but NOT @Entity
|
|
305
309
|
if required_annotations == frozenset({"Data"}):
|
|
306
310
|
if "Data" in found and "Entity" not in found:
|
|
307
|
-
return FileClassification(path, category, "high", relevance, why,
|
|
311
|
+
return FileClassification(path, category, "high", relevance, why, evidence)
|
|
308
312
|
continue
|
|
309
313
|
# For compound rules (Service+Transactional, Controller+RequestMapping): all required
|
|
310
314
|
if required_annotations <= found:
|
|
311
|
-
return FileClassification(path, category, "high", relevance, why,
|
|
315
|
+
return FileClassification(path, category, "high", relevance, why, evidence)
|
|
312
316
|
return None
|
|
313
317
|
|
|
@@ -25,36 +25,50 @@ def read_config(path: Path) -> dict:
|
|
|
25
25
|
return {}
|
|
26
26
|
|
|
27
27
|
|
|
28
|
-
def is_installed(config: dict) -> bool:
|
|
28
|
+
def is_installed(config: dict, servers_key: str = _MCP_SERVERS_KEY) -> bool:
|
|
29
29
|
"""True if the ASK Engine entry (canonical `ask` or legacy `sourcecode`) is
|
|
30
|
-
already present in
|
|
31
|
-
servers = config.get(
|
|
30
|
+
already present in the client's servers map."""
|
|
31
|
+
servers = config.get(servers_key, {})
|
|
32
32
|
return _ENTRY_NAME in servers or _LEGACY_ENTRY_NAME in servers
|
|
33
33
|
|
|
34
34
|
|
|
35
|
-
def
|
|
36
|
-
"""
|
|
37
|
-
|
|
38
|
-
|
|
35
|
+
def registered_entry(config: dict, servers_key: str = _MCP_SERVERS_KEY) -> dict:
|
|
36
|
+
"""The ASK Engine server entry as currently registered, or an empty dict."""
|
|
37
|
+
servers = config.get(servers_key, {})
|
|
38
|
+
if not isinstance(servers, dict):
|
|
39
|
+
return {}
|
|
40
|
+
entry = servers.get(_ENTRY_NAME) or servers.get(_LEGACY_ENTRY_NAME) or {}
|
|
41
|
+
return entry if isinstance(entry, dict) else {}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def apply_entry(
|
|
45
|
+
config: dict,
|
|
46
|
+
servers_key: str = _MCP_SERVERS_KEY,
|
|
47
|
+
entry_extra: tuple[tuple[str, str], ...] = (),
|
|
48
|
+
) -> dict:
|
|
49
|
+
"""Return new config dict with the canonical `ask` entry merged into the client's
|
|
50
|
+
servers map. Any legacy `sourcecode` entry is migrated away (removed) so the client
|
|
51
|
+
launches a single server. *entry_extra* carries per-client fields (e.g. VS Code's
|
|
52
|
+
explicit `type: stdio`)."""
|
|
39
53
|
config = dict(config)
|
|
40
|
-
servers: dict = dict(config.get(
|
|
54
|
+
servers: dict = dict(config.get(servers_key, {}))
|
|
41
55
|
servers.pop(_LEGACY_ENTRY_NAME, None)
|
|
42
|
-
servers[_ENTRY_NAME] = _ENTRY_VALUE
|
|
43
|
-
config[
|
|
56
|
+
servers[_ENTRY_NAME] = {**_ENTRY_VALUE, **dict(entry_extra)}
|
|
57
|
+
config[servers_key] = servers
|
|
44
58
|
return config
|
|
45
59
|
|
|
46
60
|
|
|
47
|
-
def remove_entry(config: dict) -> dict:
|
|
61
|
+
def remove_entry(config: dict, servers_key: str = _MCP_SERVERS_KEY) -> dict:
|
|
48
62
|
"""Return new config dict with the ASK Engine entry removed — both the canonical
|
|
49
63
|
`ask` key and any legacy `sourcecode` key."""
|
|
50
64
|
config = dict(config)
|
|
51
|
-
servers: dict = dict(config.get(
|
|
65
|
+
servers: dict = dict(config.get(servers_key, {}))
|
|
52
66
|
servers.pop(_ENTRY_NAME, None)
|
|
53
67
|
servers.pop(_LEGACY_ENTRY_NAME, None)
|
|
54
68
|
if servers:
|
|
55
|
-
config[
|
|
56
|
-
elif
|
|
57
|
-
del config[
|
|
69
|
+
config[servers_key] = servers
|
|
70
|
+
elif servers_key in config:
|
|
71
|
+
del config[servers_key]
|
|
58
72
|
return config
|
|
59
73
|
|
|
60
74
|
|
|
@@ -38,6 +38,38 @@ _CLIENT_REGISTRY: List[Dict[str, Any]] = [
|
|
|
38
38
|
"win32": "Cursor",
|
|
39
39
|
},
|
|
40
40
|
},
|
|
41
|
+
{
|
|
42
|
+
"name": "Windsurf",
|
|
43
|
+
"slug": "windsurf",
|
|
44
|
+
"paths": {
|
|
45
|
+
"darwin": "~/.codeium/windsurf/mcp_config.json",
|
|
46
|
+
"linux": "~/.codeium/windsurf/mcp_config.json",
|
|
47
|
+
"win32": "{USERPROFILE}/.codeium/windsurf/mcp_config.json",
|
|
48
|
+
},
|
|
49
|
+
"process": {
|
|
50
|
+
"darwin": "Windsurf",
|
|
51
|
+
"linux": "windsurf",
|
|
52
|
+
"win32": "Windsurf",
|
|
53
|
+
},
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
# VS Code keys its servers under "servers" (not "mcpServers") and wants an
|
|
57
|
+
# explicit transport on each entry — hence the per-client overrides below.
|
|
58
|
+
"name": "VS Code",
|
|
59
|
+
"slug": "vscode",
|
|
60
|
+
"paths": {
|
|
61
|
+
"darwin": "~/Library/Application Support/Code/User/mcp.json",
|
|
62
|
+
"linux": "~/.config/Code/User/mcp.json",
|
|
63
|
+
"win32": "{APPDATA}/Code/User/mcp.json",
|
|
64
|
+
},
|
|
65
|
+
"process": {
|
|
66
|
+
"darwin": "Code Helper",
|
|
67
|
+
"linux": "code",
|
|
68
|
+
"win32": "Code",
|
|
69
|
+
},
|
|
70
|
+
"servers_key": "servers",
|
|
71
|
+
"entry_extra": (("type", "stdio"),),
|
|
72
|
+
},
|
|
41
73
|
]
|
|
42
74
|
|
|
43
75
|
|
|
@@ -48,6 +80,10 @@ class MCPClient:
|
|
|
48
80
|
app_installed: bool # True if the config file (or its parent dir) exists
|
|
49
81
|
process_name: str # OS process name for connectivity check
|
|
50
82
|
slug: str # --target identifier (e.g. "claude-desktop")
|
|
83
|
+
# Config-shape differences between clients. Defaults match the common shape
|
|
84
|
+
# (Claude Desktop / Cursor / Windsurf), so only divergent clients declare them.
|
|
85
|
+
servers_key: str = "mcpServers" # top-level key holding the servers map
|
|
86
|
+
entry_extra: tuple[tuple[str, str], ...] = () # extra fields on the server entry
|
|
51
87
|
|
|
52
88
|
|
|
53
89
|
def _resolve(template: str) -> Path:
|
|
@@ -79,6 +115,8 @@ def detect_clients() -> list[MCPClient]:
|
|
|
79
115
|
app_installed=app_installed,
|
|
80
116
|
process_name=process_name,
|
|
81
117
|
slug=entry["slug"],
|
|
118
|
+
servers_key=entry.get("servers_key", "mcpServers"),
|
|
119
|
+
entry_extra=tuple(entry.get("entry_extra", ())),
|
|
82
120
|
))
|
|
83
121
|
return clients
|
|
84
122
|
|
|
@@ -21,7 +21,7 @@ def build_install_plan(clients: list[MCPClient]) -> list[ClientAction]:
|
|
|
21
21
|
config = read_config(client.config_path)
|
|
22
22
|
actions.append(ClientAction(
|
|
23
23
|
client=client,
|
|
24
|
-
already_installed=is_installed(config),
|
|
24
|
+
already_installed=is_installed(config, client.servers_key),
|
|
25
25
|
will_create_file=not client.config_path.exists(),
|
|
26
26
|
))
|
|
27
27
|
return actions
|
|
@@ -34,7 +34,7 @@ def build_remove_plan(clients: list[MCPClient]) -> list[ClientAction]:
|
|
|
34
34
|
config = read_config(client.config_path)
|
|
35
35
|
actions.append(ClientAction(
|
|
36
36
|
client=client,
|
|
37
|
-
already_installed=is_installed(config),
|
|
37
|
+
already_installed=is_installed(config, client.servers_key),
|
|
38
38
|
will_create_file=False,
|
|
39
39
|
))
|
|
40
40
|
return actions
|
sourcecode/mcp/server.py
CHANGED
|
@@ -28,21 +28,54 @@ from sourcecode.error_schema import (
|
|
|
28
28
|
)
|
|
29
29
|
from sourcecode.mcp.runner import CommandError, run_command
|
|
30
30
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
31
|
+
def _record_tool_invocation(name: Any, success: bool, started: float) -> None:
|
|
32
|
+
"""Count one MCP tool invocation (Bloque C / C3 — adoption).
|
|
33
|
+
|
|
34
|
+
Aggregate only: which of our own tools ran, whether it succeeded, and a
|
|
35
|
+
duration bucket. Never the arguments — those carry repository paths — and
|
|
36
|
+
never any result content. Honours the same opt-out as every other event
|
|
37
|
+
(`ask telemetry disable`, SOURCECODE_TELEMETRY=0, DO_NOT_TRACK=1); when
|
|
38
|
+
telemetry is off, `record` returns before building anything.
|
|
39
|
+
"""
|
|
40
|
+
try:
|
|
41
|
+
import time as _time
|
|
42
|
+
|
|
43
|
+
from sourcecode import telemetry as _tel
|
|
44
|
+
|
|
45
|
+
_tel.record(
|
|
46
|
+
"mcp_tool_invoked",
|
|
47
|
+
cmd="mcp",
|
|
48
|
+
tool=str(name) if name else None,
|
|
49
|
+
duration_s=_time.monotonic() - started,
|
|
50
|
+
success=success,
|
|
51
|
+
)
|
|
52
|
+
except Exception:
|
|
53
|
+
pass # telemetry must never affect a tool call
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# Patch FastMCP's Tool.run to (a) count the invocation and (b) intercept
|
|
57
|
+
# pydantic.ValidationError and return structured JSON instead of the raw
|
|
58
|
+
# "Error executing tool X: 1 validation error..." plain-text string that FastMCP
|
|
59
|
+
# produces by default.
|
|
34
60
|
try:
|
|
35
|
-
import
|
|
61
|
+
import time as _time_mod
|
|
62
|
+
|
|
36
63
|
from mcp.server.fastmcp.tools.base import Tool as _FastMCPTool
|
|
37
64
|
|
|
65
|
+
try:
|
|
66
|
+
import pydantic as _pydantic
|
|
67
|
+
except Exception: # pragma: no cover — counting must not depend on pydantic
|
|
68
|
+
_pydantic = None # type: ignore[assignment]
|
|
69
|
+
|
|
38
70
|
_orig_tool_run = _FastMCPTool.run
|
|
39
71
|
|
|
40
72
|
async def _patched_tool_run(self, arguments, context=None, convert_result=False): # type: ignore[override]
|
|
73
|
+
_started = _time_mod.monotonic()
|
|
41
74
|
try:
|
|
42
|
-
|
|
75
|
+
_result = await _orig_tool_run(self, arguments, context=context, convert_result=convert_result)
|
|
43
76
|
except Exception as _exc:
|
|
44
77
|
_cause = getattr(_exc, "__cause__", None)
|
|
45
|
-
if isinstance(_cause, _pydantic.ValidationError):
|
|
78
|
+
if _pydantic is not None and isinstance(_cause, _pydantic.ValidationError):
|
|
46
79
|
_errors = _cause.errors()
|
|
47
80
|
_missing = [str(e.get("loc", ("?",))[0]) for e in _errors if e.get("type") == "missing"]
|
|
48
81
|
_msg = f"Missing required field: {_missing[0]}" if _missing else "Argument validation failed"
|
|
@@ -56,8 +89,14 @@ try:
|
|
|
56
89
|
expected=f"{self.name} arguments with required field '{_missing[0]}'" if _missing else f"{self.name} arguments",
|
|
57
90
|
),
|
|
58
91
|
}
|
|
92
|
+
_record_tool_invocation(self.name, False, _started)
|
|
59
93
|
return _payload
|
|
94
|
+
_record_tool_invocation(self.name, False, _started)
|
|
60
95
|
raise
|
|
96
|
+
_record_tool_invocation(
|
|
97
|
+
self.name, not getattr(_result, "isError", False), _started
|
|
98
|
+
)
|
|
99
|
+
return _result
|
|
61
100
|
|
|
62
101
|
_FastMCPTool.run = _patched_tool_run # type: ignore[method-assign]
|
|
63
102
|
except Exception:
|
sourcecode/mcp_nudge.py
CHANGED
|
@@ -41,7 +41,7 @@ except Exception: # pragma: no cover
|
|
|
41
41
|
def detect_clients() -> list: # type: ignore[misc]
|
|
42
42
|
return []
|
|
43
43
|
|
|
44
|
-
def is_installed(config: dict) -> bool: # type: ignore[misc]
|
|
44
|
+
def is_installed(config: dict, servers_key: str = "mcpServers") -> bool: # type: ignore[misc]
|
|
45
45
|
return False
|
|
46
46
|
|
|
47
47
|
def read_config(path: Path) -> dict: # type: ignore[misc]
|
|
@@ -60,7 +60,8 @@ def nudge_mcp_if_needed() -> None:
|
|
|
60
60
|
return
|
|
61
61
|
|
|
62
62
|
needs_nudge = any(
|
|
63
|
-
c.app_installed
|
|
63
|
+
c.app_installed
|
|
64
|
+
and not is_installed(read_config(c.config_path), getattr(c, "servers_key", "mcpServers"))
|
|
64
65
|
for c in clients
|
|
65
66
|
)
|
|
66
67
|
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""readiness_timeline.py — migration-readiness time series.
|
|
2
|
+
|
|
3
|
+
Persists a compact ``readiness-snapshot-v1`` artifact per capture and projects a
|
|
4
|
+
trend across N snapshots ordered by capture time. The moat is the *quantified,
|
|
5
|
+
budgetable* readiness score and estimated effort tracked over time — the
|
|
6
|
+
"days-remaining" line a migration lead can put in front of management.
|
|
7
|
+
|
|
8
|
+
Discipline (matches architectural_baseline / the product's measured-not-ROI rule):
|
|
9
|
+
* Movement only. The trend reports first→last deltas; it never labels a change
|
|
10
|
+
"improving" or "degrading" — the reader decides what a rising score means.
|
|
11
|
+
* Honest gaps. A delta is emitted only when BOTH endpoints carry the value;
|
|
12
|
+
a ``None`` readiness (no applicable migration dimension) yields no delta, and
|
|
13
|
+
a change in ``applicable_dimensions`` flags the score as not strictly
|
|
14
|
+
comparable rather than silently diffing incomparable numbers.
|
|
15
|
+
* Deterministic projection. A fixed set of snapshots always yields the same
|
|
16
|
+
trend; only the wall-clock capture time (a real event) varies between runs.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
from datetime import datetime, timezone
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, Optional
|
|
24
|
+
|
|
25
|
+
READINESS_SNAPSHOT_SCHEMA = "readiness-snapshot-v1"
|
|
26
|
+
READINESS_TREND_SCHEMA = "readiness-trend-v1"
|
|
27
|
+
|
|
28
|
+
# Budgetable score fields lifted from a migrate-check report, tracked over time.
|
|
29
|
+
_SCORE_FIELDS: tuple[str, ...] = (
|
|
30
|
+
"readiness_score",
|
|
31
|
+
"jakarta_readiness",
|
|
32
|
+
"boot3_readiness",
|
|
33
|
+
"hibernate_readiness",
|
|
34
|
+
"jdk_modernization",
|
|
35
|
+
)
|
|
36
|
+
# Numeric tracks that carry a first→last delta (scores + the two headline totals).
|
|
37
|
+
_DELTA_FIELDS: tuple[str, ...] = _SCORE_FIELDS + ("blocking_count", "estimated_effort_days")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _utc_now() -> str:
|
|
41
|
+
# Microsecond precision so two captures in the same second stay distinct and
|
|
42
|
+
# the lexicographic ISO sort remains chronological (same rule as baselines).
|
|
43
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%fZ")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _tool_version() -> str:
|
|
47
|
+
try:
|
|
48
|
+
from sourcecode import __version__
|
|
49
|
+
return str(__version__)
|
|
50
|
+
except Exception:
|
|
51
|
+
return ""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def build_snapshot(report: dict, *, ref: Optional[str] = None) -> dict:
|
|
55
|
+
"""Freeze the budgetable readiness fields of a migrate-check report dict.
|
|
56
|
+
|
|
57
|
+
``report`` is the dict from ``MigrateCheckReport.to_dict()``. Only the
|
|
58
|
+
timeseries-relevant fields are kept — the snapshot is a point on a line, not
|
|
59
|
+
a full re-storable report.
|
|
60
|
+
"""
|
|
61
|
+
snap: dict[str, Any] = {
|
|
62
|
+
"schema_version": READINESS_SNAPSHOT_SCHEMA,
|
|
63
|
+
"captured_at": _utc_now(),
|
|
64
|
+
"ref": ref or "",
|
|
65
|
+
"git_head": report.get("git_head") or "",
|
|
66
|
+
"repo_id": report.get("repo_id") or "",
|
|
67
|
+
"tool_version": _tool_version(),
|
|
68
|
+
"readiness_aggregate": report.get("readiness_aggregate"),
|
|
69
|
+
"applicable_dimensions": sorted(report.get("applicable_dimensions") or []),
|
|
70
|
+
"blocking_count": report.get("blocking_count"),
|
|
71
|
+
"estimated_effort_days": report.get("estimated_effort_days"),
|
|
72
|
+
}
|
|
73
|
+
for f in _SCORE_FIELDS:
|
|
74
|
+
snap[f] = report.get(f)
|
|
75
|
+
return snap
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _snapshot_filename(snapshot: dict) -> str:
|
|
79
|
+
"""Timestamp-stemmed filename (microsecond-unique) so re-scans never collide."""
|
|
80
|
+
stem = snapshot.get("captured_at") or "snapshot"
|
|
81
|
+
safe = "".join(c if (c.isalnum() or c in "-_.") else "-" for c in str(stem))
|
|
82
|
+
return f"{safe}.json"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def write_snapshot(snapshot: dict, out_dir: Path) -> Path:
|
|
86
|
+
"""Write `snapshot` as deterministic JSON under `out_dir`; return the path."""
|
|
87
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
88
|
+
path = out_dir / _snapshot_filename(snapshot)
|
|
89
|
+
path.write_text(
|
|
90
|
+
json.dumps(snapshot, sort_keys=True, indent=2) + "\n", encoding="utf-8"
|
|
91
|
+
)
|
|
92
|
+
return path
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def load_snapshot(path: Path) -> dict:
|
|
96
|
+
"""Load and validate one readiness-snapshot artifact."""
|
|
97
|
+
data = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
98
|
+
if not isinstance(data, dict) or data.get("schema_version") != READINESS_SNAPSHOT_SCHEMA:
|
|
99
|
+
raise ValueError(
|
|
100
|
+
f"{path}: not a {READINESS_SNAPSHOT_SCHEMA} artifact "
|
|
101
|
+
f"(schema_version={data.get('schema_version') if isinstance(data, dict) else type(data).__name__})."
|
|
102
|
+
)
|
|
103
|
+
return data
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def load_snapshots_dir(directory: Path) -> list[dict]:
|
|
107
|
+
"""Load every readiness snapshot in `directory`, sorted by capture time.
|
|
108
|
+
|
|
109
|
+
Malformed / foreign JSON files are skipped, not fatal — a snapshots directory
|
|
110
|
+
can hold unrelated files without breaking the trend.
|
|
111
|
+
"""
|
|
112
|
+
out: list[dict] = []
|
|
113
|
+
for p in sorted(Path(directory).glob("*.json")):
|
|
114
|
+
try:
|
|
115
|
+
out.append(load_snapshot(p))
|
|
116
|
+
except (ValueError, json.JSONDecodeError, OSError):
|
|
117
|
+
continue
|
|
118
|
+
out.sort(key=lambda s: (str(s.get("captured_at", "")), str(s.get("git_head", ""))))
|
|
119
|
+
return out
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def build_readiness_trend(snapshots: list[dict]) -> dict:
|
|
123
|
+
"""Project a readiness trend across time-ordered snapshots.
|
|
124
|
+
|
|
125
|
+
Emits one point per snapshot plus first→last deltas for the numeric tracks.
|
|
126
|
+
Movement only — no improving/degrading verdict.
|
|
127
|
+
"""
|
|
128
|
+
points = [
|
|
129
|
+
{
|
|
130
|
+
"captured_at": s.get("captured_at"),
|
|
131
|
+
"ref": s.get("ref") or "",
|
|
132
|
+
"git_head": s.get("git_head") or "",
|
|
133
|
+
"tool_version": s.get("tool_version") or "",
|
|
134
|
+
"readiness_aggregate": s.get("readiness_aggregate"),
|
|
135
|
+
"applicable_dimensions": list(s.get("applicable_dimensions") or []),
|
|
136
|
+
**{f: s.get(f) for f in _DELTA_FIELDS},
|
|
137
|
+
}
|
|
138
|
+
for s in snapshots
|
|
139
|
+
]
|
|
140
|
+
|
|
141
|
+
deltas: dict[str, Any] = {}
|
|
142
|
+
if len(snapshots) >= 2:
|
|
143
|
+
first, last = snapshots[0], snapshots[-1]
|
|
144
|
+
for f in _DELTA_FIELDS:
|
|
145
|
+
a, b = first.get(f), last.get(f)
|
|
146
|
+
if isinstance(a, (int, float)) and isinstance(b, (int, float)):
|
|
147
|
+
deltas[f] = round(b - a, 3) if isinstance(b - a, float) else b - a
|
|
148
|
+
else:
|
|
149
|
+
deltas[f] = None # honest: value missing at an endpoint → no delta
|
|
150
|
+
|
|
151
|
+
# Comparability: the readiness_score is a min over applicable dimensions, so it
|
|
152
|
+
# is only strictly comparable when the applicable set is unchanged end-to-end.
|
|
153
|
+
dims_changed = False
|
|
154
|
+
if len(snapshots) >= 2:
|
|
155
|
+
dims_changed = (
|
|
156
|
+
sorted(snapshots[0].get("applicable_dimensions") or [])
|
|
157
|
+
!= sorted(snapshots[-1].get("applicable_dimensions") or [])
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
notes = ["Movement only — the series reports change, not an improving/degrading verdict."]
|
|
161
|
+
if dims_changed:
|
|
162
|
+
notes.append(
|
|
163
|
+
"applicable_dimensions changed across the series — readiness_score endpoints "
|
|
164
|
+
"are not strictly comparable (the score is a min over a different dimension set)."
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
return {
|
|
168
|
+
"schema_version": READINESS_TREND_SCHEMA,
|
|
169
|
+
"count": len(snapshots),
|
|
170
|
+
"span": (
|
|
171
|
+
{"from": snapshots[0].get("captured_at"), "to": snapshots[-1].get("captured_at")}
|
|
172
|
+
if snapshots else {"from": None, "to": None}
|
|
173
|
+
),
|
|
174
|
+
"points": points,
|
|
175
|
+
"deltas": deltas,
|
|
176
|
+
"comparable": (not dims_changed) if len(snapshots) >= 2 else None,
|
|
177
|
+
"notes": notes,
|
|
178
|
+
}
|
sourcecode/telemetry/__init__.py
CHANGED
|
@@ -78,6 +78,7 @@ def record(
|
|
|
78
78
|
success: bool = True,
|
|
79
79
|
error_kind: Optional[str] = None,
|
|
80
80
|
feature: Optional[str] = None,
|
|
81
|
+
tool: Optional[str] = None,
|
|
81
82
|
repo_size: Optional[str] = None,
|
|
82
83
|
) -> None:
|
|
83
84
|
"""Record a telemetry event. Fire-and-forget — never blocks or raises.
|
|
@@ -107,6 +108,7 @@ def record(
|
|
|
107
108
|
success=success,
|
|
108
109
|
error_kind=error_kind,
|
|
109
110
|
feature=feature,
|
|
111
|
+
tool=tool,
|
|
110
112
|
install=get_install_id(),
|
|
111
113
|
session=_SESSION,
|
|
112
114
|
)
|
sourcecode/telemetry/consent.py
CHANGED
|
@@ -20,7 +20,8 @@ _NOTICE = """\
|
|
|
20
20
|
Anonymous usage metrics help improve sourcecode.
|
|
21
21
|
|
|
22
22
|
Collected: tool version, Python version, OS, commands used,
|
|
23
|
-
flags used, approximate repo size,
|
|
23
|
+
flags used, MCP tool names invoked, approximate repo size,
|
|
24
|
+
execution duration, errors.
|
|
24
25
|
|
|
25
26
|
Never collected: source code, file paths, file names, secrets,
|
|
26
27
|
tokens, environment variables, or any repository content.
|
sourcecode/telemetry/events.py
CHANGED
|
@@ -60,6 +60,8 @@ class TelemetryEvent:
|
|
|
60
60
|
success — True/False
|
|
61
61
|
error_kind — exception class name only (no message, no traceback)
|
|
62
62
|
feature — gated feature / task name (closed categorical set) or None
|
|
63
|
+
tool — MCP tool name for mcp_tool_invoked events, or None. A product
|
|
64
|
+
identifier from our own tool registry, never user input.
|
|
63
65
|
install — stable anonymous install UUID (random, no PII); enables
|
|
64
66
|
unique-user / conversion / retention metrics
|
|
65
67
|
session — 8-char random hex, ephemeral, NOT persisted
|
|
@@ -79,5 +81,6 @@ class TelemetryEvent:
|
|
|
79
81
|
success: bool = True
|
|
80
82
|
error_kind: Optional[str] = None
|
|
81
83
|
feature: Optional[str] = None
|
|
84
|
+
tool: Optional[str] = None
|
|
82
85
|
install: str = ""
|
|
83
86
|
session: str = ""
|
sourcecode/telemetry/filters.py
CHANGED
|
@@ -56,7 +56,7 @@ _SAFE_FLAGS: frozenset[str] = frozenset({
|
|
|
56
56
|
|
|
57
57
|
_SAFE_OS: frozenset[str] = frozenset({"linux", "macos", "windows", "other"})
|
|
58
58
|
_SAFE_ARCH: frozenset[str] = frozenset({"x64", "arm64", "other"})
|
|
59
|
-
_SAFE_CMD: frozenset[str] = frozenset({"analyze", "prepare-context", "telemetry", "unknown"})
|
|
59
|
+
_SAFE_CMD: frozenset[str] = frozenset({"analyze", "prepare-context", "telemetry", "mcp", "unknown"})
|
|
60
60
|
_SAFE_EVENTS: frozenset[str] = frozenset({
|
|
61
61
|
"command_executed",
|
|
62
62
|
"execution_completed",
|
|
@@ -65,6 +65,7 @@ _SAFE_EVENTS: frozenset[str] = frozenset({
|
|
|
65
65
|
"telemetry_disabled",
|
|
66
66
|
"gate_blocked",
|
|
67
67
|
"activation",
|
|
68
|
+
"mcp_tool_invoked",
|
|
68
69
|
})
|
|
69
70
|
# Closed set of gated features / task names. Used to learn which capability
|
|
70
71
|
# drives Pro demand. All values are fixed product identifiers — no user data.
|
|
@@ -109,6 +110,20 @@ def _safe_error_kind(value: str | None) -> str | None:
|
|
|
109
110
|
return name[:64] if name else None
|
|
110
111
|
|
|
111
112
|
|
|
113
|
+
def _safe_tool(value: str | None) -> str | None:
|
|
114
|
+
"""Keep an MCP tool name only if it looks like one of our own identifiers.
|
|
115
|
+
|
|
116
|
+
Tool names are product identifiers minted by the MCP registry (snake_case
|
|
117
|
+
ASCII, e.g. `get_compact_context`) — a caller cannot invoke a name we did not
|
|
118
|
+
register, so no user data can reach this field. The shape is still enforced
|
|
119
|
+
here rather than trusted: anything with a path separator, whitespace, an
|
|
120
|
+
unusual character or excess length is dropped outright.
|
|
121
|
+
"""
|
|
122
|
+
if not value:
|
|
123
|
+
return None
|
|
124
|
+
return value if re.match(r"^[a-z][a-z0-9_]{0,47}$", value) else None
|
|
125
|
+
|
|
126
|
+
|
|
112
127
|
def _safe_session(value: str) -> str:
|
|
113
128
|
"""Session ID must be a short hex string only."""
|
|
114
129
|
if re.match(r"^[0-9a-f]{1,16}$", value):
|
|
@@ -151,6 +166,10 @@ def sanitize(event: TelemetryEvent) -> dict[str, Any]:
|
|
|
151
166
|
if event.feature:
|
|
152
167
|
safe["feature"] = _safe_str(event.feature, _SAFE_FEATURES, "other")
|
|
153
168
|
|
|
169
|
+
tool = _safe_tool(event.tool)
|
|
170
|
+
if tool:
|
|
171
|
+
safe["tool"] = tool
|
|
172
|
+
|
|
154
173
|
install = _safe_install(event.install)
|
|
155
174
|
if install:
|
|
156
175
|
safe["install"] = install
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Token accounting for agent-facing responses (Bloque C / C1).
|
|
2
|
+
|
|
3
|
+
Two numbers, both measured, neither invented:
|
|
4
|
+
|
|
5
|
+
- ``emitted`` — tokens the response itself costs the agent.
|
|
6
|
+
- ``baseline_raw_read`` — tokens the same agent would have spent reading, raw,
|
|
7
|
+
the files this response is derived from and points at (the counterfactual).
|
|
8
|
+
|
|
9
|
+
``saved = max(0, baseline_raw_read - emitted)``.
|
|
10
|
+
|
|
11
|
+
Honesty contract (F-1 / measured-not-ROI):
|
|
12
|
+
|
|
13
|
+
- It is an ESTIMATE and declares itself as one (``basis``, ``token_model``).
|
|
14
|
+
- The token model is ``chars/4`` — an approximation, declared, no tokenizer
|
|
15
|
+
dependency and no per-vendor tokenizer branching.
|
|
16
|
+
- It is an UPPER BOUND of the saving, clamped at 0. Never inflated to the whole
|
|
17
|
+
repository, never expressed as money or ROI.
|
|
18
|
+
- It is deterministic: a pure function of (response, files on disk).
|
|
19
|
+
|
|
20
|
+
VAI: file selection is structural — a string counts as a citation only when it
|
|
21
|
+
resolves to a real file under the repo root; anchors come from the existing
|
|
22
|
+
manifest primitives in :mod:`sourcecode.scanner`. No branching on proprietary
|
|
23
|
+
names.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import re
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from typing import Any, Iterable
|
|
30
|
+
|
|
31
|
+
#: Declared token model. Approximation, not a tokenizer.
|
|
32
|
+
TOKEN_MODEL = "chars/4"
|
|
33
|
+
|
|
34
|
+
#: Declared counterfactual basis.
|
|
35
|
+
ECONOMY_BASIS = "counterfactual_raw_read_estimate:cited_files+build_anchors"
|
|
36
|
+
|
|
37
|
+
#: Bound on how many strings are inspected while harvesting citations.
|
|
38
|
+
_MAX_STRINGS_SCANNED = 20_000
|
|
39
|
+
|
|
40
|
+
#: Bound on how many distinct files are stat-ed for the baseline.
|
|
41
|
+
_MAX_FILES_COUNTED = 2_000
|
|
42
|
+
|
|
43
|
+
#: Trailing ``:123`` / ``:123:45`` location suffixes on cited paths.
|
|
44
|
+
_LOC_SUFFIX = re.compile(r":\d+(?::\d+)?$")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def estimate_tokens(text: str | bytes) -> int:
|
|
48
|
+
"""Estimate tokens for *text* under the declared ``chars/4`` model.
|
|
49
|
+
|
|
50
|
+
Uses UTF-8 byte length so multi-byte content is not under-counted.
|
|
51
|
+
"""
|
|
52
|
+
if isinstance(text, str):
|
|
53
|
+
raw = text.encode("utf-8")
|
|
54
|
+
else:
|
|
55
|
+
raw = text
|
|
56
|
+
return len(raw) // 4
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _iter_strings(node: Any, budget: list[int]) -> Iterable[str]:
|
|
60
|
+
"""Yield every string in a nested dict/list payload, bounded by *budget*."""
|
|
61
|
+
if budget[0] <= 0:
|
|
62
|
+
return
|
|
63
|
+
if isinstance(node, str):
|
|
64
|
+
budget[0] -= 1
|
|
65
|
+
yield node
|
|
66
|
+
elif isinstance(node, dict):
|
|
67
|
+
for value in node.values():
|
|
68
|
+
yield from _iter_strings(value, budget)
|
|
69
|
+
if budget[0] <= 0:
|
|
70
|
+
return
|
|
71
|
+
elif isinstance(node, (list, tuple)):
|
|
72
|
+
for value in node:
|
|
73
|
+
yield from _iter_strings(value, budget)
|
|
74
|
+
if budget[0] <= 0:
|
|
75
|
+
return
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def collect_cited_files(payload: Any, root: Path) -> list[str]:
|
|
79
|
+
"""Repo-relative paths cited by *payload* that resolve to real files.
|
|
80
|
+
|
|
81
|
+
A string is a citation only if it resolves to an existing regular file
|
|
82
|
+
inside *root* — structural test, no name heuristics. Fully-qualified class
|
|
83
|
+
names, prose and identifiers therefore never count.
|
|
84
|
+
"""
|
|
85
|
+
try:
|
|
86
|
+
root = root.resolve()
|
|
87
|
+
except OSError:
|
|
88
|
+
return []
|
|
89
|
+
seen: set[str] = set()
|
|
90
|
+
for text in _iter_strings(payload, [_MAX_STRINGS_SCANNED]):
|
|
91
|
+
candidate = _LOC_SUFFIX.sub("", text.strip())
|
|
92
|
+
if not candidate or len(candidate) > 512 or "\n" in candidate:
|
|
93
|
+
continue
|
|
94
|
+
if "/" not in candidate and "\\" not in candidate:
|
|
95
|
+
continue # bare names are ambiguous — never guessed
|
|
96
|
+
try:
|
|
97
|
+
path = Path(candidate)
|
|
98
|
+
resolved = (root / path).resolve() if not path.is_absolute() else path.resolve()
|
|
99
|
+
rel = resolved.relative_to(root)
|
|
100
|
+
except (OSError, ValueError):
|
|
101
|
+
continue
|
|
102
|
+
if not resolved.is_file():
|
|
103
|
+
continue
|
|
104
|
+
seen.add(str(rel))
|
|
105
|
+
if len(seen) >= _MAX_FILES_COUNTED:
|
|
106
|
+
break
|
|
107
|
+
return sorted(seen)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def collect_anchor_files(root: Path) -> list[str]:
|
|
111
|
+
"""Build manifests at depth 0–1 — the files any agent reads to orient.
|
|
112
|
+
|
|
113
|
+
Reuses the scanner's manifest primitives rather than a private name list.
|
|
114
|
+
"""
|
|
115
|
+
try:
|
|
116
|
+
from sourcecode.scanner import DEFAULT_EXCLUDES, find_manifest_paths
|
|
117
|
+
except Exception:
|
|
118
|
+
return []
|
|
119
|
+
try:
|
|
120
|
+
root = root.resolve()
|
|
121
|
+
paths = find_manifest_paths(root, DEFAULT_EXCLUDES)
|
|
122
|
+
except Exception:
|
|
123
|
+
return []
|
|
124
|
+
out: set[str] = set()
|
|
125
|
+
for p in paths:
|
|
126
|
+
try:
|
|
127
|
+
out.add(str(Path(p).resolve().relative_to(root)))
|
|
128
|
+
except (OSError, ValueError):
|
|
129
|
+
continue
|
|
130
|
+
return sorted(out)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _sum_file_tokens(root: Path, rel_paths: Iterable[str]) -> tuple[int, int]:
|
|
134
|
+
"""(tokens, files_counted) for *rel_paths*, by on-disk size under the model."""
|
|
135
|
+
total = 0
|
|
136
|
+
counted = 0
|
|
137
|
+
for rel in rel_paths:
|
|
138
|
+
try:
|
|
139
|
+
size = (root / rel).stat().st_size
|
|
140
|
+
except OSError:
|
|
141
|
+
continue
|
|
142
|
+
total += size // 4
|
|
143
|
+
counted += 1
|
|
144
|
+
return total, counted
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def token_economy(
|
|
148
|
+
rendered: str,
|
|
149
|
+
payload: Any,
|
|
150
|
+
root: Path,
|
|
151
|
+
*,
|
|
152
|
+
extra_files: Iterable[str] = (),
|
|
153
|
+
) -> dict[str, Any]:
|
|
154
|
+
"""Measured token economy block for a response.
|
|
155
|
+
|
|
156
|
+
Args:
|
|
157
|
+
rendered: the response exactly as it will be emitted (pre-injection).
|
|
158
|
+
payload: the parsed response, harvested for file citations.
|
|
159
|
+
root: repository root the citations are relative to.
|
|
160
|
+
extra_files: additional repo-relative paths to count in the baseline.
|
|
161
|
+
|
|
162
|
+
Returns a dict destined for ``_meta.token_economy``.
|
|
163
|
+
"""
|
|
164
|
+
try:
|
|
165
|
+
root = Path(root).resolve()
|
|
166
|
+
except OSError:
|
|
167
|
+
root = Path(root)
|
|
168
|
+
|
|
169
|
+
files = set(collect_cited_files(payload, root))
|
|
170
|
+
files.update(collect_anchor_files(root))
|
|
171
|
+
for extra in extra_files:
|
|
172
|
+
files.add(str(extra))
|
|
173
|
+
|
|
174
|
+
baseline, counted = _sum_file_tokens(root, sorted(files))
|
|
175
|
+
emitted = estimate_tokens(rendered)
|
|
176
|
+
return {
|
|
177
|
+
"emitted": emitted,
|
|
178
|
+
"baseline_raw_read": baseline,
|
|
179
|
+
"saved": max(0, baseline - emitted),
|
|
180
|
+
"files_counted": counted,
|
|
181
|
+
"basis": ECONOMY_BASIS,
|
|
182
|
+
"token_model": TOKEN_MODEL,
|
|
183
|
+
"estimate": True,
|
|
184
|
+
"note": (
|
|
185
|
+
"Upper-bound estimate: tokens an agent would spend reading the cited "
|
|
186
|
+
"files and build manifests raw, minus this response. Measured from "
|
|
187
|
+
"file sizes under the declared token model — not exact, not a "
|
|
188
|
+
"monetary figure. 'emitted' excludes this block."
|
|
189
|
+
),
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def inject_token_economy(content: str, root: Path) -> str:
|
|
194
|
+
"""Add ``_meta.token_economy`` to a JSON response string.
|
|
195
|
+
|
|
196
|
+
Returns *content* unchanged on any parse failure or non-dict payload, so a
|
|
197
|
+
YAML/plain response passes straight through.
|
|
198
|
+
"""
|
|
199
|
+
import json as _json
|
|
200
|
+
|
|
201
|
+
try:
|
|
202
|
+
payload = _json.loads(content)
|
|
203
|
+
if not isinstance(payload, dict):
|
|
204
|
+
return content
|
|
205
|
+
economy = token_economy(content, payload, root)
|
|
206
|
+
meta = payload.get("_meta")
|
|
207
|
+
if not isinstance(meta, dict):
|
|
208
|
+
meta = {}
|
|
209
|
+
meta["token_economy"] = economy
|
|
210
|
+
payload["_meta"] = meta
|
|
211
|
+
return _json.dumps(payload, indent=2, ensure_ascii=False)
|
|
212
|
+
except Exception:
|
|
213
|
+
return content
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
sourcecode/__init__.py,sha256=
|
|
1
|
+
sourcecode/__init__.py,sha256=28Itwr5b8-1VcEBeVmo63jQmi5T8llZbG2Sve5wnruA,309
|
|
2
2
|
sourcecode/adaptive_scanner.py,sha256=yJBKjNpkY6bpueYJ2YnRezen3sYZDecEt7WaaNWdqug,9466
|
|
3
3
|
sourcecode/archetype.py,sha256=UrmiONHfqLl7UfGzjECoQBO28hcM_1rvzn0hCt0EtjU,35033
|
|
4
4
|
sourcecode/architectural_baseline.py,sha256=7QzJri4pbL3nzAn9gNutZW6mZR8HHD6H2C2H1EEwYPE,17904
|
|
@@ -13,7 +13,7 @@ sourcecode/canonical_ir.py,sha256=LdP_Ri3rl0M5p-MY0ypVvQwhq1WuUF1FmaHid7zyzEg,29
|
|
|
13
13
|
sourcecode/change_plan.py,sha256=MNgNyu4zrLGvBBaXPwyCh-T2ZGaUQX3Hm4W87uuhZ3w,7750
|
|
14
14
|
sourcecode/cir_graphs.py,sha256=9G0HHj1kw2325IDyzo2OpX73BNswEckecf4MZUXB4JM,12078
|
|
15
15
|
sourcecode/classifier.py,sha256=JBzPwSSrDG-tUHAbcKB678HRbjLpD-ohzbzzO62mgpo,20114
|
|
16
|
-
sourcecode/cli.py,sha256=
|
|
16
|
+
sourcecode/cli.py,sha256=BgfbJPIpItFNmw1HzcasWxi1emhgNJ4ne7wPmQIQfdE,370433
|
|
17
17
|
sourcecode/code_notes_analyzer.py,sha256=EJemNCNc9Dn-1RZYu-aNbK0ELzmsyC4s6FdHi3XyNEI,9392
|
|
18
18
|
sourcecode/compare.py,sha256=jdePg0dNCxFpysiTGMsROL5XABLjy_zVUgVeySvdb0o,7514
|
|
19
19
|
sourcecode/confidence_analyzer.py,sha256=vnbPI-20FnHdjO6STxHW8fbaxmB4A7y58io63ibFZjc,21586
|
|
@@ -37,7 +37,7 @@ sourcecode/error_schema.py,sha256=uwosfNaSujtYm11_732Hu92z5ITV040fQDaIyefSvR4,16
|
|
|
37
37
|
sourcecode/evidence_provider.py,sha256=GSSL44JEaouO5AHks2sB3d1YvC9xIKIld1yBYxZpXxo,4277
|
|
38
38
|
sourcecode/explain.py,sha256=HnHWVTNNf9fzeR3FP-A-eXKeKZvBcUEuODgIO3rJQkQ,22312
|
|
39
39
|
sourcecode/file_chunker.py,sha256=3vkM3mDQ5eE_yTPvUgjyjpGFBIjkW6_mrBmIbrylnA8,16444
|
|
40
|
-
sourcecode/file_classifier.py,sha256=
|
|
40
|
+
sourcecode/file_classifier.py,sha256=pJCeN9KqWpAwKMCgGP4KDsBjuWMeo4zlbj5fv3hk9dA,15587
|
|
41
41
|
sourcecode/filter_surface.py,sha256=qunis3-yY_nhOIj5K3mVFtbSyKfc6ugIX23OSxTolwE,7079
|
|
42
42
|
sourcecode/format_contract.py,sha256=U83J_LtrpMQSN3_H_HBBxpanlo7xnfvbZw7sP3ew8_M,3678
|
|
43
43
|
sourcecode/fqn_utils.py,sha256=XLU7zDkNBXz_RZkIUNfpPmp1nekWtqP-fxV92tDV1vg,2158
|
|
@@ -47,7 +47,7 @@ sourcecode/graph_evidence.py,sha256=rENNsYRZeNstX_ExNCLlbHJAruFQwxo5d00x6wO3xwI,
|
|
|
47
47
|
sourcecode/hibernate_strat.py,sha256=h0leIhlWvSjYq3F99LxvLIDLrJ-xPYxWAREG4LkqZ-4,61190
|
|
48
48
|
sourcecode/jdk_exports.py,sha256=fCrlwNAXUT9gge_joq6kMnY3zJxYB2pxqy-0w3o3MJI,874
|
|
49
49
|
sourcecode/license.py,sha256=keFuwNxdAtvK2Ds91Wl79GMYxuxWYnN5Wbw1qBpaoUI,24896
|
|
50
|
-
sourcecode/mcp_nudge.py,sha256=
|
|
50
|
+
sourcecode/mcp_nudge.py,sha256=lKemOqK_wny2u7Ymcr2Idi5Kx8pXY02jCi-_nJYLGMg,2992
|
|
51
51
|
sourcecode/metrics_analyzer.py,sha256=m0ENgtqKeBL17kUIK3fmGkgo7UfXBNHxCMj0H_Y5K7c,22750
|
|
52
52
|
sourcecode/migrate_check.py,sha256=hsKNwipl-bSqRsYqG1O_CLbXpMlqtnN-Bif3hpHpnjg,108676
|
|
53
53
|
sourcecode/openapi_surface.py,sha256=BTt0K-woZbkbWTN77IkqeBm_Okag9owR0848fmot8sk,16207
|
|
@@ -60,6 +60,7 @@ sourcecode/pr_impact.py,sha256=dCDVw83EDbyVf6F9ZmEQmsFz8ruVH7d4mpeKQCIZHM0,16805
|
|
|
60
60
|
sourcecode/prepare_context.py,sha256=GkdI_a0RG8Y8dh8x6SpCNKZwWi7qs8sqAbFSYJegBxY,225473
|
|
61
61
|
sourcecode/progress.py,sha256=qn30sWaHOkjTgXsSBmiPkz7Rsbwc5oSlIe6JNEMYp_k,3149
|
|
62
62
|
sourcecode/ranking_engine.py,sha256=ZAucq_YX2KkWUuAZf4P0lhtQ_38vEFnUhuGtSZd1S0E,12970
|
|
63
|
+
sourcecode/readiness_timeline.py,sha256=T5NKLuRHS83lPslE2kMs_RSVENd8jZJ11N1c9gpkpH8,7244
|
|
63
64
|
sourcecode/reconciliation.py,sha256=GU-1PTcVr8zcbtC7BASfpHcZndP9AdboXPNQiBc0fzo,34251
|
|
64
65
|
sourcecode/redactor.py,sha256=SB4hwIvg8h-hvcqKcDWaZvA-aSyn-at-BIRwa0tUv5E,3227
|
|
65
66
|
sourcecode/relevance_scorer.py,sha256=0AgEt4KrV73nioMqBgjhGjtY7L2C7L7cSyKtj3IKcrw,9408
|
|
@@ -85,6 +86,7 @@ sourcecode/spring_security_audit.py,sha256=Rk-aSohezdc7YDYbSoJquVnwpkDB8ty1BCD-4
|
|
|
85
86
|
sourcecode/spring_semantic.py,sha256=jteQ1PkY9ArFJv0embg_jBIdbOxqrk9mQ2Xz8OF_FKA,14214
|
|
86
87
|
sourcecode/spring_tx_analyzer.py,sha256=lp0h5Pzzd3fPxHAAagMmG8PqR1-v2uVpFIG1o7tCWQs,41148
|
|
87
88
|
sourcecode/summarizer.py,sha256=sr0-tfecFKCr-fSkPPWbl-t9HC7SY2ZxkjnnXX8DB2A,26621
|
|
89
|
+
sourcecode/token_estimate.py,sha256=ZP3C54aQLExuf32Yvrjok2TAS2oSSyNVRA_HjhsD6Eg,7165
|
|
88
90
|
sourcecode/tree_utils.py,sha256=8GAkIfQAsvtEudIeW1l4ooH_oRtrWR8cpJQJsEa_Pfw,2093
|
|
89
91
|
sourcecode/type_usage_surface.py,sha256=51IrKRQoIoRnlsiDjHnqpJBn2rc6E59aRhgS0HTzAF0,4428
|
|
90
92
|
sourcecode/validation_inference.py,sha256=KN3tEBQzxNinwlo7buer6lx5Kqdn71JAs_wuu44McfU,17475
|
|
@@ -118,12 +120,12 @@ sourcecode/mcp/__init__.py,sha256=XU4HfRGbdid8wdUA0x_4f7uKZD1z3mv_XUY_WU_T9Mw,17
|
|
|
118
120
|
sourcecode/mcp/orchestrator.py,sha256=diVoQgn24QmgPL3Ev8Sp6hsvh02OqY3MktHXOzrlodo,36525
|
|
119
121
|
sourcecode/mcp/registry.py,sha256=OCUCRLPRM176FCw850rF7dNh74f5Zj0h3a9lpxrf8mI,68171
|
|
120
122
|
sourcecode/mcp/runner.py,sha256=RGimCwhLyHxjCvH7q72y12HWEnATXZLC-NBWhkzlu0I,2852
|
|
121
|
-
sourcecode/mcp/server.py,sha256=
|
|
123
|
+
sourcecode/mcp/server.py,sha256=P6JqQrlat2jP8OK0ysxSWNwqfUYF-bTc3X05FszR1_4,65180
|
|
122
124
|
sourcecode/mcp/onboarding/__init__.py,sha256=sj2PWqEBmMc4zBNkomg89WtL0M6S7A9yb7_wAuSWNP4,66
|
|
123
|
-
sourcecode/mcp/onboarding/applier.py,sha256=
|
|
125
|
+
sourcecode/mcp/onboarding/applier.py,sha256=Sx9vHTaXr_M1Bs4uMjO7qa_a2_p4J9oz93oJsKQ7ds4,3298
|
|
124
126
|
sourcecode/mcp/onboarding/backup.py,sha256=ihqGOR8QTX8HASRSEDyfFyXr5bkXrygPHamv4p9KTmk,1452
|
|
125
|
-
sourcecode/mcp/onboarding/detector.py,sha256=
|
|
126
|
-
sourcecode/mcp/onboarding/planner.py,sha256=
|
|
127
|
+
sourcecode/mcp/onboarding/detector.py,sha256=_WMlVecYE8jvNof3yQliICTldNg_rygnYZRHC2FAAd8,4875
|
|
128
|
+
sourcecode/mcp/onboarding/planner.py,sha256=GEKLxVP244hLT1hHEoxMUAz3QIOg8vZOEst9Bh3VqTs,1372
|
|
127
129
|
sourcecode/retrieval/__init__.py,sha256=h5pKDzXOlu_NQy-YxOaqMSD6UDNoYsmsYSoNLRRZ5uM,5478
|
|
128
130
|
sourcecode/retrieval/context.py,sha256=hjm_RxReZar6GZvVVg4XNR9sVMOUuilbi1Jt7iUoerU,11116
|
|
129
131
|
sourcecode/retrieval/errors.py,sha256=U-zqxzlzCWe3NYCNOBASTSPn1xS8j9D7LfwpGBiznL4,1452
|
|
@@ -142,14 +144,14 @@ sourcecode/retrieval/steps_impact.py,sha256=M0AwU5Ue7nkj-A7DYgQI2BP0BV1b0TFZRrYI
|
|
|
142
144
|
sourcecode/retrieval/steps_intf.py,sha256=Jx5Gnm7UT1jOE7YloD0ee3d8PuliX5FgpdvlHh36OBs,9341
|
|
143
145
|
sourcecode/retrieval/steps_struct.py,sha256=jO9L49hziTlS_xhuhCGUavgyAL3ZAnJSOZB-UA4GaRM,13893
|
|
144
146
|
sourcecode/retrieval/steps_txsec.py,sha256=ltyxFAFLHfUglGLGz7i3tY_bovBoKo1d6R2t1xZTcnk,13608
|
|
145
|
-
sourcecode/telemetry/__init__.py,sha256=
|
|
147
|
+
sourcecode/telemetry/__init__.py,sha256=a-cy2Gypx9fzJfBHK-e3hstYi6FLNlUj-k62yWyugMY,3461
|
|
146
148
|
sourcecode/telemetry/config.py,sha256=u_zctgmA1pVT3Ai5jm-pdLY_5kJTRg4Xq2bvIbGBQE4,3727
|
|
147
|
-
sourcecode/telemetry/consent.py,sha256=
|
|
148
|
-
sourcecode/telemetry/events.py,sha256=
|
|
149
|
-
sourcecode/telemetry/filters.py,sha256=
|
|
149
|
+
sourcecode/telemetry/consent.py,sha256=GuhCiA_x2n3LwGBkZnvyIwtFcf5Zwa8O0DdLqkATIr0,2220
|
|
150
|
+
sourcecode/telemetry/events.py,sha256=4_yeO58U-Cwc1Qb27VB0_EjhmroY0k91n3_VGxeALB8,2776
|
|
151
|
+
sourcecode/telemetry/filters.py,sha256=RzxauTz8HliO4BllQnXEXc7zTeqdCZi5MgqGEDuW7OQ,6570
|
|
150
152
|
sourcecode/telemetry/transport.py,sha256=4gGHsq0WeY9VywEZXA3vUxykfiYnw9uuqfjAAec7F8o,1681
|
|
151
|
-
sourcecode-2.6.
|
|
152
|
-
sourcecode-2.6.
|
|
153
|
-
sourcecode-2.6.
|
|
154
|
-
sourcecode-2.6.
|
|
155
|
-
sourcecode-2.6.
|
|
153
|
+
sourcecode-2.6.10.dist-info/METADATA,sha256=RQ1LFmY3uZZd9zTw3iA3rkChqyLlb5DjgH7VvpRTFXg,10852
|
|
154
|
+
sourcecode-2.6.10.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
|
|
155
|
+
sourcecode-2.6.10.dist-info/entry_points.txt,sha256=-JEAdChrK5We51kZcb7OaDcyil-dHBjBPL-NhuO-QY8,89
|
|
156
|
+
sourcecode-2.6.10.dist-info/licenses/LICENSE,sha256=7DdHrU9Z_3e7dSvq4ISijZNjnuHo5NIHNiHDouMQ9JU,10491
|
|
157
|
+
sourcecode-2.6.10.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|