sourcecode 2.5.15__tar.gz → 2.5.16__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sourcecode-2.5.15 → sourcecode-2.5.16}/CHANGELOG.md +42 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/PKG-INFO +1 -1
- {sourcecode-2.5.15 → sourcecode-2.5.16}/pyproject.toml +1 -1
- sourcecode-2.5.16/scripts/perf_harness.py +252 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/__init__.py +1 -1
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/ast_extractor.py +12 -11
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/cli.py +11 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/context_graph.py +35 -0
- sourcecode-2.5.16/src/sourcecode/perf.py +268 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/repository_ir.py +12 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/.github/workflows/build-windows.yml +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/.gitignore +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/.ruff.toml +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/CONTRIBUTING.md +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/LICENSE +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/README.md +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/SECURITY.md +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/raw +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/scripts/compare_integration_engines.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/scripts/customer_smoke_test.sh +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/scripts/generate_jdk_exports.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/adaptive_scanner.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/archetype.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/architecture_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/architecture_summary.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/cache.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/call_surface.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/caller_metrics.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/canonical_ir.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/cir_graphs.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/classifier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/code_notes_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/confidence_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/context_cache.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/context_scorer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/context_summarizer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/contract_model.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/contract_pipeline.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/coverage_parser.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/dependency_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/__init__.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/base.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/csproj_parser.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/dart.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/dotnet.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/elixir.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/go.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/heuristic.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/hybrid.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/java.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/jvm_ext.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/nodejs.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/parsers.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/php.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/project.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/python.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/ruby.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/rust.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/systems.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/terraform.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/detectors/tooling.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/doc_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/dynamic_argument_surface.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/endpoint_literals.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/endpoint_metrics.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/entrypoint_classifier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/env_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/error_schema.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/evidence_provider.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/explain.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/file_chunker.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/file_classifier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/format_contract.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/fqn_utils.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/git_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/graph_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/graph_evidence.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/hibernate_strat.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/jdk_exports.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/license.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/__init__.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/onboarding/__init__.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/onboarding/applier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/onboarding/backup.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/onboarding/detector.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/onboarding/planner.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/orchestrator.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/registry.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/runner.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp/server.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/mcp_nudge.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/metrics_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/migrate_check.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/openapi_surface.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/output_budget.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/path_filters.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/pr_comment_renderer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/pr_impact.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/prepare_context.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/progress.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/ranking_engine.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/reconciliation.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/redactor.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/relevance_scorer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/rename_refactor.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/repo_classifier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/__init__.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/context.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/errors.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/executor.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/planner.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/query.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/request.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/resolution.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/result.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/retriever.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/runtime.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_endpoint.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_graph.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_impact.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_intf.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_struct.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/retrieval/steps_txsec.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/ris.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/runtime_classifier.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/scanner.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/schema.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/security_config.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/security_posture.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/semantic_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/semantic_impact_engine.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/semantic_integration_engine.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/semantic_services.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/serializer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_event_topology.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_findings.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_impact.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_model.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_security_audit.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_semantic.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/spring_tx_analyzer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/summarizer.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/__init__.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/config.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/consent.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/events.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/filters.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/telemetry/transport.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/tree_utils.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/type_usage_surface.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/validation_inference.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/validation_surface.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/version_check.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/src/sourcecode/workspace.py +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/functions/README.md +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/functions/get-license/index.ts +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/functions/lemonsqueezy-webhook/index.ts +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/functions/telemetry/index.ts +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/sql/license_event_ordering.sql +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/sql/licensing_schema.sql +0 -0
- {sourcecode-2.5.15 → sourcecode-2.5.16}/supabase/sql/telemetry_events.sql +0 -0
|
@@ -1,5 +1,47 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [2.5.16] — 2026-07-19
|
|
4
|
+
|
|
5
|
+
**Performance campaign close-out: one real algorithmic win, cost-per-symbol
|
|
6
|
+
measurable, one candidate refuted by measurement.** The ADR-0006 harness localized
|
|
7
|
+
cold wall to the `analysis` phase (90–96% on every non-trivial repo); profiling
|
|
8
|
+
inside it found — and this release fixes — the one structural bottleneck, populates
|
|
9
|
+
the last empty baseline axes, and records a negative result that stopped a
|
|
10
|
+
non-winning change from landing.
|
|
11
|
+
|
|
12
|
+
### Performance
|
|
13
|
+
|
|
14
|
+
- **Killed an O(files × total_symbols) scan in per-file Java contract projection**
|
|
15
|
+
(`ast_extractor.py`, `context_graph.py`): `_java_contract_from_graph` ran once per
|
|
16
|
+
Java file and called the whole-repo `graph.symbols()`/`relations()` unfiltered, then
|
|
17
|
+
kept only its own file's rows. Added lazy once-built groupings
|
|
18
|
+
`ContextGraph.symbols_in_file(path)` / `relations_from(fqn)` (O(1) after build); the
|
|
19
|
+
projection now walks only edges sourced at its own types. Behavior-equivalent
|
|
20
|
+
(normalized old-vs-new on petclinic + keycloak; 1430 IR/spring/endpoint/impact tests
|
|
21
|
+
green). Cold wall p50, same host, speedup tracking file count as the O(files × N)
|
|
22
|
+
model predicts: neo4j 33.6→16.7s (2.0×), broadleaf 27.9→15.4s (1.8×),
|
|
23
|
+
keycloak 20.8→14.1s (1.5×), openmrs 7.9→5.8s (1.4×), ofbiz 19.4→15.0s (1.3×).
|
|
24
|
+
|
|
25
|
+
### Added
|
|
26
|
+
|
|
27
|
+
- **`repo.classes` / `repo.methods` populated in the perf baseline** from repo-ir's
|
|
28
|
+
symbol taxonomy (`repository_ir.py`, `scripts/perf_harness.py`): `analysis_meta` now
|
|
29
|
+
emits a deterministic `symbol_kinds` breakdown plus two rollups —
|
|
30
|
+
`classes` = class+interface+enum+annotation+record, `methods` = method+constructor —
|
|
31
|
+
and the harness reads them via one `repo-ir --summary-only` pass off the measured path
|
|
32
|
+
(never in `wall_ms`; repo-ir failure falls back to 0, an honest "unmeasured"). Closes
|
|
33
|
+
the last empty axes in the `perf-baseline-v1` schema, enabling cost-per-symbol.
|
|
34
|
+
|
|
35
|
+
### Notes
|
|
36
|
+
|
|
37
|
+
- **A body-fact-scanning optimization was investigated and reverted after measurement
|
|
38
|
+
showed no win.** cProfile flagged per-method body re-slicing (`_method_body` 44508× on
|
|
39
|
+
ofbiz) as a lead; a byte-identical single-pass rewrite (0 mismatches / 5555 files)
|
|
40
|
+
measured 7674→7721 ms — no improvement. The profiler had flagged a call-count-heavy but
|
|
41
|
+
wall-cheap region; the real cost is inherent regex/char scanning, already past its
|
|
42
|
+
algorithmic win. The change was dropped rather than accrue complexity on a hot,
|
|
43
|
+
output-critical path for no gain.
|
|
44
|
+
|
|
3
45
|
## [2.5.15] — 2026-07-19
|
|
4
46
|
|
|
5
47
|
**Bootstrap detection recovers the real entry point on multi-`main` repos, without
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Performance harness — runs the reference fleet under the ADR-0006 contract.
|
|
3
|
+
|
|
4
|
+
For every (command x repo x mode) cell it runs the CLI N times, records
|
|
5
|
+
wall-clock, peak RSS and (when the in-process recorder is armed) per-phase
|
|
6
|
+
timings, then writes one ``perf-baseline-v1`` JSON artifact.
|
|
7
|
+
|
|
8
|
+
Methodology is fixed by ADR-0006 §5: cold and warm are separate cells and never
|
|
9
|
+
averaged; the first warm run is discarded as cache fill; central tendency is the
|
|
10
|
+
median; p95 is always reported; cross-machine numbers are recorded, never diffed.
|
|
11
|
+
|
|
12
|
+
Usage:
|
|
13
|
+
python3 scripts/perf_harness.py --quick # petclinic only, 2 runs
|
|
14
|
+
python3 scripts/perf_harness.py # full fleet, 7 runs
|
|
15
|
+
python3 scripts/perf_harness.py --repos keycloak eureka --runs 5
|
|
16
|
+
python3 scripts/perf_harness.py --dry-run # print the plan, run nothing
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import resource
|
|
25
|
+
import shutil
|
|
26
|
+
import subprocess
|
|
27
|
+
import sys
|
|
28
|
+
import tempfile
|
|
29
|
+
import time
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
# Import the frozen collector from the package under test.
|
|
33
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
|
|
34
|
+
from sourcecode import __version__ as TOOL_VERSION # noqa: E402
|
|
35
|
+
from sourcecode import perf # noqa: E402
|
|
36
|
+
|
|
37
|
+
# ── Fleet definition ────────────────────────────────────────────────────────
|
|
38
|
+
# Directory name under the fleet root. Repos are pinned by their checked-out
|
|
39
|
+
# commit at run time (recorded in each artifact), so a size change is never
|
|
40
|
+
# mistaken for a perf change.
|
|
41
|
+
DEFAULT_FLEET_ROOT = Path(
|
|
42
|
+
os.environ.get("PERF_FLEET_ROOT", "/Users/user/Documents/workspace/testing")
|
|
43
|
+
)
|
|
44
|
+
FLEET: dict[str, str] = {
|
|
45
|
+
"petclinic": "spring-petclinic",
|
|
46
|
+
"broadleaf": "BroadleafCommerce",
|
|
47
|
+
"openmrs": "openmrs-core",
|
|
48
|
+
"keycloak": "keycloak",
|
|
49
|
+
"eureka": "eureka",
|
|
50
|
+
"jobrunr": "jobrunr",
|
|
51
|
+
"ofbiz": "ofbiz-framework",
|
|
52
|
+
"alfresco": "alfresco-community-repo",
|
|
53
|
+
"jenkins": "jenkins",
|
|
54
|
+
"neo4j": "neo4j",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
# Commands to profile. The empty string is the root analysis (`ask <repo>`);
|
|
58
|
+
# named entries are subcommands (`ask <cmd> <repo>`).
|
|
59
|
+
DEFAULT_COMMANDS = [""]
|
|
60
|
+
|
|
61
|
+
ASK_BIN = os.environ.get("ASK_BIN", "ask")
|
|
62
|
+
BYTES_PER_KB = 1024
|
|
63
|
+
# darwin ru_maxrss is bytes; linux reports kilobytes.
|
|
64
|
+
RSS_UNIT = 1 if sys.platform == "darwin" else BYTES_PER_KB
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def git_commit(repo_path: Path) -> str:
|
|
68
|
+
try:
|
|
69
|
+
out = subprocess.run(
|
|
70
|
+
["git", "-C", str(repo_path), "rev-parse", "--short", "HEAD"],
|
|
71
|
+
capture_output=True, text=True, timeout=15,
|
|
72
|
+
)
|
|
73
|
+
return out.stdout.strip() or "unknown"
|
|
74
|
+
except (subprocess.SubprocessError, OSError):
|
|
75
|
+
return "unknown"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def repo_size(repo_path: Path) -> dict[str, int]:
|
|
79
|
+
"""Repo scaling axes for cost-per-unit analysis.
|
|
80
|
+
|
|
81
|
+
``files`` / ``bytes`` are a cheap rglob over every ``*.java`` (the frozen file
|
|
82
|
+
axis — includes tests). ``classes`` / ``methods`` come from repo-ir's
|
|
83
|
+
``analysis_meta`` rollup (the IR owns the symbol taxonomy). This is one
|
|
84
|
+
``repo-ir --summary-only`` pass off the measured path; it never touches
|
|
85
|
+
``wall_ms``. If repo-ir fails the symbol counts fall back to 0 — an honest
|
|
86
|
+
"unmeasured", never a fabricated number.
|
|
87
|
+
"""
|
|
88
|
+
files = 0
|
|
89
|
+
total = 0
|
|
90
|
+
for p in repo_path.rglob("*.java"):
|
|
91
|
+
try:
|
|
92
|
+
total += p.stat().st_size
|
|
93
|
+
files += 1
|
|
94
|
+
except OSError:
|
|
95
|
+
continue
|
|
96
|
+
classes, methods = _repo_symbol_counts(repo_path)
|
|
97
|
+
return {"files": files, "bytes": total, "classes": classes, "methods": methods}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _repo_symbol_counts(repo_path: Path) -> tuple[int, int]:
|
|
101
|
+
"""(classes, methods) from ``repo-ir --summary-only``; (0, 0) on any failure."""
|
|
102
|
+
with tempfile.TemporaryDirectory(prefix="perf-ir-") as td:
|
|
103
|
+
out = Path(td) / "ir.json"
|
|
104
|
+
try:
|
|
105
|
+
subprocess.run(
|
|
106
|
+
[ASK_BIN, "repo-ir", str(repo_path), "--summary-only", "-o", str(out)],
|
|
107
|
+
capture_output=True, timeout=1800, check=False,
|
|
108
|
+
)
|
|
109
|
+
meta = json.loads(out.read_text(encoding="utf-8")).get("analysis_meta", {})
|
|
110
|
+
return int(meta.get("classes", 0) or 0), int(meta.get("methods", 0) or 0)
|
|
111
|
+
except (subprocess.SubprocessError, OSError, ValueError):
|
|
112
|
+
return 0, 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def build_argv(command: str, repo_path: Path, cold: bool) -> list[str]:
|
|
116
|
+
argv = [ASK_BIN]
|
|
117
|
+
if command:
|
|
118
|
+
argv.append(command)
|
|
119
|
+
argv.append(str(repo_path))
|
|
120
|
+
argv += ["--format", "json"]
|
|
121
|
+
if cold:
|
|
122
|
+
argv.append("--no-cache")
|
|
123
|
+
return argv
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def run_once(argv: list[str], env: dict[str, str], perf_log: Path) -> tuple[float, float, dict | None]:
|
|
127
|
+
"""Run the CLI once. Returns (wall_ms, rss_bytes, phase_ms | None)."""
|
|
128
|
+
if perf_log.exists():
|
|
129
|
+
perf_log.unlink()
|
|
130
|
+
rss_before = resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss
|
|
131
|
+
t0 = time.perf_counter()
|
|
132
|
+
subprocess.run(argv, env=env, capture_output=True, timeout=1800)
|
|
133
|
+
wall_ms = (time.perf_counter() - t0) * 1000.0
|
|
134
|
+
rss_after = resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss
|
|
135
|
+
# ru_maxrss for RUSAGE_CHILDREN is the peak among reaped children; the delta
|
|
136
|
+
# approximates this run's peak. Approximate by construction — documented.
|
|
137
|
+
rss_bytes = max(rss_after - rss_before, rss_after) * RSS_UNIT if rss_after else 0.0
|
|
138
|
+
phase_ms = _read_phase_log(perf_log)
|
|
139
|
+
return wall_ms, float(rss_bytes), phase_ms
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _read_phase_log(perf_log: Path) -> dict | None:
|
|
143
|
+
if not perf_log.exists():
|
|
144
|
+
return None
|
|
145
|
+
try:
|
|
146
|
+
last = perf_log.read_text(encoding="utf-8").strip().splitlines()[-1]
|
|
147
|
+
return json.loads(last).get("phase_ms") or None
|
|
148
|
+
except (OSError, IndexError, ValueError):
|
|
149
|
+
return None
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def measure_cell(command: str, repo_path: Path, mode: str, runs: int) -> dict:
|
|
153
|
+
cold = mode == "cold"
|
|
154
|
+
walls: list[float] = []
|
|
155
|
+
rsss: list[float] = []
|
|
156
|
+
phases: list[dict] = []
|
|
157
|
+
|
|
158
|
+
with tempfile.TemporaryDirectory(prefix="perf-cache-") as cache_dir, \
|
|
159
|
+
tempfile.TemporaryDirectory(prefix="perf-log-") as log_dir:
|
|
160
|
+
perf_log = Path(log_dir) / "phases.jsonl"
|
|
161
|
+
base_env = dict(os.environ)
|
|
162
|
+
base_env[perf.ENV_ENABLE] = "1"
|
|
163
|
+
base_env[perf.ENV_LOG] = str(perf_log)
|
|
164
|
+
base_env["SOURCECODE_CACHE_DIR"] = cache_dir
|
|
165
|
+
|
|
166
|
+
# Warm mode: one un-recorded priming run to populate the cache, then the
|
|
167
|
+
# first measured warm run is still discarded (belt and suspenders).
|
|
168
|
+
n = runs + 1 if mode == "warm" else runs
|
|
169
|
+
for i in range(n):
|
|
170
|
+
argv = build_argv(command, repo_path, cold=cold)
|
|
171
|
+
if cold:
|
|
172
|
+
# Each cold run gets a pristine cache dir.
|
|
173
|
+
for child in Path(cache_dir).iterdir():
|
|
174
|
+
shutil.rmtree(child, ignore_errors=True) if child.is_dir() else child.unlink()
|
|
175
|
+
wall, rss, phase = run_once(argv, base_env, perf_log)
|
|
176
|
+
if mode == "warm" and i == 0:
|
|
177
|
+
continue # discard cache-fill run
|
|
178
|
+
walls.append(wall)
|
|
179
|
+
rsss.append(rss)
|
|
180
|
+
if phase:
|
|
181
|
+
phases.append(phase)
|
|
182
|
+
|
|
183
|
+
return perf.build_baseline_cell(
|
|
184
|
+
tool_version=TOOL_VERSION,
|
|
185
|
+
command=command or "ask",
|
|
186
|
+
repo={"name": repo_path.name, "commit": git_commit(repo_path), **repo_size(repo_path)},
|
|
187
|
+
mode=mode,
|
|
188
|
+
wall_ms_runs=walls,
|
|
189
|
+
rss_bytes_runs=rsss,
|
|
190
|
+
phase_ms_runs=phases or None,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def main() -> int:
|
|
195
|
+
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
196
|
+
ap.add_argument("--repos", nargs="*", help="Subset of fleet names (default: all).")
|
|
197
|
+
ap.add_argument("--commands", nargs="*", default=DEFAULT_COMMANDS,
|
|
198
|
+
help="Commands to profile; '' = root analysis (default).")
|
|
199
|
+
ap.add_argument("--modes", nargs="*", default=["cold", "warm"], choices=["cold", "warm"])
|
|
200
|
+
ap.add_argument("--runs", type=int, default=7, help="Measured runs per cell (>=5 per ADR-0006).")
|
|
201
|
+
ap.add_argument("--fleet-root", type=Path, default=DEFAULT_FLEET_ROOT)
|
|
202
|
+
ap.add_argument("--out", type=Path, default=None,
|
|
203
|
+
help="Output dir (default: docs/perf/baselines/<tool_version>).")
|
|
204
|
+
ap.add_argument("--quick", action="store_true", help="petclinic only, 2 runs — smoke test.")
|
|
205
|
+
ap.add_argument("--dry-run", action="store_true", help="Print the plan, run nothing.")
|
|
206
|
+
args = ap.parse_args()
|
|
207
|
+
|
|
208
|
+
if args.quick:
|
|
209
|
+
args.repos = ["petclinic"]
|
|
210
|
+
args.runs = 2
|
|
211
|
+
|
|
212
|
+
repos = args.repos or list(FLEET)
|
|
213
|
+
out_dir = args.out or (Path(__file__).resolve().parent.parent / "docs" / "perf" / "baselines" / TOOL_VERSION)
|
|
214
|
+
|
|
215
|
+
plan = [
|
|
216
|
+
(cmd, name, mode)
|
|
217
|
+
for cmd in args.commands
|
|
218
|
+
for name in repos
|
|
219
|
+
for mode in args.modes
|
|
220
|
+
]
|
|
221
|
+
print(f"perf-harness {TOOL_VERSION} — {len(plan)} cells, {args.runs} runs each")
|
|
222
|
+
print(f"fleet root: {args.fleet_root}")
|
|
223
|
+
print(f"output: {out_dir}")
|
|
224
|
+
if args.dry_run:
|
|
225
|
+
for cmd, name, mode in plan:
|
|
226
|
+
print(f" [{mode:4}] {name:10} {cmd or 'ask'}")
|
|
227
|
+
return 0
|
|
228
|
+
|
|
229
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
230
|
+
for cmd, name, mode in plan:
|
|
231
|
+
subdir = FLEET.get(name, name)
|
|
232
|
+
repo_path = args.fleet_root / subdir
|
|
233
|
+
if not repo_path.is_dir():
|
|
234
|
+
print(f" SKIP {name}: {repo_path} not found", file=sys.stderr)
|
|
235
|
+
continue
|
|
236
|
+
print(f" [{mode:4}] {name:10} {cmd or 'ask'} ...", end="", flush=True)
|
|
237
|
+
try:
|
|
238
|
+
cell = measure_cell(cmd, repo_path, mode, args.runs)
|
|
239
|
+
except subprocess.TimeoutExpired:
|
|
240
|
+
print(" TIMEOUT", file=sys.stderr)
|
|
241
|
+
continue
|
|
242
|
+
fname = f"{name}__{cmd or 'ask'}__{mode}.json"
|
|
243
|
+
(out_dir / fname).write_text(json.dumps(cell, indent=2) + "\n", encoding="utf-8")
|
|
244
|
+
w = cell["wall_ms"]
|
|
245
|
+
print(f" p50={w['p50']}ms p95={w['p95']}ms -> {fname}")
|
|
246
|
+
|
|
247
|
+
print(f"done. artifacts in {out_dir}")
|
|
248
|
+
return 0
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
if __name__ == "__main__":
|
|
252
|
+
raise SystemExit(main())
|
|
@@ -967,25 +967,26 @@ def _java_contract_from_graph(rel_path: str, graph: "Any") -> FileContract:
|
|
|
967
967
|
nodes/edges instead of re-parsing source. Method signatures are the graph's
|
|
968
968
|
canonical compact form (`(Type)->Return`); annotations are node-grounded.
|
|
969
969
|
"""
|
|
970
|
-
file_syms =
|
|
970
|
+
file_syms = graph.symbols_in_file(rel_path)
|
|
971
971
|
type_syms = [s for s in file_syms if s.kind in ("class", "interface", "enum", "annotation")]
|
|
972
972
|
type_fqns = {s.fqn for s in type_syms}
|
|
973
973
|
|
|
974
974
|
# extends / implements edges out of this file's types (targets are simple
|
|
975
975
|
# names for out-of-repo supertypes, resolved FQNs for in-repo ones — the
|
|
976
|
-
# same forms the old regex captured for the unresolved case).
|
|
976
|
+
# same forms the old regex captured for the unresolved case). Only edges
|
|
977
|
+
# sourced at this file's types matter, so index straight to them instead of
|
|
978
|
+
# scanning the whole-repo relation set once per file.
|
|
977
979
|
extends_of: dict[str, list[str]] = {}
|
|
978
980
|
implements_of: dict[str, list[str]] = {}
|
|
979
981
|
import_targets: set[str] = set()
|
|
980
|
-
for
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
import_targets.add(r.target)
|
|
982
|
+
for fqn in type_fqns:
|
|
983
|
+
for r in graph.relations_from(fqn):
|
|
984
|
+
if r.kind == "extends":
|
|
985
|
+
extends_of.setdefault(r.source, []).append(r.target)
|
|
986
|
+
elif r.kind == "implements":
|
|
987
|
+
implements_of.setdefault(r.source, []).append(r.target)
|
|
988
|
+
elif r.kind == "imports":
|
|
989
|
+
import_targets.add(r.target)
|
|
989
990
|
|
|
990
991
|
exports: list[ExportRecord] = []
|
|
991
992
|
types: list[TypeDefinition] = []
|
|
@@ -15,6 +15,7 @@ from sourcecode import __version__
|
|
|
15
15
|
from sourcecode.error_schema import INVALID_INPUT_CODE, build_error_envelope
|
|
16
16
|
from sourcecode.entrypoint_classifier import is_production_entry_point, normalize_entry_point
|
|
17
17
|
from sourcecode.progress import Progress
|
|
18
|
+
from sourcecode import perf
|
|
18
19
|
from sourcecode.caller_metrics import CALLER_METRIC_RECONCILIATION
|
|
19
20
|
from sourcecode.repository_ir import extract_java_endpoints as _extract_java_endpoints
|
|
20
21
|
|
|
@@ -1703,6 +1704,7 @@ def main(
|
|
|
1703
1704
|
_progress = Progress()
|
|
1704
1705
|
_progress.start("scanning files")
|
|
1705
1706
|
|
|
1707
|
+
_perf_discovery = perf.start()
|
|
1706
1708
|
scanner = AdaptiveScanner(target, topology=_topology, base_depth=effective_depth,
|
|
1707
1709
|
extra_excludes=_extra_excludes)
|
|
1708
1710
|
raw_tree = scanner.scan_tree()
|
|
@@ -1710,6 +1712,8 @@ def main(
|
|
|
1710
1712
|
_progress.update("parsing manifests")
|
|
1711
1713
|
# 2. Filter .env and *.secret entries from file tree (SEC-02, all levels)
|
|
1712
1714
|
file_tree = filter_sensitive_files(raw_tree, redactor)
|
|
1715
|
+
perf.stop("discovery", _perf_discovery)
|
|
1716
|
+
_perf_detection = perf.start()
|
|
1713
1717
|
detector = ProjectDetector(build_default_detectors())
|
|
1714
1718
|
workspace_analysis = WorkspaceAnalyzer().analyze(target, manifests)
|
|
1715
1719
|
|
|
@@ -1951,6 +1955,8 @@ def main(
|
|
|
1951
1955
|
else None
|
|
1952
1956
|
)
|
|
1953
1957
|
|
|
1958
|
+
perf.stop("detection", _perf_detection)
|
|
1959
|
+
_perf_analysis = perf.start()
|
|
1954
1960
|
# 3. Build schema
|
|
1955
1961
|
# Compute analyzer fingerprints: short hashes of each analyzer's key rule
|
|
1956
1962
|
# constants so that a rule change is always visible in the output, regardless
|
|
@@ -2411,8 +2417,10 @@ def main(
|
|
|
2411
2417
|
err=True,
|
|
2412
2418
|
)
|
|
2413
2419
|
|
|
2420
|
+
perf.stop("analysis", _perf_analysis)
|
|
2414
2421
|
# 4. Serialize
|
|
2415
2422
|
_progress.update("serializing output")
|
|
2423
|
+
_perf_serialize = perf.start()
|
|
2416
2424
|
if _is_contract_mode and not agent:
|
|
2417
2425
|
from sourcecode.serializer import contract_view as _contract_view
|
|
2418
2426
|
_depth = _CONTRACT_DEPTH.get(mode, "minimal")
|
|
@@ -2455,6 +2463,9 @@ def main(
|
|
|
2455
2463
|
raw_dict = redact_dict(raw_dict)
|
|
2456
2464
|
content = _serialize_dict(raw_dict, format)
|
|
2457
2465
|
|
|
2466
|
+
perf.stop("serialize", _perf_serialize)
|
|
2467
|
+
perf.flush_recorder()
|
|
2468
|
+
|
|
2458
2469
|
# 5. Telemetry (fire-and-forget, never blocks)
|
|
2459
2470
|
try:
|
|
2460
2471
|
from sourcecode import telemetry as _tel
|
|
@@ -408,6 +408,7 @@ class ContextGraph:
|
|
|
408
408
|
"_cir", "_nodes_by_fqn", "_nodes", "_build_ms",
|
|
409
409
|
"_body_index", "_literal_index", "_guard_index", "_span_index",
|
|
410
410
|
"_class_type_index", "_relations_all", "_symbols_sorted",
|
|
411
|
+
"_symbols_by_file", "_relations_by_source",
|
|
411
412
|
)
|
|
412
413
|
|
|
413
414
|
def __init__(self, cir: CanonicalRepositoryIR, *, build_ms: float = 0.0) -> None:
|
|
@@ -439,6 +440,12 @@ class ContextGraph:
|
|
|
439
440
|
# filtered cheaply thereafter.
|
|
440
441
|
self._relations_all: Optional[tuple[Relation, ...]] = None
|
|
441
442
|
self._symbols_sorted: Optional[tuple[Symbol, ...]] = None
|
|
443
|
+
# Per-key groupings for per-file consumers (`_java_contract_from_graph`
|
|
444
|
+
# runs once per file). Without these, each file filtered the whole-repo
|
|
445
|
+
# symbol/relation set — the residual O(files × total) scan that the
|
|
446
|
+
# memoized projections above did not remove. Built once, O(1) lookup.
|
|
447
|
+
self._symbols_by_file: Optional[dict[str, list[Symbol]]] = None
|
|
448
|
+
self._relations_by_source: Optional[dict[str, list[Relation]]] = None
|
|
442
449
|
|
|
443
450
|
# -- construction -------------------------------------------------------
|
|
444
451
|
|
|
@@ -518,6 +525,34 @@ class ContextGraph:
|
|
|
518
525
|
and (name_contains is None or name_contains in s.fqn)
|
|
519
526
|
]
|
|
520
527
|
|
|
528
|
+
def symbols_in_file(self, source_file: str) -> list[Symbol]:
|
|
529
|
+
"""All symbols declared in one file, FQN-sorted — the same order and set
|
|
530
|
+
a no-filter ``symbols()`` scan filtered to this file would produce, but
|
|
531
|
+
O(1) after a one-time O(total) grouping instead of O(total) per call.
|
|
532
|
+
"""
|
|
533
|
+
if self._symbols_by_file is None:
|
|
534
|
+
if self._symbols_sorted is None:
|
|
535
|
+
self._symbols_sorted = tuple(sorted(self._nodes, key=lambda s: s.fqn))
|
|
536
|
+
idx: dict[str, list[Symbol]] = {}
|
|
537
|
+
for s in self._symbols_sorted:
|
|
538
|
+
idx.setdefault(s.source_file, []).append(s)
|
|
539
|
+
self._symbols_by_file = idx
|
|
540
|
+
return self._symbols_by_file.get(source_file, [])
|
|
541
|
+
|
|
542
|
+
def relations_from(self, source_fqn: str) -> list[Relation]:
|
|
543
|
+
"""All relations whose source is ``source_fqn``, in the IR's stable edge
|
|
544
|
+
order. O(1) after a one-time O(edges) grouping — replaces per-file scans
|
|
545
|
+
of the whole relation set.
|
|
546
|
+
"""
|
|
547
|
+
if self._relations_by_source is None:
|
|
548
|
+
if self._relations_all is None:
|
|
549
|
+
self._relations_all = tuple(_edge_to_relation(e) for e in self._cir.call_graph)
|
|
550
|
+
idx: dict[str, list[Relation]] = {}
|
|
551
|
+
for r in self._relations_all:
|
|
552
|
+
idx.setdefault(r.source, []).append(r)
|
|
553
|
+
self._relations_by_source = idx
|
|
554
|
+
return self._relations_by_source.get(source_fqn, [])
|
|
555
|
+
|
|
521
556
|
def types(self) -> list[Symbol]:
|
|
522
557
|
"""All class/interface/enum/annotation symbols, sorted by FQN."""
|
|
523
558
|
return sorted(
|