codecortex 0.15.4__tar.gz → 0.15.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. {codecortex-0.15.4/src/codecortex.egg-info → codecortex-0.15.5}/PKG-INFO +35 -22
  2. {codecortex-0.15.4 → codecortex-0.15.5}/README.md +34 -21
  3. {codecortex-0.15.4 → codecortex-0.15.5/src/codecortex.egg-info}/PKG-INFO +35 -22
  4. {codecortex-0.15.4 → codecortex-0.15.5}/src/codecortex.egg-info/SOURCES.txt +4 -1
  5. codecortex-0.15.5/src/codeintel/__init__.py +1 -0
  6. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/status.py +10 -0
  7. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/doctor.py +71 -0
  8. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/providers/graph.py +13 -7
  9. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/server.py +6 -0
  10. codecortex-0.15.5/tests/test_docs_ci_claims.py +99 -0
  11. codecortex-0.15.5/tests/test_docs_deadcode_withdrawal.py +100 -0
  12. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_graph_failure_population.py +50 -0
  13. codecortex-0.15.5/tests/test_version_skew.py +124 -0
  14. codecortex-0.15.4/src/codeintel/__init__.py +0 -1
  15. {codecortex-0.15.4 → codecortex-0.15.5}/LICENSE +0 -0
  16. {codecortex-0.15.4 → codecortex-0.15.5}/pyproject.toml +0 -0
  17. {codecortex-0.15.4 → codecortex-0.15.5}/setup.cfg +0 -0
  18. {codecortex-0.15.4 → codecortex-0.15.5}/src/codecortex.egg-info/dependency_links.txt +0 -0
  19. {codecortex-0.15.4 → codecortex-0.15.5}/src/codecortex.egg-info/entry_points.txt +0 -0
  20. {codecortex-0.15.4 → codecortex-0.15.5}/src/codecortex.egg-info/requires.txt +0 -0
  21. {codecortex-0.15.4 → codecortex-0.15.5}/src/codecortex.egg-info/top_level.txt +0 -0
  22. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/__main__.py +0 -0
  23. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/auth.py +0 -0
  24. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/cache.py +0 -0
  25. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/__init__.py +0 -0
  26. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/_common.py +0 -0
  27. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/doctor.py +0 -0
  28. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/gen_token.py +0 -0
  29. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/graph.py +0 -0
  30. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/index.py +0 -0
  31. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/install.py +0 -0
  32. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/map.py +0 -0
  33. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/query.py +0 -0
  34. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/reset.py +0 -0
  35. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/serve.py +0 -0
  36. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/serve_http.py +0 -0
  37. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/commands/setup.py +0 -0
  38. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/config.py +0 -0
  39. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/containment.py +0 -0
  40. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/gateway.py +0 -0
  41. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/grapher.py +0 -0
  42. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/http_server.py +0 -0
  43. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/indexer.py +0 -0
  44. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/injector.py +0 -0
  45. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/installer.py +0 -0
  46. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/loc.py +0 -0
  47. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/logconfig.py +0 -0
  48. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/mapper.py +0 -0
  49. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/metrics.py +0 -0
  50. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/onboarding.py +0 -0
  51. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/outcome.py +0 -0
  52. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/policy.py +0 -0
  53. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/provider.py +0 -0
  54. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/providers/__init__.py +0 -0
  55. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/providers/lsp.py +0 -0
  56. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/providers/none.py +0 -0
  57. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/providers/semantic.py +0 -0
  58. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/py.typed +0 -0
  59. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/redact.py +0 -0
  60. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/reindexer.py +0 -0
  61. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/reset.py +0 -0
  62. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/searcher.py +0 -0
  63. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/semantic_db.py +0 -0
  64. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/source_kind.py +0 -0
  65. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/term.py +0 -0
  66. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/verify.py +0 -0
  67. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/viewer/__init__.py +0 -0
  68. {codecortex-0.15.4 → codecortex-0.15.5}/src/codeintel/viewer/graph_template.html +0 -0
  69. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_cache.py +0 -0
  70. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_chunking.py +0 -0
  71. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_cli_commands.py +0 -0
  72. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_cli_help.py +0 -0
  73. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_cold_process.py +0 -0
  74. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_config.py +0 -0
  75. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_corpus.py +0 -0
  76. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_doctor.py +0 -0
  77. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_e2e.py +0 -0
  78. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_engine_adoption.py +0 -0
  79. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_enterprise.py +0 -0
  80. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_gateway.py +0 -0
  81. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_graph_provider.py +0 -0
  82. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_graph_real.py +0 -0
  83. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_graph_stdin.py +0 -0
  84. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_grapher.py +0 -0
  85. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_hardening.py +0 -0
  86. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_http_auth.py +0 -0
  87. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_http_server.py +0 -0
  88. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_incompleteness.py +0 -0
  89. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_installer.py +0 -0
  90. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_integration.py +0 -0
  91. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_loc_census.py +0 -0
  92. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_lsp_provider.py +0 -0
  93. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_lsp_real.py +0 -0
  94. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_mapper.py +0 -0
  95. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_mcp_handshake.py +0 -0
  96. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_mcp_server.py +0 -0
  97. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_model_dimension.py +0 -0
  98. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_never_raise.py +0 -0
  99. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_onboarding.py +0 -0
  100. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_rbac.py +0 -0
  101. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_reindexer.py +0 -0
  102. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_rerank.py +0 -0
  103. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_reset.py +0 -0
  104. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_semantic_provider.py +0 -0
  105. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_source_kind.py +0 -0
  106. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_term.py +0 -0
  107. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_treesitter.py +0 -0
  108. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_verify.py +0 -0
  109. {codecortex-0.15.4 → codecortex-0.15.5}/tests/test_verify_call.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codecortex
3
- Version: 0.15.4
3
+ Version: 0.15.5
4
4
  Summary: Local-first, MCP-native code-intelligence server — graph, LSP, and semantic search behind one safe code.query tool for coding agents.
5
5
  Author: Shammai Hamilton
6
6
  License-Expression: MIT
@@ -95,23 +95,29 @@ It's one call: `code.query(op, target, engine="auto")`. In `auto` mode (the defa
95
95
  | Everything about one symbol | `context` | graph + lsp | both views merged |
96
96
  | **Impact of your uncommitted edits** | `changed` | graph | changed files → impacted symbols |
97
97
  | Refactor-risk hotspots | `hotspots` | graph | highest complexity / fan-in symbols |
98
- | Unreferenced (dead) code | `deadcode` | graph | non-test symbols with no callers, **verified against the source** — [treat as candidates, not instructions](#deadcode-is-a-candidate-list-not-a-delete-list) |
98
+ | Unreferenced (dead) code | `deadcode` | graph | **withdrawn** — measured wrong in both directions on real repos; safe-nulls with `reason: "op-withdrawn"` unless you opt in — [why, and what to use instead](#deadcode-is-withdrawn) |
99
99
 
100
100
  Pin one engine with `--engine graph│lsp│semantic`, or fan out with `--engine both` / `all` to merge results.
101
101
 
102
- #### `deadcode` is a candidate list, not a delete list
102
+ #### `deadcode` is withdrawn
103
103
 
104
- `deadcode` is the one op whose output invites a destructive action, so it gets an explicit caveat.
105
- Every hit is re-read and verified against the source before it is reported, which removes the
106
- common false positives but **no reachability analysis sees every caller.** Dynamic dispatch,
107
- registries and decorators, `getattr` lookups, entry points declared in packaging metadata, plugin
108
- discovery, reflection, and calls from languages the graph does not parse are all invisible to it.
109
- Through `0.14.x` it was systematically wrong on callback-heavy code and confident about it; that
110
- class of defect is fixed, but the underlying limit is structural and permanent.
104
+ `deadcode` is withdrawn: it returns a safe-null (`reason: "op-withdrawn"`) instead of running. It
105
+ was measured wrong in **both** directions on real repositories on one it named five candidates of
106
+ which four were live code (a rollup plugin hook; two entries of a `Record<string, fn>` reached by a
107
+ runtime string; a `predicate` passed inline to the call that consumes it), and on another it
108
+ reported "(none found)" for a 4,883-function codebase that had at least three genuinely unreferenced
109
+ private helpers. It is the one op whose output is an instruction to delete code, so it needs the
110
+ highest evidence bar of any op here, and currently has the least. It returns when a labelled corpus
111
+ measures its precision and recall — not before.
111
112
 
112
- **So: review each hit before deleting anything, and never wire `deadcode` into an agent that
113
- deletes without a human in the loop.** Used as a ranked list of *places worth looking*, it is
114
- genuinely useful. Used as a work order, it will eventually remove live code.
113
+ **Use `callers` on a specific symbol instead.** "Does anything call this?" is exactly the question
114
+ `deadcode` was trying to answer in bulk, and `callers` answers it accurately, one symbol at a time.
115
+
116
+ **Escape hatch, if you understand the risk.** Set `CODEINTEL_ENABLE_UNVERIFIED_OPS=1` to run it
117
+ anyway. It still re-reads the source before reporting a hit, which removes the *common* false
118
+ positives — but that verification is exactly what was measured wrong on the repositories above, so
119
+ **review every hit before deleting anything, and never wire it into an agent that deletes without a
120
+ human in the loop.**
115
121
 
116
122
  **Example — "who uses `safe_null_result`?"**
117
123
 
@@ -511,26 +517,33 @@ back to grep rather than crashing.
511
517
 
512
518
  | Area | Why |
513
519
  |---|---|
514
- | `deadcode` | It suggests deletions and cannot see every caller — [read the caveat](#deadcode-is-a-candidate-list-not-a-delete-list). |
520
+ | `deadcode` | Withdrawn by default (`reason: "op-withdrawn"`) measured wrong in both directions on real repos. Use `callers` on a specific symbol instead — [details](#deadcode-is-withdrawn). |
515
521
  | Non-loopback serving | `serve-http` is stdlib `http.server`. It binds loopback by default for a reason; front it with a reverse proxy and see [docs/deploy.md](docs/deploy.md). |
516
522
  | RBAC between **untrusting** tenants | It separates privilege levels among callers you already trust. It is not a wall against an adversary with write access to their own root — see the warning in [docs/deploy.md](docs/deploy.md). |
517
523
  | Unattended automation | Anything that acts on a result without a human reading it deserves a pilot first. |
518
524
 
519
525
  **On the test numbers.** The suite is large and the coverage floor is enforced, but read the figure
520
- with its caveat: the graph and LSP backends are external binaries that are **not installed in CI**,
521
- so those two engines are exercised against hand-authored mocks rather than the real wire contract,
522
- and the release canary which does assert on real answer text against a built wheel — currently
523
- covers the semantic engine only. Line coverage measures how much of the intended behavior runs, not
524
- how much of reality it has met.
526
+ with its caveat. In the **main test job** the graph and LSP backends are absent, so their live tests
527
+ skip and those engines run against hand-authored mocks rather than the real wire contract. Separate
528
+ jobs cover the contract itself: `graph-contract` installs the pinned `codebase-memory-mcp` and runs
529
+ the live graph tests and **fails if they skipped**, because a silently-skipped contract test is
530
+ how a total backend outage stayed green here once — while the nightly corpus job runs that same
531
+ real backend against pinned third-party repositories. `lsp-contract` runs the live serena tests but
532
+ is **`continue-on-error`**: serena is fetched from an upstream git HEAD this project does not
533
+ control, so a breakage there must be visible without blocking an unrelated release. Read that as
534
+ the LSP wire contract being *watched* rather than *gated*. The release canary — the only check that
535
+ asserts on real answer text from a built wheel — still covers **the semantic engine only**.
536
+
537
+ Line coverage measures how much of the intended behavior runs, not how much of reality it has met.
525
538
 
526
539
  **The honest one-paragraph version.** codeintel has been run on very few repositories its author did
527
540
  not write, and that is where its bugs have come from — every fix in `0.15.x` came from pointing it
528
541
  at an unfamiliar codebase. Its characteristic failure mode is **answering confidently from the
529
542
  wrong index rather than failing loudly**, which the never-raise contract makes harder to notice: a
530
543
  wrong answer and a right one are the same shape. Run `codeintel doctor` before trusting a repo-wide
531
- answer, treat `deadcode` as candidates for review, and if something looks off please
532
- [report it](#reporting-a-problem) — an issue from someone who is not the author is the single most
533
- useful thing this project can receive right now.
544
+ answer and `deadcode` in particular is withdrawn rather than merely caveated (see above) — and if
545
+ something looks off please [report it](#reporting-a-problem) — an issue from someone who is not the
546
+ author is the single most useful thing this project can receive right now.
534
547
 
535
548
  **Engine coverage depends on external binaries.** Semantic search works out of the box. The graph
536
549
  engine needs `codebase-memory-mcp` and the LSP engine needs `uvx` on `PATH` — without them those
@@ -58,23 +58,29 @@ It's one call: `code.query(op, target, engine="auto")`. In `auto` mode (the defa
58
58
  | Everything about one symbol | `context` | graph + lsp | both views merged |
59
59
  | **Impact of your uncommitted edits** | `changed` | graph | changed files → impacted symbols |
60
60
  | Refactor-risk hotspots | `hotspots` | graph | highest complexity / fan-in symbols |
61
- | Unreferenced (dead) code | `deadcode` | graph | non-test symbols with no callers, **verified against the source** — [treat as candidates, not instructions](#deadcode-is-a-candidate-list-not-a-delete-list) |
61
+ | Unreferenced (dead) code | `deadcode` | graph | **withdrawn** — measured wrong in both directions on real repos; safe-nulls with `reason: "op-withdrawn"` unless you opt in — [why, and what to use instead](#deadcode-is-withdrawn) |
62
62
 
63
63
  Pin one engine with `--engine graph│lsp│semantic`, or fan out with `--engine both` / `all` to merge results.
64
64
 
65
- #### `deadcode` is a candidate list, not a delete list
65
+ #### `deadcode` is withdrawn
66
66
 
67
- `deadcode` is the one op whose output invites a destructive action, so it gets an explicit caveat.
68
- Every hit is re-read and verified against the source before it is reported, which removes the
69
- common false positives but **no reachability analysis sees every caller.** Dynamic dispatch,
70
- registries and decorators, `getattr` lookups, entry points declared in packaging metadata, plugin
71
- discovery, reflection, and calls from languages the graph does not parse are all invisible to it.
72
- Through `0.14.x` it was systematically wrong on callback-heavy code and confident about it; that
73
- class of defect is fixed, but the underlying limit is structural and permanent.
67
+ `deadcode` is withdrawn: it returns a safe-null (`reason: "op-withdrawn"`) instead of running. It
68
+ was measured wrong in **both** directions on real repositories on one it named five candidates of
69
+ which four were live code (a rollup plugin hook; two entries of a `Record<string, fn>` reached by a
70
+ runtime string; a `predicate` passed inline to the call that consumes it), and on another it
71
+ reported "(none found)" for a 4,883-function codebase that had at least three genuinely unreferenced
72
+ private helpers. It is the one op whose output is an instruction to delete code, so it needs the
73
+ highest evidence bar of any op here, and currently has the least. It returns when a labelled corpus
74
+ measures its precision and recall — not before.
74
75
 
75
- **So: review each hit before deleting anything, and never wire `deadcode` into an agent that
76
- deletes without a human in the loop.** Used as a ranked list of *places worth looking*, it is
77
- genuinely useful. Used as a work order, it will eventually remove live code.
76
+ **Use `callers` on a specific symbol instead.** "Does anything call this?" is exactly the question
77
+ `deadcode` was trying to answer in bulk, and `callers` answers it accurately, one symbol at a time.
78
+
79
+ **Escape hatch, if you understand the risk.** Set `CODEINTEL_ENABLE_UNVERIFIED_OPS=1` to run it
80
+ anyway. It still re-reads the source before reporting a hit, which removes the *common* false
81
+ positives — but that verification is exactly what was measured wrong on the repositories above, so
82
+ **review every hit before deleting anything, and never wire it into an agent that deletes without a
83
+ human in the loop.**
78
84
 
79
85
  **Example — "who uses `safe_null_result`?"**
80
86
 
@@ -474,26 +480,33 @@ back to grep rather than crashing.
474
480
 
475
481
  | Area | Why |
476
482
  |---|---|
477
- | `deadcode` | It suggests deletions and cannot see every caller — [read the caveat](#deadcode-is-a-candidate-list-not-a-delete-list). |
483
+ | `deadcode` | Withdrawn by default (`reason: "op-withdrawn"`) measured wrong in both directions on real repos. Use `callers` on a specific symbol instead — [details](#deadcode-is-withdrawn). |
478
484
  | Non-loopback serving | `serve-http` is stdlib `http.server`. It binds loopback by default for a reason; front it with a reverse proxy and see [docs/deploy.md](docs/deploy.md). |
479
485
  | RBAC between **untrusting** tenants | It separates privilege levels among callers you already trust. It is not a wall against an adversary with write access to their own root — see the warning in [docs/deploy.md](docs/deploy.md). |
480
486
  | Unattended automation | Anything that acts on a result without a human reading it deserves a pilot first. |
481
487
 
482
488
  **On the test numbers.** The suite is large and the coverage floor is enforced, but read the figure
483
- with its caveat: the graph and LSP backends are external binaries that are **not installed in CI**,
484
- so those two engines are exercised against hand-authored mocks rather than the real wire contract,
485
- and the release canary which does assert on real answer text against a built wheel — currently
486
- covers the semantic engine only. Line coverage measures how much of the intended behavior runs, not
487
- how much of reality it has met.
489
+ with its caveat. In the **main test job** the graph and LSP backends are absent, so their live tests
490
+ skip and those engines run against hand-authored mocks rather than the real wire contract. Separate
491
+ jobs cover the contract itself: `graph-contract` installs the pinned `codebase-memory-mcp` and runs
492
+ the live graph tests and **fails if they skipped**, because a silently-skipped contract test is
493
+ how a total backend outage stayed green here once — while the nightly corpus job runs that same
494
+ real backend against pinned third-party repositories. `lsp-contract` runs the live serena tests but
495
+ is **`continue-on-error`**: serena is fetched from an upstream git HEAD this project does not
496
+ control, so a breakage there must be visible without blocking an unrelated release. Read that as
497
+ the LSP wire contract being *watched* rather than *gated*. The release canary — the only check that
498
+ asserts on real answer text from a built wheel — still covers **the semantic engine only**.
499
+
500
+ Line coverage measures how much of the intended behavior runs, not how much of reality it has met.
488
501
 
489
502
  **The honest one-paragraph version.** codeintel has been run on very few repositories its author did
490
503
  not write, and that is where its bugs have come from — every fix in `0.15.x` came from pointing it
491
504
  at an unfamiliar codebase. Its characteristic failure mode is **answering confidently from the
492
505
  wrong index rather than failing loudly**, which the never-raise contract makes harder to notice: a
493
506
  wrong answer and a right one are the same shape. Run `codeintel doctor` before trusting a repo-wide
494
- answer, treat `deadcode` as candidates for review, and if something looks off please
495
- [report it](#reporting-a-problem) — an issue from someone who is not the author is the single most
496
- useful thing this project can receive right now.
507
+ answer and `deadcode` in particular is withdrawn rather than merely caveated (see above) — and if
508
+ something looks off please [report it](#reporting-a-problem) — an issue from someone who is not the
509
+ author is the single most useful thing this project can receive right now.
497
510
 
498
511
  **Engine coverage depends on external binaries.** Semantic search works out of the box. The graph
499
512
  engine needs `codebase-memory-mcp` and the LSP engine needs `uvx` on `PATH` — without them those
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codecortex
3
- Version: 0.15.4
3
+ Version: 0.15.5
4
4
  Summary: Local-first, MCP-native code-intelligence server — graph, LSP, and semantic search behind one safe code.query tool for coding agents.
5
5
  Author: Shammai Hamilton
6
6
  License-Expression: MIT
@@ -95,23 +95,29 @@ It's one call: `code.query(op, target, engine="auto")`. In `auto` mode (the defa
95
95
  | Everything about one symbol | `context` | graph + lsp | both views merged |
96
96
  | **Impact of your uncommitted edits** | `changed` | graph | changed files → impacted symbols |
97
97
  | Refactor-risk hotspots | `hotspots` | graph | highest complexity / fan-in symbols |
98
- | Unreferenced (dead) code | `deadcode` | graph | non-test symbols with no callers, **verified against the source** — [treat as candidates, not instructions](#deadcode-is-a-candidate-list-not-a-delete-list) |
98
+ | Unreferenced (dead) code | `deadcode` | graph | **withdrawn** — measured wrong in both directions on real repos; safe-nulls with `reason: "op-withdrawn"` unless you opt in — [why, and what to use instead](#deadcode-is-withdrawn) |
99
99
 
100
100
  Pin one engine with `--engine graph│lsp│semantic`, or fan out with `--engine both` / `all` to merge results.
101
101
 
102
- #### `deadcode` is a candidate list, not a delete list
102
+ #### `deadcode` is withdrawn
103
103
 
104
- `deadcode` is the one op whose output invites a destructive action, so it gets an explicit caveat.
105
- Every hit is re-read and verified against the source before it is reported, which removes the
106
- common false positives but **no reachability analysis sees every caller.** Dynamic dispatch,
107
- registries and decorators, `getattr` lookups, entry points declared in packaging metadata, plugin
108
- discovery, reflection, and calls from languages the graph does not parse are all invisible to it.
109
- Through `0.14.x` it was systematically wrong on callback-heavy code and confident about it; that
110
- class of defect is fixed, but the underlying limit is structural and permanent.
104
+ `deadcode` is withdrawn: it returns a safe-null (`reason: "op-withdrawn"`) instead of running. It
105
+ was measured wrong in **both** directions on real repositories on one it named five candidates of
106
+ which four were live code (a rollup plugin hook; two entries of a `Record<string, fn>` reached by a
107
+ runtime string; a `predicate` passed inline to the call that consumes it), and on another it
108
+ reported "(none found)" for a 4,883-function codebase that had at least three genuinely unreferenced
109
+ private helpers. It is the one op whose output is an instruction to delete code, so it needs the
110
+ highest evidence bar of any op here, and currently has the least. It returns when a labelled corpus
111
+ measures its precision and recall — not before.
111
112
 
112
- **So: review each hit before deleting anything, and never wire `deadcode` into an agent that
113
- deletes without a human in the loop.** Used as a ranked list of *places worth looking*, it is
114
- genuinely useful. Used as a work order, it will eventually remove live code.
113
+ **Use `callers` on a specific symbol instead.** "Does anything call this?" is exactly the question
114
+ `deadcode` was trying to answer in bulk, and `callers` answers it accurately, one symbol at a time.
115
+
116
+ **Escape hatch, if you understand the risk.** Set `CODEINTEL_ENABLE_UNVERIFIED_OPS=1` to run it
117
+ anyway. It still re-reads the source before reporting a hit, which removes the *common* false
118
+ positives — but that verification is exactly what was measured wrong on the repositories above, so
119
+ **review every hit before deleting anything, and never wire it into an agent that deletes without a
120
+ human in the loop.**
115
121
 
116
122
  **Example — "who uses `safe_null_result`?"**
117
123
 
@@ -511,26 +517,33 @@ back to grep rather than crashing.
511
517
 
512
518
  | Area | Why |
513
519
  |---|---|
514
- | `deadcode` | It suggests deletions and cannot see every caller — [read the caveat](#deadcode-is-a-candidate-list-not-a-delete-list). |
520
+ | `deadcode` | Withdrawn by default (`reason: "op-withdrawn"`) measured wrong in both directions on real repos. Use `callers` on a specific symbol instead — [details](#deadcode-is-withdrawn). |
515
521
  | Non-loopback serving | `serve-http` is stdlib `http.server`. It binds loopback by default for a reason; front it with a reverse proxy and see [docs/deploy.md](docs/deploy.md). |
516
522
  | RBAC between **untrusting** tenants | It separates privilege levels among callers you already trust. It is not a wall against an adversary with write access to their own root — see the warning in [docs/deploy.md](docs/deploy.md). |
517
523
  | Unattended automation | Anything that acts on a result without a human reading it deserves a pilot first. |
518
524
 
519
525
  **On the test numbers.** The suite is large and the coverage floor is enforced, but read the figure
520
- with its caveat: the graph and LSP backends are external binaries that are **not installed in CI**,
521
- so those two engines are exercised against hand-authored mocks rather than the real wire contract,
522
- and the release canary which does assert on real answer text against a built wheel — currently
523
- covers the semantic engine only. Line coverage measures how much of the intended behavior runs, not
524
- how much of reality it has met.
526
+ with its caveat. In the **main test job** the graph and LSP backends are absent, so their live tests
527
+ skip and those engines run against hand-authored mocks rather than the real wire contract. Separate
528
+ jobs cover the contract itself: `graph-contract` installs the pinned `codebase-memory-mcp` and runs
529
+ the live graph tests and **fails if they skipped**, because a silently-skipped contract test is
530
+ how a total backend outage stayed green here once — while the nightly corpus job runs that same
531
+ real backend against pinned third-party repositories. `lsp-contract` runs the live serena tests but
532
+ is **`continue-on-error`**: serena is fetched from an upstream git HEAD this project does not
533
+ control, so a breakage there must be visible without blocking an unrelated release. Read that as
534
+ the LSP wire contract being *watched* rather than *gated*. The release canary — the only check that
535
+ asserts on real answer text from a built wheel — still covers **the semantic engine only**.
536
+
537
+ Line coverage measures how much of the intended behavior runs, not how much of reality it has met.
525
538
 
526
539
  **The honest one-paragraph version.** codeintel has been run on very few repositories its author did
527
540
  not write, and that is where its bugs have come from — every fix in `0.15.x` came from pointing it
528
541
  at an unfamiliar codebase. Its characteristic failure mode is **answering confidently from the
529
542
  wrong index rather than failing loudly**, which the never-raise contract makes harder to notice: a
530
543
  wrong answer and a right one are the same shape. Run `codeintel doctor` before trusting a repo-wide
531
- answer, treat `deadcode` as candidates for review, and if something looks off please
532
- [report it](#reporting-a-problem) — an issue from someone who is not the author is the single most
533
- useful thing this project can receive right now.
544
+ answer and `deadcode` in particular is withdrawn rather than merely caveated (see above) — and if
545
+ something looks off please [report it](#reporting-a-problem) — an issue from someone who is not the
546
+ author is the single most useful thing this project can receive right now.
534
547
 
535
548
  **Engine coverage depends on external binaries.** Semantic search works out of the box. The graph
536
549
  engine needs `codebase-memory-mcp` and the LSP engine needs `uvx` on `PATH` — without them those
@@ -66,6 +66,8 @@ tests/test_cli_help.py
66
66
  tests/test_cold_process.py
67
67
  tests/test_config.py
68
68
  tests/test_corpus.py
69
+ tests/test_docs_ci_claims.py
70
+ tests/test_docs_deadcode_withdrawal.py
69
71
  tests/test_doctor.py
70
72
  tests/test_e2e.py
71
73
  tests/test_engine_adoption.py
@@ -100,4 +102,5 @@ tests/test_source_kind.py
100
102
  tests/test_term.py
101
103
  tests/test_treesitter.py
102
104
  tests/test_verify.py
103
- tests/test_verify_call.py
105
+ tests/test_verify_call.py
106
+ tests/test_version_skew.py
@@ -0,0 +1 @@
1
+ __version__ = "0.15.5"
@@ -32,6 +32,16 @@ def run(args: Any) -> int:
32
32
  if status.get("healthy") is False:
33
33
  print("\n run `codeintel doctor` for the fix for each gap")
34
34
 
35
+ # Printed before anything else the user might act on. A skew means every line above describes
36
+ # the code this process loaded, not the code installed — so a fix the user can read in the
37
+ # CHANGELOG can be absent from every answer while the engines all report green.
38
+ skew = status.get("version_skew")
39
+ if isinstance(skew, dict) and skew.get("running") and skew.get("installed"):
40
+ print(
41
+ f"\n ! serving {skew['running']}, but {skew['installed']} is installed"
42
+ "\n restart the codeintel server (or the agent hosting it) to pick it up"
43
+ )
44
+
35
45
  from codeintel.config import load_config
36
46
  from codeintel.semantic_db import default_db_path
37
47
 
@@ -8,6 +8,7 @@ effects). The same report drives the CLI `doctor` command, the `code.doctor` MCP
8
8
  from __future__ import annotations
9
9
 
10
10
  import os
11
+ import pathlib
11
12
  import shutil
12
13
  from collections.abc import Callable
13
14
  from typing import Any
@@ -106,6 +107,51 @@ def _dist_version(name: str) -> str | None:
106
107
  return None
107
108
 
108
109
 
110
+ def running_version_skew() -> tuple[str, str] | None:
111
+ """``(running, on_disk)`` when this process is serving code older than what is installed.
112
+
113
+ A long-lived server holds the module it imported at startup. Upgrading the package underneath
114
+ it — `uv tool install`, `pip install -U` — replaces the files on disk and changes nothing about
115
+ the running process, which keeps answering with the old code until something restarts it. That
116
+ gap is silent and it is not hypothetical: a fix can be committed, released, installed, read in
117
+ the CHANGELOG, and still absent from every answer the user is getting, with `status` reporting
118
+ the stale version as if it were the truth.
119
+
120
+ Detected by re-reading `__version__` out of the very file this module was loaded FROM, at call
121
+ time. `importlib.metadata` would be the obvious route and is the wrong one here: it answers
122
+ "what does the installed distribution claim", which is the same number for a fresh process and
123
+ a stale one, and its path caches make the negative case unreliable. Parsed with `ast` rather
124
+ than imported or regexed — re-importing would either return the cached stale module or execute
125
+ freshly-installed code inside a process running the old version, and neither is something a
126
+ health check should do.
127
+
128
+ Never raises, and stays silent whenever it cannot be sure: a missing file, an unparseable
129
+ source, or an absent `__version__` all mean "no claim", because a false upgrade prompt costs
130
+ more trust than a missed one.
131
+ """
132
+ try:
133
+ import ast as _ast
134
+
135
+ import codeintel
136
+ running = getattr(codeintel, "__version__", None)
137
+ path = getattr(codeintel, "__file__", None)
138
+ if not isinstance(running, str) or not path:
139
+ return None
140
+ src = pathlib.Path(path).read_text(encoding="utf-8", errors="replace")
141
+ for node in _ast.parse(src).body:
142
+ if not isinstance(node, _ast.Assign):
143
+ continue
144
+ for target in node.targets:
145
+ if isinstance(target, _ast.Name) and target.id == "__version__":
146
+ on_disk = _ast.literal_eval(node.value)
147
+ if isinstance(on_disk, str) and on_disk != running:
148
+ return (running, on_disk)
149
+ return None
150
+ except Exception:
151
+ return None
152
+ return None
153
+
154
+
109
155
  def collect_versions(engines: dict) -> dict:
110
156
  """Versions of the external backends each engine depends on.
111
157
 
@@ -270,12 +316,27 @@ def run_doctor(
270
316
  healthy = all(
271
317
  e.get("status") != "fail" for n, e in engines.items() if n not in _OPTIONAL_ENGINES
272
318
  )
319
+ # A skew is reported even when every engine is green, because it makes all the other numbers
320
+ # untrustworthy: they describe the code this process loaded, not the code that is installed.
321
+ skew = running_version_skew()
273
322
  return {
274
323
  "ok": True,
275
324
  "project_root": root,
276
325
  "deep": bool(deep),
277
326
  "treesitter": treesitter,
278
327
  "versions": versions,
328
+ "version_skew": (
329
+ {
330
+ "running": skew[0],
331
+ "installed": skew[1],
332
+ "remediation": (
333
+ f"this process is serving {skew[0]} while {skew[1]} is installed — "
334
+ "restart the codeintel server (or the agent hosting it) to pick it up"
335
+ ),
336
+ }
337
+ if skew
338
+ else None
339
+ ),
279
340
  "summary": {"ready": ready, "total": len(engines), "healthy": healthy},
280
341
  "engines": engines,
281
342
  "registrations": collect_registrations(),
@@ -326,6 +387,16 @@ def render_doctor_text(report: dict) -> str:
326
387
  if rem:
327
388
  out.append(" " + c.bold(c.cyan("fix:")) + " " + rem)
328
389
 
390
+ # Rendered with the engine notes rather than in the version block: it is a gap with a fix,
391
+ # which is what this section is for, and it outranks the engine rows — those describe the
392
+ # loaded code, and a skew means the loaded code is not the installed code.
393
+ skew = report.get("version_skew")
394
+ if isinstance(skew, dict) and skew.get("running") and skew.get("installed"):
395
+ out.append(" " + c.dim("└─") + " " + c.cyan("codeintel")
396
+ + f": serving {skew['running']}, installed is {skew['installed']}")
397
+ out.append(" " + c.bold(c.cyan("fix:")) + " restart the codeintel server "
398
+ "(or the agent hosting it) to pick up the installed version")
399
+
329
400
  summ = report.get("summary", {})
330
401
  ready, total, healthy = summ.get("ready", "?"), summ.get("total", "?"), summ.get("healthy")
331
402
  count = c.bold(f"{ready} / {total}")
@@ -1070,6 +1070,15 @@ class GraphProvider:
1070
1070
  does filter them: a callee in a different language family than the caller, or in a file that
1071
1071
  is not code at all, is not a callee — it is a name collision, and dropping it costs nothing
1072
1072
  real. What survives is marked so the caller knows the resolution is name-based.
1073
+
1074
+ The family check is resolved PER ROW against that row's own `a.file_path`, not against the
1075
+ set of families across every row. `target` can itself be a bare name shared by callers in
1076
+ more than one language, and comparing a row's callee to the UNION of every caller's family
1077
+ let a genuine collision hide behind an unrelated caller: three rows for a Python `target`
1078
+ and a TypeScript `target` made `caller_families = {python, ts-js}`, so a `.ts` name-collision
1079
+ callee reached from the *Python* caller passed the check — it shares a family with the
1080
+ wrong caller. Each row already carries the one caller file it actually came from, so that is
1081
+ what it is checked against.
1073
1082
  """
1074
1083
  cypher = (
1075
1084
  f'MATCH (a)-[c:CALLS|USAGE]->(b) WHERE a.name="{_cypher_literal(target)}" '
@@ -1079,10 +1088,6 @@ class GraphProvider:
1079
1088
  if not rows:
1080
1089
  return None
1081
1090
 
1082
- caller_families = {
1083
- _lang_family(str(r.get("a.file_path") or "")) for r in rows
1084
- } - {""}
1085
-
1086
1091
  kept: list[dict] = []
1087
1092
  dropped = 0
1088
1093
  for r in rows:
@@ -1090,9 +1095,10 @@ class GraphProvider:
1090
1095
  if _is_non_code(path):
1091
1096
  dropped += 1 # a data/doc file cannot be a callee
1092
1097
  continue
1093
- fam = _lang_family(path)
1094
- if fam and caller_families and fam not in caller_families:
1095
- dropped += 1 # cross-language name collision
1098
+ callee_fam = _lang_family(path)
1099
+ caller_fam = _lang_family(str(r.get("a.file_path") or ""))
1100
+ if callee_fam and caller_fam and callee_fam != caller_fam:
1101
+ dropped += 1 # cross-language name collision, resolved against THIS row's own caller
1096
1102
  continue
1097
1103
  kept.append(r)
1098
1104
 
@@ -132,6 +132,7 @@ _STATUS_FALLBACK: dict = {
132
132
  "model": None,
133
133
  "healthy": False,
134
134
  "readiness": {},
135
+ "version_skew": None,
135
136
  }
136
137
 
137
138
 
@@ -228,6 +229,11 @@ def _code_status_handler_inner(args: dict) -> dict:
228
229
  "healthy": bool(summary.get("healthy")),
229
230
  "readiness": readiness,
230
231
  "versions": report.get("versions", {}) if isinstance(report, dict) else {},
232
+ # Null on the normal path. Non-null means every other field above describes the code
233
+ # this process loaded at startup rather than the code that is installed — so it is
234
+ # surfaced on `status`, not just in `doctor`, because `status` is what a caller checks
235
+ # when an expected fix appears to be missing.
236
+ "version_skew": report.get("version_skew") if isinstance(report, dict) else None,
231
237
  }
232
238
  except Exception:
233
239
  return dict(_STATUS_FALLBACK)
@@ -0,0 +1,99 @@
1
+ """The README's claims about CI must match the workflow that actually runs.
2
+
3
+ This guards a drift that has now happened in BOTH directions. The `deadcode` case was the
4
+ optimistic one: the docs advertised a capability the product had withdrawn. This is the pessimistic
5
+ one — the README told readers the graph and LSP backends were "not installed in CI, so those two
6
+ engines are exercised against hand-authored mocks", long after dedicated contract jobs had been
7
+ added that install the real backends and run against them. Understating assurance is a smaller sin
8
+ than overstating it, but it is the same defect: a hand-written claim about a machine-readable fact,
9
+ with nothing checking the two still agree.
10
+
11
+ Facts are derived from `.github/workflows/ci.yml`, never typed here — a hand-typed expectation is
12
+ the thing being guarded against. Parsed as text rather than with PyYAML, which is present in the
13
+ dev environment but is not a declared dependency, so importing it would make this test the reason a
14
+ clean install fails.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import re
19
+ from pathlib import Path
20
+
21
+ import pytest
22
+
23
+ ROOT = Path(__file__).resolve().parent.parent
24
+ CI = ROOT / ".github" / "workflows" / "ci.yml"
25
+ README = ROOT / "README.md"
26
+
27
+
28
+ def _job_blocks(text: str) -> dict[str, str]:
29
+ """``{job_name: body}`` for each top-level job — a 2-space key under `jobs:`."""
30
+ lines = text.splitlines()
31
+ starts: list[tuple[int, str]] = [
32
+ (i, m.group(1))
33
+ for i, line in enumerate(lines)
34
+ if (m := re.match(r"^ ([a-z0-9][a-z0-9_-]*):\s*$", line))
35
+ ]
36
+ blocks: dict[str, str] = {}
37
+ for idx, (line_no, name) in enumerate(starts):
38
+ end = starts[idx + 1][0] if idx + 1 < len(starts) else len(lines)
39
+ blocks[name] = "\n".join(lines[line_no:end])
40
+ return blocks
41
+
42
+
43
+ @pytest.fixture(scope="module")
44
+ def ci_text() -> str:
45
+ if not CI.exists():
46
+ pytest.skip("ci.yml not present")
47
+ return CI.read_text(encoding="utf-8")
48
+
49
+
50
+ @pytest.fixture(scope="module")
51
+ def readme_text() -> str:
52
+ return README.read_text(encoding="utf-8")
53
+
54
+
55
+ def test_the_workflow_actually_has_contract_jobs(ci_text):
56
+ # Non-vacuity. Every assertion below is of the form "for each contract job ..." and would pass
57
+ # trivially against a workflow that has none — which is exactly the state this file exists to
58
+ # notice, so it is asserted rather than assumed.
59
+ contract = [n for n in _job_blocks(ci_text) if "contract" in n]
60
+ assert contract, "no *-contract jobs found in ci.yml — the parser or the workflow changed"
61
+
62
+
63
+ def test_readme_names_every_contract_job(ci_text, readme_text):
64
+ missing = [
65
+ name for name in _job_blocks(ci_text)
66
+ if "contract" in name and name not in readme_text
67
+ ]
68
+ assert not missing, (
69
+ f"ci.yml runs contract job(s) {missing} that the README never mentions — the docs "
70
+ "understate what is actually verified"
71
+ )
72
+
73
+
74
+ def test_readme_does_not_claim_backends_are_absent_from_ci(ci_text, readme_text):
75
+ # The precise stale sentence, tied to the fact that falsifies it.
76
+ installs_graph_backend = "codebase-memory-mcp==" in ci_text
77
+ if installs_graph_backend:
78
+ assert "not installed in CI" not in readme_text, (
79
+ "ci.yml installs the pinned graph backend, but the README still says the backends are "
80
+ "'not installed in CI' — the claim was true before the contract jobs existed"
81
+ )
82
+
83
+
84
+ def test_readme_reflects_which_contract_jobs_actually_gate(ci_text, readme_text):
85
+ """A non-gating job must not read as a gating one.
86
+
87
+ `continue-on-error` is the difference between "a serena regression turns CI red" and "a serena
88
+ regression is a note someone may notice". A reader deciding how much to trust the LSP path
89
+ needs that distinction, so if any contract job carries the flag, the README has to say so.
90
+ """
91
+ soft = [
92
+ name for name, body in _job_blocks(ci_text).items()
93
+ if "contract" in name and re.search(r"^\s*continue-on-error:\s*true\s*$", body, re.M)
94
+ ]
95
+ if soft:
96
+ assert "continue-on-error" in readme_text, (
97
+ f"contract job(s) {soft} are continue-on-error, so they do not gate a release — the "
98
+ "README describes the contract coverage without saying which parts are non-gating"
99
+ )