algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
@@ -0,0 +1,220 @@
1
+ """CoT quality scoring (I1 + I3).
2
+
3
+ Scoring a CoT block on sequencing markers, length ratio, and verification
4
+ cadence. Used by algo-cli evals to flag under-thinking, over-thinking, and
5
+ stream-of-consciousness reasoning.
6
+
7
+ Provenance: ALGO.md I1 (CoT-Proportional Reasoning) and I3 (Sequenced
8
+ Reasoning Markers). Calibration: Fable-5 corpus (4,665 rows, 100% coverage
9
+ of cot/completion fields; median cot_ratio = 1.14, mean = 1.28).
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass
15
+ from enum import Enum
16
+
17
+
18
+ class SequencePattern(str, Enum):
19
+ """Recognized tool-sequence patterns from the Fable-5 trace audit (I7)."""
20
+
21
+ EMPTY = "empty"
22
+ TDD_EDIT_TEST_EDIT = "tdd_edit_test_edit"
23
+ VERIFY_AFTER_EDIT = "verify_after_edit"
24
+ SHELL_INSPECT_LOOP = "shell_inspect_loop"
25
+ UNCLASSIFIED = "unclassified"
26
+
27
+
28
+ class Band(str, Enum):
29
+ UNDER = "under_thinking" # cot_ratio < 0.5
30
+ IN_BAND = "in_band" # 0.5 <= cot_ratio <= 5.0 (Fable-5 p90 = 1.79, max observed 4.2)
31
+ OVER = "over_thinking" # cot_ratio > 5.0
32
+
33
+
34
+ # Sequential markers found in well-structured CoT.
35
+ MARKER_RE = re.compile(
36
+ r"\b(First|Next|Then|Finally|Step|Now)\b",
37
+ re.IGNORECASE,
38
+ )
39
+ # "First, ... Next, ..." must appear in that order to count as well-sequenced.
40
+ WELL_SEQUENCED_RE = re.compile(
41
+ r"\bFirst\b[\s\S]*?\bNext\b",
42
+ re.IGNORECASE,
43
+ )
44
+
45
+
46
+ @dataclass
47
+ class ToolSequenceQuality:
48
+ """Result of scoring a tool-call sequence for healthy verify cadence (I7)."""
49
+
50
+ tool_names: tuple[str, ...]
51
+ pattern: SequencePattern
52
+ sequence_score: float
53
+ verification_present: bool
54
+ edit_count: int
55
+ shell_count: int
56
+ read_count: int
57
+ summary: str
58
+
59
+ def to_dict(self) -> dict:
60
+ return {
61
+ "tool_names": list(self.tool_names),
62
+ "pattern": self.pattern.value,
63
+ "sequence_score": round(self.sequence_score, 2),
64
+ "verification_present": self.verification_present,
65
+ "edit_count": self.edit_count,
66
+ "shell_count": self.shell_count,
67
+ "read_count": self.read_count,
68
+ "summary": self.summary,
69
+ }
70
+
71
+
72
+ @dataclass
73
+ class CoTQuality:
74
+ """Result of scoring one CoT block against a completion string."""
75
+
76
+ cot_chars: int
77
+ completion_chars: int
78
+ cot_ratio: float
79
+ band: Band
80
+ markers: tuple[str, ...]
81
+ well_sequenced: bool
82
+ structure_score: float
83
+ summary: str
84
+
85
+ def to_dict(self) -> dict:
86
+ return {
87
+ "cot_chars": self.cot_chars,
88
+ "completion_chars": self.completion_chars,
89
+ "cot_ratio": round(self.cot_ratio, 2),
90
+ "band": self.band.value,
91
+ "markers": list(self.markers),
92
+ "well_sequenced": self.well_sequenced,
93
+ "structure_score": round(self.structure_score, 2),
94
+ "summary": self.summary,
95
+ }
96
+
97
+
98
+ def _normalize_tool_name(name: str) -> str:
99
+ lowered = (name or "").strip().lower()
100
+ if "edit" in lowered or lowered in {"write_file", "batch_edit"}:
101
+ return "edit"
102
+ if "bash" in lowered or "shell" in lowered or lowered == "run_shell":
103
+ return "bash"
104
+ if "read" in lowered or lowered in {"grep", "search_files", "find_unique_anchor"}:
105
+ return "read"
106
+ return lowered or "unknown"
107
+
108
+
109
+ def score_tool_sequence(tool_names: list[str] | tuple[str, ...]) -> ToolSequenceQuality:
110
+ """Score a sequence of tool calls for the Fable-5 TDD cadence (I7).
111
+
112
+ The strongest healthy pattern is Edit→Bash→Edit: change, verify, repair.
113
+ Bash→Bash→Read is recognized as an inspection loop, useful but weaker.
114
+ """
115
+ names = tuple(tool_names or ())
116
+ normalized = [_normalize_tool_name(name) for name in names]
117
+ edit_count = normalized.count("edit")
118
+ shell_count = normalized.count("bash")
119
+ read_count = normalized.count("read")
120
+ verification_present = shell_count > 0
121
+
122
+ pattern = SequencePattern.UNCLASSIFIED
123
+ score = 0.0
124
+ for idx in range(len(normalized) - 2):
125
+ window = normalized[idx:idx + 3]
126
+ if window == ["edit", "bash", "edit"]:
127
+ pattern = SequencePattern.TDD_EDIT_TEST_EDIT
128
+ score = 1.0
129
+ break
130
+ else:
131
+ for idx in range(len(normalized) - 1):
132
+ window = normalized[idx:idx + 2]
133
+ if window == ["edit", "bash"]:
134
+ pattern = SequencePattern.VERIFY_AFTER_EDIT
135
+ score = 0.75
136
+ break
137
+ else:
138
+ if len(normalized) >= 3 and normalized[:3] == ["bash", "bash", "read"]:
139
+ pattern = SequencePattern.SHELL_INSPECT_LOOP
140
+ score = 0.55
141
+ elif not normalized:
142
+ pattern = SequencePattern.EMPTY
143
+ score = 0.0
144
+ elif verification_present:
145
+ score = 0.35
146
+ elif read_count > 0:
147
+ score = 0.2
148
+
149
+ summary = (
150
+ f"tools={len(names)}, pattern={pattern.value}, score={score:.2f}, "
151
+ f"edit={edit_count}, shell={shell_count}, read={read_count}"
152
+ )
153
+ return ToolSequenceQuality(
154
+ tool_names=names,
155
+ pattern=pattern,
156
+ sequence_score=score,
157
+ verification_present=verification_present,
158
+ edit_count=edit_count,
159
+ shell_count=shell_count,
160
+ read_count=read_count,
161
+ summary=summary,
162
+ )
163
+
164
+
165
+ def score_cot(cot: str, completion: str) -> CoTQuality:
166
+ """Score one CoT block.
167
+
168
+ Args:
169
+ cot: The reasoning / thinking block preceding a tool call. May be empty.
170
+ completion: The actual tool call or response. May be empty.
171
+
172
+ Returns:
173
+ A CoTQuality record with band, markers, well_sequenced, and a [0, 1]
174
+ structure_score suitable for eval grading.
175
+ """
176
+ cot_len = len(cot or "")
177
+ comp_len = len(completion or "")
178
+ ratio = cot_len / max(1, comp_len)
179
+
180
+ if ratio < 0.5:
181
+ band = Band.UNDER
182
+ elif ratio > 5.0:
183
+ band = Band.OVER
184
+ else:
185
+ band = Band.IN_BAND
186
+
187
+ markers = tuple(m.group(0) for m in MARKER_RE.finditer(cot or ""))
188
+ well_seq = bool(WELL_SEQUENCED_RE.search(cot or ""))
189
+
190
+ # marker_score capped at 0.6 (3 markers); seq_score 0.4; band_bonus 0.2
191
+ marker_score = min(0.6, 0.2 * len(markers))
192
+ seq_score = 0.4 if well_seq else 0.0
193
+ band_bonus = 0.2 if band == Band.IN_BAND else 0.0
194
+ score = min(1.0, marker_score + seq_score + band_bonus)
195
+
196
+ summary = (
197
+ f"cot={cot_len}ch, completion={comp_len}ch, ratio={ratio:.2f} "
198
+ f"({band.value}), markers={len(markers)}, sequenced={well_seq}, "
199
+ f"score={score:.2f}"
200
+ )
201
+ return CoTQuality(
202
+ cot_chars=cot_len,
203
+ completion_chars=comp_len,
204
+ cot_ratio=ratio,
205
+ band=band,
206
+ markers=markers,
207
+ well_sequenced=well_seq,
208
+ structure_score=score,
209
+ summary=summary,
210
+ )
211
+
212
+
213
+ __all__ = [
214
+ "Band",
215
+ "CoTQuality",
216
+ "SequencePattern",
217
+ "ToolSequenceQuality",
218
+ "score_cot",
219
+ "score_tool_sequence",
220
+ ]
@@ -0,0 +1,401 @@
1
+ """Bounded, offline effectiveness benchmark for harness retrieval.
2
+
3
+ The benchmark reads a snapshot of the persisted harness index and owns every
4
+ BM25 object it creates. It deliberately does not call ``harness.search_index``
5
+ or clear/populate any process-global retrieval cache.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ import json
12
+ import statistics
13
+ import time
14
+ from collections.abc import Callable, Mapping, Sequence
15
+ from typing import Any
16
+
17
+ from .. import harness
18
+ from ..retrieval_algorithms import (
19
+ FULL_SORT_THRESHOLD,
20
+ BM25Index,
21
+ lexical_tokens,
22
+ stable_top_k,
23
+ )
24
+
25
+ BENCHMARK_VERSION = "harness-retrieval-v1"
26
+ CANARY_QUERIES: tuple[str, ...] = (
27
+ "rate your harness",
28
+ "harness context",
29
+ "memory recall",
30
+ "verification before completion",
31
+ "index-compute-lab",
32
+ )
33
+ CANONICAL_ALGO_ID = "algo-cli:algorithm:ALGO.md"
34
+ STABILITY_PASSES = 3
35
+ CANARY_LIMIT = 5
36
+ COLD_SAMPLE_TARGET = 5
37
+ REUSABLE_WARMUPS = 3
38
+ REUSABLE_SAMPLE_TARGET = 9
39
+ MIN_REUSABLE_SPEEDUP = 1.5
40
+ MAX_WARM_MAD_RATIO = 0.25
41
+ MAX_BENCHMARK_RECORDS = 2_048
42
+ MAX_BENCHMARK_TEXT_CHARS = 40_000
43
+
44
+ SearchFn = Callable[[str, int], Sequence[Any]]
45
+ ClockFn = Callable[[], int]
46
+
47
+
48
+ def _digest(value: Any) -> str:
49
+ encoded = json.dumps(
50
+ value,
51
+ ensure_ascii=True,
52
+ sort_keys=True,
53
+ separators=(",", ":"),
54
+ default=str,
55
+ ).encode("utf-8")
56
+ return hashlib.sha256(encoded).hexdigest()
57
+
58
+
59
+ def _load_persisted_index() -> tuple[dict[str, Any], str | None]:
60
+ """Read the live index file without invoking the global index cache."""
61
+
62
+ try:
63
+ payload = json.loads(harness.INDEX_PATH.read_text(encoding="utf-8"))
64
+ except FileNotFoundError:
65
+ return {"records": []}, f"harness index not found: {harness.INDEX_PATH}"
66
+ except (OSError, json.JSONDecodeError) as exc:
67
+ return {"records": []}, f"could not read harness index: {exc}"
68
+ if not isinstance(payload, dict):
69
+ return {"records": []}, "harness index root is not an object"
70
+ return payload, None
71
+
72
+
73
+ def _eligible_records(index: Mapping[str, Any]) -> list[dict[str, Any]]:
74
+ raw_records = index.get("records")
75
+ if not isinstance(raw_records, list):
76
+ return []
77
+ return [
78
+ record
79
+ for record in raw_records
80
+ if isinstance(record, dict) and not harness.is_excluded_from_retrieval(record)
81
+ ]
82
+
83
+
84
+ def _bounded_records(records: Sequence[dict[str, Any]]) -> list[dict[str, Any]]:
85
+ if len(records) <= MAX_BENCHMARK_RECORDS:
86
+ return list(records)
87
+ # Keep canonical/project-local evidence in the bounded corpus, then retain
88
+ # source order so repeated runs over an unchanged index remain identical.
89
+ prioritized = sorted(
90
+ enumerate(records),
91
+ key=lambda pair: (
92
+ str(pair[1].get("id") or "") != CANONICAL_ALGO_ID,
93
+ str(pair[1].get("harness") or "") != "algo-cli",
94
+ pair[0],
95
+ ),
96
+ )
97
+ return [record for _position, record in prioritized[:MAX_BENCHMARK_RECORDS]]
98
+
99
+
100
+ def _search_text(record: Mapping[str, Any]) -> str:
101
+ text = str(record.get("search_text") or "")
102
+ if text:
103
+ return text[:MAX_BENCHMARK_TEXT_CHARS]
104
+ return " ".join(
105
+ str(record.get(field) or "")
106
+ for field in (
107
+ "id",
108
+ "harness",
109
+ "kind",
110
+ "title",
111
+ "description",
112
+ "tags",
113
+ "relative_path",
114
+ "summary",
115
+ )
116
+ ).lower()[:MAX_BENCHMARK_TEXT_CHARS]
117
+
118
+
119
+ def _local_search(
120
+ records: Sequence[dict[str, Any]],
121
+ bm25: BM25Index,
122
+ query: str,
123
+ limit: int,
124
+ ) -> list[dict[str, Any]]:
125
+ terms = lexical_tokens(query)
126
+ if not terms:
127
+ return []
128
+ scored: list[tuple[float, dict[str, Any]]] = []
129
+ for lexical_score, record in zip(bm25.scores(terms), records):
130
+ score = lexical_score + float(harness.score_record(record, terms))
131
+ if score > 0.0:
132
+ scored.append((score, record))
133
+ return [
134
+ record
135
+ for _score, record in stable_top_k(
136
+ scored,
137
+ limit,
138
+ score=lambda pair: pair[0],
139
+ )
140
+ ]
141
+
142
+
143
+ def _result_id(result: Any) -> str:
144
+ if isinstance(result, Mapping):
145
+ return str(result.get("id") or "")
146
+ return str(result or "")
147
+
148
+
149
+ def _stable_top_k_parity() -> tuple[bool, str]:
150
+ """Exercise the heap branch above its adaptive crossover threshold."""
151
+
152
+ values = [
153
+ (index, (index * 2_654_435_761) % 97)
154
+ for index in range(FULL_SORT_THRESHOLD + 257)
155
+ ]
156
+ expected = sorted(values, key=lambda item: item[1], reverse=True)[:17]
157
+ actual = stable_top_k(values, 17, score=lambda item: item[1])
158
+ return actual == expected, _digest(actual)
159
+
160
+
161
+ def _measure_ns(operation: Callable[[], Any], clock_ns: ClockFn) -> int:
162
+ started = int(clock_ns())
163
+ operation()
164
+ return max(0, int(clock_ns()) - started)
165
+
166
+
167
+ def _median_absolute_deviation(values: Sequence[int], median: float) -> float:
168
+ if not values:
169
+ return 0.0
170
+ return float(statistics.median(abs(float(value) - median) for value in values))
171
+
172
+
173
+ def _milliseconds(value_ns: float) -> float:
174
+ return round(float(value_ns) / 1_000_000.0, 6)
175
+
176
+
177
+ def run_harness_retrieval_benchmark(
178
+ index: Mapping[str, Any] | None = None,
179
+ search_fn: SearchFn | None = None,
180
+ *,
181
+ clock_ns: ClockFn | None = None,
182
+ ) -> dict[str, Any]:
183
+ """Run the bounded retrieval benchmark and return JSON-serializable evidence.
184
+
185
+ Args:
186
+ index: Optional index payload. When omitted, read the persisted live index
187
+ directly without touching the harness index cache.
188
+ search_fn: Optional ``(query, limit) -> results`` function for canary
189
+ checks. Timing always uses local BM25 instances.
190
+ clock_ns: Optional monotonic nanosecond clock for deterministic tests.
191
+ """
192
+
193
+ load_error: str | None = None
194
+ if index is None:
195
+ index_payload, load_error = _load_persisted_index()
196
+ else:
197
+ index_payload = dict(index)
198
+ raw_records = index_payload.get("records")
199
+ index_record_count = len(raw_records) if isinstance(raw_records, list) else 0
200
+ eligible_records = _eligible_records(index_payload)
201
+ records = _bounded_records(eligible_records)
202
+ documents = [_search_text(record) for record in records]
203
+ reusable_index = BM25Index(documents)
204
+ active_search: SearchFn
205
+ if search_fn is None:
206
+ def active_search(query: str, limit: int) -> list[dict[str, Any]]:
207
+ return _local_search(records, reusable_index, query, limit)
208
+ else:
209
+ active_search = search_fn
210
+
211
+ ranking_passes: list[list[list[str]]] = []
212
+ search_errors: list[str] = []
213
+ for _pass_index in range(STABILITY_PASSES):
214
+ pass_rankings: list[list[str]] = []
215
+ for query in CANARY_QUERIES:
216
+ try:
217
+ results = active_search(query, CANARY_LIMIT)
218
+ ids = [_result_id(result) for result in results]
219
+ pass_rankings.append([record_id for record_id in ids if record_id])
220
+ except Exception as exc:
221
+ search_errors.append(f"{query}: {type(exc).__name__}: {exc}")
222
+ pass_rankings.append([])
223
+ ranking_passes.append(pass_rankings)
224
+
225
+ observation_count = len(CANARY_QUERIES) * STABILITY_PASSES
226
+ nonempty_observations = sum(
227
+ 1
228
+ for pass_rankings in ranking_passes
229
+ for ranking in pass_rankings
230
+ if ranking
231
+ )
232
+ stable_rankings = bool(ranking_passes) and all(
233
+ pass_rankings == ranking_passes[0]
234
+ for pass_rankings in ranking_passes[1:]
235
+ )
236
+ canonical_algo_top1 = bool(ranking_passes) and all(
237
+ pass_rankings
238
+ and pass_rankings[0]
239
+ and pass_rankings[0][0] == CANONICAL_ALGO_ID
240
+ for pass_rankings in ranking_passes
241
+ )
242
+ top_k_parity, top_k_digest = _stable_top_k_parity()
243
+
244
+ clock = clock_ns or time.perf_counter_ns
245
+ query_terms = [lexical_tokens(query) for query in CANARY_QUERIES]
246
+
247
+ def score_all_queries(bm25: BM25Index) -> list[list[float]]:
248
+ return [bm25.scores(terms) for terms in query_terms]
249
+
250
+ cold_samples_ns: list[int] = []
251
+ for _sample in range(COLD_SAMPLE_TARGET):
252
+ cold_samples_ns.append(
253
+ _measure_ns(
254
+ lambda: score_all_queries(BM25Index(documents)),
255
+ clock,
256
+ )
257
+ )
258
+
259
+ for _warmup in range(REUSABLE_WARMUPS):
260
+ score_all_queries(reusable_index)
261
+ warm_samples_ns = [
262
+ _measure_ns(lambda: score_all_queries(reusable_index), clock)
263
+ for _sample in range(REUSABLE_SAMPLE_TARGET)
264
+ ]
265
+
266
+ cold_median_ns = float(statistics.median(cold_samples_ns)) if cold_samples_ns else 0.0
267
+ warm_median_ns = float(statistics.median(warm_samples_ns)) if warm_samples_ns else 0.0
268
+ cold_mad_ns = _median_absolute_deviation(cold_samples_ns, cold_median_ns)
269
+ warm_mad_ns = _median_absolute_deviation(warm_samples_ns, warm_median_ns)
270
+ warm_mad_ratio = warm_mad_ns / warm_median_ns if warm_median_ns > 0.0 else None
271
+ speedup = cold_median_ns / warm_median_ns if warm_median_ns > 0.0 else None
272
+ sufficient_samples = (
273
+ len(cold_samples_ns) >= COLD_SAMPLE_TARGET
274
+ and len(warm_samples_ns) >= REUSABLE_SAMPLE_TARGET
275
+ and cold_median_ns > 0.0
276
+ and warm_median_ns > 0.0
277
+ )
278
+
279
+ correctness_failures: list[str] = []
280
+ if load_error:
281
+ correctness_failures.append(load_error)
282
+ if not records:
283
+ correctness_failures.append("no eligible harness records")
284
+ if search_errors:
285
+ correctness_failures.append("canary search raised an exception")
286
+ if nonempty_observations != observation_count:
287
+ correctness_failures.append(
288
+ f"nonempty canary observations {nonempty_observations}/{observation_count}"
289
+ )
290
+ if not stable_rankings:
291
+ correctness_failures.append("canary rankings changed across stability passes")
292
+ if not canonical_algo_top1:
293
+ correctness_failures.append("canonical ALGO record was not top-1")
294
+ if not top_k_parity:
295
+ correctness_failures.append("stable_top_k diverged from a full stable sort")
296
+ correctness_passed = not correctness_failures
297
+
298
+ performance_warnings: list[str] = []
299
+ if not sufficient_samples:
300
+ performance_warnings.append("timing samples were insufficient or below clock resolution")
301
+ if speedup is None or speedup < MIN_REUSABLE_SPEEDUP:
302
+ performance_warnings.append(
303
+ f"reusable BM25 speedup was below {MIN_REUSABLE_SPEEDUP:.1f}x"
304
+ )
305
+ if warm_mad_ratio is None or warm_mad_ratio > MAX_WARM_MAD_RATIO:
306
+ performance_warnings.append(
307
+ f"warm MAD ratio exceeded {MAX_WARM_MAD_RATIO:.2f}"
308
+ )
309
+
310
+ if not correctness_passed:
311
+ status = "fail"
312
+ reason = "retrieval correctness failed: " + "; ".join(correctness_failures)
313
+ elif performance_warnings:
314
+ status = "warn"
315
+ reason = "retrieval correctness passed; " + "; ".join(performance_warnings)
316
+ else:
317
+ assert speedup is not None and warm_mad_ratio is not None
318
+ status = "pass"
319
+ reason = (
320
+ "retrieval correctness passed; reusable BM25 speedup "
321
+ f"{speedup:.2f}x with warm MAD ratio {warm_mad_ratio:.3f}"
322
+ )
323
+
324
+ index_fingerprint = [
325
+ {
326
+ "id": str(record.get("id") or ""),
327
+ "harness": str(record.get("harness") or ""),
328
+ "kind": str(record.get("kind") or ""),
329
+ "relative_path": str(record.get("relative_path") or ""),
330
+ "search_text": _search_text(record),
331
+ }
332
+ for record in records
333
+ ]
334
+ cold_samples_ms = [_milliseconds(value) for value in cold_samples_ns]
335
+ warm_samples_ms = [_milliseconds(value) for value in warm_samples_ns]
336
+ return {
337
+ "benchmark_version": BENCHMARK_VERSION,
338
+ "status": status,
339
+ "reason": reason,
340
+ "correctness": {
341
+ "passed": correctness_passed,
342
+ "failures": correctness_failures,
343
+ "search_errors": search_errors,
344
+ "canary_query_count": len(CANARY_QUERIES),
345
+ "stability_passes": STABILITY_PASSES,
346
+ "observation_count": observation_count,
347
+ "nonempty_observations": nonempty_observations,
348
+ "stable_rankings": stable_rankings,
349
+ "canonical_algo_top1": canonical_algo_top1,
350
+ "stable_top_k_parity": top_k_parity,
351
+ "first_pass_rankings": {
352
+ query: ranking_passes[0][index]
353
+ for index, query in enumerate(CANARY_QUERIES)
354
+ },
355
+ },
356
+ "performance": {
357
+ "operation": "bm25_build_plus_all_queries_vs_reusable_all_queries",
358
+ "cold_sample_count": len(cold_samples_ns),
359
+ "warmup_count": REUSABLE_WARMUPS,
360
+ "warm_sample_count": len(warm_samples_ns),
361
+ "cold_samples_ms": cold_samples_ms,
362
+ "warm_samples_ms": warm_samples_ms,
363
+ "cold_median_ms": _milliseconds(cold_median_ns),
364
+ "warm_median_ms": _milliseconds(warm_median_ns),
365
+ "cold_mad_ms": _milliseconds(cold_mad_ns),
366
+ "warm_mad_ms": _milliseconds(warm_mad_ns),
367
+ "warm_mad_ratio": (
368
+ round(warm_mad_ratio, 6) if warm_mad_ratio is not None else None
369
+ ),
370
+ "speedup": round(speedup, 6) if speedup is not None else None,
371
+ "sufficient_samples": sufficient_samples,
372
+ "minimum_speedup": MIN_REUSABLE_SPEEDUP,
373
+ "maximum_warm_mad_ratio": MAX_WARM_MAD_RATIO,
374
+ },
375
+ "evidence": {
376
+ "index_record_count": index_record_count,
377
+ "eligible_record_count": len(eligible_records),
378
+ "benchmark_record_count": len(records),
379
+ "maximum_benchmark_records": MAX_BENCHMARK_RECORDS,
380
+ "maximum_text_chars": MAX_BENCHMARK_TEXT_CHARS,
381
+ "index_digest": _digest(index_fingerprint),
382
+ "canary_digest": _digest(CANARY_QUERIES),
383
+ "ranking_digest": _digest(ranking_passes),
384
+ "stable_top_k_digest": top_k_digest,
385
+ "timing_digest": _digest(
386
+ {
387
+ "cold_samples_ns": cold_samples_ns,
388
+ "warm_samples_ns": warm_samples_ns,
389
+ }
390
+ ),
391
+ },
392
+ }
393
+
394
+
395
+ __all__ = [
396
+ "BENCHMARK_VERSION",
397
+ "CANARY_QUERIES",
398
+ "CANONICAL_ALGO_ID",
399
+ "MAX_BENCHMARK_RECORDS",
400
+ "run_harness_retrieval_benchmark",
401
+ ]