algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""CoT quality scoring (I1 + I3).
|
|
2
|
+
|
|
3
|
+
Scoring a CoT block on sequencing markers, length ratio, and verification
|
|
4
|
+
cadence. Used by algo-cli evals to flag under-thinking, over-thinking, and
|
|
5
|
+
stream-of-consciousness reasoning.
|
|
6
|
+
|
|
7
|
+
Provenance: ALGO.md I1 (CoT-Proportional Reasoning) and I3 (Sequenced
|
|
8
|
+
Reasoning Markers). Calibration: Fable-5 corpus (4,665 rows, 100% coverage
|
|
9
|
+
of cot/completion fields; median cot_ratio = 1.14, mean = 1.28).
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from enum import Enum
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class SequencePattern(str, Enum):
|
|
19
|
+
"""Recognized tool-sequence patterns from the Fable-5 trace audit (I7)."""
|
|
20
|
+
|
|
21
|
+
EMPTY = "empty"
|
|
22
|
+
TDD_EDIT_TEST_EDIT = "tdd_edit_test_edit"
|
|
23
|
+
VERIFY_AFTER_EDIT = "verify_after_edit"
|
|
24
|
+
SHELL_INSPECT_LOOP = "shell_inspect_loop"
|
|
25
|
+
UNCLASSIFIED = "unclassified"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Band(str, Enum):
|
|
29
|
+
UNDER = "under_thinking" # cot_ratio < 0.5
|
|
30
|
+
IN_BAND = "in_band" # 0.5 <= cot_ratio <= 5.0 (Fable-5 p90 = 1.79, max observed 4.2)
|
|
31
|
+
OVER = "over_thinking" # cot_ratio > 5.0
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# Sequential markers found in well-structured CoT.
|
|
35
|
+
MARKER_RE = re.compile(
|
|
36
|
+
r"\b(First|Next|Then|Finally|Step|Now)\b",
|
|
37
|
+
re.IGNORECASE,
|
|
38
|
+
)
|
|
39
|
+
# "First, ... Next, ..." must appear in that order to count as well-sequenced.
|
|
40
|
+
WELL_SEQUENCED_RE = re.compile(
|
|
41
|
+
r"\bFirst\b[\s\S]*?\bNext\b",
|
|
42
|
+
re.IGNORECASE,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class ToolSequenceQuality:
|
|
48
|
+
"""Result of scoring a tool-call sequence for healthy verify cadence (I7)."""
|
|
49
|
+
|
|
50
|
+
tool_names: tuple[str, ...]
|
|
51
|
+
pattern: SequencePattern
|
|
52
|
+
sequence_score: float
|
|
53
|
+
verification_present: bool
|
|
54
|
+
edit_count: int
|
|
55
|
+
shell_count: int
|
|
56
|
+
read_count: int
|
|
57
|
+
summary: str
|
|
58
|
+
|
|
59
|
+
def to_dict(self) -> dict:
|
|
60
|
+
return {
|
|
61
|
+
"tool_names": list(self.tool_names),
|
|
62
|
+
"pattern": self.pattern.value,
|
|
63
|
+
"sequence_score": round(self.sequence_score, 2),
|
|
64
|
+
"verification_present": self.verification_present,
|
|
65
|
+
"edit_count": self.edit_count,
|
|
66
|
+
"shell_count": self.shell_count,
|
|
67
|
+
"read_count": self.read_count,
|
|
68
|
+
"summary": self.summary,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass
|
|
73
|
+
class CoTQuality:
|
|
74
|
+
"""Result of scoring one CoT block against a completion string."""
|
|
75
|
+
|
|
76
|
+
cot_chars: int
|
|
77
|
+
completion_chars: int
|
|
78
|
+
cot_ratio: float
|
|
79
|
+
band: Band
|
|
80
|
+
markers: tuple[str, ...]
|
|
81
|
+
well_sequenced: bool
|
|
82
|
+
structure_score: float
|
|
83
|
+
summary: str
|
|
84
|
+
|
|
85
|
+
def to_dict(self) -> dict:
|
|
86
|
+
return {
|
|
87
|
+
"cot_chars": self.cot_chars,
|
|
88
|
+
"completion_chars": self.completion_chars,
|
|
89
|
+
"cot_ratio": round(self.cot_ratio, 2),
|
|
90
|
+
"band": self.band.value,
|
|
91
|
+
"markers": list(self.markers),
|
|
92
|
+
"well_sequenced": self.well_sequenced,
|
|
93
|
+
"structure_score": round(self.structure_score, 2),
|
|
94
|
+
"summary": self.summary,
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _normalize_tool_name(name: str) -> str:
|
|
99
|
+
lowered = (name or "").strip().lower()
|
|
100
|
+
if "edit" in lowered or lowered in {"write_file", "batch_edit"}:
|
|
101
|
+
return "edit"
|
|
102
|
+
if "bash" in lowered or "shell" in lowered or lowered == "run_shell":
|
|
103
|
+
return "bash"
|
|
104
|
+
if "read" in lowered or lowered in {"grep", "search_files", "find_unique_anchor"}:
|
|
105
|
+
return "read"
|
|
106
|
+
return lowered or "unknown"
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def score_tool_sequence(tool_names: list[str] | tuple[str, ...]) -> ToolSequenceQuality:
|
|
110
|
+
"""Score a sequence of tool calls for the Fable-5 TDD cadence (I7).
|
|
111
|
+
|
|
112
|
+
The strongest healthy pattern is Edit→Bash→Edit: change, verify, repair.
|
|
113
|
+
Bash→Bash→Read is recognized as an inspection loop, useful but weaker.
|
|
114
|
+
"""
|
|
115
|
+
names = tuple(tool_names or ())
|
|
116
|
+
normalized = [_normalize_tool_name(name) for name in names]
|
|
117
|
+
edit_count = normalized.count("edit")
|
|
118
|
+
shell_count = normalized.count("bash")
|
|
119
|
+
read_count = normalized.count("read")
|
|
120
|
+
verification_present = shell_count > 0
|
|
121
|
+
|
|
122
|
+
pattern = SequencePattern.UNCLASSIFIED
|
|
123
|
+
score = 0.0
|
|
124
|
+
for idx in range(len(normalized) - 2):
|
|
125
|
+
window = normalized[idx:idx + 3]
|
|
126
|
+
if window == ["edit", "bash", "edit"]:
|
|
127
|
+
pattern = SequencePattern.TDD_EDIT_TEST_EDIT
|
|
128
|
+
score = 1.0
|
|
129
|
+
break
|
|
130
|
+
else:
|
|
131
|
+
for idx in range(len(normalized) - 1):
|
|
132
|
+
window = normalized[idx:idx + 2]
|
|
133
|
+
if window == ["edit", "bash"]:
|
|
134
|
+
pattern = SequencePattern.VERIFY_AFTER_EDIT
|
|
135
|
+
score = 0.75
|
|
136
|
+
break
|
|
137
|
+
else:
|
|
138
|
+
if len(normalized) >= 3 and normalized[:3] == ["bash", "bash", "read"]:
|
|
139
|
+
pattern = SequencePattern.SHELL_INSPECT_LOOP
|
|
140
|
+
score = 0.55
|
|
141
|
+
elif not normalized:
|
|
142
|
+
pattern = SequencePattern.EMPTY
|
|
143
|
+
score = 0.0
|
|
144
|
+
elif verification_present:
|
|
145
|
+
score = 0.35
|
|
146
|
+
elif read_count > 0:
|
|
147
|
+
score = 0.2
|
|
148
|
+
|
|
149
|
+
summary = (
|
|
150
|
+
f"tools={len(names)}, pattern={pattern.value}, score={score:.2f}, "
|
|
151
|
+
f"edit={edit_count}, shell={shell_count}, read={read_count}"
|
|
152
|
+
)
|
|
153
|
+
return ToolSequenceQuality(
|
|
154
|
+
tool_names=names,
|
|
155
|
+
pattern=pattern,
|
|
156
|
+
sequence_score=score,
|
|
157
|
+
verification_present=verification_present,
|
|
158
|
+
edit_count=edit_count,
|
|
159
|
+
shell_count=shell_count,
|
|
160
|
+
read_count=read_count,
|
|
161
|
+
summary=summary,
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def score_cot(cot: str, completion: str) -> CoTQuality:
|
|
166
|
+
"""Score one CoT block.
|
|
167
|
+
|
|
168
|
+
Args:
|
|
169
|
+
cot: The reasoning / thinking block preceding a tool call. May be empty.
|
|
170
|
+
completion: The actual tool call or response. May be empty.
|
|
171
|
+
|
|
172
|
+
Returns:
|
|
173
|
+
A CoTQuality record with band, markers, well_sequenced, and a [0, 1]
|
|
174
|
+
structure_score suitable for eval grading.
|
|
175
|
+
"""
|
|
176
|
+
cot_len = len(cot or "")
|
|
177
|
+
comp_len = len(completion or "")
|
|
178
|
+
ratio = cot_len / max(1, comp_len)
|
|
179
|
+
|
|
180
|
+
if ratio < 0.5:
|
|
181
|
+
band = Band.UNDER
|
|
182
|
+
elif ratio > 5.0:
|
|
183
|
+
band = Band.OVER
|
|
184
|
+
else:
|
|
185
|
+
band = Band.IN_BAND
|
|
186
|
+
|
|
187
|
+
markers = tuple(m.group(0) for m in MARKER_RE.finditer(cot or ""))
|
|
188
|
+
well_seq = bool(WELL_SEQUENCED_RE.search(cot or ""))
|
|
189
|
+
|
|
190
|
+
# marker_score capped at 0.6 (3 markers); seq_score 0.4; band_bonus 0.2
|
|
191
|
+
marker_score = min(0.6, 0.2 * len(markers))
|
|
192
|
+
seq_score = 0.4 if well_seq else 0.0
|
|
193
|
+
band_bonus = 0.2 if band == Band.IN_BAND else 0.0
|
|
194
|
+
score = min(1.0, marker_score + seq_score + band_bonus)
|
|
195
|
+
|
|
196
|
+
summary = (
|
|
197
|
+
f"cot={cot_len}ch, completion={comp_len}ch, ratio={ratio:.2f} "
|
|
198
|
+
f"({band.value}), markers={len(markers)}, sequenced={well_seq}, "
|
|
199
|
+
f"score={score:.2f}"
|
|
200
|
+
)
|
|
201
|
+
return CoTQuality(
|
|
202
|
+
cot_chars=cot_len,
|
|
203
|
+
completion_chars=comp_len,
|
|
204
|
+
cot_ratio=ratio,
|
|
205
|
+
band=band,
|
|
206
|
+
markers=markers,
|
|
207
|
+
well_sequenced=well_seq,
|
|
208
|
+
structure_score=score,
|
|
209
|
+
summary=summary,
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
__all__ = [
|
|
214
|
+
"Band",
|
|
215
|
+
"CoTQuality",
|
|
216
|
+
"SequencePattern",
|
|
217
|
+
"ToolSequenceQuality",
|
|
218
|
+
"score_cot",
|
|
219
|
+
"score_tool_sequence",
|
|
220
|
+
]
|
|
@@ -0,0 +1,401 @@
|
|
|
1
|
+
"""Bounded, offline effectiveness benchmark for harness retrieval.
|
|
2
|
+
|
|
3
|
+
The benchmark reads a snapshot of the persisted harness index and owns every
|
|
4
|
+
BM25 object it creates. It deliberately does not call ``harness.search_index``
|
|
5
|
+
or clear/populate any process-global retrieval cache.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import statistics
|
|
13
|
+
import time
|
|
14
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from .. import harness
|
|
18
|
+
from ..retrieval_algorithms import (
|
|
19
|
+
FULL_SORT_THRESHOLD,
|
|
20
|
+
BM25Index,
|
|
21
|
+
lexical_tokens,
|
|
22
|
+
stable_top_k,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
BENCHMARK_VERSION = "harness-retrieval-v1"
|
|
26
|
+
CANARY_QUERIES: tuple[str, ...] = (
|
|
27
|
+
"rate your harness",
|
|
28
|
+
"harness context",
|
|
29
|
+
"memory recall",
|
|
30
|
+
"verification before completion",
|
|
31
|
+
"index-compute-lab",
|
|
32
|
+
)
|
|
33
|
+
CANONICAL_ALGO_ID = "algo-cli:algorithm:ALGO.md"
|
|
34
|
+
STABILITY_PASSES = 3
|
|
35
|
+
CANARY_LIMIT = 5
|
|
36
|
+
COLD_SAMPLE_TARGET = 5
|
|
37
|
+
REUSABLE_WARMUPS = 3
|
|
38
|
+
REUSABLE_SAMPLE_TARGET = 9
|
|
39
|
+
MIN_REUSABLE_SPEEDUP = 1.5
|
|
40
|
+
MAX_WARM_MAD_RATIO = 0.25
|
|
41
|
+
MAX_BENCHMARK_RECORDS = 2_048
|
|
42
|
+
MAX_BENCHMARK_TEXT_CHARS = 40_000
|
|
43
|
+
|
|
44
|
+
SearchFn = Callable[[str, int], Sequence[Any]]
|
|
45
|
+
ClockFn = Callable[[], int]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _digest(value: Any) -> str:
|
|
49
|
+
encoded = json.dumps(
|
|
50
|
+
value,
|
|
51
|
+
ensure_ascii=True,
|
|
52
|
+
sort_keys=True,
|
|
53
|
+
separators=(",", ":"),
|
|
54
|
+
default=str,
|
|
55
|
+
).encode("utf-8")
|
|
56
|
+
return hashlib.sha256(encoded).hexdigest()
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _load_persisted_index() -> tuple[dict[str, Any], str | None]:
|
|
60
|
+
"""Read the live index file without invoking the global index cache."""
|
|
61
|
+
|
|
62
|
+
try:
|
|
63
|
+
payload = json.loads(harness.INDEX_PATH.read_text(encoding="utf-8"))
|
|
64
|
+
except FileNotFoundError:
|
|
65
|
+
return {"records": []}, f"harness index not found: {harness.INDEX_PATH}"
|
|
66
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
67
|
+
return {"records": []}, f"could not read harness index: {exc}"
|
|
68
|
+
if not isinstance(payload, dict):
|
|
69
|
+
return {"records": []}, "harness index root is not an object"
|
|
70
|
+
return payload, None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _eligible_records(index: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
74
|
+
raw_records = index.get("records")
|
|
75
|
+
if not isinstance(raw_records, list):
|
|
76
|
+
return []
|
|
77
|
+
return [
|
|
78
|
+
record
|
|
79
|
+
for record in raw_records
|
|
80
|
+
if isinstance(record, dict) and not harness.is_excluded_from_retrieval(record)
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _bounded_records(records: Sequence[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
85
|
+
if len(records) <= MAX_BENCHMARK_RECORDS:
|
|
86
|
+
return list(records)
|
|
87
|
+
# Keep canonical/project-local evidence in the bounded corpus, then retain
|
|
88
|
+
# source order so repeated runs over an unchanged index remain identical.
|
|
89
|
+
prioritized = sorted(
|
|
90
|
+
enumerate(records),
|
|
91
|
+
key=lambda pair: (
|
|
92
|
+
str(pair[1].get("id") or "") != CANONICAL_ALGO_ID,
|
|
93
|
+
str(pair[1].get("harness") or "") != "algo-cli",
|
|
94
|
+
pair[0],
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
return [record for _position, record in prioritized[:MAX_BENCHMARK_RECORDS]]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _search_text(record: Mapping[str, Any]) -> str:
|
|
101
|
+
text = str(record.get("search_text") or "")
|
|
102
|
+
if text:
|
|
103
|
+
return text[:MAX_BENCHMARK_TEXT_CHARS]
|
|
104
|
+
return " ".join(
|
|
105
|
+
str(record.get(field) or "")
|
|
106
|
+
for field in (
|
|
107
|
+
"id",
|
|
108
|
+
"harness",
|
|
109
|
+
"kind",
|
|
110
|
+
"title",
|
|
111
|
+
"description",
|
|
112
|
+
"tags",
|
|
113
|
+
"relative_path",
|
|
114
|
+
"summary",
|
|
115
|
+
)
|
|
116
|
+
).lower()[:MAX_BENCHMARK_TEXT_CHARS]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _local_search(
|
|
120
|
+
records: Sequence[dict[str, Any]],
|
|
121
|
+
bm25: BM25Index,
|
|
122
|
+
query: str,
|
|
123
|
+
limit: int,
|
|
124
|
+
) -> list[dict[str, Any]]:
|
|
125
|
+
terms = lexical_tokens(query)
|
|
126
|
+
if not terms:
|
|
127
|
+
return []
|
|
128
|
+
scored: list[tuple[float, dict[str, Any]]] = []
|
|
129
|
+
for lexical_score, record in zip(bm25.scores(terms), records):
|
|
130
|
+
score = lexical_score + float(harness.score_record(record, terms))
|
|
131
|
+
if score > 0.0:
|
|
132
|
+
scored.append((score, record))
|
|
133
|
+
return [
|
|
134
|
+
record
|
|
135
|
+
for _score, record in stable_top_k(
|
|
136
|
+
scored,
|
|
137
|
+
limit,
|
|
138
|
+
score=lambda pair: pair[0],
|
|
139
|
+
)
|
|
140
|
+
]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _result_id(result: Any) -> str:
|
|
144
|
+
if isinstance(result, Mapping):
|
|
145
|
+
return str(result.get("id") or "")
|
|
146
|
+
return str(result or "")
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _stable_top_k_parity() -> tuple[bool, str]:
|
|
150
|
+
"""Exercise the heap branch above its adaptive crossover threshold."""
|
|
151
|
+
|
|
152
|
+
values = [
|
|
153
|
+
(index, (index * 2_654_435_761) % 97)
|
|
154
|
+
for index in range(FULL_SORT_THRESHOLD + 257)
|
|
155
|
+
]
|
|
156
|
+
expected = sorted(values, key=lambda item: item[1], reverse=True)[:17]
|
|
157
|
+
actual = stable_top_k(values, 17, score=lambda item: item[1])
|
|
158
|
+
return actual == expected, _digest(actual)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _measure_ns(operation: Callable[[], Any], clock_ns: ClockFn) -> int:
|
|
162
|
+
started = int(clock_ns())
|
|
163
|
+
operation()
|
|
164
|
+
return max(0, int(clock_ns()) - started)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _median_absolute_deviation(values: Sequence[int], median: float) -> float:
|
|
168
|
+
if not values:
|
|
169
|
+
return 0.0
|
|
170
|
+
return float(statistics.median(abs(float(value) - median) for value in values))
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _milliseconds(value_ns: float) -> float:
|
|
174
|
+
return round(float(value_ns) / 1_000_000.0, 6)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def run_harness_retrieval_benchmark(
|
|
178
|
+
index: Mapping[str, Any] | None = None,
|
|
179
|
+
search_fn: SearchFn | None = None,
|
|
180
|
+
*,
|
|
181
|
+
clock_ns: ClockFn | None = None,
|
|
182
|
+
) -> dict[str, Any]:
|
|
183
|
+
"""Run the bounded retrieval benchmark and return JSON-serializable evidence.
|
|
184
|
+
|
|
185
|
+
Args:
|
|
186
|
+
index: Optional index payload. When omitted, read the persisted live index
|
|
187
|
+
directly without touching the harness index cache.
|
|
188
|
+
search_fn: Optional ``(query, limit) -> results`` function for canary
|
|
189
|
+
checks. Timing always uses local BM25 instances.
|
|
190
|
+
clock_ns: Optional monotonic nanosecond clock for deterministic tests.
|
|
191
|
+
"""
|
|
192
|
+
|
|
193
|
+
load_error: str | None = None
|
|
194
|
+
if index is None:
|
|
195
|
+
index_payload, load_error = _load_persisted_index()
|
|
196
|
+
else:
|
|
197
|
+
index_payload = dict(index)
|
|
198
|
+
raw_records = index_payload.get("records")
|
|
199
|
+
index_record_count = len(raw_records) if isinstance(raw_records, list) else 0
|
|
200
|
+
eligible_records = _eligible_records(index_payload)
|
|
201
|
+
records = _bounded_records(eligible_records)
|
|
202
|
+
documents = [_search_text(record) for record in records]
|
|
203
|
+
reusable_index = BM25Index(documents)
|
|
204
|
+
active_search: SearchFn
|
|
205
|
+
if search_fn is None:
|
|
206
|
+
def active_search(query: str, limit: int) -> list[dict[str, Any]]:
|
|
207
|
+
return _local_search(records, reusable_index, query, limit)
|
|
208
|
+
else:
|
|
209
|
+
active_search = search_fn
|
|
210
|
+
|
|
211
|
+
ranking_passes: list[list[list[str]]] = []
|
|
212
|
+
search_errors: list[str] = []
|
|
213
|
+
for _pass_index in range(STABILITY_PASSES):
|
|
214
|
+
pass_rankings: list[list[str]] = []
|
|
215
|
+
for query in CANARY_QUERIES:
|
|
216
|
+
try:
|
|
217
|
+
results = active_search(query, CANARY_LIMIT)
|
|
218
|
+
ids = [_result_id(result) for result in results]
|
|
219
|
+
pass_rankings.append([record_id for record_id in ids if record_id])
|
|
220
|
+
except Exception as exc:
|
|
221
|
+
search_errors.append(f"{query}: {type(exc).__name__}: {exc}")
|
|
222
|
+
pass_rankings.append([])
|
|
223
|
+
ranking_passes.append(pass_rankings)
|
|
224
|
+
|
|
225
|
+
observation_count = len(CANARY_QUERIES) * STABILITY_PASSES
|
|
226
|
+
nonempty_observations = sum(
|
|
227
|
+
1
|
|
228
|
+
for pass_rankings in ranking_passes
|
|
229
|
+
for ranking in pass_rankings
|
|
230
|
+
if ranking
|
|
231
|
+
)
|
|
232
|
+
stable_rankings = bool(ranking_passes) and all(
|
|
233
|
+
pass_rankings == ranking_passes[0]
|
|
234
|
+
for pass_rankings in ranking_passes[1:]
|
|
235
|
+
)
|
|
236
|
+
canonical_algo_top1 = bool(ranking_passes) and all(
|
|
237
|
+
pass_rankings
|
|
238
|
+
and pass_rankings[0]
|
|
239
|
+
and pass_rankings[0][0] == CANONICAL_ALGO_ID
|
|
240
|
+
for pass_rankings in ranking_passes
|
|
241
|
+
)
|
|
242
|
+
top_k_parity, top_k_digest = _stable_top_k_parity()
|
|
243
|
+
|
|
244
|
+
clock = clock_ns or time.perf_counter_ns
|
|
245
|
+
query_terms = [lexical_tokens(query) for query in CANARY_QUERIES]
|
|
246
|
+
|
|
247
|
+
def score_all_queries(bm25: BM25Index) -> list[list[float]]:
|
|
248
|
+
return [bm25.scores(terms) for terms in query_terms]
|
|
249
|
+
|
|
250
|
+
cold_samples_ns: list[int] = []
|
|
251
|
+
for _sample in range(COLD_SAMPLE_TARGET):
|
|
252
|
+
cold_samples_ns.append(
|
|
253
|
+
_measure_ns(
|
|
254
|
+
lambda: score_all_queries(BM25Index(documents)),
|
|
255
|
+
clock,
|
|
256
|
+
)
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
for _warmup in range(REUSABLE_WARMUPS):
|
|
260
|
+
score_all_queries(reusable_index)
|
|
261
|
+
warm_samples_ns = [
|
|
262
|
+
_measure_ns(lambda: score_all_queries(reusable_index), clock)
|
|
263
|
+
for _sample in range(REUSABLE_SAMPLE_TARGET)
|
|
264
|
+
]
|
|
265
|
+
|
|
266
|
+
cold_median_ns = float(statistics.median(cold_samples_ns)) if cold_samples_ns else 0.0
|
|
267
|
+
warm_median_ns = float(statistics.median(warm_samples_ns)) if warm_samples_ns else 0.0
|
|
268
|
+
cold_mad_ns = _median_absolute_deviation(cold_samples_ns, cold_median_ns)
|
|
269
|
+
warm_mad_ns = _median_absolute_deviation(warm_samples_ns, warm_median_ns)
|
|
270
|
+
warm_mad_ratio = warm_mad_ns / warm_median_ns if warm_median_ns > 0.0 else None
|
|
271
|
+
speedup = cold_median_ns / warm_median_ns if warm_median_ns > 0.0 else None
|
|
272
|
+
sufficient_samples = (
|
|
273
|
+
len(cold_samples_ns) >= COLD_SAMPLE_TARGET
|
|
274
|
+
and len(warm_samples_ns) >= REUSABLE_SAMPLE_TARGET
|
|
275
|
+
and cold_median_ns > 0.0
|
|
276
|
+
and warm_median_ns > 0.0
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
correctness_failures: list[str] = []
|
|
280
|
+
if load_error:
|
|
281
|
+
correctness_failures.append(load_error)
|
|
282
|
+
if not records:
|
|
283
|
+
correctness_failures.append("no eligible harness records")
|
|
284
|
+
if search_errors:
|
|
285
|
+
correctness_failures.append("canary search raised an exception")
|
|
286
|
+
if nonempty_observations != observation_count:
|
|
287
|
+
correctness_failures.append(
|
|
288
|
+
f"nonempty canary observations {nonempty_observations}/{observation_count}"
|
|
289
|
+
)
|
|
290
|
+
if not stable_rankings:
|
|
291
|
+
correctness_failures.append("canary rankings changed across stability passes")
|
|
292
|
+
if not canonical_algo_top1:
|
|
293
|
+
correctness_failures.append("canonical ALGO record was not top-1")
|
|
294
|
+
if not top_k_parity:
|
|
295
|
+
correctness_failures.append("stable_top_k diverged from a full stable sort")
|
|
296
|
+
correctness_passed = not correctness_failures
|
|
297
|
+
|
|
298
|
+
performance_warnings: list[str] = []
|
|
299
|
+
if not sufficient_samples:
|
|
300
|
+
performance_warnings.append("timing samples were insufficient or below clock resolution")
|
|
301
|
+
if speedup is None or speedup < MIN_REUSABLE_SPEEDUP:
|
|
302
|
+
performance_warnings.append(
|
|
303
|
+
f"reusable BM25 speedup was below {MIN_REUSABLE_SPEEDUP:.1f}x"
|
|
304
|
+
)
|
|
305
|
+
if warm_mad_ratio is None or warm_mad_ratio > MAX_WARM_MAD_RATIO:
|
|
306
|
+
performance_warnings.append(
|
|
307
|
+
f"warm MAD ratio exceeded {MAX_WARM_MAD_RATIO:.2f}"
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
if not correctness_passed:
|
|
311
|
+
status = "fail"
|
|
312
|
+
reason = "retrieval correctness failed: " + "; ".join(correctness_failures)
|
|
313
|
+
elif performance_warnings:
|
|
314
|
+
status = "warn"
|
|
315
|
+
reason = "retrieval correctness passed; " + "; ".join(performance_warnings)
|
|
316
|
+
else:
|
|
317
|
+
assert speedup is not None and warm_mad_ratio is not None
|
|
318
|
+
status = "pass"
|
|
319
|
+
reason = (
|
|
320
|
+
"retrieval correctness passed; reusable BM25 speedup "
|
|
321
|
+
f"{speedup:.2f}x with warm MAD ratio {warm_mad_ratio:.3f}"
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
index_fingerprint = [
|
|
325
|
+
{
|
|
326
|
+
"id": str(record.get("id") or ""),
|
|
327
|
+
"harness": str(record.get("harness") or ""),
|
|
328
|
+
"kind": str(record.get("kind") or ""),
|
|
329
|
+
"relative_path": str(record.get("relative_path") or ""),
|
|
330
|
+
"search_text": _search_text(record),
|
|
331
|
+
}
|
|
332
|
+
for record in records
|
|
333
|
+
]
|
|
334
|
+
cold_samples_ms = [_milliseconds(value) for value in cold_samples_ns]
|
|
335
|
+
warm_samples_ms = [_milliseconds(value) for value in warm_samples_ns]
|
|
336
|
+
return {
|
|
337
|
+
"benchmark_version": BENCHMARK_VERSION,
|
|
338
|
+
"status": status,
|
|
339
|
+
"reason": reason,
|
|
340
|
+
"correctness": {
|
|
341
|
+
"passed": correctness_passed,
|
|
342
|
+
"failures": correctness_failures,
|
|
343
|
+
"search_errors": search_errors,
|
|
344
|
+
"canary_query_count": len(CANARY_QUERIES),
|
|
345
|
+
"stability_passes": STABILITY_PASSES,
|
|
346
|
+
"observation_count": observation_count,
|
|
347
|
+
"nonempty_observations": nonempty_observations,
|
|
348
|
+
"stable_rankings": stable_rankings,
|
|
349
|
+
"canonical_algo_top1": canonical_algo_top1,
|
|
350
|
+
"stable_top_k_parity": top_k_parity,
|
|
351
|
+
"first_pass_rankings": {
|
|
352
|
+
query: ranking_passes[0][index]
|
|
353
|
+
for index, query in enumerate(CANARY_QUERIES)
|
|
354
|
+
},
|
|
355
|
+
},
|
|
356
|
+
"performance": {
|
|
357
|
+
"operation": "bm25_build_plus_all_queries_vs_reusable_all_queries",
|
|
358
|
+
"cold_sample_count": len(cold_samples_ns),
|
|
359
|
+
"warmup_count": REUSABLE_WARMUPS,
|
|
360
|
+
"warm_sample_count": len(warm_samples_ns),
|
|
361
|
+
"cold_samples_ms": cold_samples_ms,
|
|
362
|
+
"warm_samples_ms": warm_samples_ms,
|
|
363
|
+
"cold_median_ms": _milliseconds(cold_median_ns),
|
|
364
|
+
"warm_median_ms": _milliseconds(warm_median_ns),
|
|
365
|
+
"cold_mad_ms": _milliseconds(cold_mad_ns),
|
|
366
|
+
"warm_mad_ms": _milliseconds(warm_mad_ns),
|
|
367
|
+
"warm_mad_ratio": (
|
|
368
|
+
round(warm_mad_ratio, 6) if warm_mad_ratio is not None else None
|
|
369
|
+
),
|
|
370
|
+
"speedup": round(speedup, 6) if speedup is not None else None,
|
|
371
|
+
"sufficient_samples": sufficient_samples,
|
|
372
|
+
"minimum_speedup": MIN_REUSABLE_SPEEDUP,
|
|
373
|
+
"maximum_warm_mad_ratio": MAX_WARM_MAD_RATIO,
|
|
374
|
+
},
|
|
375
|
+
"evidence": {
|
|
376
|
+
"index_record_count": index_record_count,
|
|
377
|
+
"eligible_record_count": len(eligible_records),
|
|
378
|
+
"benchmark_record_count": len(records),
|
|
379
|
+
"maximum_benchmark_records": MAX_BENCHMARK_RECORDS,
|
|
380
|
+
"maximum_text_chars": MAX_BENCHMARK_TEXT_CHARS,
|
|
381
|
+
"index_digest": _digest(index_fingerprint),
|
|
382
|
+
"canary_digest": _digest(CANARY_QUERIES),
|
|
383
|
+
"ranking_digest": _digest(ranking_passes),
|
|
384
|
+
"stable_top_k_digest": top_k_digest,
|
|
385
|
+
"timing_digest": _digest(
|
|
386
|
+
{
|
|
387
|
+
"cold_samples_ns": cold_samples_ns,
|
|
388
|
+
"warm_samples_ns": warm_samples_ns,
|
|
389
|
+
}
|
|
390
|
+
),
|
|
391
|
+
},
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
__all__ = [
|
|
396
|
+
"BENCHMARK_VERSION",
|
|
397
|
+
"CANARY_QUERIES",
|
|
398
|
+
"CANONICAL_ALGO_ID",
|
|
399
|
+
"MAX_BENCHMARK_RECORDS",
|
|
400
|
+
"run_harness_retrieval_benchmark",
|
|
401
|
+
]
|