woods 2.0.0.beta1 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +400 -1
  3. data/CONTRIBUTING.md +224 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +233 -13
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +158 -2
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +15 -7
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +71 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/atomic_file.rb +133 -3
  56. data/lib/woods/builder.rb +21 -5
  57. data/lib/woods/cache/cache_middleware.rb +28 -7
  58. data/lib/woods/cache/cache_store.rb +4 -5
  59. data/lib/woods/change_set.rb +5 -4
  60. data/lib/woods/console/credential_index.rb +20 -2
  61. data/lib/woods/console/credential_scanner.rb +14 -14
  62. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  63. data/lib/woods/console/embedded_executor.rb +1 -1
  64. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  65. data/lib/woods/console/rack_middleware.rb +22 -13
  66. data/lib/woods/console/server.rb +18 -16
  67. data/lib/woods/dependency_graph.rb +65 -13
  68. data/lib/woods/embedding/corpus.rb +94 -0
  69. data/lib/woods/embedding/indexer.rb +90 -46
  70. data/lib/woods/embedding/openai.rb +17 -6
  71. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  72. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  73. data/lib/woods/export/typed_reader.rb +56 -0
  74. data/lib/woods/extractor.rb +557 -228
  75. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  76. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  77. data/lib/woods/extractors/caching_extractor.rb +3 -1
  78. data/lib/woods/extractors/concern_extractor.rb +64 -6
  79. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  80. data/lib/woods/extractors/controller_extractor.rb +13 -4
  81. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  82. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  83. data/lib/woods/extractors/engine_extractor.rb +3 -1
  84. data/lib/woods/extractors/event_extractor.rb +4 -2
  85. data/lib/woods/extractors/factory_extractor.rb +3 -1
  86. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  87. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  88. data/lib/woods/extractors/job_extractor.rb +6 -19
  89. data/lib/woods/extractors/lib_extractor.rb +3 -1
  90. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  91. data/lib/woods/extractors/manager_extractor.rb +3 -1
  92. data/lib/woods/extractors/method_parameters.rb +53 -0
  93. data/lib/woods/extractors/middleware_argument.rb +65 -0
  94. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  95. data/lib/woods/extractors/migration_extractor.rb +3 -1
  96. data/lib/woods/extractors/model_extractor.rb +39 -33
  97. data/lib/woods/extractors/package_extractor.rb +24 -4
  98. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  99. data/lib/woods/extractors/policy_extractor.rb +3 -1
  100. data/lib/woods/extractors/poro_extractor.rb +3 -1
  101. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  102. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  103. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  104. data/lib/woods/extractors/route_extractor.rb +3 -1
  105. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  106. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  107. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  108. data/lib/woods/extractors/service_extractor.rb +3 -1
  109. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  110. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  111. data/lib/woods/extractors/source_nesting.rb +1 -1
  112. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  113. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  114. data/lib/woods/extractors/validator_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  116. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  117. data/lib/woods/flow_assembler.rb +87 -8
  118. data/lib/woods/flow_precomputer.rb +44 -7
  119. data/lib/woods/gem_mapper.rb +2 -0
  120. data/lib/woods/git_history.rb +116 -0
  121. data/lib/woods/graph_analyzer.rb +195 -63
  122. data/lib/woods/hooks/context_cli.rb +54 -0
  123. data/lib/woods/hooks/context_event.rb +88 -0
  124. data/lib/woods/hooks/context_hint.rb +73 -0
  125. data/lib/woods/hooks/context_impact.rb +77 -0
  126. data/lib/woods/hooks/context_output.rb +47 -0
  127. data/lib/woods/hooks/context_state.rb +102 -0
  128. data/lib/woods/hooks/refresh.rb +79 -0
  129. data/lib/woods/hooks/rule_projection.rb +78 -0
  130. data/lib/woods/input_rules.rb +19 -0
  131. data/lib/woods/mcp/bearer_auth.rb +20 -12
  132. data/lib/woods/mcp/bootstrapper.rb +62 -0
  133. data/lib/woods/mcp/index_reader.rb +323 -160
  134. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  135. data/lib/woods/mcp/origin_guard.rb +17 -9
  136. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  137. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  138. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  139. data/lib/woods/mcp/search_results.rb +74 -0
  140. data/lib/woods/mcp/server.rb +158 -37
  141. data/lib/woods/mcp/tool_contract.rb +2 -0
  142. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  143. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  144. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  145. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  146. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  147. data/lib/woods/notion/exporter.rb +56 -17
  148. data/lib/woods/obsidian/destination_plan.rb +98 -0
  149. data/lib/woods/obsidian/name_mapper.rb +19 -3
  150. data/lib/woods/obsidian/note_builder.rb +19 -10
  151. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  152. data/lib/woods/operator/pipeline_guard.rb +18 -13
  153. data/lib/woods/path_dispatcher.rb +7 -1
  154. data/lib/woods/payload_store.rb +29 -15
  155. data/lib/woods/railtie.rb +3 -3
  156. data/lib/woods/railtie_support.rb +12 -12
  157. data/lib/woods/rake_helpers.rb +392 -0
  158. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  159. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  160. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  161. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  162. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  163. data/lib/woods/resilience/index_validator.rb +112 -23
  164. data/lib/woods/retrieval/context_assembler.rb +50 -15
  165. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  166. data/lib/woods/retrieval/lexical_index.rb +119 -0
  167. data/lib/woods/retrieval/ranker.rb +4 -2
  168. data/lib/woods/retrieval/scope.rb +108 -0
  169. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  170. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  171. data/lib/woods/retrieval/search_executor.rb +86 -27
  172. data/lib/woods/retrieval/source_evidence.rb +200 -0
  173. data/lib/woods/retriever.rb +98 -22
  174. data/lib/woods/ruby_analyzer/trace_enricher.rb +80 -38
  175. data/lib/woods/session_tracer/middleware.rb +10 -12
  176. data/lib/woods/session_tracer/redis_store.rb +22 -6
  177. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  178. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  179. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  180. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  181. data/lib/woods/source_inputs/handoff.rb +102 -0
  182. data/lib/woods/source_inputs/launcher.rb +157 -0
  183. data/lib/woods/source_inputs/manifest.rb +124 -0
  184. data/lib/woods/source_inputs/private_key.rb +55 -0
  185. data/lib/woods/source_inputs/scanner.rb +171 -0
  186. data/lib/woods/source_inputs/scopes.rb +71 -0
  187. data/lib/woods/source_inputs/session.rb +214 -0
  188. data/lib/woods/source_inputs/status.rb +84 -0
  189. data/lib/woods/source_inputs/verifier.rb +107 -0
  190. data/lib/woods/storage/metadata_store.rb +25 -25
  191. data/lib/woods/storage/pgvector.rb +29 -8
  192. data/lib/woods/storage/qdrant.rb +17 -7
  193. data/lib/woods/storage/vector_store.rb +18 -6
  194. data/lib/woods/tasks.rb +3 -2
  195. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  196. data/lib/woods/unblocked/exporter.rb +59 -70
  197. data/lib/woods/version.rb +1 -1
  198. data/lib/woods/watch/boot_snapshot.rb +52 -0
  199. data/lib/woods/watch/daemon.rb +136 -28
  200. data/lib/woods/watch/listen_watcher.rb +4 -0
  201. data/lib/woods/watch/polling_watcher.rb +5 -1
  202. data/lib/woods/watch/status.rb +20 -15
  203. data/lib/woods/watch/tree_scan.rb +21 -13
  204. data/lib/woods/watch/watcher.rb +4 -1
  205. data/lib/woods.rb +135 -11
  206. data/plugin/.claude-plugin/plugin.json +1 -1
  207. data/plugin/hooks/adapters/normalize.jq +15 -0
  208. data/plugin/hooks/adapters/normalize.rb +63 -0
  209. data/plugin/hooks/hooks.json +20 -0
  210. data/plugin/hooks/woods-context.sh +50 -0
  211. data/plugin/hooks/woods-input-rules.sh +159 -0
  212. data/plugin/hooks/woods-opencode.mjs +65 -0
  213. data/plugin/hooks/woods-post-edit.sh +2 -225
  214. data/plugin/hooks/woods-refresh.sh +260 -0
  215. data/plugin/hooks/woods-session-start.sh +47 -55
  216. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  217. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  218. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  219. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  220. data/plugin/skills/woods-setup/SKILL.md +107 -6
  221. metadata +84 -5
data/docs/DOCKER_SETUP.md CHANGED
@@ -87,6 +87,12 @@ docker compose exec app bundle exec rake woods:extract_framework
87
87
 
88
88
  Run the watcher as its own development service or process-manager entry, not as a one-off terminal command. Docker Desktop bind mounts may not deliver reliable native filesystem events; set `WOODS_WATCH_POLL=1` for polling when needed. The watcher updates structural generations automatically, while semantic vectors still require `woods:embed_incremental`.
89
89
 
90
+ When host-side tasks or one-off containers read the daemon's shared index,
91
+ `WOODS_WATCH_TRUST_FOREIGN_HOST=1` lets those readers trust its recent heartbeat.
92
+ Set it in each reader process; Docker does not forward host variables by default.
93
+ See [cross-host liveness](WATCH_DAEMON.md#cross-host-liveness) for the 15-minute
94
+ crash-detection bound and single-supervisor requirement.
95
+
90
96
  ### Index persistence
91
97
 
92
98
  Persist `tmp/woods/` if the index should survive container replacement. A bind mount also makes it available to optional host-side tools:
@@ -185,6 +191,16 @@ Use this only when the application bundle, a supported Ruby, and the Woods execu
185
191
  | **Survives container replacement** | Only with a bind/named volume | Yes, on host disk |
186
192
  | **Needs Ruby/Woods bundle on host** | No | Yes |
187
193
 
194
+ ### Index filesystem performance
195
+
196
+ Bind mounts backed by virtiofs or FUSE can make each hardlink, rename and
197
+ metadata lookup costly. Profile `payload seed`, writes and `payload prune`
198
+ before tuning extraction. A container volume can reduce these costs, but it
199
+ changes host visibility: use a distinct index location per worktree, run MCP
200
+ where that path is visible, and update export/archive paths together. A single
201
+ shared volume for every worktree would mix their indexes. See
202
+ [incremental profiling](INCREMENTAL_EXTRACTION.md#profiling-fixed-costs).
203
+
188
204
  ## Console Server Setup
189
205
 
190
206
  The Console Server queries live Rails state. There are two launch paths for the same embedded server.
@@ -194,13 +210,16 @@ Before either path can start, deliberately enable live-data access in the Rails
194
210
  ```ruby
195
211
  Woods.configure do |config|
196
212
  config.console_mcp_enabled = true
197
- config.console_mcp_token = ENV["WOODS_CONSOLE_MCP_TOKEN"]
213
+ config.console_mcp_http_enabled = false # stdio-only
198
214
  end
199
215
  ```
200
216
 
201
217
  The process exits with status 1 while this master switch is false. Review [Console MCP setup and security](CONSOLE_MCP_SETUP.md) before enabling it.
202
218
 
203
- Stdio does not send the bearer token, but production Rails boot still requires `WOODS_CONSOLE_MCP_TOKEN` to contain at least 32 characters whenever Console is enabled. Provide it to the container through the application's normal secret mechanism. Outside production, omitting it warns and leaves the Console HTTP endpoint guarded with 401.
219
+ This explicitly disables HTTP Console while retaining stdio access; no HTTP
220
+ token is needed at boot. Existing configurations default to HTTP enabled.
221
+ For HTTP deployment, enable the HTTP flag and configure its token, origins
222
+ and TLS using the [Console setup guide](CONSOLE_MCP_SETUP.md#option-c-http-rack-middleware).
204
223
 
205
224
  ### Comparison
206
225
 
data/docs/EVALUATION.md CHANGED
@@ -14,6 +14,177 @@ Three harnesses, three questions.
14
14
 
15
15
  `woods:evaluate:baseline[grep|random|file_level]` scores a naive strategy on the same query set for comparison.
16
16
 
17
+ ## Paired lexical comparison (#404 / #227)
18
+
19
+ `bench/evaluation/lexical_comparison.rb` runs **all 28 existing questions at their
20
+ existing budgets** through semantic retrieval, lexical-only retrieval, a fixed
21
+ query-seeded graph experiment, and the existing identifier-substring grep baseline.
22
+ Unlike the strategy groups below, these conditions answer the same questions.
23
+ The semantic condition replays the same captured real MiniLM vectors; labels,
24
+ source snapshot and the original quality floors are unchanged.
25
+
26
+ ```bash
27
+ bundle exec ruby -Ilib bench/evaluation/lexical_comparison.rb /tmp/woods-paired.json
28
+ # Use the pinned tokenizer environment described below:
29
+ python bench/evaluation/capture_tokens.py /tmp/woods-paired.json lexical_comparison_capture.json
30
+ ```
31
+
32
+ The second command writes `bench/evaluation/lexical_comparison_capture.json`.
33
+ It retains every query's outcomes, context hash and exact token count, including
34
+ misses. Raw returned contexts remain in `/tmp/woods-paired.json`. The capture binds
35
+ its corpus/vector hashes and tokenizer provenance. It is reviewed experimental
36
+ evidence, not a replacement for the required semantic gate.
37
+
38
+ Initial Ruby 4.0.6 comparison (five warm pipeline repetitions per question):
39
+
40
+ | Condition | Precision@5 | Recall | MRR | Mean actual context tokens | Median query latency (ms) |
41
+ |---|---:|---:|---:|---:|---:|
42
+ | Existing semantic pipeline | 0.536 | 0.580 | 0.857 | 1,000.7 | 3.019 |
43
+ | Explicit lexical | 0.492 | 0.666 | 0.780 | 1,152.6 | 0.407 |
44
+ | Lexical + query-seeded graph experiment | 0.396 | 0.476 | 0.750 | 1,159.8 | 3.292 |
45
+ | Existing identifier grep baseline | 0.342 | 0.288 | 0.336 | 592.9 | 0.628 |
46
+
47
+ Lexical retrieval improves recall on this small set while losing precision and
48
+ first-hit rank against the semantic pipeline. This supports an explicit offline
49
+ option, not semantic equivalence or a new default. The seeded graph condition
50
+ loses on these measures and stays **evaluation-only**. It uses 20 iterations,
51
+ restart probability 0.15, normalized positive lexical seeds, bidirectional
52
+ recorded relationships within type eligibility, and no seeds for a no-match
53
+ query. These choices were fixed before scoring; no weights were tuned to labels.
54
+ The grep baseline matches identifiers, not source text; it retains its existing
55
+ ordering, with eligibility applied before its limit and the same budgeted renderer.
56
+
57
+ Latencies are warm in-process measurements over 62 units. They exclude lexical
58
+ snapshot construction, live embedding/network latency and Rails extraction; they
59
+ are not host-scale performance claims. Exact tokens count returned context only,
60
+ using the existing `cl100k_base` capture machinery. Agent task outcomes were not
61
+ measured. Ambiguous typed identities, concern/callback metadata and unrelated-hub
62
+ controls also have targeted regressions; those synthetic controls do not replace
63
+ representative host evaluation. Broader production/provider/agent evidence remains
64
+ open under #227.
65
+
66
+ ## Checked-in Canopy retrieval gate
67
+
68
+ The required CI `coverage` job runs this offline command before its test suite:
69
+
70
+ ```bash
71
+ bundle exec ruby -Ilib bench/evaluation/runner.rb
72
+ # Report: tmp/retrieval-evaluation.json (uploaded by CI)
73
+ ```
74
+
75
+ `bench/evaluation/` contains a versioned, deliberately small baseline:
76
+
77
+ - **Corpus:** 62 runtime-extracted Canopy units and 28 annotated queries covering
78
+ publishing, billing, newsletters, and support. Source and relationships come
79
+ from the committed fictional `woods-testbed` Rails application; no application
80
+ records are included. Exact application/Woods revisions are in `corpus.json`.
81
+ This snapshot selects models, services, jobs, controllers, Pundit policies, and
82
+ mailers; it does not represent every extractor or a large production index.
83
+ - **Embeddings:** real `all-MiniLM-L6-v2` ONNX inference, captured with a pinned
84
+ model revision and file hashes. CI replays those vectors through the current
85
+ production classifier, stores, graph, ranker, type fallback, and assembler.
86
+ No hash-based fake embeddings or expected-answer vectors are used.
87
+ - **Gate:** per-strategy precision at 5, recall, and MRR must remain at least 95%
88
+ of the first observed score (floors rounded down to six decimal places).
89
+ Corpus/vector/token-provenance mismatches, missing strategies, and unexpected
90
+ strategy selection also fail. The baseline format is developer-only and is
91
+ **not** the `EVAL_BASELINE_FILE` aggregate-threshold format.
92
+
93
+ B-190/B-191 recapture on Ruby 4.0.6 (five warmed pipeline repetitions per query; Ruby 3.3.1
94
+ and 3.4.10 replay the same answers):
95
+
96
+ | Strategy | Queries | Precision@5 | Recall | MRR | Mean actual context tokens |
97
+ |---|---:|---:|---:|---:|---:|
98
+ | Keyword | 4 | 0.292 | 0.500 | 0.750 | 1,020.8 |
99
+ | Vector | 4 | 0.313 | 0.375 | 0.625 | 1,041.2 |
100
+ | Graph | 8 | 0.813 | 0.519 | 1.000 | 1,041.1 |
101
+ | Hybrid | 4 | 0.750 | 0.396 | 1.000 | 1,058.8 |
102
+ | Direct with type filtering | 4 | 0.375 | 0.750 | 0.625 | 551.0 |
103
+ | Within-type vector fallback | 4 | 0.400 | 1.000 | 1.000 | 1,250.8 |
104
+
105
+ Precision@5 divides relevant hits by the actual returned slice size (up to five),
106
+ not always by five. Recall divides retrieved relevant units by all annotated
107
+ relevant units. MRR is the mean reciprocal rank of the first relevant hit; a
108
+ query with no relevant hit contributes zero.
109
+
110
+ `profiles.json` binds six measured Ruby runtimes to a shared context capture:
111
+ Ruby 3.0.7, 3.1.7, 3.2.11, 3.3.1, 3.4.10, and 4.0.6. Other Ruby engines or
112
+ major/minor versions fail with a recapture instruction. All six produce identical
113
+ retrieved identifiers, context bytes, token counts and quality metrics for every
114
+ query. The original positive per-runtime quality floors remain unchanged in
115
+ `baseline.json` and `baseline_legacy.json`.
116
+
117
+ B-191 orders equal PageRank scores lexically by identifier. B-192 orders hybrid
118
+ candidates by descending score, identifier, then source before truncation and
119
+ RRF; the same ordering selects the three vector seeds for graph expansion.
120
+ Only `hybrid-2` changes from the prior Ruby 3.3–4.0 capture: `Newsletter::DeliverJob`
121
+ now precedes `Newsletter::Delivery`, matching the prior Ruby 3.0–3.2 answer.
122
+ No quality metric or exact token count changes. This is production retrieval
123
+ ordering, with no answer reordering in the evaluation harness.
124
+
125
+ These are separate annotated query groups, **not a controlled comparison of six
126
+ strategies on identical questions**. Relevance annotations were written from
127
+ fixture source and its functional contract before scoring. Initial graph and
128
+ direct probes remain in the corpus; additional probes exercise the supported
129
+ snake-case graph roots and actual within-type fallback. The B-190 fix now
130
+ resolves all four originally empty graph queries, including `trace Billing::Invoice`
131
+ and `trace ReviewAssignment`. The B-190 change affected those four graph answers. B-191 additionally
132
+ changes vector and within-type fallback ranks or membership; all original queries,
133
+ relevance labels, vectors, and quality floors remain unchanged. Graph precision/recall/MRR increased from 0.417 / 0.238 / 0.500 to
134
+ 0.813 / 0.519 / 1.000. Low semantic recall and editorial misses are
135
+ also retained. The floors prevent further regression; they are not release
136
+ quality targets or evidence that retrieval is already good enough.
137
+
138
+ `capture_report.json` records per-query ranks, quality, latency ranges, and exact
139
+ `cl100k_base` counts of the returned context. Those counts are bound to context
140
+ SHA-256 values; the tokenizer vocabulary asset hash, pattern, special-token map,
141
+ and package version are recorded alongside them. Changed context requires explicit recapture and review. Woods'
142
+ `tokens_used` remains a separate character-ratio estimate. Neither number counts
143
+ agent prompts, generated answers, or total agent usage. Vector capture records
144
+ actual model-input token counts (including special tokens after truncation),
145
+ separately from output-context tokens. The model truncates inputs to 512 tokens,
146
+ so long source units are only partially represented.
147
+
148
+ Pipeline latency excludes model loading, downloads, live embedding requests, and
149
+ network time. `vectors.json` separately records the warm embedding batch time,
150
+ input counts, model/package provenance, and SHA-256 of each model asset. Timing
151
+ is observational and has no shared-runner CI threshold. This baseline does not
152
+ validate OpenAI/Ollama quality, their latency, production-scale storage, or agent
153
+ outcomes. The baseline scope of #227 is complete; broader provider and
154
+ production-scale evaluation remain separate follow-up work.
155
+
156
+ ### Reproduce or review a new capture
157
+
158
+ Ordinary CI needs only the Ruby bundle and the checked-in files. Model refresh is
159
+ an explicit maintenance operation with isolated Python dependencies:
160
+
161
+ ```bash
162
+ uv venv /tmp/woods-eval-venv
163
+ uv pip install --python /tmp/woods-eval-venv/bin/python -r bench/evaluation/requirements.txt
164
+ /tmp/woods-eval-venv/bin/python bench/evaluation/capture.py /tmp/woods-eval-model
165
+ ```
166
+
167
+ This overwrites `vectors.json` with real inference from the pinned Apache-2.0
168
+ model. To capture changed context after an intentional pipeline change:
169
+
170
+ ```bash
171
+ bundle exec ruby -Ilib -r./bench/evaluation/runner -e \
172
+ 'r=RetrievalBaseline::Runner.new; File.write("/tmp/retrieval-capture.json", JSON.pretty_generate(r.run))'
173
+ /tmp/woods-eval-venv/bin/python bench/evaluation/capture_tokens.py /tmp/retrieval-capture.json
174
+ ```
175
+
176
+ Replay all six measured Ruby lines before replacing the shared capture. Retain
177
+ the exact capture runtime in the report and preserve each runtime’s original
178
+ quality floors. A newly observed runtime difference needs separate reviewed
179
+ captures rather than being hidden by answer reordering.
180
+
181
+ Review every changed answer and score before updating the matching baseline digests or
182
+ floors. Never lower a floor merely to make CI pass. Keep labels independent of
183
+ rankings and preserve hard or failed queries. When changing the application
184
+ snapshot, use a fresh runtime `woods:extract` into a disposable output directory,
185
+ retain only deliberate source/relationship fields, and record both repository
186
+ SHAs. Do not copy a host's live index or application records into the repository.
187
+
17
188
  ## Agent-level ablation
18
189
 
19
190
  Retrieval scores say whether the right units come back. They do not say whether an agent finishes the job. The ablation runs the same task set twice per task, once with the Woods MCP server available and once without, and compares resolution rate, tokens, cost, and turns.
@@ -80,7 +251,7 @@ bin/rails "woods:evaluate:ablation[config/eval_ablation.json]"
80
251
 
81
252
  The report holds every attempt (`task_id`, `condition`, `resolved`, `total_tokens`, `cost_usd`, `turns`, `duration_ms`, `error`, `provenance`) and a summary per condition (`resolution_rate`, `mean_tokens`, `mean_cost_usd`, `mean_turns`, `tasks`, `errors`) plus a `delta` (on minus off, present only when both conditions ran). Tokens are the sum of input, output, cache-creation, and cache-read tokens. `mean_tokens` is `null` when every trial in that condition errored before the agent produced JSON.
82
253
 
83
- A timeout (`AblationRunner.new(..., timeout: seconds)`, default 600) bounds every command a trial runs, applied independently to each one rather than as a single budget shared across the trial: worktree add/remove, the optional `reset`, the agent invocation, and the check. A timed-out `reset` or `check` counts as an error the same way a timed-out agent invocation does. When the default subprocess executor is in use, a timed-out command is terminated (`TERM`, then `KILL` if still alive after a short grace period) so it never outlives the trial that started it.
254
+ A timeout (`AblationRunner.new(..., timeout: seconds)`, default 600) bounds every command a trial runs, applied independently to each one rather than as a single budget shared across the trial: worktree add/remove, the optional `reset`, the agent invocation, and the check. A timed-out `reset` or `check` counts as an error the same way a timed-out agent invocation does. The default subprocess executor creates an owned process group. On timeout it sends `TERM`, then `KILL` after a short grace period if necessary, to that group, including ordinary descendants whose parent already exited. Commands must not deliberately detach into separate sessions/groups; those escape this cleanup. Custom executors exposing only a PID retain single-process cleanup; executors exposing no PID must handle their own subprocess lifecycle.
84
255
 
85
256
  ### Where to get tasks
86
257
 
@@ -89,3 +260,295 @@ The Rails Foundation's "Agents on Rails" benchmark (announced 2026-08-13, built
89
260
  ### Reading the numbers
90
261
 
91
262
  A negative `delta.mean_tokens` with an equal or higher `delta.resolution_rate` is the result that sells the index, on the tasks actually run. It does not, by itself, establish that the index causes the difference: run at least ten tasks, two conditions on five tasks is noise, and remember that agent runs are not perfectly reproducible even at a fixed baseline SHA. Keep the baseline SHA and Woods generation fixed across a comparison (both are recorded per result), and do not let the `on` agent run `woods:extract` mid-task.
263
+
264
+ ## Explicit scope comparison
265
+
266
+ `bench/evaluation/scope_comparison.rb` reuses #227's 62-unit Canopy extraction,
267
+ 28 original questions, captured MiniLM vectors, unchanged gold labels, and the
268
+ same 1200-token budgets. Fixed domain-word rules choose the existing billing,
269
+ newsletter, or support directories across models, controllers, jobs, and use
270
+ cases. Other questions are unscoped controls. This assigns **no package ownership**
271
+ to Canopy. Expected units outside a requested directory remain in the gold labels,
272
+ so restricting scope can reduce recall for questions about cross-boundary behavior.
273
+
274
+ ```bash
275
+ bundle exec ruby -Ilib bench/evaluation/scope_comparison.rb /tmp/woods-scope-raw.json
276
+ python bench/evaluation/capture_tokens.py /tmp/woods-scope-raw.json scope_comparison_capture.json
277
+ ```
278
+
279
+ Use the pinned Python requirements in this directory's existing evaluation runbook.
280
+ The capture records per-query scope, outcomes, context hashes, exact cl100k counts,
281
+ and the existing precision@5, recall, and MRR metrics. Warm replay latency includes
282
+ scope preparation and excludes provider/network time. This is a filtering and
283
+ ranking comparison, not a coding-task or live-host performance claim.
284
+
285
+ `spec/retrieval/package_scope_spec.rb` separately builds a synthetic Packwerk
286
+ filesystem fixture and obtains root, nested, and sibling ownership from the real
287
+ `PackageExtractor`. It verifies package boundaries without inventing ownership in
288
+ a captured host corpus. Live pgvector/Qdrant tests cover a small eligible package
289
+ behind globally stronger matches, typed chunks, and old vectors without metadata
290
+ payload fields. Broader agent task-quality evidence remains unmeasured by this
291
+ scope comparison.
292
+
293
+ The reviewed capture uses 11 scoped questions and 17 unchanged controls per mode.
294
+ On the 11 questions with a requested directory scope:
295
+
296
+ | Mode | Scope | Precision@5 | Recall | MRR | Mean exact context tokens |
297
+ |---|---|---:|---:|---:|---:|
298
+ | Semantic | none | 0.5758 | 0.6530 | 0.9545 | 907.55 |
299
+ | Semantic | explicit paths | 0.6364 | 0.5621 | 0.9091 | 820.73 |
300
+ | Lexical | none | 0.5394 | 0.7364 | 0.8636 | 1201.45 |
301
+ | Lexical | explicit paths | 0.6970 | 0.6455 | 0.8030 | 936.73 |
302
+
303
+ Scope improves precision here while losing cross-boundary expected units. It is
304
+ an explicit intent filter, not a default ranking improvement. Scoped lexical
305
+ median query time was 2.30 ms versus 0.41 ms unscoped on these 11 replay questions;
306
+ preparing the full metadata snapshot is visible even on this small corpus.
307
+ Exact token counts measure the returned context, not MCP metadata or agent prompts;
308
+ Woods' 1200-token budget is an estimate and can differ from cl100k counts.
309
+
310
+ ## Compact evidence comparison (#403)
311
+
312
+ `bench/evaluation/evidence_comparison.rb` replays the same #227 Canopy corpus,
313
+ 28 questions, expected unit labels, captured MiniLM vectors and per-question
314
+ budgets with full, compact and outline evidence, separately for semantic and
315
+ lexical retrieval. Only within-unit evidence formatting varies within each
316
+ strategy. The checked capture binds all 168 returned contexts to exact
317
+ `tiktoken 0.11.0` `cl100k_base` counts; full-mode hashes match the approved prior
318
+ comparison. The original labels are unit-level labels, not method/span relevance
319
+ or task-correctness labels.
320
+
321
+ | Strategy / evidence | P@5 | Unit recall | MRR | Mean exact context tokens |
322
+ |---|---:|---:|---:|---:|
323
+ | Semantic / full | 0.5363 | 0.5798 | 0.8571 | 1000.68 |
324
+ | Semantic / compact | 0.4786 | 0.6476 | 0.8571 | 1034.68 |
325
+ | Semantic / outline | 0.4536 | 0.8304 | 0.8571 | 976.86 |
326
+ | Lexical / full | 0.4917 | 0.6661 | 0.7798 | 1152.64 |
327
+ | Lexical / compact | 0.4881 | 0.7077 | 0.7887 | 1063.79 |
328
+ | Lexical / outline | 0.3857 | 0.7964 | 0.7887 | 1014.75 |
329
+
330
+ These results do not justify a default change. Outlines can expose more unit names
331
+ without their method bodies; 19 of 100 returned semantic compact units and 13 of
332
+ 103 lexical compact units have no selected source span (the response explicitly
333
+ reports omissions and may contain runtime metadata). Per-unit selected/omitted
334
+ span counts remain in the capture. Increased unit recall is not evidence that an
335
+ agent saw the implementation it needed. Compact semantic context consumed more
336
+ actual tokens on average. Structured provenance and MCP envelopes are additional
337
+ output tokens, outside the context-only counts above.
338
+
339
+ Median warm query time across questions was 3.24/11.80/11.44 ms for semantic
340
+ full/compact/outline and 0.46/4.44/6.08 ms for lexical modes. This measures local
341
+ replay and includes source parsing; semantic provider/network time is excluded.
342
+ It does not establish live-host performance.
343
+
344
+ A separate deterministic long-prefix regression fixes the target method before
345
+ selection: 80 unrelated methods precede `refund(payment)`. Under the same small
346
+ context allowance the compact selector includes its entire body and exact
347
+ published byte/hash coordinates. This is a synthetic retention test, separate
348
+ from the real-corpus unit metrics above. Unicode, shared-line declarations, nested
349
+ blocks, commented inlined concern display, inherited metadata, unknown physical
350
+ locations and unknown generation are covered by source-evidence contract tests.
351
+ Installed stdio and HTTP checks exercise generation attribution and typed,
352
+ SHA-guarded full lookup on a real Rails extraction.
353
+
354
+ Reproduce the corpus comparison without changing labels:
355
+
356
+ ```bash
357
+ bundle exec ruby -Ilib bench/evaluation/evidence_comparison.rb /tmp/woods-evidence-raw.json
358
+ python bench/evaluation/capture_tokens.py /tmp/woods-evidence-raw.json evidence_comparison_capture.json
359
+ ```
360
+
361
+ Use the same tokenizer version/vocabulary recorded by the capture. Raw contexts
362
+ are an external artifact; the checked file retains hashes, exact counts, unit
363
+ outcomes and source-span counts. Candidate ranking and ownership diversity are
364
+ outside this experiment.
365
+
366
+ ### Public Writebook task attempts
367
+
368
+ [`public_task_evidence_capture.json`](../bench/evaluation/public_task_evidence_capture.json)
369
+ records twelve actual agent attempts on the Rails Foundation's public
370
+ [`ai-evals`](https://github.com/rails/ai-evals) Writebook tasks at revision
371
+ `5327efe54dd767d662ab89332f6a4285fef07bd8` (Writebook 1.2.1):
372
+ `ar-announce-once`, `aj-enqueue-after-commit`, and `ar-archive-book-access`.
373
+ Each task ran twice per condition with the same task prompt, client 2.1.267,
374
+ resolved model `claude-sonnet-5`, 40-turn cap and Woods source revision
375
+ `33abd6da8e072c7f510d052292aaff4260c30c7f`. Hidden verifiers remained outside the
376
+ agent workspace until the attempt finished. Baseline failure/reference success
377
+ prechecks and prompt/artifact hashes are retained in the capture.
378
+
379
+ The intervention changed the default evidence mode at the MCP boundary. Full
380
+ lookup kept its existing complete output; compact lookup used a 1200-estimated-token
381
+ budget. Explicit evidence choices, metadata-only reads and full-source follow-ups
382
+ remained available. This measures a default-mode change, not a forced mode on
383
+ every call. All actual evidence calls used `lookup` (26 full-condition calls and
384
+ 20 compact-condition calls); there were no `codebase_retrieve` calls, query hints,
385
+ or explicit full follow-ups. Consequently this does not establish retrieval
386
+ ranking or query-guided compact evidence benefits.
387
+
388
+ | Condition | Hidden task suites passed | Median MCP text tokens (cl100k) | Median agent wall time |
389
+ |---|---:|---:|---:|
390
+ | Full | 5/6 | 8661 | 180.3 s |
391
+ | Compact | 4/6 | 2316 | 132.8 s |
392
+
393
+ Both compact announcement attempts failed the go-live-then-return-to-draft edge;
394
+ the second full attempt passed it. Both conditions passed both repetitions of the
395
+ other tasks. Two compact `Accessable` lookups returned not-found errors; these are
396
+ retained as valid tool errors, not discarded trials. The earlier 20-turn calibration
397
+ pair hit the turn cap and failed the same announcement edge in both modes; the
398
+ fixed 40-turn primary protocol and all failed/setup pilots are recorded separately.
399
+
400
+ Compact delivered less MCP text in this small sample but passed one fewer task.
401
+ That is a reason to keep it explicit and retain full-source verification, not to
402
+ change the default. Two trials could overlap, so wall times are descriptive rather
403
+ than isolated performance estimates. `cl100k_base` measures representation size,
404
+ not the Claude tokenizer or billing. The capture separately records serialized MCP
405
+ response counts, client-reported usage and list-price estimates; these must not be
406
+ substituted for model-visible context or billed charges. It contains metrics,
407
+ arguments and hashes, without shipping application source or agent transcripts.
408
+
409
+
410
+ ## Source-owner overlap experiment (#412)
411
+
412
+ `bench/evaluation/owner_overlap_comparison.rb` evaluates an offline selector after
413
+ ranking, using the production context assembler's full-source appender, section
414
+ allocations, truncation and token estimator. It does not add a runtime option.
415
+ The strongest result stays first. A later candidate moves ahead only when the
416
+ higher-scoring pending span is already completely covered by delivered source.
417
+ Several disjoint methods from one file remain separate; typed units are never
418
+ merged. Exact-target/pinpoint queries and sections with one verified owner bypass
419
+ the policy. Unknown origins block promotion across their position. Truncated
420
+ parents establish no complete span coverage. Candidate scores remain unchanged.
421
+
422
+ Origins require a unique exact byte substring in an explicitly supplied immutable
423
+ source snapshot. Display paths alone, ambiguous repeated text, and synthesized
424
+ inlined concern display do not prove physical ownership. Original-file and
425
+ published-source hashes bind the verified spans. There are no query-time host-file
426
+ reads. These are offline fixture origins, not newly published runtime provenance.
427
+
428
+ The fixed fixtures define necessary methods and budgets before selection. Seven
429
+ synthetic cases and one real Woods source case use actual `RubyAnalyzer` output;
430
+ the latter is static source evidence, not a booted Rails application. Synthetic
431
+ executable oracles load only delivered snippets into an isolated namespace and
432
+ assert their fixed multi-method behavior. The wrapper supplies the declared class
433
+ name for a method; it supplies no absent implementation. These deterministic task
434
+ checks are separate from an LLM answer/task benchmark.
435
+
436
+ | Fixture | Necessary-method recall, baseline → overlap | Executable fixture, baseline → overlap | Exact context tokens, baseline → overlap |
437
+ |---|---|---|---|
438
+ | Crowded class/method overlap | 0.5 → 1.0 | fail → pass | 312 → 308 |
439
+ | Single owner | 1.0 → 1.0 | pass → pass | 312 → 312 |
440
+ | Exact target / pinpoint (each) | 0.5 → 0.5 | fail → fail | 312 → 312 |
441
+ | Truncated parent | 0.0 → 0.0 | fail → fail | 165 → 165 |
442
+ | Disjoint methods in one file | 1.0 → 1.0 | pass → pass | 97 → 97 |
443
+ | Unknown inlined display origin | 0.5 → 0.5 | fail → fail | 325 → 325 |
444
+ | Woods storage identity / filename source | 0.5 → 1.0 | not measured | 336 → 366 |
445
+
446
+ Relevant-owner coverage equals necessary-method recall in these fixtures because
447
+ each required owner has one method, except the disjoint-method case where both
448
+ methods share one owner and coverage remains 1.0. The real-source case retains
449
+ `StorageIdentity.key` and `FilenameUtils#collision_safe_filename`; that is source
450
+ retention, not proof that every helper needed to execute a change is present.
451
+ The budget is Woods' existing character estimate: for example, the 230-token
452
+ crowded fixture delivers 312/308 cl100k tokens. Exact counts measure context text,
453
+ not prompts, provenance envelopes or agent output. No new token calibration or
454
+ section-budget borrowing is introduced.
455
+
456
+ The same #227 28 Canopy questions, gold labels, vectors and budgets serve as a
457
+ negative control. All 62 units have distinct display paths and lack verified
458
+ original spans. Both conditions therefore return identical contexts, matching
459
+ the approved semantic full capture byte for byte. This demonstrates safe bypass;
460
+ it cannot demonstrate owner-selection benefit on a runtime Rails index.
461
+ Equal scores, typed collisions, unknown-origin barriers and separate
462
+ primary/supporting/framework allocations have additional regression coverage.
463
+
464
+ **Decision: keep this evaluation-only.** Fixed overlap cases benefit, but a
465
+ truncated strongest class still consumes the whole allowance before complementary
466
+ evidence can be selected. Real runtime provenance and actual agent answer/task
467
+ correctness under this selector remain unmeasured. The Writebook task attempts
468
+ above tested evidence formatting, not this policy, and cannot fill that gap.
469
+ A production proposal needs trustworthy published physical spans and a paired
470
+ representative task evaluation; no default change, owner quota, or near-relevance
471
+ promotion is justified by these fixtures.
472
+
473
+ ```bash
474
+ bundle exec ruby -Ilib bench/evaluation/owner_overlap_comparison.rb /tmp/woods-owner-overlap-raw.json
475
+ python bench/evaluation/capture_tokens.py /tmp/woods-owner-overlap-raw.json owner_overlap_capture.json
476
+ ```
477
+
478
+ The checked capture retains all 72 results, unchanged #227 gold labels, fixed-case
479
+ requirements, source hashes, actual context counts/hashes, original scores and
480
+ selection/deferral/budget-exhaustion traces. It retains failed cases. Use the
481
+ recorded `tiktoken` version and vocabulary; keep raw contexts outside the checkout.
482
+
483
+
484
+ ## Matched optional context-hook tasks (#406)
485
+
486
+ **Both conditions passed all four task attempts.** This small comparison does not
487
+ establish a task-quality or efficiency improvement. Hints remained optional.
488
+ The context condition returned less MCP text in aggregate, added its own context,
489
+ and had a higher median agent wall time; individual repetitions varied.
490
+
491
+ `bench/evaluation/hook_context_capture.json` records all eight preselected trials,
492
+ returned-context hashes and actual token counts, official-client usage, emitted
493
+ and delivered hints, callback latency and hidden-verifier outcomes. Two tasks
494
+ from Rails Foundation's public `ai-evals` at
495
+ `5327efe54dd767d662ab89332f6a4285fef07bd8` used disposable Writebook 1.2.1/Ruby 3.4.7
496
+ images. Seeded defects failed and reference fixes passed before agent trials.
497
+ Both arms used Claude Code 2.1.267, `claude-sonnet-5`, a 40-turn limit, the same
498
+ prompt and source snapshot, full MCP evidence, and lexical retrieval. Refresh
499
+ was disabled in both arms to isolate the independent context switch. Each
500
+ condition ran twice per task; two trials ran concurrently. The hidden verifier
501
+ was copied only after the agent exited.
502
+
503
+ | Task attempt | Manual agent wall (s) | Context agent wall (s) | Hidden verifier, manual / context |
504
+ |---|---:|---:|---|
505
+ | `aj-enqueue-after-commit`, rep 1 | 111.3 | 113.6 | pass / pass |
506
+ | `aj-enqueue-after-commit`, rep 2 | 173.4 | 154.9 | pass / pass |
507
+ | `ar-atomic-import`, rep 1 | 105.3 | 210.4 | pass / pass |
508
+ | `ar-atomic-import`, rep 2 | 126.5 | 119.9 | pass / pass |
509
+
510
+ | Measure | Manual tools | Context enabled |
511
+ |---|---:|---:|
512
+ | Tasks passed | 4 / 4 | 4 / 4 |
513
+ | Median agent wall (s) | 118.9 | 137.4 |
514
+ | Total MCP text, actual cl100k tokens | 45,834 | 39,550 |
515
+ | Additional delivered hint tokens, cl100k | 0 | 1,914 |
516
+ | Emitted / delivered hints | 0 / 0 | 16 / 16 |
517
+ | Client-reported input tokens | 160 | 158 |
518
+ | Client-reported cache creation tokens | 217,920 | 206,056 |
519
+ | Client-reported cache read tokens | 5,715,709 | 5,467,210 |
520
+ | Client-reported output tokens | 40,884 | 42,000 |
521
+
522
+ The 16 context callbacks took 0.590–0.818s (median
523
+ 0.613s); the largest complete JSON envelope was 1,067 bytes.
524
+ Disabled callbacks took 1.75ms median and emitted nothing.
525
+ Every emitted hint appeared in the official client's `hook_additional_context`
526
+ attachments. A separate native-plugin probe also matched SessionStart and
527
+ PostToolUse hint strings inside captured model requests. Attachment delivery does
528
+ not establish that the model used a hint. Callback measurement wraps the actual
529
+ entrypoint and excludes the instrumentation wrapper's own startup; agent wall
530
+ includes the whole client run. OS scheduling and cold startup can still cause a
531
+ silent timeout under the fixed deadline.
532
+
533
+ An independent review checked all 44 rendered relationship claims against each
534
+ trial's typed graph and found no incorrect claim. Useful owners were often
535
+ already inspected; broad transitive candidates were sometimes redundant. One
536
+ atomic-import run explicitly followed the DemoContent candidate and found no
537
+ matching call site. New test files received honest unresolved-path notices.
538
+ Source freshness was initially unknown because of `unverified_consumption_scopes`,
539
+ then edits were marked drifted; truncated and pre-refresh qualifications remained
540
+ visible. Correct candidates are not proof of task benefit or complete impact.
541
+
542
+ Exact cl100k counts measure representation size, separately from the client's
543
+ repeated input/cache/output accounting. The client's summed list-price estimates
544
+ were $2.424 / $2.338; these are not billed charges or evidence of a
545
+ repeatable saving. This two-task sample and overlapping execution cannot establish
546
+ isolated performance or general quality. Raw traces remain local review artifacts;
547
+ the checked capture binds their hashes. `diff.patch` covers tracked files only;
548
+ new untracked tests are preserved through native Write/Edit records and separately
549
+ labelled transcript reconstructions, not claimed as disk-captured final trees.
550
+
551
+ The measured hook implementation is `21a578d6`; subsequent main integration left
552
+ its helper and entrypoint bytes unchanged. The earlier smoke setup missing new
553
+ bundle executable stubs is retained as an invalid setup attempt. The fixed
554
+ primary trials all completed and none was discarded or replaced.