woods 2.0.0.beta2 → 2.0.0.beta4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +339 -1
  3. data/CONTRIBUTING.md +188 -12
  4. data/README.md +93 -174
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +109 -8
  7. data/docs/AGENT_SETUP.md +98 -7
  8. data/docs/BACKEND_MATRIX.md +25 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +267 -16
  11. data/docs/CONSOLE_MCP_SETUP.md +80 -7
  12. data/docs/DOCKER_SETUP.md +22 -3
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +45 -6
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +147 -7
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +276 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +37 -22
  22. data/docs/MCP_WORKTREE_SETUP.md +43 -83
  23. data/docs/NOTION_INTEGRATION.md +13 -0
  24. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  25. data/docs/PUBLISHED_INDEX.md +72 -0
  26. data/docs/README.md +7 -0
  27. data/docs/RETRIEVAL_GUIDE.md +273 -12
  28. data/docs/RUNTIME_TRACING.md +71 -0
  29. data/docs/SOURCE_FRESHNESS.md +143 -0
  30. data/docs/TROUBLESHOOTING.md +129 -18
  31. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  32. data/docs/UPGRADING_TO_2.md +48 -22
  33. data/docs/WATCH_DAEMON.md +277 -67
  34. data/exe/woods-agent-config +6 -0
  35. data/exe/woods-extract +5 -0
  36. data/exe/woods-hook-context +6 -0
  37. data/exe/woods-mcp-start +14 -9
  38. data/lib/generators/woods/pgvector_generator.rb +8 -2
  39. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  40. data/lib/tasks/woods.rake +47 -397
  41. data/lib/woods/agent_configuration/applier.rb +135 -0
  42. data/lib/woods/agent_configuration/cli.rb +101 -0
  43. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  44. data/lib/woods/agent_configuration/document.rb +105 -0
  45. data/lib/woods/agent_configuration/error.rb +7 -0
  46. data/lib/woods/agent_configuration/launcher.rb +75 -0
  47. data/lib/woods/agent_configuration/layout.rb +72 -0
  48. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  49. data/lib/woods/agent_configuration/plan.rb +98 -0
  50. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  51. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  52. data/lib/woods/agent_configuration/planner.rb +63 -0
  53. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  54. data/lib/woods/agent_configuration/preflight.rb +100 -0
  55. data/lib/woods/agent_configuration/recovery.rb +49 -0
  56. data/lib/woods/ast/node.rb +2 -0
  57. data/lib/woods/ast/parser.rb +38 -5
  58. data/lib/woods/builder.rb +21 -5
  59. data/lib/woods/cache/cache_middleware.rb +28 -7
  60. data/lib/woods/cache/cache_store.rb +4 -5
  61. data/lib/woods/change_set.rb +5 -4
  62. data/lib/woods/console/credential_index.rb +20 -2
  63. data/lib/woods/console/credential_scanner.rb +18 -17
  64. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  65. data/lib/woods/console/dispatch_pipeline.rb +7 -0
  66. data/lib/woods/console/embedded_executor.rb +32 -10
  67. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  68. data/lib/woods/console/rack_middleware.rb +22 -13
  69. data/lib/woods/console/server.rb +18 -16
  70. data/lib/woods/console/sql_noise_stripper.rb +9 -7
  71. data/lib/woods/console/sql_table_scanner.rb +47 -7
  72. data/lib/woods/console/sql_validator.rb +49 -9
  73. data/lib/woods/console/sqlite_read_guard.rb +46 -0
  74. data/lib/woods/coordination/pipeline_lock.rb +3 -2
  75. data/lib/woods/dependency_graph.rb +65 -13
  76. data/lib/woods/embedding/corpus.rb +94 -0
  77. data/lib/woods/embedding/indexer.rb +114 -60
  78. data/lib/woods/embedding/openai.rb +17 -6
  79. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  80. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  81. data/lib/woods/export/typed_reader.rb +56 -0
  82. data/lib/woods/extractor.rb +277 -149
  83. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  84. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  85. data/lib/woods/extractors/caching_extractor.rb +3 -1
  86. data/lib/woods/extractors/concern_extractor.rb +64 -6
  87. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  88. data/lib/woods/extractors/controller_extractor.rb +13 -4
  89. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  90. data/lib/woods/extractors/declared_parent.rb +55 -0
  91. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  92. data/lib/woods/extractors/engine_extractor.rb +3 -1
  93. data/lib/woods/extractors/event_extractor.rb +4 -2
  94. data/lib/woods/extractors/factory_extractor.rb +3 -1
  95. data/lib/woods/extractors/graphql_extractor.rb +10 -13
  96. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  97. data/lib/woods/extractors/job_extractor.rb +6 -19
  98. data/lib/woods/extractors/lib_extractor.rb +13 -9
  99. data/lib/woods/extractors/mailer_extractor.rb +26 -15
  100. data/lib/woods/extractors/manager_extractor.rb +3 -1
  101. data/lib/woods/extractors/method_parameters.rb +53 -0
  102. data/lib/woods/extractors/middleware_argument.rb +65 -0
  103. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  104. data/lib/woods/extractors/migration_extractor.rb +3 -1
  105. data/lib/woods/extractors/model_extractor.rb +26 -34
  106. data/lib/woods/extractors/package_extractor.rb +24 -4
  107. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  108. data/lib/woods/extractors/policy_extractor.rb +3 -1
  109. data/lib/woods/extractors/poro_extractor.rb +13 -9
  110. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  111. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  112. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  113. data/lib/woods/extractors/route_extractor.rb +3 -1
  114. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  115. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  116. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  117. data/lib/woods/extractors/service_extractor.rb +3 -1
  118. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  119. data/lib/woods/extractors/shared_utility_methods.rb +48 -19
  120. data/lib/woods/extractors/source_nesting.rb +1 -1
  121. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  122. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  123. data/lib/woods/extractors/validator_extractor.rb +3 -1
  124. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  125. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  126. data/lib/woods/gem_mapper.rb +2 -0
  127. data/lib/woods/git_history.rb +116 -0
  128. data/lib/woods/graph_analyzer.rb +35 -6
  129. data/lib/woods/hooks/context_cli.rb +54 -0
  130. data/lib/woods/hooks/context_event.rb +88 -0
  131. data/lib/woods/hooks/context_hint.rb +73 -0
  132. data/lib/woods/hooks/context_impact.rb +77 -0
  133. data/lib/woods/hooks/context_output.rb +47 -0
  134. data/lib/woods/hooks/context_state.rb +102 -0
  135. data/lib/woods/hooks/refresh.rb +79 -0
  136. data/lib/woods/hooks/rule_projection.rb +78 -0
  137. data/lib/woods/input_rules.rb +19 -0
  138. data/lib/woods/mcp/bearer_auth.rb +22 -13
  139. data/lib/woods/mcp/bootstrapper.rb +79 -4
  140. data/lib/woods/mcp/config_resolver.rb +2 -1
  141. data/lib/woods/mcp/index_reader.rb +334 -162
  142. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  143. data/lib/woods/mcp/origin_guard.rb +17 -9
  144. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  145. data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
  146. data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
  147. data/lib/woods/mcp/search_results.rb +74 -0
  148. data/lib/woods/mcp/server.rb +178 -63
  149. data/lib/woods/mcp/tool_contract.rb +3 -1
  150. data/lib/woods/mcp/tool_response_renderer.rb +41 -0
  151. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  152. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  153. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  154. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  155. data/lib/woods/mcp/traversal_response.rb +22 -0
  156. data/lib/woods/notion/exporter.rb +56 -17
  157. data/lib/woods/obsidian/destination_plan.rb +98 -0
  158. data/lib/woods/obsidian/name_mapper.rb +19 -3
  159. data/lib/woods/obsidian/note_builder.rb +19 -10
  160. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  161. data/lib/woods/operator/pipeline_guard.rb +18 -13
  162. data/lib/woods/path_dispatcher.rb +13 -6
  163. data/lib/woods/payload_store.rb +27 -26
  164. data/lib/woods/published_index/typed_unit_reader.rb +40 -3
  165. data/lib/woods/published_index.rb +2 -2
  166. data/lib/woods/railtie.rb +3 -3
  167. data/lib/woods/railtie_support.rb +12 -12
  168. data/lib/woods/rake_helpers.rb +382 -0
  169. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  170. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  171. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  172. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  173. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  174. data/lib/woods/resilience/index_validator.rb +112 -23
  175. data/lib/woods/retrieval/context_assembler.rb +50 -15
  176. data/lib/woods/retrieval/lexical_assembler.rb +84 -0
  177. data/lib/woods/retrieval/lexical_index.rb +120 -0
  178. data/lib/woods/retrieval/ranker.rb +4 -2
  179. data/lib/woods/retrieval/scope.rb +108 -0
  180. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  181. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  182. data/lib/woods/retrieval/search_executor.rb +86 -27
  183. data/lib/woods/retrieval/source_evidence.rb +200 -0
  184. data/lib/woods/retriever.rb +98 -22
  185. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  186. data/lib/woods/session_tracer/file_store.rb +6 -1
  187. data/lib/woods/session_tracer/middleware.rb +10 -12
  188. data/lib/woods/session_tracer/redis_store.rb +22 -6
  189. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  190. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  191. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  192. data/lib/woods/source_inputs/consumer_errors.rb +31 -0
  193. data/lib/woods/source_inputs/handoff.rb +102 -0
  194. data/lib/woods/source_inputs/launcher.rb +157 -0
  195. data/lib/woods/source_inputs/manifest.rb +124 -0
  196. data/lib/woods/source_inputs/private_key.rb +55 -0
  197. data/lib/woods/source_inputs/scanner.rb +171 -0
  198. data/lib/woods/source_inputs/scopes.rb +71 -0
  199. data/lib/woods/source_inputs/session.rb +214 -0
  200. data/lib/woods/source_inputs/status.rb +84 -0
  201. data/lib/woods/source_inputs/verifier.rb +107 -0
  202. data/lib/woods/storage/metadata_store.rb +25 -25
  203. data/lib/woods/storage/pgvector.rb +35 -10
  204. data/lib/woods/storage/qdrant.rb +17 -7
  205. data/lib/woods/storage/vector_store.rb +18 -6
  206. data/lib/woods/tasks.rb +3 -2
  207. data/lib/woods/temporal/json_snapshot_store.rb +58 -9
  208. data/lib/woods/unblocked/exporter.rb +59 -70
  209. data/lib/woods/version.rb +1 -1
  210. data/lib/woods/watch/boot_snapshot.rb +52 -0
  211. data/lib/woods/watch/daemon.rb +154 -32
  212. data/lib/woods/watch/listen_watcher.rb +4 -0
  213. data/lib/woods/watch/polling_watcher.rb +5 -1
  214. data/lib/woods/watch/status.rb +20 -15
  215. data/lib/woods/watch/tree_scan.rb +21 -13
  216. data/lib/woods/watch/watcher.rb +4 -1
  217. data/lib/woods.rb +50 -11
  218. data/plugin/.claude-plugin/plugin.json +1 -1
  219. data/plugin/hooks/adapters/normalize.jq +15 -0
  220. data/plugin/hooks/adapters/normalize.rb +63 -0
  221. data/plugin/hooks/hooks.json +20 -0
  222. data/plugin/hooks/woods-context.sh +50 -0
  223. data/plugin/hooks/woods-input-rules.sh +159 -0
  224. data/plugin/hooks/woods-opencode.mjs +65 -0
  225. data/plugin/hooks/woods-post-edit.sh +2 -225
  226. data/plugin/hooks/woods-refresh.sh +260 -0
  227. data/plugin/hooks/woods-session-start.sh +47 -55
  228. data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
  229. data/plugin/skills/woods-diagnose/SKILL.md +319 -1
  230. data/plugin/skills/woods-investigate/SKILL.md +145 -0
  231. data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
  232. data/plugin/skills/woods-setup/SKILL.md +110 -6
  233. metadata +87 -5
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'woods/console/sql_noise_stripper'
4
+ require 'woods/console/sqlite_read_guard'
4
5
 
5
6
  # @see Woods
6
7
  module Woods
@@ -28,11 +29,11 @@ module Woods
28
29
  # named constants), not imperative logic.
29
30
  class SqlValidator # rubocop:disable Metrics/ClassLength
30
31
  # SQL dialects whose normalization the lock-clause check can run over.
31
- KNOWN_DIALECTS = %i[postgres mysql].freeze
32
+ KNOWN_DIALECTS = %i[postgres mysql sqlite].freeze
32
33
 
33
34
  # The dialect the statement will execute under, when the caller knows
34
35
  # it. `nil` (the default) keeps the conservative union: every check
35
- # runs against both dialect normalizations, which can reject a
36
+ # runs against all supported dialect normalizations, which can reject a
36
37
  # statement that is valid under one dialect's quote grammar (a MySQL
37
38
  # `\'` escape hides prose that the PostgreSQL view reads as SQL). When
38
39
  # the execution boundary knows the adapter, passing the matching
@@ -206,9 +207,9 @@ module Woods
206
207
  # function calls.
207
208
  FUNCTION_SCAN_EXCLUDED_KEYWORDS = %w[
208
209
  IN EXISTS NOT AND OR VALUES WHERE HAVING ON IS BETWEEN CASE WHEN
209
- THEN ELSE END FROM JOIN USING WITH SELECT DISTINCT ALL ANY SOME
210
- UNION INTERSECT EXCEPT ORDER GROUP BY ASC DESC LIMIT OFFSET AS INTO
211
- OVER PARTITION FILTER WITHIN RETURNING EXPLAIN
210
+ THEN ELSE FROM JOIN USING SELECT DISTINCT ALL ANY SOME
211
+ UNION INTERSECT EXCEPT ORDER GROUP BY LIMIT OFFSET AS INTO
212
+ OVER FILTER RETURNING EXPLAIN
212
213
  ].freeze
213
214
 
214
215
  # EXPLAIN is a statement leader, so `EXPLAIN (FORMAT JSON) SELECT` is an
@@ -318,6 +319,7 @@ module Woods
318
319
 
319
320
  return validate_dialect_variants!(sql) if unknown_grammar?
320
321
 
322
+ SqliteReadGuard.validate!(sql) if dialect == :sqlite
321
323
  normalized = sql.strip
322
324
 
323
325
  # Reject multiple statements (semicolons not inside string literals)
@@ -582,10 +584,9 @@ module Woods
582
584
  match = Regexp.last_match
583
585
  quoted = !(match[1] || match[2]).nil?
584
586
  identifier = match[1] || match[2] || match[3]
585
- # A bare keyword before `(` is grammar (IN/EXISTS/…), not a call; a
586
- # quoted name before `(` is always a call, so keyword exclusion
587
- # applies only to the bare form.
588
- next if !quoted && FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(identifier.upcase)
587
+ # Some engines also allow keyword spellings as function names.
588
+ # Only exempt those names in their supported grammar positions.
589
+ next if !quoted && function_keyword_grammar?(identifier.upcase, stripped[0...match.begin(0)])
589
590
  next if ALLOWED_FUNCTIONS.include?(identifier.downcase)
590
591
 
591
592
  raise SqlValidationError,
@@ -594,6 +595,45 @@ module Woods
594
595
  end
595
596
  end
596
597
 
598
+ # Recognize keyword grammar without exempting identically named calls.
599
+ # The prefix is already stripped of comments and literal contents.
600
+ #
601
+ # @param keyword [String] Uppercase bare identifier before `(`.
602
+ # @param prefix [String] SQL preceding the identifier.
603
+ # @return [Boolean]
604
+ def function_keyword_grammar?(keyword, prefix)
605
+ return false unless FUNCTION_SCAN_EXCLUDED_KEYWORDS.include?(keyword)
606
+
607
+ case keyword
608
+ when 'EXPLAIN' then prefix.strip.empty?
609
+ when 'BY' then prefix.match?(/\b(?:ORDER|GROUP|PARTITION)\s*\z/i)
610
+ when 'OVER', 'FILTER' then prefix.match?(/\)\s*\z/)
611
+ when 'ANY', 'SOME' then comparison_quantifier?(prefix)
612
+ when 'OFFSET' then offset_grammar?(prefix)
613
+ else true
614
+ end
615
+ end
616
+
617
+ # SQLite has no quantified ANY/SOME comparison grammar; both spellings
618
+ # can instead invoke application-defined functions after an operator.
619
+ def comparison_quantifier?(prefix)
620
+ return true if dialect == :postgres # Reserved quantifiers, including LIKE ANY.
621
+
622
+ dialect != :sqlite && prefix.match?(/[=<>]\s*\z/)
623
+ end
624
+
625
+ # PostgreSQL reserves bare OFFSET (it cannot name a function), so its
626
+ # standalone OFFSET clause remains supported. Other dialects only accept
627
+ # parenthesized offsets after a completed LIMIT operand. Calls
628
+ # at an expression's start or after an operator are never offset syntax.
629
+ # Keep the supported operand boundary conservative: literals, bind
630
+ # numbers, parenthesized expressions, or PostgreSQL's LIMIT ALL.
631
+ def offset_grammar?(prefix)
632
+ return true if dialect == :postgres
633
+
634
+ prefix.match?(/(?:[0-9)'"`]|\bLIMIT\s+ALL)\s*\z/i)
635
+ end
636
+
597
637
  # Check if the SQL contains a forbidden keyword at a statement-leader
598
638
  # position after stripping comments and string literals. This catches
599
639
  # comment-hidden injections like "SELECT 1 --;\nDELETE FROM users",
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'woods/console/sql_noise_stripper'
4
+ require 'woods/console/sql_table_scanner'
5
+
6
+ module Woods
7
+ module Console
8
+ # Refuse SQLite-specific syntax outside the scanner's supported grammar.
9
+ # This is a conservative read boundary, not a general-purpose SQL parser.
10
+ module SqliteReadGuard
11
+ SIMPLE_IDENTIFIER = /\A[A-Za-z_][A-Za-z0-9_]*\z/
12
+ UNSUPPORTED_SYNTAX = /[^\x00-\x7F]|\$|\[|''\s*\(|\b(?:FROM|JOIN)(?=['"`(])/i
13
+ QUOTED_IDENTIFIER = /"(?:[^"]|"")*"|`(?:[^`]|``)*`/
14
+
15
+ # @param sql [String] SQL to execute on SQLite
16
+ # @raise [SqlValidationError] when an identifier or table factor cannot be checked
17
+ # @return [void]
18
+ def self.validate!(sql)
19
+ view = SqlNoiseStripper.strip_noise(sql, dialect: :sqlite)
20
+ refuse! if view.match?(UNSUPPORTED_SYNTAX)
21
+ view.scan(QUOTED_IDENTIFIER) { |quoted| refuse! unless quoted[1...-1].match?(SIMPLE_IDENTIFIER) }
22
+ SqlTableScanner.relation_factors(view).each do |factor|
23
+ refuse! unless supported_factor?(factor.strip)
24
+ end
25
+ end
26
+
27
+ # A SELECT/WITH subquery has its own independently scanned table factors.
28
+ # Parenthesized table groups and string-quoted names are refused rather
29
+ # than silently omitted from the blocked-table scan.
30
+ def self.supported_factor?(factor)
31
+ return true if factor.match?(/\A\(\s*(?:SELECT|WITH)\b/i)
32
+
33
+ match = SqlTableScanner::LEAD_IDENT.match(factor)
34
+ match && factor[match.end(0)..].match?(/\A(?:\s|,|\)|\z)/)
35
+ end
36
+ private_class_method :supported_factor?
37
+
38
+ def self.refuse!
39
+ raise SqlValidationError,
40
+ 'Rejected: unsupported SQLite identifier or table-reference syntax. ' \
41
+ 'Use simple bare or double-quoted identifiers and SELECT subqueries.'
42
+ end
43
+ private_class_method :refuse!
44
+ end
45
+ end
46
+ end
@@ -30,8 +30,9 @@ module Woods
30
30
  # The transaction-guard filename for a lock of +name+, exposed so
31
31
  # cleanup code that empties a lock directory (woods:clean) can skip the
32
32
  # guard during its sweep — deleting a flock'd guard out from under a
33
- # contender's critical section would split the flock across two inodes —
34
- # and remove it only after the lock is released.
33
+ # contender's critical section would split the flock across two inodes.
34
+ # The guard stays after cleanup and lock release: a new contender may
35
+ # already be holding it before its lock file exists.
35
36
  #
36
37
  # @param name [String]
37
38
  # @return [String]
@@ -32,9 +32,14 @@ module Woods
32
32
  # enforce_dependencies: package units (Task 7)
33
33
  # commit_count, change_frequency: git enrichment (Task 5)
34
34
  NODE_ATTRIBUTE_KEYS = %i[
35
- database table foreign_key_tables package enforce_dependencies commit_count change_frequency
35
+ database table foreign_key_tables package enforce_dependencies commit_count change_frequency kind
36
36
  ].freeze
37
37
 
38
+ # These extractors describe a whole source file rather than one constant.
39
+ # Other types remain unclassified: a path-like identifier is not evidence
40
+ # that a unit is a profile, and an unmarked unit need not name a constant.
41
+ FILE_PROFILE_TYPES = %i[caching configuration test_mapping rails_source gem_source].freeze
42
+
38
43
  # Optional edge keys beyond target and via. `through` is the through
39
44
  # association name; `through_db` is the through model's resolved
40
45
  # database, emitted only when present; `disable_joins` is emitted only
@@ -80,14 +85,14 @@ module Woods
80
85
  }.merge(node_attributes_from(unit))
81
86
 
82
87
  (@edges[unit.identifier] ||= {})[unit.type] =
83
- unit.dependencies.map { |d| { target: d[:target], via: d[:via] }.merge(self.class.edge_attributes(d)) }
88
+ self.class.normalize_edges(unit.dependencies, strict: true)
84
89
  (@file_map[unit.file_path] ||= Set.new).add(unit.identifier) if unit.file_path
85
90
 
86
91
  # Type index for filtering (Set-based for O(1) insert)
87
92
  (@type_index[unit.type] ||= Set.new).add(unit.identifier)
88
93
 
89
94
  # Build reverse edges (Set-based for O(1) insert)
90
- unit.dependencies.each do |dep|
95
+ @edges[unit.identifier][unit.type].each do |dep|
91
96
  (@reverse[dep[:target]] ||= Set.new).add(unit.identifier)
92
97
  (@reverse_via[[dep[:target], dep[:via]]] ||= Set.new).add(unit.identifier)
93
98
  end
@@ -645,7 +650,7 @@ module Woods
645
650
  # @return [Hash{Symbol => Object}] normalized, nil-free
646
651
  def node_attributes_from(unit)
647
652
  metadata = unit.respond_to?(:metadata) && unit.metadata.is_a?(Hash) ? unit.metadata : {}
648
- attrs = {}
653
+ attrs = self.class.file_profile_attributes(unit.type)
649
654
  attrs[:database] = metadata[:database] unless metadata[:database].nil?
650
655
  attrs[:table] = metadata[:table_name] if unit.type == :model && !metadata[:table_name].nil?
651
656
  tables = Array(metadata[:foreign_keys]).filter_map { |fk| fk[:to_table] || fk['to_table'] if fk.is_a?(Hash) }
@@ -729,14 +734,15 @@ module Woods
729
734
  # `nodes` and `edges` stay keyed on the bare identifier, holding the
730
735
  # **primary** node — the type that sorts first. Identifiers registered
731
736
  # under more than one type put their remaining nodes in `variants`, a flat
732
- # array that is **omitted entirely when empty**. So the serialized form of
733
- # a graph with no collisions is byte-for-byte what it has always been, and
734
- # a `dependency_graph.json` written before this change loads through
737
+ # array that is **omitted entirely when empty**. A
738
+ # `dependency_graph.json` written before this change loads through
735
739
  # {.from_h} unmodified: it simply carries no `variants`.
736
740
  #
737
741
  # `reverse`, `file_map` and `type_index` need no new shape. They are
738
742
  # already identifier-valued sets, so a Scenic view `reports` and a factory
739
743
  # `reports` each contribute to their own type bucket and their own path.
744
+ # The additive `reverse_via` index preserves typed source identities and
745
+ # complete relationship records without changing the legacy `reverse` map.
740
746
  #
741
747
  # @return [Hash] Complete graph data
742
748
  def to_h
@@ -747,6 +753,7 @@ module Woods
747
753
  nodes: key_sorted(self.class.relativize_nodes(primary_nodes, root)),
748
754
  edges: key_sorted(primary_edges),
749
755
  reverse: key_sorted(@reverse.transform_values { |ids| ids.to_a.sort }),
756
+ reverse_via: published_reverse_via,
750
757
  file_map: key_sorted(self.class.relativize_file_map(@file_map, root)),
751
758
  type_index: key_sorted(@type_index.transform_values { |ids| ids.to_a.sort }),
752
759
  stats: {
@@ -770,6 +777,23 @@ module Woods
770
777
  end
771
778
  private :key_sorted
772
779
 
780
+ # Reverse buckets are derived from every typed forward registration, rather
781
+ # than the bare-name in-memory filter index, which cannot preserve variants
782
+ # or distinguish association attributes. Legacy nil relationships stay nil.
783
+ def published_reverse_via
784
+ buckets = {}
785
+ @edges.each do |source, by_type|
786
+ by_type.each do |type, edges|
787
+ edges.each do |edge|
788
+ record = { source: source, source_type: type }.merge(edge.except(:target))
789
+ (buckets[edge[:target]] ||= []) << record
790
+ end
791
+ end
792
+ end
793
+ key_sorted(buckets.transform_values { |records| records.sort_by { |record| JSON.generate(record) } })
794
+ end
795
+ private :published_reverse_via
796
+
773
797
  # A copy whose edge containers are the caller's alone.
774
798
  #
775
799
  # `dup` copies only the top level, so `to_h[:edges][id]` used to be the
@@ -783,6 +807,9 @@ module Woods
783
807
  def detached_snapshot(memo)
784
808
  snapshot = memo.dup
785
809
  snapshot[:edges] = memo[:edges].transform_values { |list| list.map(&:dup) }
810
+ snapshot[:reverse_via] = memo[:reverse_via].transform_values do |records|
811
+ records.map { |record| record.transform_values { |value| value.is_a?(String) ? value.dup : value } }
812
+ end
786
813
  if memo.key?(:variants)
787
814
  snapshot[:variants] = memo[:variants].map do |record|
788
815
  record.merge(edges: record[:edges].map(&:dup))
@@ -794,7 +821,12 @@ module Woods
794
821
 
795
822
  # @return [Hash{String => Hash}] identifier => primary node
796
823
  def primary_nodes
797
- @nodes.transform_values { |nodes| primary_of(nodes) }
824
+ # Late enrichment and JSON loading can insert optional fields in a
825
+ # different order. Canonicalize like variant_records so a graph's JSON
826
+ # fingerprint is stable across full and incremental publication.
827
+ @nodes.transform_values do |nodes|
828
+ primary_of(nodes).slice(:type, :file_path, :namespace, *NODE_ATTRIBUTE_KEYS)
829
+ end
798
830
  end
799
831
  private :primary_nodes
800
832
 
@@ -969,10 +1001,17 @@ module Woods
969
1001
  # @param node [Hash]
970
1002
  # @return [Hash{Symbol => Object}]
971
1003
  def self.persisted_node_attributes(node)
972
- NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
1004
+ attributes = NODE_ATTRIBUTE_KEYS.each_with_object({}) do |key, attrs|
973
1005
  value = node.key?(key) ? node[key] : node[key.to_s]
974
1006
  attrs[key] = normalize_node_attribute(key, value) unless value.nil?
975
1007
  end
1008
+ attributes.merge(file_profile_attributes(node[:type] || node['type']))
1009
+ end
1010
+
1011
+ # Backfill the additive marker when an older graph is republished, so
1012
+ # unchanged nodes in an incremental run agree with a fresh full graph.
1013
+ def self.file_profile_attributes(type)
1014
+ FILE_PROFILE_TYPES.include?(type&.to_sym) ? { kind: 'file_profile' } : {}
976
1015
  end
977
1016
 
978
1017
  # Edge attributes present in a dependency or persisted edge hash.
@@ -1032,7 +1071,10 @@ module Woods
1032
1071
  graph.instance_variable_set(:@edges, edges)
1033
1072
 
1034
1073
  raw_reverse = data[:reverse] || data['reverse'] || {}
1035
- graph.instance_variable_set(:@reverse, raw_reverse.transform_values { |v| v.is_a?(Set) ? v : Set.new(v) })
1074
+ reverse = raw_reverse.each_with_object({}) do |(target, sources), index|
1075
+ (index[target.to_s] ||= Set.new).merge(sources)
1076
+ end
1077
+ graph.instance_variable_set(:@reverse, reverse)
1036
1078
 
1037
1079
  raw_file_map = data[:file_map] || data['file_map'] || {}
1038
1080
  graph.instance_variable_set(:@file_map, absolutize_file_map(normalize_file_map(raw_file_map), root))
@@ -1126,7 +1168,12 @@ module Woods
1126
1168
  }.merge(persisted_node_attributes(node))
1127
1169
  end
1128
1170
 
1129
- # Normalize edge data from either old format (bare strings) or new format (hashes).
1171
+ # Normalize fresh dependencies and persisted edges to the same identity:
1172
+ # string targets and symbol relationship labels. Extractors may emit
1173
+ # symbolic external targets (such as :http_api); retaining those symbols
1174
+ # after a JSON-loaded baseline splits reverse indexes on re-registration.
1175
+ # Unit dependency metadata stays unchanged; only graph records normalize.
1176
+ # Also accepts the old bare-string edge format.
1130
1177
  #
1131
1178
  # ROUND-TRIP INVARIANT (do not break when refactoring):
1132
1179
  # DependencyGraph#to_h -> JSON.generate -> JSON.parse -> DependencyGraph.from_h
@@ -1147,15 +1194,20 @@ module Woods
1147
1194
  # explicit data conversion.
1148
1195
  #
1149
1196
  # @param edges [Array] Edge entries — either strings or hashes
1197
+ # @param strict [Boolean] require hashes for fresh units; legacy loads stay tolerant
1150
1198
  # @return [Array<Hash>] Normalized edges with :target and :via keys
1151
- def self.normalize_edges(edges)
1199
+ def self.normalize_edges(edges, strict: false)
1200
+ raise ArgumentError, 'Fresh graph dependencies must be an array' if strict && !edges.is_a?(Array)
1201
+
1152
1202
  return [] unless edges.is_a?(Array)
1153
1203
 
1154
1204
  edges.map do |edge|
1205
+ raise ArgumentError, 'Fresh graph dependencies must be hashes' if strict && !edge.is_a?(Hash)
1206
+
1155
1207
  if edge.is_a?(String)
1156
1208
  { target: edge, via: nil }
1157
1209
  elsif edge.is_a?(Hash)
1158
- { target: edge[:target] || edge['target'], via: (edge[:via] || edge['via'])&.to_sym }
1210
+ { target: (edge[:target] || edge['target'])&.to_s, via: (edge[:via] || edge['via'])&.to_sym }
1159
1211
  .merge(edge_attributes(edge))
1160
1212
  else
1161
1213
  { target: edge.to_s, via: nil }
@@ -0,0 +1,94 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+ require_relative '../atomic_file'
5
+ require_relative '../generation'
6
+
7
+ module Woods
8
+ class Error < StandardError; end unless defined?(Woods::Error)
9
+
10
+ module Embedding
11
+ # Collect the complete extraction input before an embedding run can mutate
12
+ # stores. Native publications use their authoritative listings, never a glob
13
+ # that could mistake a missing/corrupt payload for intentional deletion.
14
+ class Corpus
15
+ class Incomplete < Woods::Error; end
16
+
17
+ def initialize(output_dir)
18
+ @output_dir = output_dir.to_s
19
+ end
20
+
21
+ def load
22
+ native? ? published_units : legacy_units
23
+ rescue IOError, SystemCallError, JSON::ParserError, EncodingError, ArgumentError => e
24
+ raise Incomplete, "Embedding input incomplete: #{e.message}. Repair or rebuild extraction before embedding."
25
+ end
26
+
27
+ private
28
+
29
+ def native?
30
+ path = File.join(@output_dir, Generation::FILENAME)
31
+ if File.exist?(path)
32
+ marker = JSON.parse(AtomicFile.read(path))
33
+ raise IOError, 'invalid generation marker' unless marker.is_a?(Hash)
34
+ end
35
+
36
+ if !marker || marker['payload'].nil?
37
+ if File.directory?(File.join(@output_dir, 'payloads'))
38
+ raise IOError, 'missing publication pointer beside payloads'
39
+ end
40
+
41
+ return false
42
+ end
43
+
44
+ validate_marker(marker)
45
+ true
46
+ end
47
+
48
+ def validate_marker(marker)
49
+ strings_valid = %w[payload token].all? { |key| marker[key].is_a?(String) && !marker[key].empty? }
50
+ return if strings_valid && marker['number'].is_a?(Integer) && marker['number'].positive?
51
+
52
+ raise IOError, 'invalid published generation marker'
53
+ end
54
+
55
+ def published_units
56
+ require_relative '../mcp/index_reader'
57
+
58
+ reader = MCP::IndexReader.new(@output_dir)
59
+ reader.with_pinned_generation do
60
+ # Generation's compatibility fallback to the root is useful for
61
+ # readers, but cannot prove completeness for destructive reconciliation.
62
+ if reader.payload_dir.expand_path == Pathname.new(@output_dir).expand_path
63
+ raise IOError,
64
+ 'published payload missing or invalid'
65
+ end
66
+
67
+ validate_counts(reader.manifest)
68
+ reader.each_unit.to_a
69
+ end
70
+ end
71
+
72
+ def validate_counts(manifest)
73
+ counts = manifest.is_a?(Hash) && manifest['counts']
74
+ valid = counts.is_a?(Hash) && counts.all? do |dir, count|
75
+ MCP::IndexReader::TYPE_DIRS.include?(dir) && count.is_a?(Integer) && count >= 0
76
+ end
77
+ raise IOError, 'invalid published manifest counts' unless valid
78
+ end
79
+
80
+ # Pre-pointer indexes may use arbitrary filenames and have no manifest
81
+ # or authoritative listing. Keep that input contract, but never silently
82
+ # omit unreadable JSON before reconciling persisted identities.
83
+ def legacy_units
84
+ Dir.glob(File.join(@output_dir, '**', '*.json')).filter_map do |path|
85
+ relative = path.delete_prefix("#{@output_dir}/")
86
+ next if relative.start_with?('dumps/', 'payloads/') || File.basename(path) == 'checkpoint.json'
87
+
88
+ data = JSON.parse(AtomicFile.read(path))
89
+ data if data.is_a?(Hash) && data.key?('type') && data.key?('identifier')
90
+ end
91
+ end
92
+ end
93
+ end
94
+ end