woods 2.0.0.beta2 → 2.0.0.beta3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +262 -1
  3. data/CONTRIBUTING.md +173 -9
  4. data/README.md +7 -3
  5. data/SECURITY.md +9 -6
  6. data/docs/AGENT_GUIDE.md +83 -4
  7. data/docs/AGENT_SETUP.md +82 -1
  8. data/docs/BACKEND_MATRIX.md +20 -0
  9. data/docs/CLIENT_HOOKS.md +111 -0
  10. data/docs/CONFIGURATION_REFERENCE.md +199 -14
  11. data/docs/CONSOLE_MCP_SETUP.md +35 -5
  12. data/docs/DOCKER_SETUP.md +21 -2
  13. data/docs/EVALUATION.md +464 -1
  14. data/docs/EXTRACTOR_REFERENCE.md +36 -5
  15. data/docs/FAQ.md +11 -12
  16. data/docs/GETTING_STARTED.md +17 -5
  17. data/docs/INCREMENTAL_EXTRACTION.md +117 -1
  18. data/docs/INDEX_LAYOUT.md +382 -0
  19. data/docs/INTERNALS.md +7 -2
  20. data/docs/MCP_SERVERS.md +221 -5
  21. data/docs/MCP_TOOL_COOKBOOK.md +33 -18
  22. data/docs/NOTION_INTEGRATION.md +13 -0
  23. data/docs/OBSIDIAN_INTEGRATION.md +57 -9
  24. data/docs/PUBLISHED_INDEX.md +55 -0
  25. data/docs/README.md +7 -0
  26. data/docs/RETRIEVAL_GUIDE.md +253 -11
  27. data/docs/RUNTIME_TRACING.md +71 -0
  28. data/docs/SOURCE_FRESHNESS.md +143 -0
  29. data/docs/TROUBLESHOOTING.md +117 -5
  30. data/docs/UNBLOCKED_INTEGRATION.md +25 -0
  31. data/docs/UPGRADING_TO_2.md +44 -22
  32. data/docs/WATCH_DAEMON.md +259 -59
  33. data/exe/woods-agent-config +6 -0
  34. data/exe/woods-extract +5 -0
  35. data/exe/woods-hook-context +6 -0
  36. data/lib/generators/woods/templates/woods.rb.tt +1 -3
  37. data/lib/tasks/woods.rake +47 -397
  38. data/lib/woods/agent_configuration/applier.rb +133 -0
  39. data/lib/woods/agent_configuration/cli.rb +101 -0
  40. data/lib/woods/agent_configuration/cli_options.rb +29 -0
  41. data/lib/woods/agent_configuration/document.rb +105 -0
  42. data/lib/woods/agent_configuration/error.rb +7 -0
  43. data/lib/woods/agent_configuration/launcher.rb +75 -0
  44. data/lib/woods/agent_configuration/layout.rb +59 -0
  45. data/lib/woods/agent_configuration/managed_section.rb +62 -0
  46. data/lib/woods/agent_configuration/plan.rb +98 -0
  47. data/lib/woods/agent_configuration/plan_diff.rb +38 -0
  48. data/lib/woods/agent_configuration/planned_files.rb +61 -0
  49. data/lib/woods/agent_configuration/planner.rb +63 -0
  50. data/lib/woods/agent_configuration/planner_validation.rb +77 -0
  51. data/lib/woods/agent_configuration/preflight.rb +100 -0
  52. data/lib/woods/agent_configuration/recovery.rb +49 -0
  53. data/lib/woods/ast/node.rb +2 -0
  54. data/lib/woods/ast/parser.rb +38 -5
  55. data/lib/woods/builder.rb +21 -5
  56. data/lib/woods/cache/cache_middleware.rb +28 -7
  57. data/lib/woods/cache/cache_store.rb +4 -5
  58. data/lib/woods/change_set.rb +5 -4
  59. data/lib/woods/console/credential_index.rb +20 -2
  60. data/lib/woods/console/credential_scanner.rb +14 -14
  61. data/lib/woods/console/credential_scanner_registry.rb +36 -0
  62. data/lib/woods/console/embedded_executor.rb +1 -1
  63. data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
  64. data/lib/woods/console/rack_middleware.rb +22 -13
  65. data/lib/woods/console/server.rb +18 -16
  66. data/lib/woods/dependency_graph.rb +65 -13
  67. data/lib/woods/embedding/corpus.rb +94 -0
  68. data/lib/woods/embedding/indexer.rb +90 -46
  69. data/lib/woods/embedding/openai.rb +17 -6
  70. data/lib/woods/evaluation/ablation_executor.rb +6 -1
  71. data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
  72. data/lib/woods/export/typed_reader.rb +56 -0
  73. data/lib/woods/extractor.rb +232 -137
  74. data/lib/woods/extractors/action_cable_extractor.rb +3 -1
  75. data/lib/woods/extractors/behavioral_profile.rb +9 -7
  76. data/lib/woods/extractors/caching_extractor.rb +3 -1
  77. data/lib/woods/extractors/concern_extractor.rb +64 -6
  78. data/lib/woods/extractors/configuration_extractor.rb +7 -3
  79. data/lib/woods/extractors/controller_extractor.rb +13 -4
  80. data/lib/woods/extractors/database_view_extractor.rb +3 -1
  81. data/lib/woods/extractors/decorator_extractor.rb +3 -1
  82. data/lib/woods/extractors/engine_extractor.rb +3 -1
  83. data/lib/woods/extractors/event_extractor.rb +4 -2
  84. data/lib/woods/extractors/factory_extractor.rb +3 -1
  85. data/lib/woods/extractors/graphql_extractor.rb +8 -2
  86. data/lib/woods/extractors/i18n_extractor.rb +3 -1
  87. data/lib/woods/extractors/job_extractor.rb +6 -19
  88. data/lib/woods/extractors/lib_extractor.rb +3 -1
  89. data/lib/woods/extractors/mailer_extractor.rb +20 -5
  90. data/lib/woods/extractors/manager_extractor.rb +3 -1
  91. data/lib/woods/extractors/method_parameters.rb +53 -0
  92. data/lib/woods/extractors/middleware_argument.rb +65 -0
  93. data/lib/woods/extractors/middleware_extractor.rb +9 -3
  94. data/lib/woods/extractors/migration_extractor.rb +3 -1
  95. data/lib/woods/extractors/model_extractor.rb +39 -33
  96. data/lib/woods/extractors/package_extractor.rb +24 -4
  97. data/lib/woods/extractors/phlex_extractor.rb +3 -1
  98. data/lib/woods/extractors/policy_extractor.rb +3 -1
  99. data/lib/woods/extractors/poro_extractor.rb +3 -1
  100. data/lib/woods/extractors/pundit_extractor.rb +3 -1
  101. data/lib/woods/extractors/rails_source_extractor.rb +4 -2
  102. data/lib/woods/extractors/rake_task_extractor.rb +4 -2
  103. data/lib/woods/extractors/route_extractor.rb +3 -1
  104. data/lib/woods/extractors/route_helper_resolver.rb +10 -33
  105. data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
  106. data/lib/woods/extractors/serializer_extractor.rb +4 -2
  107. data/lib/woods/extractors/service_extractor.rb +3 -1
  108. data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
  109. data/lib/woods/extractors/shared_utility_methods.rb +27 -15
  110. data/lib/woods/extractors/source_nesting.rb +1 -1
  111. data/lib/woods/extractors/state_machine_extractor.rb +3 -1
  112. data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
  113. data/lib/woods/extractors/validator_extractor.rb +3 -1
  114. data/lib/woods/extractors/view_component_extractor.rb +3 -1
  115. data/lib/woods/extractors/view_template_extractor.rb +3 -1
  116. data/lib/woods/gem_mapper.rb +2 -0
  117. data/lib/woods/git_history.rb +116 -0
  118. data/lib/woods/graph_analyzer.rb +35 -6
  119. data/lib/woods/hooks/context_cli.rb +54 -0
  120. data/lib/woods/hooks/context_event.rb +88 -0
  121. data/lib/woods/hooks/context_hint.rb +73 -0
  122. data/lib/woods/hooks/context_impact.rb +77 -0
  123. data/lib/woods/hooks/context_output.rb +47 -0
  124. data/lib/woods/hooks/context_state.rb +102 -0
  125. data/lib/woods/hooks/refresh.rb +79 -0
  126. data/lib/woods/hooks/rule_projection.rb +78 -0
  127. data/lib/woods/input_rules.rb +19 -0
  128. data/lib/woods/mcp/bearer_auth.rb +20 -12
  129. data/lib/woods/mcp/bootstrapper.rb +62 -0
  130. data/lib/woods/mcp/index_reader.rb +323 -160
  131. data/lib/woods/mcp/initialization_guidance.rb +27 -0
  132. data/lib/woods/mcp/origin_guard.rb +17 -9
  133. data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
  134. data/lib/woods/mcp/renderers/markdown_renderer.rb +8 -1
  135. data/lib/woods/mcp/renderers/plain_renderer.rb +7 -1
  136. data/lib/woods/mcp/search_results.rb +74 -0
  137. data/lib/woods/mcp/server.rb +158 -37
  138. data/lib/woods/mcp/tool_contract.rb +2 -0
  139. data/lib/woods/mcp/tool_response_renderer.rb +25 -0
  140. data/lib/woods/mcp/traversal_evidence.rb +113 -0
  141. data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
  142. data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
  143. data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
  144. data/lib/woods/notion/exporter.rb +56 -17
  145. data/lib/woods/obsidian/destination_plan.rb +98 -0
  146. data/lib/woods/obsidian/name_mapper.rb +19 -3
  147. data/lib/woods/obsidian/note_builder.rb +19 -10
  148. data/lib/woods/obsidian/vault_exporter.rb +88 -32
  149. data/lib/woods/operator/pipeline_guard.rb +18 -13
  150. data/lib/woods/path_dispatcher.rb +7 -1
  151. data/lib/woods/payload_store.rb +27 -26
  152. data/lib/woods/railtie.rb +3 -3
  153. data/lib/woods/railtie_support.rb +12 -12
  154. data/lib/woods/rake_helpers.rb +392 -0
  155. data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
  156. data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
  157. data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
  158. data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
  159. data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
  160. data/lib/woods/resilience/index_validator.rb +112 -23
  161. data/lib/woods/retrieval/context_assembler.rb +50 -15
  162. data/lib/woods/retrieval/lexical_assembler.rb +73 -0
  163. data/lib/woods/retrieval/lexical_index.rb +119 -0
  164. data/lib/woods/retrieval/ranker.rb +4 -2
  165. data/lib/woods/retrieval/scope.rb +108 -0
  166. data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
  167. data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
  168. data/lib/woods/retrieval/search_executor.rb +86 -27
  169. data/lib/woods/retrieval/source_evidence.rb +200 -0
  170. data/lib/woods/retriever.rb +98 -22
  171. data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
  172. data/lib/woods/session_tracer/middleware.rb +10 -12
  173. data/lib/woods/session_tracer/redis_store.rb +22 -6
  174. data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
  175. data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
  176. data/lib/woods/session_tracer/unit_resolver.rb +63 -0
  177. data/lib/woods/source_inputs/consumer_errors.rb +27 -0
  178. data/lib/woods/source_inputs/handoff.rb +102 -0
  179. data/lib/woods/source_inputs/launcher.rb +157 -0
  180. data/lib/woods/source_inputs/manifest.rb +124 -0
  181. data/lib/woods/source_inputs/private_key.rb +55 -0
  182. data/lib/woods/source_inputs/scanner.rb +171 -0
  183. data/lib/woods/source_inputs/scopes.rb +71 -0
  184. data/lib/woods/source_inputs/session.rb +214 -0
  185. data/lib/woods/source_inputs/status.rb +84 -0
  186. data/lib/woods/source_inputs/verifier.rb +107 -0
  187. data/lib/woods/storage/metadata_store.rb +25 -25
  188. data/lib/woods/storage/pgvector.rb +29 -8
  189. data/lib/woods/storage/qdrant.rb +17 -7
  190. data/lib/woods/storage/vector_store.rb +18 -6
  191. data/lib/woods/tasks.rb +3 -2
  192. data/lib/woods/temporal/json_snapshot_store.rb +29 -8
  193. data/lib/woods/unblocked/exporter.rb +59 -70
  194. data/lib/woods/version.rb +1 -1
  195. data/lib/woods/watch/boot_snapshot.rb +52 -0
  196. data/lib/woods/watch/daemon.rb +136 -28
  197. data/lib/woods/watch/listen_watcher.rb +4 -0
  198. data/lib/woods/watch/polling_watcher.rb +5 -1
  199. data/lib/woods/watch/status.rb +20 -15
  200. data/lib/woods/watch/tree_scan.rb +21 -13
  201. data/lib/woods/watch/watcher.rb +4 -1
  202. data/lib/woods.rb +50 -11
  203. data/plugin/.claude-plugin/plugin.json +1 -1
  204. data/plugin/hooks/adapters/normalize.jq +15 -0
  205. data/plugin/hooks/adapters/normalize.rb +63 -0
  206. data/plugin/hooks/hooks.json +20 -0
  207. data/plugin/hooks/woods-context.sh +50 -0
  208. data/plugin/hooks/woods-input-rules.sh +159 -0
  209. data/plugin/hooks/woods-opencode.mjs +65 -0
  210. data/plugin/hooks/woods-post-edit.sh +2 -225
  211. data/plugin/hooks/woods-refresh.sh +260 -0
  212. data/plugin/hooks/woods-session-start.sh +47 -55
  213. data/plugin/skills/woods-agent-enable/SKILL.md +13 -0
  214. data/plugin/skills/woods-diagnose/SKILL.md +288 -1
  215. data/plugin/skills/woods-investigate/SKILL.md +106 -0
  216. data/plugin/skills/woods-mcp-config/SKILL.md +89 -1
  217. data/plugin/skills/woods-setup/SKILL.md +107 -6
  218. metadata +84 -5
@@ -0,0 +1,100 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'set'
4
+
5
+ module Woods
6
+ module MCP
7
+ # A generation-scoped view retaining typed edge ownership. Only node/variant
8
+ # indexes are prepared eagerly; relationship arrays remain lazy and every
9
+ # examined relationship consumes the caller's work budget.
10
+ class TraversalEvidenceIndex
11
+ ATTRIBUTES = %w[via through through_db disable_joins].freeze
12
+
13
+ def initialize(graph)
14
+ @graph = graph
15
+ @nodes = graph.fetch('nodes', {})
16
+ @variants = Array(graph['variants']).group_by { |record| record['identifier'] }
17
+ @types = @nodes.transform_values { |node| [node['type']].compact }
18
+ @variants.each do |identifier, records|
19
+ @types[identifier] = ((@types[identifier] || []) + records.map { |record| record['type'] }).compact.uniq.sort
20
+ end
21
+ @types.transform_values!(&:freeze).freeze
22
+ @multi_database = @nodes.values.filter_map { |node| node['database'] }.uniq.size > 1
23
+ end
24
+
25
+ def include?(identifier)
26
+ @nodes.key?(identifier)
27
+ end
28
+
29
+ def types(identifier)
30
+ @types.fetch(identifier, [])
31
+ end
32
+
33
+ def identity(identifier)
34
+ candidates = types(identifier)
35
+ return { identifier: identifier, type: candidates.first } if candidates.size == 1
36
+
37
+ { identifier: identifier, type: nil, candidate_types: candidates,
38
+ resolution: candidates.empty? ? 'unresolved' : 'ambiguous' }
39
+ end
40
+
41
+ def node(identifier, depth)
42
+ metadata = @nodes[identifier]
43
+ result = { type: metadata&.dig('type'), depth: depth, deps: [] }
44
+ result[:types] = types(identifier) if types(identifier).size > 1
45
+ result[:database] = metadata&.dig('database') if @multi_database
46
+ result
47
+ end
48
+
49
+ def each_edge(identifier, direction, budget, &block)
50
+ if direction == :forward
51
+ each_forward(identifier, budget, &block)
52
+ elsif @graph.key?('reverse_via')
53
+ each_reverse_record(identifier, budget, &block)
54
+ else
55
+ each_legacy_reverse(identifier, budget, &block)
56
+ end
57
+ end
58
+
59
+ private
60
+
61
+ def each_forward(identifier, budget, &block)
62
+ primary = @nodes[identifier]
63
+ each_owned_edge(identifier, primary && primary['type'], @graph.fetch('edges', {})[identifier], budget, &block)
64
+ (@variants[identifier] || []).each do |record|
65
+ each_owned_edge(identifier, record['type'], record['edges'], budget, &block)
66
+ end
67
+ end
68
+
69
+ def each_owned_edge(identifier, type, edges, budget)
70
+ (edges || []).each do |stored|
71
+ budget.consume_edge
72
+ stored = { 'target' => stored } unless stored.is_a?(Hash)
73
+ yield record(identifier, type, stored.fetch('target'), stored)
74
+ end
75
+ end
76
+
77
+ def each_reverse_record(identifier, budget)
78
+ (@graph.fetch('reverse_via')[identifier] || []).each do |stored|
79
+ budget.consume_edge
80
+ yield record(stored.fetch('source'), stored.fetch('source_type'), identifier, stored)
81
+ end
82
+ end
83
+
84
+ def each_legacy_reverse(identifier, budget)
85
+ (@graph.fetch('reverse', {})[identifier] || []).each do |source|
86
+ budget.consume_edge
87
+ each_forward(source, budget) do |edge|
88
+ yield edge if edge[:target][:identifier] == identifier
89
+ end
90
+ end
91
+ end
92
+
93
+ def record(source, type, target, stored)
94
+ result = { source: { identifier: source, type: type }, target: identity(target) }
95
+ ATTRIBUTES.each { |attribute| result[attribute.to_sym] = stored[attribute] }
96
+ result
97
+ end
98
+ end
99
+ end
100
+ end
@@ -0,0 +1,41 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'set'
4
+
5
+ module Woods
6
+ module MCP
7
+ # Retain the paged rows' ancestors once, together with their witness edges
8
+ # and other recorded relationships between visible/context endpoints.
9
+ module TraversalEvidencePage
10
+ def self.apply(result)
11
+ explanation = result[:explanation]
12
+ return result unless explanation
13
+
14
+ visible = result.fetch(:nodes).keys.to_set
15
+ witnesses = explanation.fetch(:witnesses)
16
+ needed = ancestors(visible, witnesses)
17
+ page_witnesses = witnesses.slice(*(witnesses.keys & needed.to_a))
18
+ witness_edges = page_witnesses.values.map { |witness| witness[:edge_id] }.compact.to_set
19
+ edges = explanation.fetch(:edges).select do |id, edge|
20
+ source = edge[:source][:identifier]
21
+ target = edge[:target][:identifier]
22
+ witness_edges.include?(id) ||
23
+ (needed.include?(source) && needed.include?(target) && (visible.include?(source) || visible.include?(target)))
24
+ end
25
+ result[:explanation] = explanation.merge(edges: edges, witnesses: page_witnesses.to_h do |identifier, witness|
26
+ [identifier, witness.merge(context: !visible.include?(identifier))]
27
+ end)
28
+ result
29
+ end
30
+
31
+ def self.ancestors(visible, witnesses)
32
+ needed = Set.new
33
+ visible.each do |identifier|
34
+ identifier = witnesses.fetch(identifier)[:parent] while identifier && needed.add?(identifier)
35
+ end
36
+ needed
37
+ end
38
+ private_class_method :ancestors
39
+ end
40
+ end
41
+ end
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Woods
4
+ module MCP
5
+ # Equivalent evidence for all text renderers; JSON retains the same records.
6
+ module TraversalEvidenceText
7
+ def self.lines(explanation)
8
+ return [] unless explanation
9
+
10
+ lines = ['', "Recorded relationships (original source -> target; traversal: #{value(explanation, :direction)}):",
11
+ "Root identity: #{identity(value(explanation, :root))}"]
12
+ value(explanation, :edges).each do |id, edge|
13
+ attributes = TraversalEvidenceIndex::ATTRIBUTES.map do |attribute|
14
+ stored = value(edge, attribute)
15
+ "#{attribute}=#{stored.nil? ? 'unknown' : stored}"
16
+ end
17
+ lines << "#{id}: #{identity(value(edge, :source))} -> #{identity(value(edge, :target))}; #{attributes.join('; ')}"
18
+ end
19
+ lines << 'Witnesses: direct = recorded root relationship; transitive = inferred reachability, not observed execution.'
20
+ value(explanation, :witnesses).each do |identifier, witness|
21
+ lines << witness_line(identifier, witness)
22
+ end
23
+ lines
24
+ end
25
+
26
+ def self.identity(endpoint)
27
+ type = value(endpoint, :type)
28
+ label = if type
29
+ type
30
+ else
31
+ candidates = value(endpoint, :candidate_types) || []
32
+ "#{value(endpoint, :resolution)}; candidate types: #{candidates.empty? ? 'none' : candidates.join(', ')}"
33
+ end
34
+ "#{value(endpoint, :identifier)} (#{label})"
35
+ end
36
+
37
+ def self.witness_line(identifier, witness)
38
+ parent = value(witness, :parent) || 'none'
39
+ edge = value(witness, :edge_id) || 'none'
40
+ complete = value(witness, :typed_path_complete) ? 'yes' : 'no'
41
+ context = value(witness, :context) ? 'yes' : 'no'
42
+ "#{identifier}: #{value(witness, :impact)}; parent=#{parent}; edge=#{edge}; " \
43
+ "typed path complete=#{complete}; context only=#{context}"
44
+ end
45
+
46
+ def self.value(hash, key)
47
+ hash.key?(key.to_sym) ? hash[key.to_sym] : hash[key.to_s]
48
+ end
49
+ private_class_method :identity, :witness_line, :value
50
+ end
51
+ end
52
+ end
@@ -8,6 +8,7 @@ require_relative 'mappers/column_mapper'
8
8
  require_relative 'mappers/migration_mapper'
9
9
  require_relative 'rate_limiter'
10
10
  require_relative 'sync_manifest'
11
+ require_relative '../export/typed_reader'
11
12
 
12
13
  module Woods
13
14
  module Notion
@@ -83,6 +84,7 @@ module Woods
83
84
  @database_ids = config.notion_database_ids || {}
84
85
  @client = client || Client.new(api_token: api_token)
85
86
  @reader = reader || build_reader(index_dir)
87
+ @typed_reader = Export::TypedReader.new(@reader)
86
88
  @manifest = manifest || build_manifest(index_dir)
87
89
  @force_full = force_full.nil? ? env_force? : force_full
88
90
  @page_id_cache = {}
@@ -94,7 +96,7 @@ module Woods
94
96
  # @return [Hash] { data_models: Integer, columns: Integer,
95
97
  # skipped: Integer, errors: Array<String> }
96
98
  def sync_all
97
- with_pinned_index do
99
+ with_prepared_index do
98
100
  model_stats = @database_ids[:data_models] ? sync_data_models : empty_stats
99
101
  warn_columns_without_data_models if @database_ids[:columns] && !@database_ids[:data_models]
100
102
  column_stats = @database_ids[:columns] ? sync_columns : empty_stats
@@ -124,12 +126,20 @@ module Woods
124
126
  #
125
127
  # @return [Hash] { synced: Integer, skipped: Integer, errors: Array<String> }
126
128
  def sync_data_models
129
+ return empty_stats unless @database_ids[:data_models]
130
+
131
+ with_prepared_index(%w[model migration]) { sync_data_models_prepared }
132
+ end
133
+
134
+ def sync_data_models_prepared
127
135
  database_id = @database_ids[:data_models]
128
136
  return empty_stats unless database_id
129
137
 
138
+ prepared = false
130
139
  begin
131
140
  migration_dates = load_migration_dates
132
141
  shared_tables = shared_table_names
142
+ prepared = true
133
143
  stats = sync_units('model', database_id, 'Table Name', SCOPE_DATA_MODELS) do |unit_data|
134
144
  properties = Mappers::ModelMapper.new.map(unit_data)
135
145
  # Enrichment reads the bare table name from the title — run it
@@ -141,9 +151,10 @@ module Woods
141
151
  @manifest.prune(SCOPE_DATA_MODELS, stats.delete(:current_keys))
142
152
  stats
143
153
  ensure
144
- save_manifest
154
+ save_manifest if prepared
145
155
  end
146
156
  end
157
+ private :sync_data_models_prepared
147
158
 
148
159
  # Sync column data to the Columns Notion database.
149
160
  #
@@ -160,14 +171,23 @@ module Woods
160
171
  #
161
172
  # @return [Hash] { synced: Integer, skipped: Integer, errors: Array<String> }
162
173
  def sync_columns
174
+ return empty_stats unless @database_ids[:columns]
175
+
176
+ with_prepared_index(['model']) { sync_columns_prepared }
177
+ end
178
+
179
+ def sync_columns_prepared
163
180
  database_id = @database_ids[:columns]
164
181
  return empty_stats unless database_id
165
182
 
183
+ prepared = false
166
184
  begin
167
185
  totals = { synced: 0, skipped: 0, errors: [] }
168
186
  current_keys = []
187
+ groups = column_groups
188
+ prepared = true
169
189
 
170
- column_groups.each do |group|
190
+ groups.each do |group|
171
191
  result = sync_table_columns(group, database_id, current_keys)
172
192
  totals[:synced] += result[:synced]
173
193
  totals[:skipped] += result[:skipped]
@@ -177,12 +197,40 @@ module Woods
177
197
  @manifest.prune(SCOPE_COLUMNS, current_keys)
178
198
  totals
179
199
  ensure
180
- save_manifest
200
+ save_manifest if prepared
181
201
  end
182
202
  end
203
+ private :sync_columns_prepared
183
204
 
184
205
  private
185
206
 
207
+ # Reuse one validated snapshot through preprocessing and remote mapping.
208
+ # Standalone public sync methods have the same pin as sync_all.
209
+ def with_prepared_index(types = nil)
210
+ return yield if @published_units
211
+
212
+ with_pinned_index do
213
+ types ||= configured_types
214
+ @published_units = types.empty? ? [] : @typed_reader.all(only: types)
215
+ begin
216
+ yield
217
+ ensure
218
+ @published_units = nil
219
+ end
220
+ end
221
+ end
222
+
223
+ def units_for(type)
224
+ @published_units.select { |unit| unit['type'] == type }
225
+ end
226
+
227
+ def configured_types
228
+ types = []
229
+ types << 'model' if @database_ids[:data_models] || @database_ids[:columns]
230
+ types << 'migration' if @database_ids[:data_models]
231
+ types
232
+ end
233
+
186
234
  # Sync all units of a type, yielding each for property mapping.
187
235
  #
188
236
  # @param type [String] Unit type to list
@@ -196,9 +244,8 @@ module Woods
196
244
  def sync_units(type, database_id, title_property, scope)
197
245
  stats = { synced: 0, skipped: 0, errors: [], current_keys: [] }
198
246
 
199
- @reader.list_units(type: type).each do |entry|
200
- unit_data = @reader.find_unit(entry['identifier'])
201
- next unless unit_data
247
+ units_for(type).each do |unit_data|
248
+ entry = unit_data
202
249
 
203
250
  begin
204
251
  properties, legacy_title = yield(unit_data)
@@ -227,12 +274,7 @@ module Woods
227
274
  #
228
275
  # @yield [Hash, Hash] Index entry and full unit data
229
276
  def each_model_unit
230
- @reader.list_units(type: 'model').each do |entry|
231
- unit_data = @reader.find_unit(entry['identifier'])
232
- next unless unit_data
233
-
234
- yield(entry, unit_data)
235
- end
277
+ units_for('model').each { |unit_data| yield(unit_data, unit_data) }
236
278
  end
237
279
 
238
280
  # Run a multi-read export body against one index generation.
@@ -436,10 +478,7 @@ module Woods
436
478
  # @return [Hash<String, String>] { table_name => latest_date }
437
479
  def load_migration_dates
438
480
  mapper = Mappers::MigrationMapper.new
439
- units = @reader.list_units(type: 'migration').filter_map { |e| @reader.find_unit(e['identifier']) }
440
- mapper.latest_changes(units)
441
- rescue StandardError
442
- {}
481
+ mapper.latest_changes(units_for('migration'))
443
482
  end
444
483
 
445
484
  # Upsert a Notion page: find by title, update if exists, create if not.
@@ -0,0 +1,98 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'digest'
4
+ require 'json'
5
+ require_relative 'errors'
6
+
7
+ module Woods
8
+ module Obsidian
9
+ # Preflight every output before changing the vault. The receipt deliberately
10
+ # covers only the fixed machine assets; Markdown retains its public marker.
11
+ # This is not a transaction against concurrent external editors.
12
+ class DestinationPlan
13
+ RECEIPT = '_woods/ownership.json'
14
+ ASSETS = %w[_woods/manifest.json _woods/dependency_graph.json _woods/graph_analysis.json
15
+ .obsidian/app.json .obsidian/types.json .obsidian/graph.json Units.base].freeze
16
+ MAX_RECEIPT_BYTES = 4096
17
+
18
+ def initialize(vault, files, managed_note:)
19
+ @vault = vault
20
+ @files = files
21
+ @managed_note = managed_note
22
+ end
23
+
24
+ def commit
25
+ previous = read_receipt
26
+ files = @files.dup
27
+ digests = files.filter_map do |path, content|
28
+ rel = relative(path)
29
+ [rel, Digest::SHA256.hexdigest(content)] if ASSETS.include?(rel)
30
+ end.to_h
31
+ digests = previous.merge(digests)
32
+ files[@vault.join(RECEIPT)] = "#{JSON.pretty_generate('schema_version' => 1, 'files' => digests.sort.to_h)}\n"
33
+ snapshots = files.to_h do |path, content|
34
+ current = existing_bytes(path)
35
+ rel = relative(path)
36
+ permitted = current.nil? || current == content.b ||
37
+ (path.extname == '.md' && @managed_note.call(path)) ||
38
+ (ASSETS.include?(rel) && previous[rel] == Digest::SHA256.hexdigest(current)) ||
39
+ rel == RECEIPT
40
+ refuse(rel) unless permitted
41
+ [path, current]
42
+ end
43
+ files.each do |path, content|
44
+ refuse(relative(path), 'changed after preflight') unless existing_bytes(path) == snapshots[path]
45
+ AtomicFile.write(path, content)
46
+ end
47
+ end
48
+
49
+ private
50
+
51
+ def read_receipt
52
+ bytes = existing_bytes(@vault.join(RECEIPT))
53
+ return {} unless bytes
54
+
55
+ refuse(RECEIPT, 'invalid ownership receipt') if bytes.bytesize > MAX_RECEIPT_BYTES
56
+ data = JSON.parse(bytes)
57
+ valid = data.is_a?(Hash) && data.keys.sort == %w[files schema_version] &&
58
+ data['schema_version'] == 1 && data['files'].is_a?(Hash) &&
59
+ data['files'].all? do |path, digest|
60
+ ASSETS.include?(path) && digest.is_a?(String) && digest.match?(/\A[0-9a-f]{64}\z/)
61
+ end
62
+ refuse(RECEIPT, 'invalid ownership receipt') unless valid
63
+ data['files']
64
+ rescue JSON::ParserError
65
+ refuse(RECEIPT, 'invalid ownership receipt')
66
+ end
67
+
68
+ # Reject child symlinks, directories and special files before any read,
69
+ # including a FIFO at the receipt path. The vault root itself can be a
70
+ # symlink, as can ancestors such as macOS /tmp.
71
+ def existing_bytes(path)
72
+ rel = relative(path)
73
+ cursor = @vault
74
+ parts = Pathname.new(rel).each_filename.to_a
75
+ parts.each_with_index do |part, index|
76
+ cursor = cursor.join(part)
77
+ next unless cursor.exist? || cursor.symlink?
78
+
79
+ refuse(rel, 'symlink destination') if cursor.symlink?
80
+ expected = index == parts.size - 1 ? cursor.file? : cursor.directory?
81
+ refuse(rel, 'not a regular destination') unless expected
82
+ end
83
+ path.exist? ? File.binread(path) : nil
84
+ end
85
+
86
+ def relative(path)
87
+ rel = path.relative_path_from(@vault).to_s
88
+ refuse(rel, 'outside the vault') if Pathname.new(rel).each_filename.include?('..')
89
+ rel
90
+ end
91
+
92
+ def refuse(path, reason = 'unmanaged or modified destination')
93
+ raise ExportError, "refusing #{path}: #{reason}; no sweep performed. " \
94
+ 'Inspect and back up the conflicting file, then move it aside or export to a new directory.'
95
+ end
96
+ end
97
+ end
98
+ end
@@ -42,7 +42,13 @@ module Woods
42
42
 
43
43
  # @param id_to_dir [Hash{String=>String}] map of unit identifier -> type
44
44
  # folder name (e.g. "User" => "models")
45
- def initialize(id_to_dir)
45
+ # @param options [Hash] optional internal-key display identifiers and the
46
+ # complete graph's ambiguous identifier set. Positional hash support is
47
+ # retained for existing NameMapper.new('User' => 'models') callers.
48
+ def initialize(id_to_dir, options = {})
49
+ @identifiers = options.fetch(:identifiers, {})
50
+ @ambiguous_identifiers = options.fetch(:ambiguous_identifiers, Set.new)
51
+ @reference_keys = id_to_dir.keys.to_h { |key| [@identifiers.fetch(key, key), key] }
46
52
  @map = {}
47
53
  @paths = {}
48
54
  build(id_to_dir)
@@ -63,6 +69,14 @@ module Woods
63
69
  "[[#{entry[:target]}|#{entry[:alias]}]]"
64
70
  end
65
71
 
72
+ # Resolve a human metadata reference only when its identifier is unambiguous
73
+ # in the complete graph, including units excluded from this export.
74
+ def wikilink_for_identifier(identifier)
75
+ return nil if @ambiguous_identifiers.include?(identifier)
76
+
77
+ wikilink(@reference_keys[identifier])
78
+ end
79
+
66
80
  # @return [Hash{String=>String}] inverse path -> id map (sorted)
67
81
  def paths_to_ids
68
82
  @paths
@@ -74,9 +88,11 @@ module Woods
74
88
  taken = Hash.new { |h, dir| h[dir] = Set.new(RESERVED_BASENAMES) }
75
89
  id_to_dir.keys.sort.each do |id|
76
90
  dir = id_to_dir[id]
77
- basename = assign_basename(id, taken[dir])
91
+ identifier = @identifiers.fetch(id, id)
92
+ basename = assign_basename(identifier, taken[dir])
78
93
  path = "#{dir}/#{basename}#{EXTENSION}"
79
- @map[id] = { dir: dir, basename: basename, path: path, target: "#{dir}/#{basename}", alias: alias_for(id) }
94
+ @map[id] =
95
+ { dir: dir, basename: basename, path: path, target: "#{dir}/#{basename}", alias: alias_for(identifier) }
80
96
  @paths[path] = id
81
97
  end
82
98
  end
@@ -28,12 +28,15 @@ module Woods
28
28
  ASSOCIATION_ORDER = %w[belongs_to has_one has_many has_and_belongs_to_many].freeze
29
29
 
30
30
  # @param name_mapper [NameMapper] resolves ids to wikilinks/paths
31
- # @param nodes [Hash] graph nodes ({ id => { 'type' => , ... } }) for edge-endpoint types
31
+ # @param nodes [Hash] note-key => graph node for edge-endpoint types
32
32
  # @param pagerank [Hash{String=>Float}] persisted pagerank scores
33
33
  # @param analysis [Hash, nil] parsed graph_analysis.json (hubs/cycles/orphans/bridges) or nil
34
34
  # @param include_source [Boolean] embed scrubbed source code
35
35
  # @param scanner [#scan, nil] credential scanner (required when include_source)
36
- def initialize(name_mapper:, nodes:, pagerank: {}, analysis: nil, include_source: false, scanner: nil)
36
+ def initialize(name_mapper:, nodes:, pagerank: {}, analysis: nil, include_source: false, scanner: nil,
37
+ identifiers: {}, ambiguous_identifiers: Set.new)
38
+ @identifiers = identifiers
39
+ @ambiguous_identifiers = ambiguous_identifiers
37
40
  @mapper = name_mapper
38
41
  @nodes = nodes || {}
39
42
  @pagerank = pagerank || {}
@@ -45,10 +48,10 @@ module Woods
45
48
  @cycle_members = cycle_member_set(analysis)
46
49
  end
47
50
 
48
- # @param id [String] bare unit identifier (graph node key)
51
+ # @param id [String, Array<String>] note key: bare id for standalone use, or [identifier, type]
49
52
  # @param unit [Hash] parsed unit JSON (used only for body facts)
50
53
  # @param depends_on [Array<Hash>] [{ target:, via: }] filtered to the emitted set
51
- # @param used_by [Array<String>] unique dependent ids filtered to the emitted set
54
+ # @param used_by [Array<String, Array<String>>] unique dependent note keys filtered to the emitted set
52
55
  # @return [String] the rendered note. A failed source scrub omits the source
53
56
  # section (the note is still written); build never returns nil.
54
57
  def build(id:, unit:, depends_on:, used_by:)
@@ -74,7 +77,7 @@ module Woods
74
77
  def frontmatter(id, unit, depends_on, used_by)
75
78
  fm = {
76
79
  'woods_managed' => true,
77
- 'id' => id,
80
+ 'id' => @identifiers.fetch(id, id),
78
81
  'type' => unit['type'],
79
82
  'file' => unit['file_path'],
80
83
  'source_hash' => unit['source_hash'],
@@ -82,17 +85,23 @@ module Woods
82
85
  'dependency_count' => depends_on.size,
83
86
  'dependent_count' => used_by.size,
84
87
  'tags' => tags_for(id, unit),
85
- 'aliases' => [id]
88
+ 'aliases' => [@identifiers.fetch(id, id)]
86
89
  }.compact
87
90
  "#{fm.to_yaml}---"
88
91
  end
89
92
 
93
+ def analysis_identifier(id)
94
+ identifier = @identifiers.fetch(id, id)
95
+ identifier unless @ambiguous_identifiers.include?(identifier)
96
+ end
97
+
90
98
  def pagerank_for(id)
91
- score = @pagerank[id]
99
+ score = @pagerank[analysis_identifier(id)]
92
100
  score&.round(6)
93
101
  end
94
102
 
95
103
  def tags_for(id, unit)
104
+ id = analysis_identifier(id)
96
105
  tags = ['woods/unit', "woods/#{unit['type']}"]
97
106
  tags << 'woods/hub' if @hubs.include?(id)
98
107
  tags << 'woods/orphan' if @orphans.include?(id)
@@ -124,11 +133,11 @@ module Woods
124
133
  end
125
134
 
126
135
  def callout(id, used_by)
127
- if @hubs.include?(id)
136
+ if @hubs.include?(analysis_identifier(id))
128
137
  pr = pagerank_for(id)
129
138
  suffix = pr ? " (PageRank #{format('%.4f', pr)})" : ''
130
139
  "> [!warning] Hub — high blast radius\n> #{used_by.size} units depend on this#{suffix}."
131
- elsif @cycle_members.include?(id)
140
+ elsif @cycle_members.include?(analysis_identifier(id))
132
141
  "> [!warning] Part of a dependency cycle\n> This unit participates in a circular dependency."
133
142
  end
134
143
  end
@@ -181,7 +190,7 @@ module Woods
181
190
  return [] unless items
182
191
 
183
192
  items.filter_map do |assoc|
184
- link = @mapper.wikilink(assoc[:target])
193
+ link = @mapper.wikilink_for_identifier(assoc[:target])
185
194
  next unless link
186
195
 
187
196
  assoc[:dependent] ? "#{link} (dependent: #{assoc[:dependent]})" : link