woods 2.0.0.beta2 → 2.0.0.beta4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +339 -1
- data/CONTRIBUTING.md +188 -12
- data/README.md +93 -174
- data/SECURITY.md +9 -6
- data/docs/AGENT_GUIDE.md +109 -8
- data/docs/AGENT_SETUP.md +98 -7
- data/docs/BACKEND_MATRIX.md +25 -0
- data/docs/CLIENT_HOOKS.md +111 -0
- data/docs/CONFIGURATION_REFERENCE.md +267 -16
- data/docs/CONSOLE_MCP_SETUP.md +80 -7
- data/docs/DOCKER_SETUP.md +22 -3
- data/docs/EVALUATION.md +464 -1
- data/docs/EXTRACTOR_REFERENCE.md +45 -6
- data/docs/FAQ.md +11 -12
- data/docs/GETTING_STARTED.md +17 -5
- data/docs/INCREMENTAL_EXTRACTION.md +147 -7
- data/docs/INDEX_LAYOUT.md +382 -0
- data/docs/INTERNALS.md +7 -2
- data/docs/MCP_SERVERS.md +276 -5
- data/docs/MCP_TOOL_COOKBOOK.md +37 -22
- data/docs/MCP_WORKTREE_SETUP.md +43 -83
- data/docs/NOTION_INTEGRATION.md +13 -0
- data/docs/OBSIDIAN_INTEGRATION.md +57 -9
- data/docs/PUBLISHED_INDEX.md +72 -0
- data/docs/README.md +7 -0
- data/docs/RETRIEVAL_GUIDE.md +273 -12
- data/docs/RUNTIME_TRACING.md +71 -0
- data/docs/SOURCE_FRESHNESS.md +143 -0
- data/docs/TROUBLESHOOTING.md +129 -18
- data/docs/UNBLOCKED_INTEGRATION.md +25 -0
- data/docs/UPGRADING_TO_2.md +48 -22
- data/docs/WATCH_DAEMON.md +277 -67
- data/exe/woods-agent-config +6 -0
- data/exe/woods-extract +5 -0
- data/exe/woods-hook-context +6 -0
- data/exe/woods-mcp-start +14 -9
- data/lib/generators/woods/pgvector_generator.rb +8 -2
- data/lib/generators/woods/templates/woods.rb.tt +1 -3
- data/lib/tasks/woods.rake +47 -397
- data/lib/woods/agent_configuration/applier.rb +135 -0
- data/lib/woods/agent_configuration/cli.rb +101 -0
- data/lib/woods/agent_configuration/cli_options.rb +29 -0
- data/lib/woods/agent_configuration/document.rb +105 -0
- data/lib/woods/agent_configuration/error.rb +7 -0
- data/lib/woods/agent_configuration/launcher.rb +75 -0
- data/lib/woods/agent_configuration/layout.rb +72 -0
- data/lib/woods/agent_configuration/managed_section.rb +62 -0
- data/lib/woods/agent_configuration/plan.rb +98 -0
- data/lib/woods/agent_configuration/plan_diff.rb +38 -0
- data/lib/woods/agent_configuration/planned_files.rb +61 -0
- data/lib/woods/agent_configuration/planner.rb +63 -0
- data/lib/woods/agent_configuration/planner_validation.rb +77 -0
- data/lib/woods/agent_configuration/preflight.rb +100 -0
- data/lib/woods/agent_configuration/recovery.rb +49 -0
- data/lib/woods/ast/node.rb +2 -0
- data/lib/woods/ast/parser.rb +38 -5
- data/lib/woods/builder.rb +21 -5
- data/lib/woods/cache/cache_middleware.rb +28 -7
- data/lib/woods/cache/cache_store.rb +4 -5
- data/lib/woods/change_set.rb +5 -4
- data/lib/woods/console/credential_index.rb +20 -2
- data/lib/woods/console/credential_scanner.rb +18 -17
- data/lib/woods/console/credential_scanner_registry.rb +36 -0
- data/lib/woods/console/dispatch_pipeline.rb +7 -0
- data/lib/woods/console/embedded_executor.rb +32 -10
- data/lib/woods/console/encrypted_credential_snapshot.rb +16 -0
- data/lib/woods/console/rack_middleware.rb +22 -13
- data/lib/woods/console/server.rb +18 -16
- data/lib/woods/console/sql_noise_stripper.rb +9 -7
- data/lib/woods/console/sql_table_scanner.rb +47 -7
- data/lib/woods/console/sql_validator.rb +49 -9
- data/lib/woods/console/sqlite_read_guard.rb +46 -0
- data/lib/woods/coordination/pipeline_lock.rb +3 -2
- data/lib/woods/dependency_graph.rb +65 -13
- data/lib/woods/embedding/corpus.rb +94 -0
- data/lib/woods/embedding/indexer.rb +114 -60
- data/lib/woods/embedding/openai.rb +17 -6
- data/lib/woods/evaluation/ablation_executor.rb +6 -1
- data/lib/woods/evaluation/ablation_timed_executor.rb +22 -4
- data/lib/woods/export/typed_reader.rb +56 -0
- data/lib/woods/extractor.rb +277 -149
- data/lib/woods/extractors/action_cable_extractor.rb +3 -1
- data/lib/woods/extractors/behavioral_profile.rb +9 -7
- data/lib/woods/extractors/caching_extractor.rb +3 -1
- data/lib/woods/extractors/concern_extractor.rb +64 -6
- data/lib/woods/extractors/configuration_extractor.rb +7 -3
- data/lib/woods/extractors/controller_extractor.rb +13 -4
- data/lib/woods/extractors/database_view_extractor.rb +3 -1
- data/lib/woods/extractors/declared_parent.rb +55 -0
- data/lib/woods/extractors/decorator_extractor.rb +3 -1
- data/lib/woods/extractors/engine_extractor.rb +3 -1
- data/lib/woods/extractors/event_extractor.rb +4 -2
- data/lib/woods/extractors/factory_extractor.rb +3 -1
- data/lib/woods/extractors/graphql_extractor.rb +10 -13
- data/lib/woods/extractors/i18n_extractor.rb +3 -1
- data/lib/woods/extractors/job_extractor.rb +6 -19
- data/lib/woods/extractors/lib_extractor.rb +13 -9
- data/lib/woods/extractors/mailer_extractor.rb +26 -15
- data/lib/woods/extractors/manager_extractor.rb +3 -1
- data/lib/woods/extractors/method_parameters.rb +53 -0
- data/lib/woods/extractors/middleware_argument.rb +65 -0
- data/lib/woods/extractors/middleware_extractor.rb +9 -3
- data/lib/woods/extractors/migration_extractor.rb +3 -1
- data/lib/woods/extractors/model_extractor.rb +26 -34
- data/lib/woods/extractors/package_extractor.rb +24 -4
- data/lib/woods/extractors/phlex_extractor.rb +3 -1
- data/lib/woods/extractors/policy_extractor.rb +3 -1
- data/lib/woods/extractors/poro_extractor.rb +13 -9
- data/lib/woods/extractors/pundit_extractor.rb +3 -1
- data/lib/woods/extractors/rails_source_extractor.rb +4 -2
- data/lib/woods/extractors/rake_task_extractor.rb +4 -2
- data/lib/woods/extractors/route_extractor.rb +3 -1
- data/lib/woods/extractors/route_helper_resolver.rb +10 -33
- data/lib/woods/extractors/scheduled_job_extractor.rb +41 -15
- data/lib/woods/extractors/serializer_extractor.rb +4 -2
- data/lib/woods/extractors/service_extractor.rb +3 -1
- data/lib/woods/extractors/shared_dependency_scanner.rb +2 -2
- data/lib/woods/extractors/shared_utility_methods.rb +48 -19
- data/lib/woods/extractors/source_nesting.rb +1 -1
- data/lib/woods/extractors/state_machine_extractor.rb +3 -1
- data/lib/woods/extractors/test_mapping_extractor.rb +3 -1
- data/lib/woods/extractors/validator_extractor.rb +3 -1
- data/lib/woods/extractors/view_component_extractor.rb +3 -1
- data/lib/woods/extractors/view_template_extractor.rb +3 -1
- data/lib/woods/gem_mapper.rb +2 -0
- data/lib/woods/git_history.rb +116 -0
- data/lib/woods/graph_analyzer.rb +35 -6
- data/lib/woods/hooks/context_cli.rb +54 -0
- data/lib/woods/hooks/context_event.rb +88 -0
- data/lib/woods/hooks/context_hint.rb +73 -0
- data/lib/woods/hooks/context_impact.rb +77 -0
- data/lib/woods/hooks/context_output.rb +47 -0
- data/lib/woods/hooks/context_state.rb +102 -0
- data/lib/woods/hooks/refresh.rb +79 -0
- data/lib/woods/hooks/rule_projection.rb +78 -0
- data/lib/woods/input_rules.rb +19 -0
- data/lib/woods/mcp/bearer_auth.rb +22 -13
- data/lib/woods/mcp/bootstrapper.rb +79 -4
- data/lib/woods/mcp/config_resolver.rb +2 -1
- data/lib/woods/mcp/index_reader.rb +334 -162
- data/lib/woods/mcp/initialization_guidance.rb +27 -0
- data/lib/woods/mcp/origin_guard.rb +17 -9
- data/lib/woods/mcp/published_lexical_retriever.rb +115 -0
- data/lib/woods/mcp/renderers/markdown_renderer.rb +22 -9
- data/lib/woods/mcp/renderers/plain_renderer.rb +18 -8
- data/lib/woods/mcp/search_results.rb +74 -0
- data/lib/woods/mcp/server.rb +178 -63
- data/lib/woods/mcp/tool_contract.rb +3 -1
- data/lib/woods/mcp/tool_response_renderer.rb +41 -0
- data/lib/woods/mcp/traversal_evidence.rb +113 -0
- data/lib/woods/mcp/traversal_evidence_index.rb +100 -0
- data/lib/woods/mcp/traversal_evidence_page.rb +41 -0
- data/lib/woods/mcp/traversal_evidence_text.rb +52 -0
- data/lib/woods/mcp/traversal_response.rb +22 -0
- data/lib/woods/notion/exporter.rb +56 -17
- data/lib/woods/obsidian/destination_plan.rb +98 -0
- data/lib/woods/obsidian/name_mapper.rb +19 -3
- data/lib/woods/obsidian/note_builder.rb +19 -10
- data/lib/woods/obsidian/vault_exporter.rb +88 -32
- data/lib/woods/operator/pipeline_guard.rb +18 -13
- data/lib/woods/path_dispatcher.rb +13 -6
- data/lib/woods/payload_store.rb +27 -26
- data/lib/woods/published_index/typed_unit_reader.rb +40 -3
- data/lib/woods/published_index.rb +2 -2
- data/lib/woods/railtie.rb +3 -3
- data/lib/woods/railtie_support.rb +12 -12
- data/lib/woods/rake_helpers.rb +382 -0
- data/lib/woods/resilience/graph_invariant_validator/membership_checks.rb +71 -0
- data/lib/woods/resilience/graph_invariant_validator/node_checks.rb +61 -0
- data/lib/woods/resilience/graph_invariant_validator/reverse_relationship_checks.rb +46 -0
- data/lib/woods/resilience/graph_invariant_validator.rb +119 -0
- data/lib/woods/resilience/index_validator/graph_checks.rb +80 -0
- data/lib/woods/resilience/index_validator.rb +112 -23
- data/lib/woods/retrieval/context_assembler.rb +50 -15
- data/lib/woods/retrieval/lexical_assembler.rb +84 -0
- data/lib/woods/retrieval/lexical_index.rb +120 -0
- data/lib/woods/retrieval/ranker.rb +4 -2
- data/lib/woods/retrieval/scope.rb +108 -0
- data/lib/woods/retrieval/scoped_graph_store.rb +32 -0
- data/lib/woods/retrieval/scoped_vector_store.rb +55 -0
- data/lib/woods/retrieval/search_executor.rb +86 -27
- data/lib/woods/retrieval/source_evidence.rb +200 -0
- data/lib/woods/retriever.rb +98 -22
- data/lib/woods/ruby_analyzer/trace_enricher.rb +77 -38
- data/lib/woods/session_tracer/file_store.rb +6 -1
- data/lib/woods/session_tracer/middleware.rb +10 -12
- data/lib/woods/session_tracer/redis_store.rb +22 -6
- data/lib/woods/session_tracer/session_flow_assembler.rb +23 -17
- data/lib/woods/session_tracer/solid_cache_coordination.rb +6 -4
- data/lib/woods/session_tracer/unit_resolver.rb +63 -0
- data/lib/woods/source_inputs/consumer_errors.rb +31 -0
- data/lib/woods/source_inputs/handoff.rb +102 -0
- data/lib/woods/source_inputs/launcher.rb +157 -0
- data/lib/woods/source_inputs/manifest.rb +124 -0
- data/lib/woods/source_inputs/private_key.rb +55 -0
- data/lib/woods/source_inputs/scanner.rb +171 -0
- data/lib/woods/source_inputs/scopes.rb +71 -0
- data/lib/woods/source_inputs/session.rb +214 -0
- data/lib/woods/source_inputs/status.rb +84 -0
- data/lib/woods/source_inputs/verifier.rb +107 -0
- data/lib/woods/storage/metadata_store.rb +25 -25
- data/lib/woods/storage/pgvector.rb +35 -10
- data/lib/woods/storage/qdrant.rb +17 -7
- data/lib/woods/storage/vector_store.rb +18 -6
- data/lib/woods/tasks.rb +3 -2
- data/lib/woods/temporal/json_snapshot_store.rb +58 -9
- data/lib/woods/unblocked/exporter.rb +59 -70
- data/lib/woods/version.rb +1 -1
- data/lib/woods/watch/boot_snapshot.rb +52 -0
- data/lib/woods/watch/daemon.rb +154 -32
- data/lib/woods/watch/listen_watcher.rb +4 -0
- data/lib/woods/watch/polling_watcher.rb +5 -1
- data/lib/woods/watch/status.rb +20 -15
- data/lib/woods/watch/tree_scan.rb +21 -13
- data/lib/woods/watch/watcher.rb +4 -1
- data/lib/woods.rb +50 -11
- data/plugin/.claude-plugin/plugin.json +1 -1
- data/plugin/hooks/adapters/normalize.jq +15 -0
- data/plugin/hooks/adapters/normalize.rb +63 -0
- data/plugin/hooks/hooks.json +20 -0
- data/plugin/hooks/woods-context.sh +50 -0
- data/plugin/hooks/woods-input-rules.sh +159 -0
- data/plugin/hooks/woods-opencode.mjs +65 -0
- data/plugin/hooks/woods-post-edit.sh +2 -225
- data/plugin/hooks/woods-refresh.sh +260 -0
- data/plugin/hooks/woods-session-start.sh +47 -55
- data/plugin/skills/woods-agent-enable/SKILL.md +19 -0
- data/plugin/skills/woods-diagnose/SKILL.md +319 -1
- data/plugin/skills/woods-investigate/SKILL.md +145 -0
- data/plugin/skills/woods-mcp-config/SKILL.md +90 -2
- data/plugin/skills/woods-setup/SKILL.md +110 -6
- metadata +87 -5
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'set'
|
|
4
|
+
require_relative 'graph_invariant_validator/membership_checks'
|
|
5
|
+
require_relative 'graph_invariant_validator/node_checks'
|
|
6
|
+
require_relative 'graph_invariant_validator/reverse_relationship_checks'
|
|
7
|
+
|
|
8
|
+
module Woods
|
|
9
|
+
module Resilience
|
|
10
|
+
# Checks raw published graph data against typed per-directory index entries.
|
|
11
|
+
# It deliberately does not use DependencyGraph.from_h: normalization there
|
|
12
|
+
# can discard malformed records before a validator has seen them.
|
|
13
|
+
#
|
|
14
|
+
# Target identifiers are not necessarily graph nodes (external/unresolved
|
|
15
|
+
# dependencies are valid), and reverse/file/type indexes contain bare names,
|
|
16
|
+
# not typed endpoints. Expected membership is the union over all variants.
|
|
17
|
+
class GraphInvariantValidator
|
|
18
|
+
include MembershipChecks
|
|
19
|
+
include NodeChecks
|
|
20
|
+
include ReverseRelationshipChecks
|
|
21
|
+
|
|
22
|
+
def initialize(graph:, index_entries:)
|
|
23
|
+
@graph = graph
|
|
24
|
+
@index_entries = index_entries
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# @return [Array<String>] semantic errors, without changing either input
|
|
28
|
+
def validate
|
|
29
|
+
@errors = []
|
|
30
|
+
return ['dependency_graph.json: expected an object'] unless @graph.is_a?(Hash)
|
|
31
|
+
|
|
32
|
+
@typed_nodes = {}
|
|
33
|
+
@primary_types = {}
|
|
34
|
+
@expected_reverse = membership_map
|
|
35
|
+
@expected_reverse_via = Hash.new { |hash, key| hash[key] = [] }
|
|
36
|
+
@expected_files = membership_map
|
|
37
|
+
@expected_types = membership_map
|
|
38
|
+
collect_nodes
|
|
39
|
+
validate_edges
|
|
40
|
+
validate_reverse_relationships
|
|
41
|
+
validate_membership('reverse', @expected_reverse)
|
|
42
|
+
validate_membership('file_map', @expected_files, scalar: true)
|
|
43
|
+
validate_membership('type_index', @expected_types)
|
|
44
|
+
validate_index_agreement
|
|
45
|
+
@errors
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
private
|
|
49
|
+
|
|
50
|
+
def membership_map
|
|
51
|
+
Hash.new { |hash, key| hash[key] = Set.new }
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def error(label, message)
|
|
55
|
+
@errors << "dependency_graph.json #{label}: #{message}"
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def object_section(name)
|
|
59
|
+
value = @graph[name]
|
|
60
|
+
return value if value.is_a?(Hash)
|
|
61
|
+
|
|
62
|
+
error(name, 'expected an object')
|
|
63
|
+
{}
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def name?(value)
|
|
67
|
+
value.is_a?(String) && !value.empty?
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def validate_edges
|
|
71
|
+
edges = object_section('edges')
|
|
72
|
+
@primary_types.each_key do |identifier|
|
|
73
|
+
error('edges', "missing edge list for #{identifier.inspect}") unless edges.key?(identifier)
|
|
74
|
+
end
|
|
75
|
+
edges.each do |identifier, list|
|
|
76
|
+
label = "edges[#{identifier.inspect}]"
|
|
77
|
+
error(label, 'source has no primary node') unless @primary_types.key?(identifier)
|
|
78
|
+
validate_edge_list(identifier, @primary_types[identifier], list, label)
|
|
79
|
+
end
|
|
80
|
+
@variants.each_with_index do |record, index|
|
|
81
|
+
next unless record.is_a?(Hash)
|
|
82
|
+
|
|
83
|
+
validate_edge_list(record['identifier'], record['type'], record['edges'], "variants[#{index}].edges")
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def validate_edge_list(source, source_type, list, label)
|
|
88
|
+
unless list.is_a?(Array)
|
|
89
|
+
error(label, 'expected an array')
|
|
90
|
+
return
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
list.each_with_index do |edge, index|
|
|
94
|
+
validate_edge(source, source_type, edge, "#{label}[#{index}]")
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def validate_edge(source, source_type, edge, label)
|
|
99
|
+
target = edge.is_a?(Hash) ? edge['target'] : edge
|
|
100
|
+
unless name?(target)
|
|
101
|
+
error(label, 'expected a target identifier or an object with a target identifier')
|
|
102
|
+
return
|
|
103
|
+
end
|
|
104
|
+
validate_edge_attributes(edge, label) if edge.is_a?(Hash)
|
|
105
|
+
@expected_reverse[target].add(source) if name?(source)
|
|
106
|
+
record_reverse_relationship(target, source, source_type, edge)
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def validate_edge_attributes(edge, label)
|
|
110
|
+
%w[via through through_db].each do |key|
|
|
111
|
+
error(label, "#{key} must be a string or null") unless edge[key].nil? || edge[key].is_a?(String)
|
|
112
|
+
end
|
|
113
|
+
return if [nil, true, false].include?(edge['disable_joins'])
|
|
114
|
+
|
|
115
|
+
error(label, 'disable_joins must be a boolean or null')
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Woods
|
|
4
|
+
module Resilience
|
|
5
|
+
class IndexValidator
|
|
6
|
+
# Reuse the published reader's retention pin but inspect raw graph and
|
|
7
|
+
# unit artifacts independently, without normalizing away bad records.
|
|
8
|
+
module GraphChecks
|
|
9
|
+
private
|
|
10
|
+
|
|
11
|
+
def with_validation_payload
|
|
12
|
+
root = Pathname.new(@index_dir)
|
|
13
|
+
if root.join('manifest.json').exist? || root.join('generation.json').exist?
|
|
14
|
+
Woods::PublishedIndex::GenerationCatalog.pointer(root)
|
|
15
|
+
reader = Woods::MCP::IndexReader.new(root)
|
|
16
|
+
reader.with_pinned_generation do
|
|
17
|
+
@validation_payload = reader.payload_dir.to_s
|
|
18
|
+
@validation_manifest = reader.manifest
|
|
19
|
+
yield
|
|
20
|
+
end
|
|
21
|
+
else
|
|
22
|
+
@validation_payload = @index_dir
|
|
23
|
+
yield
|
|
24
|
+
end
|
|
25
|
+
ensure
|
|
26
|
+
@validation_payload = nil
|
|
27
|
+
@validation_manifest = nil
|
|
28
|
+
@graph_index_entries = nil
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def static_source_map?
|
|
32
|
+
manifest = @validation_manifest
|
|
33
|
+
manifest.is_a?(Hash) && manifest['provenance'].is_a?(Hash) &&
|
|
34
|
+
manifest['provenance']['mode'] == 'woods_static_ruby_source'
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def validation_index_entries(path, errors)
|
|
38
|
+
entries = JSON.parse(Woods::AtomicFile.read(path))
|
|
39
|
+
unless entries.is_a?(Array)
|
|
40
|
+
errors << "#{path}: expected an array"
|
|
41
|
+
return []
|
|
42
|
+
end
|
|
43
|
+
entries.select do |entry|
|
|
44
|
+
valid = entry.is_a?(Hash) && entry['identifier'].is_a?(String) && !entry['identifier'].empty?
|
|
45
|
+
errors << "#{path}: entry requires a nonempty identifier" unless valid
|
|
46
|
+
valid
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def collect_graph_index_entry(type_dir, entry, data, errors)
|
|
51
|
+
directory = File.basename(type_dir)
|
|
52
|
+
types = Woods::MCP::IndexReader::UNIT_TYPES_BY_DIR.fetch(directory)
|
|
53
|
+
typed_entry = graph_index_entry(entry, data, types)
|
|
54
|
+
@graph_index_entries << typed_entry
|
|
55
|
+
return unless data
|
|
56
|
+
|
|
57
|
+
file = find_unit_file(type_dir, entry['identifier'])
|
|
58
|
+
unless data['identifier'] == entry['identifier'] && types.include?(data['type'])
|
|
59
|
+
errors << "#{file}: expected typed unit #{typed_entry['type']}:#{entry['identifier']}"
|
|
60
|
+
return
|
|
61
|
+
end
|
|
62
|
+
validate_graph_unit_path(file, entry, data, errors)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def graph_index_entry(entry, data, types)
|
|
66
|
+
type = data && types.include?(data['type']) ? data['type'] : types.first
|
|
67
|
+
result = entry.merge('type' => type)
|
|
68
|
+
result['file_path'] = data['file_path'] if data&.key?('file_path') && !entry.key?('file_path')
|
|
69
|
+
result
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def validate_graph_unit_path(file, entry, data, errors)
|
|
73
|
+
return unless entry.key?('file_path') && data.key?('file_path') && entry['file_path'] != data['file_path']
|
|
74
|
+
|
|
75
|
+
errors << "#{file}: file_path differs from #{File.dirname(file)}/_index.json for #{entry['identifier']}"
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
end
|
|
@@ -2,10 +2,16 @@
|
|
|
2
2
|
|
|
3
3
|
require 'json'
|
|
4
4
|
require 'set'
|
|
5
|
+
require 'rubygems/version'
|
|
6
|
+
require_relative '../version'
|
|
5
7
|
require_relative '../filename_utils'
|
|
6
8
|
require_relative '../atomic_file'
|
|
7
9
|
|
|
8
10
|
require_relative '../generation'
|
|
11
|
+
require_relative '../published_index'
|
|
12
|
+
require_relative 'graph_invariant_validator'
|
|
13
|
+
require_relative 'index_validator/graph_checks'
|
|
14
|
+
require_relative '../source_inputs/manifest'
|
|
9
15
|
|
|
10
16
|
module Woods
|
|
11
17
|
module Resilience
|
|
@@ -16,6 +22,8 @@ module Woods
|
|
|
16
22
|
# - All files referenced in the index exist on disk
|
|
17
23
|
# - Content hashes (source_hash) match the actual source_code
|
|
18
24
|
# - No stale unit files exist that aren't listed in the index
|
|
25
|
+
# - Typed graph identities and reverse/file/type memberships agree
|
|
26
|
+
# All checks share one pinned generation; the validator never repairs it.
|
|
19
27
|
#
|
|
20
28
|
# **This class knows nothing about vectors or embedding dimensions.** Six
|
|
21
29
|
# documents used to credit it with detecting dimension mismatches; it never
|
|
@@ -23,10 +31,8 @@ module Woods
|
|
|
23
31
|
# `Tasks.verify_store_dimensions!` before a durable embed run and by
|
|
24
32
|
# {Woods::Storage::Snapshotter::Vector} at MCP boot.
|
|
25
33
|
#
|
|
26
|
-
#
|
|
27
|
-
#
|
|
28
|
-
# check inline rather than calling this — deliberate duplication left alone
|
|
29
|
-
# for now, since the task's output format is user-facing.
|
|
34
|
+
# Shared by the `woods:validate` rake task and worktree integration checks.
|
|
35
|
+
# Writer-version mismatches are advisory; structural errors still fail.
|
|
30
36
|
#
|
|
31
37
|
# @example
|
|
32
38
|
# validator = IndexValidator.new(index_dir: "tmp/woods")
|
|
@@ -34,6 +40,7 @@ module Woods
|
|
|
34
40
|
# puts report.errors if !report.valid?
|
|
35
41
|
class IndexValidator # rubocop:disable Metrics/ClassLength
|
|
36
42
|
include Woods::FilenameUtils
|
|
43
|
+
include GraphChecks
|
|
37
44
|
|
|
38
45
|
# Report produced by {#validate}.
|
|
39
46
|
#
|
|
@@ -84,13 +91,19 @@ module Woods
|
|
|
84
91
|
return ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
|
|
85
92
|
end
|
|
86
93
|
|
|
87
|
-
|
|
88
|
-
|
|
94
|
+
with_validation_payload do
|
|
95
|
+
@graph_index_entries = []
|
|
96
|
+
payload_type_dirs(errors).each do |type_dir|
|
|
97
|
+
validate_type_directory(type_dir, warnings, errors)
|
|
98
|
+
end
|
|
99
|
+
validate_flow_artifacts(errors)
|
|
100
|
+
validate_against_manifest(warnings, errors)
|
|
89
101
|
end
|
|
90
|
-
validate_flow_artifacts(errors)
|
|
91
|
-
validate_against_manifest(warnings, errors)
|
|
92
102
|
|
|
93
103
|
ValidationReport.new(valid?: errors.empty?, warnings: warnings, errors: errors)
|
|
104
|
+
rescue IOError, SystemCallError, ArgumentError, JSON::ParserError, Woods::PublishedIndex::CorruptPointerError => e
|
|
105
|
+
errors << "Cannot read published index: #{e.class}: #{e.message}"
|
|
106
|
+
ValidationReport.new(valid?: false, warnings: warnings, errors: errors)
|
|
94
107
|
end
|
|
95
108
|
|
|
96
109
|
# The checks `woods:validate` used to carry inline: manifest counts
|
|
@@ -107,19 +120,72 @@ module Woods
|
|
|
107
120
|
manifest_path = File.join(payload, 'manifest.json')
|
|
108
121
|
return unless File.exist?(manifest_path)
|
|
109
122
|
|
|
110
|
-
manifest =
|
|
123
|
+
manifest = read_validation_manifest(manifest_path, errors)
|
|
124
|
+
return unless manifest
|
|
125
|
+
|
|
126
|
+
validate_writer_version(manifest['woods_version'], warnings)
|
|
111
127
|
unresolvable = Hash.new { |hash, key| hash[key] = [] }
|
|
112
128
|
|
|
113
|
-
(
|
|
129
|
+
manifest.fetch('counts', {}).each do |type, expected_count|
|
|
114
130
|
validate_manifest_type(payload, type, expected_count, unresolvable, warnings, errors)
|
|
115
131
|
end
|
|
116
132
|
|
|
117
133
|
warn_unresolvable_paths(warnings, unresolvable)
|
|
118
134
|
validate_dependency_graph(payload, errors)
|
|
135
|
+
validate_source_inputs(payload, errors)
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
# Optional for old generations; malformed new provenance is an artifact
|
|
139
|
+
# integrity error. Source drift itself is advisory and belongs to status.
|
|
140
|
+
def validate_source_inputs(payload, errors)
|
|
141
|
+
path = File.join(payload, SourceInputs::Manifest::FILE_NAME)
|
|
142
|
+
return unless File.exist?(path)
|
|
143
|
+
|
|
144
|
+
File.open(path, File::RDONLY | File::NONBLOCK) do |file|
|
|
145
|
+
raise SourceInputs::Manifest::Invalid unless file.stat.file?
|
|
146
|
+
|
|
147
|
+
SourceInputs::Manifest.parse(file.read(SourceInputs::Manifest::MAX_BYTES + 1))
|
|
148
|
+
end
|
|
149
|
+
rescue SourceInputs::Manifest::Invalid, SystemCallError, IOError
|
|
150
|
+
errors << 'Invalid source_inputs.json provenance artifact'
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
def read_validation_manifest(path, errors)
|
|
154
|
+
manifest = JSON.parse(Woods::AtomicFile.read(path))
|
|
155
|
+
unless manifest.is_a?(Hash)
|
|
156
|
+
errors << 'manifest.json: expected an object'
|
|
157
|
+
return
|
|
158
|
+
end
|
|
159
|
+
unless manifest.fetch('counts', {}).is_a?(Hash)
|
|
160
|
+
errors << 'manifest.json counts: expected an object'
|
|
161
|
+
return
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
manifest
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
# The version records the last publisher, not the producer of every
|
|
168
|
+
# retained unit. Missing provenance is normal for older indexes.
|
|
169
|
+
def validate_writer_version(value, warnings)
|
|
170
|
+
return if value.nil?
|
|
171
|
+
|
|
172
|
+
unless value.is_a?(String) && !value.strip.empty?
|
|
173
|
+
warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
|
|
174
|
+
return
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
writer = Gem::Version.new(value)
|
|
178
|
+
reader = Gem::Version.new(Woods::VERSION)
|
|
179
|
+
return if writer.segments.first == reader.segments.first
|
|
180
|
+
|
|
181
|
+
warnings << "Index last published by Woods #{value}; reader is Woods #{Woods::VERSION}. " \
|
|
182
|
+
'Major versions differ. Run a full woods:extract before relying on compatibility.'
|
|
183
|
+
rescue ArgumentError
|
|
184
|
+
warnings << 'Invalid manifest woods_version; writer provenance is unknown. Run a full woods:extract.'
|
|
119
185
|
end
|
|
120
186
|
|
|
121
|
-
# Split
|
|
122
|
-
#
|
|
187
|
+
# Split unresolvable app-tree paths from gem-owned paths so each warning
|
|
188
|
+
# can distinguish a different filesystem environment from a changed bundle.
|
|
123
189
|
#
|
|
124
190
|
# @param warnings [Array<String>]
|
|
125
191
|
# @param unresolvable [Hash{String => Array<Array(String, Boolean)>}]
|
|
@@ -143,15 +209,16 @@ module Woods
|
|
|
143
209
|
|
|
144
210
|
# Absolute paths outside the app root belong to a gem — an engine model,
|
|
145
211
|
# a framework source — and resolve only where that gem is installed at
|
|
146
|
-
# the extracting path.
|
|
147
|
-
#
|
|
212
|
+
# the extracting path. A changed bundle requires fresh extraction; a
|
|
213
|
+
# reader on another filesystem instead needs the original gem paths.
|
|
148
214
|
def warn_unresolvable_gem_paths(warnings, type, identifiers)
|
|
149
215
|
return if identifiers.empty?
|
|
150
216
|
|
|
151
217
|
warnings << "#{type}: #{identifiers.size} unit(s) whose file_path lies outside the app root " \
|
|
152
218
|
"and is absent here (e.g. #{identifiers.first(3).join(', ')}). Gem-owned units " \
|
|
153
219
|
'(engine models, framework sources) resolve only where that gem is installed at ' \
|
|
154
|
-
'the extracting path.'
|
|
220
|
+
'the extracting path. After a bundle update, run woods:extract in a fresh process ' \
|
|
221
|
+
'with the updated bundle, then woods:validate.'
|
|
155
222
|
end
|
|
156
223
|
|
|
157
224
|
# rubocop:disable-next Metrics/ParameterLists
|
|
@@ -175,6 +242,10 @@ module Woods
|
|
|
175
242
|
# @param errors [Array<String>]
|
|
176
243
|
def validate_unit_file(file, type, unresolvable, errors)
|
|
177
244
|
data = JSON.parse(Woods::AtomicFile.read(file))
|
|
245
|
+
unless data.is_a?(Hash)
|
|
246
|
+
errors << "#{file}: expected a unit object"
|
|
247
|
+
return
|
|
248
|
+
end
|
|
178
249
|
errors << "#{file}: missing identifier" unless data['identifier']
|
|
179
250
|
errors << "#{file}: missing source_code" unless data['source_code']
|
|
180
251
|
file_path = data['file_path']
|
|
@@ -211,13 +282,14 @@ module Woods
|
|
|
211
282
|
return
|
|
212
283
|
end
|
|
213
284
|
|
|
214
|
-
JSON.parse(Woods::AtomicFile.read(graph_path))
|
|
285
|
+
graph = JSON.parse(Woods::AtomicFile.read(graph_path))
|
|
286
|
+
errors.concat(GraphInvariantValidator.new(graph: graph, index_entries: @graph_index_entries).validate)
|
|
215
287
|
rescue JSON::ParserError
|
|
216
288
|
errors << 'dependency_graph.json: invalid JSON'
|
|
217
289
|
end
|
|
218
290
|
|
|
219
291
|
def payload_dir
|
|
220
|
-
Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
|
|
292
|
+
@validation_payload || Woods::Generation.new(output_dir: @index_dir).payload_dir.to_s
|
|
221
293
|
end
|
|
222
294
|
|
|
223
295
|
private
|
|
@@ -292,12 +364,18 @@ module Woods
|
|
|
292
364
|
# failure RAISES: {#payload_type_dirs} converts it to a validation
|
|
293
365
|
# error rather than degrading to a silently empty allowlist.
|
|
294
366
|
#
|
|
367
|
+
# The explicit Woods static-map provenance additionally admits the
|
|
368
|
+
# GemMapper's own type families; arbitrary directories stay excluded.
|
|
295
369
|
# `flows/` is deliberately absent: it holds `flow_index.json` and
|
|
296
370
|
# per-flow documents, which {#validate_flow_artifacts} owns.
|
|
297
371
|
#
|
|
298
372
|
# @return [Array<String>]
|
|
299
373
|
def type_directory_allowlist
|
|
300
|
-
|
|
374
|
+
directories = self.class.unit_type_directories
|
|
375
|
+
return directories unless static_source_map?
|
|
376
|
+
|
|
377
|
+
require_relative '../gem_mapper'
|
|
378
|
+
directories | Woods::GemMapper::TYPE_DIRECTORIES.values
|
|
301
379
|
end
|
|
302
380
|
|
|
303
381
|
# Validate the flows/ artifact family (G-2): `flow_index.json` parses,
|
|
@@ -361,13 +439,12 @@ module Woods
|
|
|
361
439
|
return
|
|
362
440
|
end
|
|
363
441
|
|
|
364
|
-
index_entries = JSON.parse(Woods::AtomicFile.read(index_path))
|
|
365
442
|
indexed_identifiers = Set.new
|
|
366
|
-
|
|
367
|
-
index_entries.each do |entry|
|
|
443
|
+
validation_index_entries(index_path, errors).each do |entry|
|
|
368
444
|
identifier = entry['identifier']
|
|
369
445
|
indexed_identifiers << identifier
|
|
370
|
-
validate_index_entry(type_dir, type_name, identifier, errors)
|
|
446
|
+
data = validate_index_entry(type_dir, type_name, identifier, errors)
|
|
447
|
+
collect_graph_index_entry(type_dir, entry, data, errors)
|
|
371
448
|
end
|
|
372
449
|
|
|
373
450
|
check_stale_files(type_dir, type_name, indexed_identifiers, warnings)
|
|
@@ -413,11 +490,23 @@ module Woods
|
|
|
413
490
|
# @param errors [Array<String>] Accumulated errors
|
|
414
491
|
def validate_content_hash(unit_file, identifier, errors)
|
|
415
492
|
data = JSON.parse(Woods::AtomicFile.read(unit_file))
|
|
493
|
+
unless data.is_a?(Hash)
|
|
494
|
+
errors << "#{unit_file}: expected a unit object"
|
|
495
|
+
return
|
|
496
|
+
end
|
|
497
|
+
check_content_hash(data, identifier, errors)
|
|
498
|
+
data
|
|
499
|
+
end
|
|
500
|
+
|
|
501
|
+
def check_content_hash(data, identifier, errors)
|
|
416
502
|
source_code = data['source_code']
|
|
417
503
|
stored_hash = data['source_hash']
|
|
418
|
-
|
|
419
504
|
return unless source_code && stored_hash
|
|
420
505
|
|
|
506
|
+
unless source_code.is_a?(String) && stored_hash.is_a?(String)
|
|
507
|
+
errors << "#{identifier}: source_code and source_hash must be strings"
|
|
508
|
+
return
|
|
509
|
+
end
|
|
421
510
|
expected_hash = Digest::SHA256.hexdigest(source_code)
|
|
422
511
|
return if stored_hash == expected_hash
|
|
423
512
|
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative 'search_executor'
|
|
4
4
|
require_relative '../token_utils'
|
|
5
|
+
require_relative 'source_evidence'
|
|
5
6
|
|
|
6
7
|
module Woods
|
|
7
8
|
module Retrieval
|
|
@@ -103,7 +104,18 @@ module Woods
|
|
|
103
104
|
# @param structural_context [String, nil] Optional codebase overview text
|
|
104
105
|
# @param budget [Integer, nil] Override token budget; falls back to @budget
|
|
105
106
|
# @return [AssembledContext] Token-budgeted context with source attribution
|
|
106
|
-
def assemble(
|
|
107
|
+
def assemble(**options)
|
|
108
|
+
# Pipeline assemblers are shared by concurrent Ruby/HTTP callers. Keep
|
|
109
|
+
# per-request metadata, query, mode and generation on a private worker.
|
|
110
|
+
dup.send(:assemble_request, **options)
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def assemble_request(candidates:, classification:, structural_context: nil, budget: nil,
|
|
114
|
+
evidence: 'full', query: nil, generation: nil)
|
|
115
|
+
SourceEvidence.validate_mode!(evidence)
|
|
116
|
+
@evidence_mode = evidence
|
|
117
|
+
@evidence_query = query
|
|
118
|
+
@evidence_generation = generation
|
|
107
119
|
effective_budget = budget || @budget
|
|
108
120
|
sections = []
|
|
109
121
|
sources = []
|
|
@@ -132,6 +144,22 @@ module Woods
|
|
|
132
144
|
build_result(sections, sources, effective_budget, @skipped_missing_metadata)
|
|
133
145
|
end
|
|
134
146
|
|
|
147
|
+
private :assemble_request
|
|
148
|
+
|
|
149
|
+
# Estimate token count. Prefers the injected {TokenCounter} — which
|
|
150
|
+
# loads the provider's real tokenizer and returns exact counts — and
|
|
151
|
+
# falls back to the configured chars-per-token ratio when no counter
|
|
152
|
+
# is wired.
|
|
153
|
+
#
|
|
154
|
+
# @param text [String]
|
|
155
|
+
# @return [Integer]
|
|
156
|
+
def estimate_tokens(text)
|
|
157
|
+
return 0 if text.nil? || text.empty?
|
|
158
|
+
return @token_counter.count(text) if @token_counter
|
|
159
|
+
|
|
160
|
+
(text.length / @chars_per_token).ceil
|
|
161
|
+
end
|
|
162
|
+
|
|
135
163
|
private
|
|
136
164
|
|
|
137
165
|
# Suffix the Indexer appends when a single unit is split into multiple
|
|
@@ -293,6 +321,10 @@ module Woods
|
|
|
293
321
|
return tokens_used
|
|
294
322
|
end
|
|
295
323
|
|
|
324
|
+
if @evidence_mode != 'full'
|
|
325
|
+
return append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
|
|
326
|
+
end
|
|
327
|
+
|
|
296
328
|
text = format_unit(unit, candidate)
|
|
297
329
|
tokens = estimate_tokens(text)
|
|
298
330
|
remaining = budget - tokens_used
|
|
@@ -308,6 +340,23 @@ module Woods
|
|
|
308
340
|
end
|
|
309
341
|
end
|
|
310
342
|
|
|
343
|
+
def append_compact_candidate(parts, sources, candidate, unit, budget, tokens_used)
|
|
344
|
+
header = "## #{unit_field(unit, :identifier)} (#{unit_field(unit, :type)})\n" \
|
|
345
|
+
"File: #{unit_field(unit, :file_path)}\n\n"
|
|
346
|
+
remaining = budget - tokens_used
|
|
347
|
+
return tokens_used unless remaining.positive?
|
|
348
|
+
|
|
349
|
+
evidence = SourceEvidence.new(unit: unit, query: @evidence_query, generation: @evidence_generation)
|
|
350
|
+
.render(mode: @evidence_mode, budget: remaining,
|
|
351
|
+
counter: ->(text) { estimate_tokens(header + text) })
|
|
352
|
+
return tokens_used if evidence.text.empty?
|
|
353
|
+
|
|
354
|
+
text = header + evidence.text
|
|
355
|
+
parts << text
|
|
356
|
+
sources << build_source_attribution(candidate, unit).merge(evidence: evidence.provenance)
|
|
357
|
+
tokens_used + estimate_tokens(text)
|
|
358
|
+
end
|
|
359
|
+
|
|
311
360
|
# Format a unit for inclusion in context.
|
|
312
361
|
#
|
|
313
362
|
# @param unit [Hash] Unit data from metadata store
|
|
@@ -400,20 +449,6 @@ module Woods
|
|
|
400
449
|
"#{text[0...target_chars]}\n... [truncated]"
|
|
401
450
|
end
|
|
402
451
|
|
|
403
|
-
# Estimate token count. Prefers the injected {TokenCounter} — which
|
|
404
|
-
# loads the provider's real tokenizer and returns exact counts — and
|
|
405
|
-
# falls back to the configured chars-per-token ratio when no counter
|
|
406
|
-
# is wired.
|
|
407
|
-
#
|
|
408
|
-
# @param text [String]
|
|
409
|
-
# @return [Integer]
|
|
410
|
-
def estimate_tokens(text)
|
|
411
|
-
return 0 if text.nil? || text.empty?
|
|
412
|
-
return @token_counter.count(text) if @token_counter
|
|
413
|
-
|
|
414
|
-
(text.length / @chars_per_token).ceil
|
|
415
|
-
end
|
|
416
|
-
|
|
417
452
|
# Effective chars-per-token for chunk-size sizing. When an exact
|
|
418
453
|
# counter is present, prefer its native ratio (e.g. 1.2 for
|
|
419
454
|
# nomic-embed-text) so truncation and estimation agree. Falls back
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'context_assembler'
|
|
4
|
+
require_relative 'lexical_index'
|
|
5
|
+
|
|
6
|
+
module Woods
|
|
7
|
+
module Retrieval
|
|
8
|
+
# Lexical context keeps matching evidence visible and charges every notice,
|
|
9
|
+
# header and truncation marker to the same character-based token estimate.
|
|
10
|
+
class LexicalAssembler
|
|
11
|
+
def estimate_tokens(text)
|
|
12
|
+
(text.length / 4.0).ceil
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
def assemble(candidates:, budget:, evidence: 'full', query: nil, generation: nil, **)
|
|
16
|
+
SourceEvidence.validate_mode!(evidence)
|
|
17
|
+
raise ArgumentError, 'budget must be a positive Integer' unless budget.is_a?(Integer) && budget.positive?
|
|
18
|
+
|
|
19
|
+
# Reserve the largest included count before selecting evidence. Replacing
|
|
20
|
+
# it afterward can only shrink the text; do not refill or reorder sources.
|
|
21
|
+
notice = count_notice(candidates.size, candidates.size)
|
|
22
|
+
context = notice[0, budget * 4]
|
|
23
|
+
sources = []
|
|
24
|
+
candidates.each do |candidate|
|
|
25
|
+
unit = candidate.metadata
|
|
26
|
+
header = "\n\n## #{unit['identifier']} (#{unit['type']})\nFile: #{unit['file_path']}\n" \
|
|
27
|
+
"Matched: #{candidate.matched_fields.join(', ')}\n\n"
|
|
28
|
+
remaining = (budget * 4) - context.length - header.length
|
|
29
|
+
next unless remaining.positive?
|
|
30
|
+
|
|
31
|
+
if evidence != 'full'
|
|
32
|
+
selected = SourceEvidence.new(unit: unit, query: query, generation: generation)
|
|
33
|
+
.render(mode: evidence, budget: budget,
|
|
34
|
+
counter: ->(text) { estimate_tokens(context + header + text) })
|
|
35
|
+
next if selected.text.empty?
|
|
36
|
+
|
|
37
|
+
context += header + selected.text
|
|
38
|
+
sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
|
|
39
|
+
score: candidate.score, matched_fields: candidate.matched_fields,
|
|
40
|
+
evidence: selected.provenance }
|
|
41
|
+
next
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
source = evidence_text(candidate)
|
|
45
|
+
truncated = source.length > remaining
|
|
46
|
+
marker = "\n[Published evidence truncated; use lookup for the full unit.]"
|
|
47
|
+
next if truncated && remaining <= marker.length
|
|
48
|
+
|
|
49
|
+
context += header + (truncated ? source[0, remaining - marker.length] + marker : source)
|
|
50
|
+
sources << { identifier: unit['identifier'], type: unit['type'], file_path: unit['file_path'],
|
|
51
|
+
score: candidate.score, matched_fields: candidate.matched_fields, truncated: truncated }
|
|
52
|
+
break if truncated
|
|
53
|
+
end
|
|
54
|
+
body = context[notice.length..].to_s
|
|
55
|
+
context = (count_notice(candidates.size, sources.size) + body)[0, budget * 4]
|
|
56
|
+
AssembledContext.new(context: context, tokens_used: estimate_tokens(context), budget: budget,
|
|
57
|
+
sources: sources, sections: [:primary], skipped_missing_metadata: 0)
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
private
|
|
61
|
+
|
|
62
|
+
def count_notice(candidates, included)
|
|
63
|
+
text = "Mode: lexical (field-aware BM25; sources included: #{included}; " \
|
|
64
|
+
"candidates considered: #{candidates}; candidate limit: #{LexicalIndex::DEFAULT_LIMIT}; " \
|
|
65
|
+
'token counts estimated).'
|
|
66
|
+
text += "\nNo lexical matches." if candidates.zero?
|
|
67
|
+
text
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def evidence_text(candidate)
|
|
71
|
+
unit = candidate.metadata
|
|
72
|
+
source = unit['source_code'].to_s
|
|
73
|
+
return source unless candidate.matched_fields.any? { |field| field.start_with?('runtime:') }
|
|
74
|
+
|
|
75
|
+
metadata = unit['metadata'].is_a?(Hash) ? unit['metadata'] : {}
|
|
76
|
+
values = LexicalIndex::RUNTIME_FIELDS.each_with_object({}) do |key, selected|
|
|
77
|
+
value = metadata[key] || unit[key]
|
|
78
|
+
selected[key] = value unless value.nil?
|
|
79
|
+
end
|
|
80
|
+
"Published runtime metadata:\n#{JSON.pretty_generate(values)}\n\n#{source}"
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|