scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -0,0 +1,317 @@
1
+ # Navigating inference provenance
2
+
3
+ This document explains how Scout-AI links persisted chats, Workflow jobs, agent logs, inference segments, and tool calls. It is intended for framework contributors and developers building provenance-aware tools.
4
+
5
+ For the detailed investigation and design rationale, see [../../research/provenance-navigation-design.md](../../research/provenance-navigation-design.md).
6
+
7
+ ## Core model
8
+
9
+ Scout-AI provenance combines two native Scout data models:
10
+
11
+ - a **Chat file** is a persisted Array of messages;
12
+ - a **Workflow Step** is a persisted task execution with dependencies, status, result, and artifacts.
13
+
14
+ These are the only structural node kinds. Agents, inference segments, tool calls, and token records are observations inside chats or jobs rather than separate runtime wrapper objects.
15
+
16
+ The structural relations are:
17
+
18
+ | Parent | Relation | Child | Meaning |
19
+ |---|---|---|---|
20
+ | chat | `job` | job | A projected response was produced by a Workflow job. |
21
+ | chat | `agent_job` | job | A delegated tool call returned an agent whose `job=` receipt names the producer job. |
22
+ | job | `dependency` | job | A normal Scout Workflow dependency. |
23
+ | job | `log` | chat | A persisted agent conversation under `.files/*.chat`, `.files/*.society/**/*.chat`, or the legacy `.files/log/**/*.chat`. |
24
+ | chat | `log` | chat | A saved agent conversation under the chat's own `.files` sidecar, same three families (root copy excluded). |
25
+ | job | `result` | chat | The job result is itself a chat file. |
26
+
27
+ Relations describe root-outward discovery. A renderer may reverse `job` or `dependency` when drawing natural data flow.
28
+
29
+ The `log` relation covers exactly three file families under `.files`, nothing else:
30
+
31
+ - `.files/*.chat` — the new top-level chat files (`agent.chat` by default, `worker.chat`/`critic.chat` for named agents);
32
+ - `.files/*.society/**/*.chat` — the new society tree (nested societies keep the plain `society` basename deeper down);
33
+ - `.files/log/**/*.chat` — the **legacy** layout, still read for back-compat; nothing writes it anymore and old files are never migrated.
34
+
35
+ Restart snapshots written by `Agent#start` live under `.files/resets/<timestamp>.chat`, directly under `.files` and **outside** all three families: they are recovery artifacts, not logs, and provenance traversal does not follow them, for jobs and for chats alike. Results of both layouts are de-duplicated and sorted, so a files dir holding both layouts is visited exactly once per chat.
36
+
37
+ The two `log` parents are deliberately asymmetric:
38
+
39
+ - a **job** root includes its own top-level `<job>.files/<name>.chat` (`agent.chat` and friends, and the legacy `<job>.files/log/agent.chat`) as a real log node; renderers such as `scout-ai llm prov` hide it from the tree because it duplicates the job node itself;
40
+ - a **chat** root excludes its root copy — every top-level `<save_file>.files/<name>.chat` and the legacy `<save_file>.files/log/agent.chat` — because the save mechanism writes a full copy of the root conversation there and including it would duplicate the root as its own child. The exclusion is for the **top level** only: society conversations under `<name>.society/<agent>/<conversation>/agent.chat` (and the legacy `log/society/<agent>/<conversation>/agent.chat`) are also named `agent.chat` and **are** included.
41
+
42
+ Imported and continued chats are **not** provenance relations. They are a chat-compilation concern resolved during `Chat.parse` and `LLM.chat`. The persisted `.chat` file already contains the full inlined conversation. Provenance traversal therefore never follows `import`, `continue`, or `last` chat references.
43
+
44
+ ## Safe persisted-chat loading
45
+
46
+ Provenance inspection uses `Chat.load(file)`. It parses the persisted messages without compiling the chat. It therefore does not execute `task`, `job`, `file`, `import`, tool, or other control roles.
47
+
48
+ Do not use `LLM.chat` to inspect historical evidence: that method compiles control roles for inference.
49
+
50
+ ## Structural traversal
51
+
52
+ `Chat.traverse_provenance` is the authoritative traversal primitive. It accepts a chat file or Step and yields native `Path` and `Step` values:
53
+
54
+ Chat.traverse_provenance(root, root_type: :chat) do |
55
+ kind, object, parent_kind, parent, relation, first_visit
56
+ |
57
+ # kind is :chat or :job
58
+ end
59
+
60
+ Without a block it returns an Enumerator.
61
+
62
+ The root has nil parent and relation. Every structural edge is yielded. When a shared dependency or cycle reaches an already visited node, `first_visit` is false and the node is not expanded again. Node identity includes both kind and path, because a chat-producing Step and its result chat can share a filesystem path.
63
+
64
+ A chat node expands its own `.files` sidecar logs with the `log` relation, exactly like a job does; a job node expands `dependency`, `log`, and `result`. Node identity is `[kind, realpath]`, so a file reachable through both a job log glob and a chat sidecar glob collapses to a single node (first visit wins).
65
+
66
+ `root_type` decides how the root is loaded. The prov CLI always passes it explicitly, from its own job detection (`.info` sidecar present). Callers that hand over a bare path string should know that traversal infers `:chat` when `Step.type` is empty for that path, which is the case for plain persisted chat files; pass `root_type: :job` whenever the root is known to be a Step, so the node is loaded with `Step.load` and expanded through the job relations instead of the chat ones.
67
+
68
+ By default, loading and resolution errors are raised. Analytical callers that need partial results can supply `on_error`:
69
+
70
+ warnings = []
71
+ records = Chat.traverse_provenance(
72
+ root,
73
+ on_error: ->(error, kind, object, relation, reference) {
74
+ warnings << [error, kind, object, relation, reference]
75
+ }
76
+ ).to_a
77
+
78
+ This distinguishes absent evidence from evidence that could not be read.
79
+
80
+ The `follow` option can restrict traversal to selected relations. The supported values are `job`, `dependency`, `log`, `result`, and `agent_job`.
81
+
82
+ ### Collectors
83
+
84
+ Thin collectors use the same traversal:
85
+
86
+ - `Chat.provenance_chat_files(root)` returns every discovered chat path;
87
+ - `Chat.provenance_jobs(root)` returns every discovered Step;
88
+ - `Chat.provenance_edges(root)` returns typed structural edges;
89
+ - `Chat.tokens(root)` sums direct inference usage from discovered chats.
90
+
91
+ `Chat.provenance` remains as a compatibility collector. New code should use the traversal or typed edges because the compatibility Hash does not represent job nodes and relation types fully.
92
+
93
+ ### Direct-neighbour helpers
94
+
95
+ Direct readers do not recurse:
96
+
97
+ - `Chat.direct_job_chat_files(job)` returns chat logs owned directly by a job;
98
+ - `Chat.direct_chat_sidecar_files(path)` returns chat logs owned directly by a persisted chat's `.files` sidecar (all three families above), excluding the top-level root copies `<save_file>.files/<name>.chat` and the legacy `<save_file>.files/log/agent.chat`;
99
+ - `Chat.job_result_chat_file(job)` returns a chat result when present.
100
+
101
+ Recursion belongs only to `traverse_provenance`.
102
+
103
+ ## Meta messages and inference segments
104
+
105
+ A meta message has role `meta` and content serialized as key/value pairs. Two important forms are:
106
+
107
+ 1. **Direct inference metadata**, containing fields such as `pt`, `ct`, and `tt`.
108
+ 2. **Job projection metadata**, containing `job=<path>` and no direct inference cost.
109
+
110
+ `Chat.project(job, messages)` prepends exactly one producer marker (`job=<path>`, no token fields) and then keeps the response messages in their original order, with the per-inference metas inline, adjacent to the function calls they produced. The projected copy is therefore self-contained for attribution: it carries the same `inference_id` and token fields as the agent log, so a reader does not need to go back to the log to know what a delegated response cost.
111
+
112
+ Consequences of this contract:
113
+
114
+ - Duplicated evidence is the norm. The same inference appears in the agent log, in the job result chat, and in any parent conversation that consumed the job chat. `Chat.trace_indices` collapses the copies by `inference_id` (falling back to the digest-based lineage id for legacy metas without `inference_id`), so `Chat.token_totals` counts each inference once. Legacy lineages cannot always be merged across chats; precise deduplication relies on `inference_id`, which every new inference carries.
115
+ - The marker is deliberately **separate** from the inference metas: `Chat.direct_entries` excludes metas carrying `job=`, so folding the producer path into an inference meta would silently drop that segment from direct token counting.
116
+ - `reas` (reasoning summaries) are stripped from projected copies by default, keeping `inference_id` and the token fields while avoiding the bulk of the projection size cost. Set `chat.project.keep_reas` (env `CHAT_PROJECT_KEEP_REAS`) to keep them.
117
+ - Projection is idempotent: re-projecting an already-projected chat (the consumption path in `LLM::Agent#ask`) keeps exactly one `job=` marker and never duplicates inference metas.
118
+
119
+ ### Token fields
120
+
121
+ | Field | Meaning |
122
+ |---|---|
123
+ | `pt`, `ct`, `tt` | Prompt, completion, and total tokens for one request. |
124
+ | `cct`, `cwt`, `rt` | Cache-hit, cache-write, and reasoning tokens for one request. |
125
+ | `*_c` | Running total represented by this chat. It is a checkpoint, not an additive event. |
126
+ | `*_s` | Process/thread session snapshot. It is not attributable by itself. |
127
+ | `inference_id` | Scout-generated identity for one actual backend request. |
128
+ | `provider_response_id` | Provider response identity when available. |
129
+ | `job` | Producer Step for a projected response segment. |
130
+ | `orphan` | Request produced no persisted message (reasoning-only round; its segment covers zero messages). Real cost, marked explicitly at meta-creation time by the backend. |
131
+ | `reas` | Optional reasoning summary; stripped from projected copies unless `chat.project.keep_reas` is set. |
132
+
133
+ Every new direct inference receives a locally generated `inference_id`. This distinguishes genuinely repeated requests even when their conversation, response, and token counts are identical. Copied chat history retains the original ID and is counted once.
134
+
135
+ Legacy chats without an inference ID fall back to conversational lineage deduplication. Reports that require precision should expose whether an entry used `inference_id` or `legacy_lineage` deduplication.
136
+
137
+ Never sum `*_c` or `*_s` snapshots. Sum direct fields from deduplicated direct inference segments.
138
+
139
+ ## Message identity and location
140
+
141
+ Scout-AI distinguishes two concepts:
142
+
143
+ - a **lineage ID** identifies equivalent conversational content;
144
+ - a **message address** identifies one persisted location as `[chat_path, index]`.
145
+
146
+ `chat.message_index(source: path)` includes both. Meta messages do not advance conversational lineage because providers do not receive them.
147
+
148
+ Use lineage IDs for detecting copied history. Use addresses to retrieve exact persisted messages.
149
+
150
+ ## Response tracing
151
+
152
+ `Chat.trace_chats(chats)` groups messages into response segments. A meta message opens a segment; another meta or a user/system turn closes it.
153
+
154
+ `Chat.trace_chat_sources(path_to_chat)` is the source-aware form. Its records include:
155
+
156
+ - `lineage_id`;
157
+ - `inference_id` when present;
158
+ - `deduplication`, either `inference_id` or `legacy_lineage`;
159
+ - `meta_address`;
160
+ - covered message lineage IDs;
161
+ - covered `message_addresses`;
162
+ - parsed metadata;
163
+ - orphan status.
164
+
165
+ An orphan segment covers zero messages: the request produced no persisted message (typically a reasoning-only round whose output was consumed internally before the meta was written). Such metas are marked `orphan=true` at creation time by the backend, so the persisted meta is self-explanatory; `trace_indices` derives the same fact independently. The marker is inert for accounting — orphan requests still carry real token cost.
166
+
167
+ `Chat.direct_entries(chats)` selects direct inference segments. `Chat.token_totals(chats)` sums all canonical direct token fields.
168
+
169
+ ## Tool-call analysis
170
+
171
+ `Chat.tool_calls(chat, source: path)` pairs `function_call` and `mcp_call` messages with `function_call_output` messages by call ID. It returns plain Hash records containing call/output addresses, parsed records, arguments, and output content.
172
+
173
+ Pairing is structural. Success interpretation is separate:
174
+
175
+ call = Chat.tool_calls(chat, source: path).first
176
+ status = Chat.tool_call_status(call)
177
+
178
+ The common status policy treats:
179
+
180
+ - missing output as unknown;
181
+ - JSON containing `exception` as failure;
182
+ - JSON containing non-zero `exit_status` as failure;
183
+ - any other persisted output as success.
184
+
185
+ Provider call IDs are scoped to a chat; do not assume they are globally unique across files.
186
+
187
+ Calls named `ask` or `hand_off_to_*` provide semantic evidence of delegation. Workflow-backed calls have structural job/log links. A socialized call's association with a society log may still be inferred from naming conventions, so reports should label that association as inferred rather than authoritative.
188
+
189
+ ## Delegated agent receipts (`meta` / legacy `agent_meta`)
190
+
191
+ When a tool returns an `LLM::Agent`, `LLM.process_calls` embeds the child agent's inference evidence in the parent `function_call_output` JSON envelope. The envelope is generic: it is produced for any tool returning an agent, not only for `ask`.
192
+
193
+ Two receipt formats exist, and the reader accepts both:
194
+
195
+ - **Current format — the `meta` key.** The writer deserializes the child agent's `meta` messages (`LLM.meta_receipt_from_messages`) and emits an Array of plain field Hashes, each already parsed:
196
+
197
+ ```
198
+ function_call_output: {"name":"ask","content":"child answer","id":"call_1","meta":[{"pt":100,"ct":50,"tt":150,"inference_id":"aaa"},{"job":"Worker/ask/Default_x"}]}
199
+ ```
200
+
201
+ An entry carries either the child's direct inference metadata (`pt`, `ct`, `tt`, ..., `inference_id`) or a producer reference (`job=<path>` as a field). Entries that would carry no fields are dropped by the writer.
202
+
203
+ - **Legacy format — the `agent_meta` key.** Older data stores serialized meta messages:
204
+
205
+ ```
206
+ "agent_meta":[{"role":"meta","content":"pt=100 ct=50 tt=150 inference_id=aaa"}, ...]
207
+ ```
208
+
209
+ This shape is **no longer written**, but it is still read: each `content` String is parsed with `Chat.parse_meta`, so historical chats remain fully addressable.
210
+
211
+ When both keys are present in one envelope the current `meta` key wins and the legacy one is ignored (the current writer emits exactly one of the two).
212
+
213
+ See [../../research/agent-meta-provenance-integration-plan.md](../../research/agent-meta-provenance-integration-plan.md) for the design record of the original (serialized) shape; the `meta` key replaced it in commit `efd8ebbc`.
214
+
215
+ Receipts are embedded provenance evidence, **not** parent-chat messages, and must never be injected into the parent chat. `Chat#meta`, `chat.role_messages(:meta)`, and `Chat.token_totals([chat])` keep describing the local chat only. A receipt is read as an observation attached to the paired tool output.
216
+
217
+ ### Receipt extraction helpers
218
+
219
+ - `Chat.agent_meta_evidence(chat, source: nil, warnings: nil)` returns one Hash per valid receipt entry across the paired tool outputs of a chat. Pairing is delegated to `Chat.tool_calls`; raw text is never scanned. Records carry `origin: :agent_meta` (for **both** formats), the `meta` fields (already deserialized for current-format entries, parsed from the legacy content String for legacy entries), `source`, `output_address`, `evidence_address`, `call_id`, `tool_name`, `agent_meta_index`, and `raw_message`. The evidence address suffix mirrors the persisted key: `[:meta, index]` for current-format entries and `[:agent_meta, index]` for legacy ones, so an address always points at the JSON element that is actually on disk. `raw_message` is nil for current-format entries (they are born deserialized) and `{role:, content:}` for legacy entries.
220
+ - `Chat.meta_evidence(chat, source: nil, warnings: nil)` returns the local `meta` messages (`origin: :chat_meta`, with `meta_address`) followed by the receipt records.
221
+ - `Chat.agent_meta_job_references(chat, source: nil, warnings: nil)` filters receipt records whose parsed meta has a `job` key and adds the reference at the top level as `job:`.
222
+
223
+ Malformed receipts (the receipt key not holding an Array; an entry that is not a Hash, has the wrong role, has non-String content, parses to nothing, or — current format only — is a field Hash with no fields) are skipped and never reinterpreted as provenance. When the caller supplies a `warnings` Array, each malformed item appends one warning Hash with the reason, the output address, `call_id`, `tool_name`, and the raw entry. Warning reasons are:
224
+
225
+ | Reason | Applies to | Meaning |
226
+ |---|---|---|
227
+ | `:not_an_array` | both | The receipt value is not an Array. |
228
+ | `:not_a_hash` | both | An entry is not a Hash. |
229
+ | `:invalid_role` | legacy | An entry's role is not `meta`. |
230
+ | `:invalid_content` | legacy | An entry's content is not a String. |
231
+ | `:unparseable_meta` | legacy | The content String parses to no fields. |
232
+ | `:empty_meta` | current | An already-deserialized field Hash carries no fields. |
233
+
234
+ Malformed warnings mirror the persisted key in their `evidence_address`, and always carry `origin: :agent_meta` regardless of format.
235
+
236
+ ### The `agent_job` relation
237
+
238
+ `Chat::PROVENANCE_RELATIONS` includes `agent_job`: chat to delegated producer job, resolved from `job=` receipts. The child is a normal Step and follows `dependency`, `log`, and `result` as usual. A job reference whose Step path and `.info` sidecar both do not exist is not followed and is reported instead.
239
+
240
+ Diagnostics go through `Chat.provenance_error` with relation `:agent_job`; the error itself is a plain `ScoutException` whose message is built by `Chat.agent_meta_error_message`, and every structured fact (enclosing chat path, tool output address, receipt address, call id, tool name, malformed entry, reference, reason) travels in the `on_error` reference Hash. In strict mode (no `on_error`) a malformed receipt raises; with `on_error` each problem is reported once per receipt, while the rest of the chat's provenance still expands. Only output JSON that parses to a Hash carrying an explicit receipt key (current `meta`, legacy `agent_meta`) is ever inspected: unparseable tool outputs are never scanned for the substring `agent_meta`.
241
+
242
+ ### Provenance-aware token accounting
243
+
244
+ `Chat.provenance_token_events(root, warnings: nil, **options)` returns one Hash per deduplicated direct inference event across discovered chats and their receipts, with `inference_id`, `identity`, `deduplication`, canonical `meta`, `tokens`, `evidence`, and `conflict`:
245
+
246
+ - identity priority: `inference_id`, then `provider_response_id`, then conversational lineage for chat-side legacy metas, then the receipt evidence address itself (`:receipt_unresolved`, never merged, so legacy receipt data may overcount);
247
+ - canonical evidence: `:chat_meta` beats `:agent_meta`; within one origin the first in discovery order wins. Tokens come from the canonical evidence only, never from a sum of duplicates;
248
+ - conflicting evidence that shares an identity keeps every evidence record with its own meta, counts only the canonical one, sets `conflict: true`, and appends an `:identity_conflict` warning when a `warnings` Array was supplied. `strict: true` raises instead.
249
+
250
+ Deduplication happens exactly once, inside this collector: chat-side records are collected with `deduplicate: false`, so an inference persisted in several saved chats/logs plus one receipt yields a single event whose `evidence` array lists every address.
251
+
252
+ A missing `provider_response_id` in one evidence record and a present one in another is *incomplete evidence*, not a conflict: such events set `incomplete_evidence: true`, are counted normally, and never trigger conflict warnings. Only two different non-empty `provider_response_id` values (or disagreement on `pt`/`ct`/`tt`) conflict. A conflicting event still contributes its canonical tokens, so any total containing conflicts is best-effort, not authoritative; `Chat.provenance_token_totals(root, conflicts: hash)` reports that flag explicitly.
253
+
254
+ `Chat.provenance_token_totals(root, scope:)` sums event tokens by scope. Scopes are **evidence coverage**, not a partition of cost: an event stored both in a saved child log and in a receipt belongs to `:chat_evidence` and to `:receipt_evidence`, so those two must never be summed together. `:deduplicated_total` (default) counts every event once and `:receipt_only` is the disjoint delegated contribution with no saved-chat evidence. `Chat.tokens(root)` delegates to the collector, so provenance aggregates include receipt-only child usage without double counting when the same child inference is also persisted in a job.
255
+
256
+ ## Workflow failures and partial provenance
257
+
258
+ Traversal visits jobs regardless of `done?`. Error and aborted Steps may still have dependencies, Step info, results, or partial agent logs. Status and exception details remain authoritative in Step info.
259
+
260
+ Backend request failures currently preserve emergency chat/options/meta snapshots through the backend exception mechanism. If these snapshots are later moved under an owning Step's files directory or linked from Step info, normal provenance traversal can expose them without introducing a separate Session abstraction.
261
+
262
+ ## `scout-ai llm prov`
263
+
264
+ The `prov` command consumes `Chat.traverse_provenance` once and then separates:
265
+
266
+ 1. discovery of nodes and typed edges;
267
+ 2. direct token analysis;
268
+ 3. tree, compact flow, DOT, and plot rendering.
269
+
270
+ Usage:
271
+
272
+ scout-ai llm prov path/to/chat
273
+ scout-ai llm prov path/to/chat --flow
274
+ scout-ai llm prov path/to/chat --dot flow.dot
275
+ scout-ai llm prov path/to/chat --plot flow.svg
276
+ scout-ai llm prov path/to/chat --evidence
277
+
278
+ Root classification uses the `.info` sidecar only: a path is a job iff `<path>.info` exists, and is loaded with `Step.load`. The presence of a `.files` sidecar is **not** evidence of a job, because saved agent chats also carry one; a chat root is simply `Path.setup`'d.
279
+
280
+ The default tree is a spanning-tree presentation of a DAG. Repeated nodes are displayed as seen references rather than recursively expanded. Compact and graphical flows choose natural data-flow arrow direction during rendering without changing traversal semantics.
281
+
282
+ Job token values describe direct chat logs owned by that job. They do not silently include the complete dependency subtree.
283
+
284
+ Delegated calls are reported from receipts (current `meta` key or legacy `agent_meta` key; both formats are accepted transparently). The tree labels a job reached through a receipt as `delegated-job`, adds one `delegated receipt: N events, total=<tt>, <tools>` annotation line under chats that carry receipts, and `--component` prints `scope local:` / `scope receipt:` / `scope aggregate:` lines when receipt evidence exists. Flow and DOT render receipt edges as `delegated_result`.
285
+
286
+ `--evidence` prints the deduplicated direct inference events behind the totals: identity, raw token values, evidence locations (parent output address plus call id), and status (`counted once`, `receipt-only`, `legacy unresolved`, `conflict`), followed by receipt-only, legacy-unresolved, identity-conflict, and job-projection sections. Receipt addresses print as `base:idx[meta,i]` for current-format entries and `base:idx[agent_meta,i]` for legacy ones, mirroring the persisted key. Receipt problems and identity conflicts are listed in the trailing warnings block.
287
+
288
+ ## ChatAnalyst
289
+
290
+ ChatAnalyst uses the same core traversal. It does not define Session, ChatGraph, ProvenanceContext, or node wrapper classes. Each Workflow task collects temporary report state in ordinary Hashes and Arrays and applies shared Chat operations for message indexing, tracing, tool-call pairing, and token accounting.
291
+
292
+ This keeps responsibilities separate:
293
+
294
+ - Scout-AI owns persisted structural navigation and Chat analysis primitives;
295
+ - ChatAnalyst owns agent-oriented JSON reports;
296
+ - `prov` owns human and Graphviz rendering;
297
+ - Workflow Step info owns execution lifecycle and failure provenance.
298
+
299
+ ChatAnalyst is expected to consume the receipt primitives above — `Chat.agent_meta_evidence`, `Chat.meta_evidence`, `Chat.agent_meta_job_references`, and the `:agent_job` relation — instead of re-deriving delegated inference evidence. That update is pending and out of this change.
300
+
301
+ ## Key source files
302
+
303
+ | File | Responsibility |
304
+ |---|---|
305
+ | `lib/scout/llm/chat/provenance.rb` | Structural traversal, direct neighbours, collectors, and token events. |
306
+ | `lib/scout/llm/chat/agent_meta.rb` | Delegated-agent receipt evidence extraction. |
307
+ | `lib/scout/llm/chat/process/meta.rb` | Meta parsing, lineage, source-aware tracing, projections, and token totals. |
308
+ | `lib/scout/llm/chat/tool_calls.rb` | Tool-call pairing and common status interpretation. |
309
+ | `lib/scout/llm/backends/default.rb` | Direct token metadata and inference identities. |
310
+ | `lib/scout/llm/agent/workflow.rb` | Agent log persistence and chat-task projection. |
311
+ | `scout_commands/llm/prov` | Tree, flow, DOT, and plot rendering. |
312
+
313
+ ## Cross-references
314
+
315
+ - [ChatLifecycle.md](ChatLifecycle.md) — Chat data and compilation.
316
+ - [DelegationInternals.md](DelegationInternals.md) — Socialized and delegated agents.
317
+ - [../../research/provenance-navigation-design.md](../../research/provenance-navigation-design.md) — Investigation, alternatives, and migration rationale.
@@ -0,0 +1,345 @@
1
+ # Building Agents
2
+
3
+ This page explains how to create, configure, and use agents in Scout-AI. It is
4
+ intended for workflow authors who want to build reusable, stateful AI
5
+ assistants with tools and delegation.
6
+
7
+ **You should read this if:** you understand chats and want to take the next
8
+ step to persistent, tool-using agents.
9
+
10
+ ---
11
+
12
+ ## What an agent is
13
+
14
+ An agent is a **stateful wrapper** around a chat. It remembers:
15
+
16
+ - **A start chat** — the system prompt and initial context that prefix every
17
+ conversation.
18
+ - **Tools** — workflows, knowledge bases, and MCP servers the agent can call.
19
+ - **Options** — which endpoint and model to use, output format, etc.
20
+
21
+ You interact with an agent through a DSL that feels like building a chat: you
22
+ add messages, call `chat` to get a response, and the agent maintains the
23
+ conversation history for you.
24
+
25
+ ---
26
+
27
+ ## Creating an agent in Ruby
28
+
29
+ The simplest way is the `LLM.agent` factory:
30
+
31
+ ```ruby
32
+ require 'scout-ai'
33
+
34
+ agent = LLM.agent(endpoint: :openai)
35
+ ```
36
+
37
+ ### Setting the system prompt
38
+
39
+ The system prompt lives on `start_chat`:
40
+
41
+ ```ruby
42
+ agent.start_chat.system "You are a helpful assistant that answers concisely."
43
+ ```
44
+
45
+ ### Starting a conversation
46
+
47
+ Call `start` to create a new conversation branch:
48
+
49
+ ```ruby
50
+ agent.start
51
+ ```
52
+
53
+ If the agent has a `save_file` **and** a prior non-empty saved chat, `start`
54
+ also snapshots the old conversation to
55
+ `<save_file>.files/resets/<timestamp>.chat` (colons stripped from the
56
+ timestamp, `_1`, `_2`, … suffixes on name collisions) before clearing it. The
57
+ snapshot is lazy — no reset directory is created when there is nothing to
58
+ snapshot — and non-fatal if it fails. Reset snapshots live directly under
59
+ `.files`, in none of the three directories the `:log` relation sweeps
60
+ (`*.files/*.chat`, `*.files/*.society/**`, the legacy `*.files/log/**`), so
61
+ provenance traversal ignores them: they are recovery artifacts, not logs.
62
+
63
+ ### Adding messages
64
+
65
+ Use role methods (forwarded to the underlying chat):
66
+
67
+ ```ruby
68
+ agent.user "What is the capital of France?"
69
+ ```
70
+
71
+ ### Getting a response
72
+
73
+ Call `chat` to send the conversation to the model and get a response:
74
+
75
+ ```ruby
76
+ puts agent.chat # => "Paris"
77
+ ```
78
+
79
+ `chat` appends the model's response to the conversation, so subsequent calls
80
+ have full history:
81
+
82
+ ```ruby
83
+ agent.user "And its population?"
84
+ puts agent.chat # => "Approximately 2.2 million in the city proper."
85
+ ```
86
+
87
+ `chat` also triggers an **auto-save**: when `agent.save_file` is set, the full
88
+ conversation is serialized recursively at the end of every `chat` call (see
89
+ below). A bare `agent.ask(...)` does **not** auto-save — only `chat` does.
90
+ `LLM.ask` goes through `agent.chat`, so it inherits the same behavior.
91
+
92
+ ---
93
+
94
+ ## Persisting agents: `save_file`, `save`, and auto-save
95
+
96
+ Agents persist their conversations on demand, through three cooperating
97
+ pieces:
98
+
99
+ - **`agent.save_file = path`** designates where this agent's own chat lives
100
+ (for example `job.files/agent.chat` for a `chat_task` job, or
101
+ `chat.files/agent.society/<agent_name>/<conversation>/agent.chat` for a
102
+ socialized specialist; named agents write `worker.chat` /
103
+ `worker.society`).
104
+ - **`agent.save(path = nil)`** writes the agent's **full** `current_chat` —
105
+ not just the new messages — to `save_file` (or to `path`, when given). It
106
+ raises `ScoutException` when neither is set. Files whose content is already
107
+ current are reported but not rewritten; a changed file is refreshed, and the
108
+ method returns the sorted list of absolute paths whose content is now
109
+ current. While saving, every reachable child agent gets its `save_file`
110
+ assigned too, so later independent child turns auto-save in place.
111
+ - **Auto-save on `chat`**: an `ensure` hook in `Agent#chat` performs a full
112
+ recursive save, but only when `save_file` is set.
113
+
114
+ The canonical layout is decided by *location*, not configuration:
115
+
116
+ - a root chat saved at `p.chat` puts its society at
117
+ `p.chat.files/agent.society/<agent_name>/<conversation>/agent.chat`
118
+ (`<name>.society` for named agents; only the depth-0 society directory is
119
+ name-derived, deeper levels keep the plain `society` basename);
120
+ - a nested chat already at `.../society/<a>/<c>/<file>` stores its own
121
+ children in the **sibling** directory `.../society/<a>/<c>/society/…` — a
122
+ nested `agent.chat` never grows a second `.files` tree;
123
+ - `chat_task` jobs and `scout-ai agent ask` always write the agent's own chat
124
+ at `<chat_or_job_path>.files/<name>.chat`: `agent.chat` for the default
125
+ agent, `worker.chat`/`critic.chat` for named ones.
126
+
127
+ **Legacy layout (read-only).** Older versions wrote
128
+ `<path>.files/log/agent.chat` and `log/society/…`. Nothing writes there
129
+ anymore, old files are not migrated, and provenance still reads them.
130
+
131
+ Saving is lazy (nothing is created eagerly; parent directories are created on
132
+ demand) and non-fatal (a failure logs a warning and the run continues). Cycle
133
+ protection is built in: already-visited paths, already-seen agents, and a
134
+ depth limit of 32 (warn, never raise) keep recursive saves from looping.
135
+
136
+ ---
137
+
138
+ ## Agent as a named directory
139
+
140
+ Instead of configuring an agent in Ruby, you can define it as a directory:
141
+
142
+ ```
143
+ Agent/
144
+ Researcher/
145
+ start_chat # system prompt + tool declarations
146
+ workflow.rb # optional: Scout workflow providing tools
147
+ knowledge_base/ # optional: KB for retrieval
148
+ python/ # optional: Python workflow tasks
149
+ ```
150
+
151
+ ### The start_chat file
152
+
153
+ This is a chat file (see [WritingChats.md](WritingChats.md)) containing the
154
+ system prompt and any initial configuration:
155
+
156
+ ```text
157
+ system:
158
+
159
+ You are a research assistant. Use the search tool to find information.
160
+ Always cite your sources.
161
+
162
+ endpoint: anthropic
163
+ model: claude-sonnet-4-20250514
164
+
165
+ introduce: SearchWorkflow
166
+ ```
167
+
168
+ ### Loading and using a named agent
169
+
170
+ From Ruby:
171
+
172
+ ```ruby
173
+ agent = LLM.load_agent('Researcher')
174
+ agent.start
175
+ agent.user "Find papers about protein folding."
176
+ puts agent.chat
177
+ ```
178
+
179
+ From the CLI:
180
+
181
+ ```bash
182
+ scout-ai agent ask Researcher "Find papers about protein folding."
183
+ ```
184
+
185
+ ### Agent discovery locations
186
+
187
+ Scout-AI looks for named agents in several places (first match wins):
188
+
189
+ 1. `Scout.workflows[name]`
190
+ 2. `Scout.Agent[name]`
191
+ 3. `Scout.var.Agent[name]`
192
+ 4. `Scout.chats.Agent[name]`
193
+ 5. `Scout.chats[name]`
194
+
195
+ ---
196
+
197
+ ## Giving agents tools
198
+
199
+ ### Workflow tools
200
+
201
+ If your agent has a workflow, all its tasks become callable tools:
202
+
203
+ ```ruby
204
+ agent = LLM::Agent.new(workflow: 'Baking', endpoint: :openai)
205
+ agent.start
206
+ agent.user "Bake muffins using the tool"
207
+ puts agent.chat # the model calls the 'bake' task automatically
208
+ ```
209
+
210
+ You can also define a workflow inline:
211
+
212
+ ```ruby
213
+ agent.workflow do
214
+ task :greet => :string do |name = nil|
215
+ "Hello, #{name}!"
216
+ end
217
+ end
218
+ ```
219
+
220
+ ### Declaring tools in the start chat
221
+
222
+ Use `tool:` or `introduce:` roles in the start_chat file:
223
+
224
+ ```text
225
+ system:
226
+
227
+ You are a code analyst.
228
+
229
+ introduce: CodeAnalyzer
230
+ ```
231
+
232
+ ### Knowledge base and MCP tools
233
+
234
+ ```text
235
+ kb: my_database [genes proteins]
236
+ mcp: https://api.example.com/mcp/
237
+ ```
238
+
239
+ See [ToolCalling.md](ToolCalling.md) for the complete tool declaration syntax.
240
+
241
+ ---
242
+
243
+ ## Options
244
+
245
+ Options control which endpoint, model, and format the agent uses. Set them
246
+ at construction:
247
+
248
+ ```ruby
249
+ agent = LLM.agent(endpoint: :anthropic, model: 'claude-sonnet-4-20250514')
250
+ ```
251
+
252
+ Or through the DSL:
253
+
254
+ ```ruby
255
+ agent.option :model, 'gpt-4o'
256
+ agent.option :temperature, 0.7
257
+ ```
258
+
259
+ Common options:
260
+
261
+ | Option | Purpose |
262
+ |--------|---------|
263
+ | `endpoint:` | Named endpoint configuration |
264
+ | `model:` | Model identifier |
265
+ | `format:` | Output format (`:json`, `:text`, or a JSON schema hash) |
266
+ | `persist:` | Whether to cache inference results (default `true`) |
267
+
268
+ ---
269
+
270
+ ## Structured outputs
271
+
272
+ ### JSON extraction
273
+
274
+ ```ruby
275
+ agent.start
276
+ agent.user 'Return {"content": ["apple", "banana", "cherry"]}'
277
+ result = agent.json # => ["apple", "banana", "cherry"]
278
+ ```
279
+
280
+ ### JSON with a schema
281
+
282
+ ```ruby
283
+ schema = {
284
+ name: 'answer',
285
+ type: 'object',
286
+ properties: {
287
+ judgement: { type: :boolean },
288
+ notes: { type: :string }
289
+ },
290
+ required: [:judgement]
291
+ }
292
+
293
+ agent.json_format(schema)
294
+ ```
295
+
296
+ ### Iteration helpers
297
+
298
+ `iterate` extracts a list from the model and processes each item:
299
+
300
+ ```ruby
301
+ agent.iterate("List 3 action items") do |action|
302
+ puts "- #{action}"
303
+ end
304
+ ```
305
+
306
+ ---
307
+
308
+ ## Error handling
309
+
310
+ Set a `process_exception` callback to intercept errors during inference:
311
+
312
+ ```ruby
313
+ agent.process_exception = Proc.new do |exception|
314
+ if exception.message =~ /rate limit/i
315
+ sleep 5
316
+ true # retry
317
+ else
318
+ false # re-raise
319
+ end
320
+ end
321
+ ```
322
+
323
+ ---
324
+
325
+ ## Common mistakes
326
+
327
+ - **Forgetting to call `start`**: Without `start`, messages go to a lazy
328
+ default branch. Calling `start` explicitly makes the lifecycle clear.
329
+ - **Putting user messages on `start_chat`**: `start_chat` is the *seed* — it
330
+ should contain system prompts and configuration, not the actual question.
331
+ - **Expecting `ask` to append to the conversation**: Use `chat` for the
332
+ stateful pattern (ask + append + return text). `ask` is the lower-level
333
+ primitive.
334
+ - **Defining wrapper methods on Agent**: The agent forwards unknown methods
335
+ to its chat automatically. You don't need to write wrappers.
336
+
337
+ ---
338
+
339
+ ## Next steps
340
+
341
+ - [ToolCalling.md](ToolCalling.md) — detailed tool configuration.
342
+ - [Delegation.md](Delegation.md) — multi-agent systems.
343
+ - [RunningInference.md](RunningInference.md) — endpoint configuration.
344
+ - [MultiAgentWorkflows.md](MultiAgentWorkflows.md) — orchestrating agents in
345
+ Scout workflows.