aia 1.1.1 → 2.0.0.0.pre.alpha

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (169) hide show
  1. checksums.yaml +4 -4
  2. data/.envrc +5 -1
  3. data/.loki +231 -0
  4. data/.quality/flay_baseline.txt +1 -0
  5. data/.quality/flog_baseline.txt +29 -0
  6. data/.quality/reek_baseline.txt +80 -0
  7. data/.rubocop.yml +116 -0
  8. data/.version +1 -1
  9. data/CHANGELOG.md +259 -50
  10. data/IMPLEMENTATION_PLAN.md +506 -0
  11. data/README.md +266 -238
  12. data/Rakefile +118 -5
  13. data/architecture_review.md +314 -0
  14. data/bin/aia +16 -0
  15. data/docs/AGENTS.md +40 -0
  16. data/docs/advanced-prompting.md +67 -3
  17. data/docs/cli-reference.md +312 -56
  18. data/docs/configuration.md +130 -19
  19. data/docs/contributing.md +56 -2
  20. data/docs/directives-reference.md +593 -78
  21. data/docs/faq.md +85 -3
  22. data/docs/guides/available-models.md +1 -1
  23. data/docs/guides/basic-usage.md +6 -6
  24. data/docs/guides/chat.md +40 -16
  25. data/docs/guides/crew.md +239 -0
  26. data/docs/guides/executable-prompts.md +1 -1
  27. data/docs/guides/index.md +1 -0
  28. data/docs/guides/models.md +15 -0
  29. data/docs/index.md +29 -2
  30. data/docs/installation.md +44 -17
  31. data/docs/mcp-integration.md +40 -0
  32. data/docs/prompt_management.md +85 -86
  33. data/docs/security.md +47 -0
  34. data/docs/special_projects_guide.md +386 -0
  35. data/docs/tools-and-mcp-examples.md +23 -0
  36. data/docs/workflows-and-pipelines.md +84 -7
  37. data/examples/.gitignore +1 -0
  38. data/examples/00_setup_aia.sh +27 -44
  39. data/examples/11_multi_model.sh +4 -14
  40. data/examples/12_token_usage.sh +3 -12
  41. data/examples/18_tools.sh +10 -2
  42. data/examples/22_chat_mode.sh +0 -10
  43. data/examples/23_verify.sh +139 -0
  44. data/examples/24_decompose.sh +139 -0
  45. data/examples/25_spawn.sh +139 -0
  46. data/examples/26_debate.sh +97 -0
  47. data/examples/27_mention_routing.sh +157 -0
  48. data/examples/28_model_switching.sh +106 -0
  49. data/examples/29_agent_harness.sh +177 -0
  50. data/examples/README.md +65 -0
  51. data/examples/advanced_multi_robot_capabilities_without_examples.md +106 -0
  52. data/examples/aia_config.yml +1 -1
  53. data/examples/aia_config_orchestrator.yml +45 -0
  54. data/examples/common.sh +18 -6
  55. data/examples/context/tech_stack.md +2 -2
  56. data/examples/prompts_dir/roles/orchestrator.md +21 -0
  57. data/examples/requirements/sinatra_taskflow_app.md +139 -0
  58. data/examples/rules/01_classify_ruby.rb +16 -0
  59. data/examples/rules/02_prefer_claude_for_code.rb +19 -0
  60. data/examples/rules/03_gate_prompt_length.rb +19 -0
  61. data/examples/rules/04_tool_selection.rb +41 -0
  62. data/examples/rules/README.md +30 -0
  63. data/examples/run_all.sh +48 -15
  64. data/examples/tools/word_count_tool.rb +1 -1
  65. data/lib/AGENTS.md +57 -0
  66. data/lib/aia/chat_loop.rb +304 -164
  67. data/lib/aia/config/cli_parser.rb +174 -111
  68. data/lib/aia/config/defaults.yml +62 -33
  69. data/lib/aia/config/mcp_parser.rb +39 -46
  70. data/lib/aia/config/model_spec.rb +34 -2
  71. data/lib/aia/config/validator.rb +108 -142
  72. data/lib/aia/config.rb +110 -145
  73. data/lib/aia/content_extractor.rb +153 -0
  74. data/lib/aia/cost_calculator.rb +38 -0
  75. data/lib/aia/crew.rb +164 -0
  76. data/lib/aia/debate_handler.rb +166 -0
  77. data/lib/aia/delegate_handler.rb +112 -0
  78. data/lib/aia/directive.rb +33 -18
  79. data/lib/aia/directive_processor.rb +16 -7
  80. data/lib/aia/directives/configuration_directives.rb +160 -20
  81. data/lib/aia/directives/context_directives.rb +38 -26
  82. data/lib/aia/directives/execution_directives.rb +136 -4
  83. data/lib/aia/directives/model_directives.rb +76 -34
  84. data/lib/aia/directives/trakflow_directives.rb +44 -0
  85. data/lib/aia/directives/utility_directives.rb +203 -6
  86. data/lib/aia/directives/web_and_file_directives.rb +96 -60
  87. data/lib/aia/errors.rb +15 -0
  88. data/lib/aia/fact_asserter.rb +27 -0
  89. data/lib/aia/fzf.rb +9 -31
  90. data/lib/aia/handler_context.rb +17 -0
  91. data/lib/aia/handler_protocol.rb +19 -0
  92. data/lib/aia/history_transfer.rb +55 -0
  93. data/lib/aia/input_collector.rb +3 -3
  94. data/lib/aia/layered_orchestrator.rb +448 -0
  95. data/lib/aia/logger.rb +24 -4
  96. data/lib/aia/mcp_config_normalizer.rb +35 -0
  97. data/lib/aia/mcp_connection_manager.rb +305 -0
  98. data/lib/aia/mcp_discovery.rb +44 -0
  99. data/lib/aia/mcp_grouper.rb +33 -0
  100. data/lib/aia/mcp_utility.rb +57 -0
  101. data/lib/aia/mention_router.rb +260 -0
  102. data/lib/aia/model_alias_registry.rb +97 -0
  103. data/lib/aia/model_switch_handler.rb +100 -0
  104. data/lib/aia/network_builder.rb +155 -0
  105. data/lib/aia/network_memory_manager.rb +55 -0
  106. data/lib/aia/patches/ruby_llm_streaming_error.rb +43 -0
  107. data/lib/aia/patches/ruby_llm_tool_error.rb +96 -0
  108. data/lib/aia/pipeline_orchestrator.rb +262 -0
  109. data/lib/aia/plugin_loader.rb +170 -0
  110. data/lib/aia/plugin_monitor.rb +208 -0
  111. data/lib/aia/prompt_decomposer.rb +157 -0
  112. data/lib/aia/prompt_handler.rb +19 -39
  113. data/lib/aia/robot_builder.rb +51 -0
  114. data/lib/aia/robot_factory.rb +334 -0
  115. data/lib/aia/robot_namer.rb +116 -0
  116. data/lib/aia/session.rb +83 -17
  117. data/lib/aia/session_tracker.rb +209 -0
  118. data/lib/aia/similarity_scorer.rb +39 -0
  119. data/lib/aia/skill_utils.rb +105 -1
  120. data/lib/aia/spawn_handler.rb +129 -0
  121. data/lib/aia/spawn_spec_parser.rb +65 -0
  122. data/lib/aia/special_mode_handler.rb +302 -0
  123. data/lib/aia/startup_coordinator.rb +150 -0
  124. data/lib/aia/streaming_runner.rb +169 -0
  125. data/lib/aia/system_prompt_assembler.rb +88 -0
  126. data/lib/aia/task_coordinator.rb +202 -0
  127. data/lib/aia/task_decomposer.rb +57 -0
  128. data/lib/aia/task_executor.rb +51 -0
  129. data/lib/aia/tfidf_math.rb +27 -0
  130. data/lib/aia/tool_filter/tfidf.rb +116 -0
  131. data/lib/aia/tool_filter/wordnet_expander.rb +127 -0
  132. data/lib/aia/tool_filter.rb +82 -0
  133. data/lib/aia/tool_filter_registry.rb +30 -0
  134. data/lib/aia/tool_filter_strategy.rb +143 -0
  135. data/lib/aia/tool_loader.rb +210 -0
  136. data/lib/aia/tool_utility.rb +30 -0
  137. data/lib/aia/tools/delegate_to_foreman_tool.rb +70 -0
  138. data/lib/aia/tools/recruit_robot_tool.rb +60 -0
  139. data/lib/aia/tools/reskill_robot_tool.rb +44 -0
  140. data/lib/aia/tools/task_board_tool.rb +114 -0
  141. data/lib/aia/trakflow_bridge.rb +173 -0
  142. data/lib/aia/turn_state.rb +94 -0
  143. data/lib/aia/ui_presenter.rb +166 -198
  144. data/lib/aia/utility.rb +134 -87
  145. data/lib/aia/{history_manager.rb → variable_input_collector.rb} +8 -9
  146. data/lib/aia/verification_network.rb +58 -0
  147. data/lib/aia.rb +108 -63
  148. data/mkdocs.yml +1 -0
  149. metadata +179 -56
  150. data/justfile +0 -215
  151. data/lib/aia/adapter/chat_execution.rb +0 -242
  152. data/lib/aia/adapter/error_handler.rb +0 -68
  153. data/lib/aia/adapter/gem_activator.rb +0 -57
  154. data/lib/aia/adapter/mcp_connector.rb +0 -274
  155. data/lib/aia/adapter/modality_handlers.rb +0 -167
  156. data/lib/aia/adapter/model_registry.rb +0 -81
  157. data/lib/aia/adapter/multi_model_chat.rb +0 -218
  158. data/lib/aia/adapter/provider_configurator.rb +0 -59
  159. data/lib/aia/adapter/tool_filter.rb +0 -85
  160. data/lib/aia/adapter/tool_loader.rb +0 -90
  161. data/lib/aia/chat_processor_service.rb +0 -178
  162. data/lib/aia/prompt_pipeline.rb +0 -183
  163. data/lib/aia/ruby_llm_adapter.rb +0 -95
  164. data/lib/extensions/openstruct_merge.rb +0 -48
  165. data/lib/extensions/ruby_llm/.irbrc +0 -56
  166. data/lib/extensions/ruby_llm/modalities.rb +0 -36
  167. data/lib/extensions/ruby_llm/provider_fix.rb +0 -79
  168. data/lib/refinements/string.rb +0 -16
  169. data/main.just +0 -76
data/examples/README.md CHANGED
@@ -185,6 +185,64 @@ Docs: [Executable Prompts](https://madbomber.github.io/aia/guides/executable-pro
185
185
 
186
186
  Docs: [Chat Guide](https://madbomber.github.io/aia/guides/chat/), [Directives Reference](https://madbomber.github.io/aia/directives-reference/), [Workflows and Pipelines](https://madbomber.github.io/aia/workflows-and-pipelines/)
187
187
 
188
+ ### 23 — Verify Mode
189
+
190
+ `23_verify.sh` — Demonstrates `/verify` mode via `VerificationNetwork`. Two robots independently answer the same question with slightly different system prompts, then a third reconciler robot compares both answers and produces a final verified response. Part 1 verifies a factual question (causes of the 2008 financial crisis); Part 2 verifies a technical explanation (TCP three-way handshake).
191
+
192
+ **Requires:** `expect` (pre-installed on macOS).
193
+
194
+ Docs: [Advanced Prompting](https://madbomber.github.io/aia/advanced-prompting/), [Chat Guide](https://madbomber.github.io/aia/guides/chat/)
195
+
196
+ ### 24 — Decompose Mode
197
+
198
+ `24_decompose.sh` — Demonstrates `/decompose` mode via `PromptDecomposer`. A coordinator robot analyzes whether a prompt can be split into 2–5 independent sub-tasks. If decomposable, specialist robots run each in parallel and results are synthesized into a single coherent response. Falls back to normal mode if the prompt is not decomposable. Part 1 uses a complex multi-part TCP/UDP question; Part 2 demonstrates the fallback with a simple question.
199
+
200
+ **Requires:** `expect` (pre-installed on macOS).
201
+
202
+ Docs: [Advanced Prompting](https://madbomber.github.io/aia/advanced-prompting/), [Chat Guide](https://madbomber.github.io/aia/guides/chat/)
203
+
204
+ ### 25 — Spawn Mode
205
+
206
+ `25_spawn.sh` — Demonstrates `/spawn` mode via `SpawnHandler`. Dynamically creates a specialist robot on demand. Part 1 explicitly names a specialist type (`/spawn security-expert`) and asks a SQL injection question. Part 2 uses auto-detection (`/spawn` with no args) where the primary robot determines the needed expertise from the question content. Specialists are cached and reused within the session.
207
+
208
+ **Requires:** `expect` (pre-installed on macOS).
209
+
210
+ Docs: [Advanced Prompting](https://madbomber.github.io/aia/advanced-prompting/), [Chat Guide](https://madbomber.github.io/aia/guides/chat/)
211
+
212
+ ### 26 — Debate Mode
213
+
214
+ `26_debate.sh` — Demonstrates `/debate` mode via `DebateHandler`. Two robots debate a topic across multiple rounds. Each round, every robot sees what the other said and responds. The debate ends when a robot says CONVERGED or after 5 rounds. `SimilarityScorer` also detects convergence by comparing round-to-round similarity. Uses `qwen3` (Tobor) vs `phi4-mini` (Quark) debating REST vs GraphQL.
215
+
216
+ **Requires:** `expect` (pre-installed on macOS), `phi4-mini` model (auto-pulled if missing).
217
+
218
+ Docs: [Advanced Prompting](https://madbomber.github.io/aia/advanced-prompting/), [Working with Models](https://madbomber.github.io/aia/guides/models/)
219
+
220
+ ### 27 — @mention Routing
221
+
222
+ `27_mention_routing.sh` — Demonstrates `@mention` routing in a multi-model network. Prefixing a message with `@RobotName` directs it to a specific robot; only that robot responds. Robot names are assigned by AIA: `qwen3` becomes `Tobor`, `phi4-mini` becomes `Quark`. Part 1 routes to `@Tobor`; Part 2 routes to `@Quark`; Part 3 shows how an unknown `@mention` triggers a listing of available robot names. Part 4 opens interactive chat to mix directed and undirected turns freely.
223
+
224
+ **Requires:** `expect` (pre-installed on macOS), `phi4-mini` model (auto-pulled if missing).
225
+
226
+ Docs: [Chat Guide](https://madbomber.github.io/aia/guides/chat/), [Working with Models](https://madbomber.github.io/aia/guides/models/)
227
+
228
+ ### 28 — Model Switching
229
+
230
+ `28_model_switching.sh` — Demonstrates the `/model` directive for switching models mid-conversation. Conversation history is transferred to the new model so it has full context of everything discussed before the switch. Part 1 asks a question with `qwen3` (Tobor), switches to `phi4-mini` (Quark) via `/model ollama/phi4-mini`, then asks a follow-up that requires the previous exchange for context.
231
+
232
+ **Requires:** `expect` (pre-installed on macOS), `phi4-mini` model (auto-pulled if missing).
233
+
234
+ Docs: [Chat Guide](https://madbomber.github.io/aia/guides/chat/), [Working with Models](https://madbomber.github.io/aia/guides/models/)
235
+
236
+ ### 29 — Agent Harness
237
+
238
+ `29_agent_harness.sh` — Demonstrates AIA as a full agent harness using a three-tier orchestration model. Tobor starts with an `orchestrator` role loaded via `--role orchestrator`, making it the primary coordinator. Part 1 shows Tobor describing its coordination strategy for a complex request. Part 2 uses `/spawn security-expert` to create a specialist lead agent on demand and route a task to it. Part 3 uses `/decompose` with a multi-dimension design review — Tobor and Quark run parallel workstreams as task runners, then Tobor synthesizes the result. Part 4 opens a free orchestration session where any combination of `/spawn`, `/decompose`, `/delegate`, `/debate`, and `@mention` routing can be tried.
239
+
240
+ The orchestrator role file lives at `examples/prompts_dir/roles/orchestrator.md` and can be customized to tune Tobor's coordination behavior.
241
+
242
+ **Requires:** `expect` (pre-installed on macOS), `phi4-mini` model (auto-pulled if missing).
243
+
244
+ Docs: [Chat Guide](https://madbomber.github.io/aia/guides/chat/), [Roles](https://madbomber.github.io/aia/guides/roles/)
245
+
188
246
  ## Running All Demos
189
247
 
190
248
  `run_all.sh` runs all non-interactive demo scripts in sequence and captures the combined output. It serves as a structural integration test — since LLM responses are non-deterministic, exact diffs between runs won't match, but you can spot missing sections, crashes, or changed command output.
@@ -221,6 +279,13 @@ diff output/run_PREV.log output/run_LATEST.log
221
279
  | `15_parameters.sh` | Part 2 prompts interactively for a required parameter |
222
280
  | `20_mcp_servers.sh` | Requires Node.js/npx + MCP filesystem server |
223
281
  | `22_chat_mode.sh` | Interactive chat session (uses `expect`) |
282
+ | `23_verify.sh` | Uses `expect` for interactive chat |
283
+ | `24_decompose.sh` | Uses `expect` for interactive chat |
284
+ | `25_spawn.sh` | Uses `expect` for interactive chat |
285
+ | `26_debate.sh` | Uses `expect` for interactive chat |
286
+ | `27_mention_routing.sh` | Uses `expect` for interactive chat |
287
+ | `28_model_switching.sh` | Uses `expect` for interactive chat |
288
+ | `29_agent_harness.sh` | Uses `expect` for interactive chat; Part 4 is fully interactive |
224
289
 
225
290
  The output includes a banner with version info, per-script pass/fail status, and a summary with counts.
226
291
 
@@ -0,0 +1,106 @@
1
+ # Advanced Multi-Robot Capabilities Without Examples
2
+
3
+ AIA v2 includes several advanced multi-robot execution modes that have no demo
4
+ scripts yet. These are currently accessible only through chat-time directives
5
+ (`/verify`, `/debate`, etc.) inside a `--chat` session, but the underlying
6
+ handlers are self-contained and could equally be exposed via CLI flags, YAML
7
+ front matter, or `//` prompt-file directives in batch mode.
8
+
9
+ ## Already Covered
10
+
11
+ `11_multi_model.sh` covers:
12
+ - Multi-model comparison mode (`-m modelA,modelB`)
13
+ - `--consensus` cooperative synthesis
14
+
15
+ ## Missing Demos
16
+
17
+ | Feature | Handler | Min models | Current entry point |
18
+ |---|---|---|---|
19
+ | `/verify` self-review | `VerificationNetwork` | 1 | `/verify` in chat |
20
+ | `/decompose` parallel tasks | `PromptDecomposer` | 1 | `/decompose` in chat |
21
+ | `/spawn` specialist robots | `SpawnHandler` | 1 | `/spawn [type]` in chat |
22
+ | `/debate` multi-round | `DebateHandler` | 2 | `/debate` in chat |
23
+ | `@mention` routing | `MentionRouter` | 2 | `@robot_name question` in chat |
24
+ | Natural language model switch | `ModelSwitchHandler` | 1 | "switch to X" in chat |
25
+ | `/delegate` TrakFlow teams | `DelegateHandler` | 2 | `/delegate` in chat |
26
+
27
+ ## Feature Details
28
+
29
+ ### `/verify` Self-Review (`lib/aia/verification_network.rb`)
30
+ - Two robots independently answer the same question with slightly different system
31
+ prompts to encourage independence, then a third "reconciler" robot compares both
32
+ answers, identifies agreements and disagreements, and produces a final verified
33
+ answer.
34
+ - All three robots use the same model (`config.models.first`) — no second model needed.
35
+ - Batch-capable: the handler takes a prompt string and returns a response with no
36
+ interactive steps. Could be exposed as `--verify` CLI flag or `mode: verify`
37
+ front matter.
38
+
39
+ ### `/decompose` Parallel Tasks (`lib/aia/prompt_decomposer.rb`)
40
+ - A coordinator robot analyzes whether a complex prompt can be split into 2-5
41
+ independent sub-tasks. If decomposable, specialist robots solve each in parallel
42
+ and results are synthesized into a unified answer. Falls back to normal mode if
43
+ the prompt is not decomposable.
44
+ - Single model is sufficient.
45
+ - Batch-capable: no interactivity required. Could be `--decompose` or `mode: decompose`.
46
+
47
+ ### `/spawn` Specialist Robots (`lib/aia/spawn_handler.rb`)
48
+ - Dynamically creates a specialist robot on demand. The user can name a type
49
+ (`/spawn security-expert`) or omit it and let the primary robot auto-detect the
50
+ needed expertise. Spawned specialists are cached for reuse within the session.
51
+ Optionally creates TrakFlow tasks when TrakFlow is available.
52
+ - Single model is sufficient.
53
+ - Batch-capable: specialist type could be specified via CLI flag or front matter.
54
+
55
+ ### `/debate` Multi-Round (`lib/aia/debate_handler.rb`)
56
+ - Two or more robots debate a topic across rounds. Each round, every robot responds
57
+ to the previous arguments. Stops when any robot says "CONVERGED" or after 5
58
+ rounds. Displays formatted round-by-round output.
59
+ - Requires a 2-model network (`-m modelA,modelB`).
60
+ - Batch-capable: debate runs to completion without user input — could be `--debate`.
61
+
62
+ ### `@mention` Routing (`lib/aia/mention_router.rb`)
63
+ - Directs a message to one specific robot in a multi-model network by name.
64
+ Syntax: `@robot_name your question`. If the robot name is unknown, available
65
+ robots are listed. Shows token metrics with `--tokens`.
66
+ - Requires a 2-model network.
67
+ - Genuinely interactive: the value is in the user directing conversation flow
68
+ turn-by-turn; less meaningful as a one-shot batch operation.
69
+
70
+ ### Natural Language Model Switch (`lib/aia/model_switch_handler.rb`)
71
+ - Detects model-change intent from natural language input, e.g. "switch to llama"
72
+ or "use phi4-mini instead". Shows "Interpreted as: /model X" and prompts for
73
+ confirmation (y/n). On confirmation, rebuilds the robot with the new model and
74
+ transfers history. Also handles `model_compare` and `model_switch_capability` intents.
75
+ - Genuinely interactive: depends on the user directing the conversation.
76
+
77
+ ### `/delegate` TrakFlow Teams (`lib/aia/delegate_handler.rb`)
78
+ - A lead robot breaks the prompt into subtasks and assigns each to team members via
79
+ a TrakFlow plan. Each robot executes its subtask in order with full shared context.
80
+ Shows the plan before execution, then step-by-step results.
81
+ - Requires a 2-model network and the TrakFlow gem.
82
+ - **Blocked**: `trak_flow` gem currently has a LoadError in this environment.
83
+ Defer until the TrakFlow dependency is resolved.
84
+
85
+ ## Batch Mode Extension Opportunities
86
+
87
+ `/verify`, `/decompose`, `/debate`, and `/spawn` are ready to be lifted out of
88
+ chat-only access. `@mention` routing and natural language model switching are
89
+ genuinely conversational and are better left as chat-only features.
90
+
91
+ | Feature | Suggested CLI flag | Suggested front matter key |
92
+ |---|---|---|
93
+ | `/verify` | `--verify` | `mode: verify` |
94
+ | `/decompose` | `--decompose` | `mode: decompose` |
95
+ | `/debate` | `--debate` | `mode: debate` |
96
+ | `/spawn` | `--spawn TYPE` | `mode: spawn` / `spawn_type: TYPE` |
97
+
98
+ ## Demo Implementation Notes
99
+
100
+ - Chat-mode demos would use `expect` scripting like `22_chat_mode.sh`.
101
+ - Single-model demos can reuse `ollama/qwen3` from the setup script.
102
+ - Multi-model demos need a second model — `ollama/phi4-mini` (already used in
103
+ `11_multi_model.sh`) is the natural choice.
104
+ - Suggested numbering: `23_verify.sh`, `24_decompose.sh`, `25_spawn.sh`,
105
+ `26_debate.sh`, `27_mention_routing.sh`, `28_model_switching.sh`
106
+ - Skip `29_delegate.sh` until the TrakFlow LoadError is resolved.
@@ -13,7 +13,7 @@ prompts:
13
13
  dir: ./prompts_dir
14
14
 
15
15
  models:
16
- - name: ollama/qwen3
16
+ - name: gpt-4.1
17
17
 
18
18
  output:
19
19
  file: ~
@@ -0,0 +1,45 @@
1
+ # AIA configuration for agent harness demo (29_agent_harness.sh)
2
+ #
3
+ # Extends the base demo config with an orchestrator system prompt so Tobor
4
+ # is pre-configured as the primary coordinator from the start of the session.
5
+ # The system_prompt is injected into the robot's context at build time —
6
+ # it is NOT sent as a user message. Use this config with --chat and optionally
7
+ # -m MODEL_A,MODEL_B to add specialist network robots.
8
+ #
9
+ # Usage:
10
+ # aia -c aia_config_orchestrator.yml --chat
11
+ # aia -c aia_config_orchestrator.yml -m gpt-4.1,gpt-4.1-mini --chat
12
+
13
+ prompts:
14
+ dir: ./prompts_dir
15
+ system_prompt: |
16
+ You are Tobor, the primary AI orchestrator in this session.
17
+
18
+ Your role is to coordinate a team of specialized robots to handle complex
19
+ tasks. Think like a project director: delegate what can be parallelized,
20
+ direct what requires deep specialization, and synthesize what needs
21
+ holistic judgment.
22
+
23
+ When you receive a request, apply this decision framework:
24
+
25
+ - Direct response: Simple, self-contained questions — answer yourself.
26
+ - /spawn specialist-type: Questions requiring narrow domain expertise.
27
+ This creates a specialist lead agent on the fly and routes the task to it.
28
+ - /decompose: Complex requests with multiple independent dimensions.
29
+ This splits the task into parallel workstreams across your robot team,
30
+ then synthesizes the results into a unified response.
31
+ - /delegate: Structured multi-step tasks where steps have dependencies.
32
+ This creates a tracked execution plan with ordered hand-offs between robots.
33
+
34
+ Always be explicit about which coordination strategy you are using and why.
35
+ When you spawn or decompose, briefly state the workstreams or specialist
36
+ roles before execution so the user understands the plan.
37
+
38
+ models:
39
+ - name: gpt-4.1
40
+
41
+ output:
42
+ file: ~
43
+
44
+ flags:
45
+ chat: false
data/examples/common.sh CHANGED
@@ -12,6 +12,15 @@
12
12
  SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
13
13
  cd "${SCRIPT_DIR}"
14
14
 
15
+ # When running from inside the development repo, prefer the local bin/aia
16
+ # over any system-installed gem. bin/aia loads from ../lib/aia directly,
17
+ # so demos always run against the version in this checkout.
18
+ # Exported so child processes (including `expect`'s `spawn aia`) inherit it.
19
+ REPO_BIN="$(cd "${SCRIPT_DIR}/.." && pwd)/bin"
20
+ if [ -f "${REPO_BIN}/aia" ]; then
21
+ export PATH="${REPO_BIN}:${PATH}"
22
+ fi
23
+
15
24
  # Clear all AIA_* env vars so personal settings don't leak in.
16
25
  while IFS= read -r var; do
17
26
  unset "$var"
@@ -19,9 +28,12 @@ done < <(env | grep '^AIA_' | cut -d= -f1)
19
28
 
20
29
  CONFIG="aia_config.yml"
21
30
 
22
- # When running from the develop branch, prefer the local bin/aia over any
23
- # installed gem so that uncommitted working-tree changes are exercised.
24
- LOCAL_AIA="${SCRIPT_DIR}/../bin/aia"
25
- if [[ -x "${LOCAL_AIA}" ]]; then
26
- export PATH="${SCRIPT_DIR}/../bin:${PATH}"
27
- fi
31
+ # Drain leaked escape sequences from the terminal input buffer.
32
+ # Reline queries cursor position (\e[6n) inside the pty; the
33
+ # terminal's responses (\e[row;colR) can leak into bash's stdin
34
+ # after expect exits. Call this after every expect block.
35
+ drain_terminal() {
36
+ sleep 0.2
37
+ stty sane 2>/dev/null || true
38
+ while IFS= read -r -t 0.2 -n 100 _ 2>/dev/null; do :; done
39
+ }
@@ -1,7 +1,7 @@
1
1
  # Tech Stack
2
2
 
3
- - Language: Ruby 3.3+
4
- - LLM interface: ruby_llm gem
3
+ - Language: Ruby 4.0+
4
+ - LLM interface: robot_lab gem (orchestration) + ruby_llm gem (LLM abstraction)
5
5
  - Prompt storage: prompt_manager gem (Markdown files)
6
6
  - CLI parsing: optparse (stdlib)
7
7
  - Config: myway_config gem (YAML + XDG)
@@ -0,0 +1,21 @@
1
+ You are Tobor, the primary AI orchestrator in this session.
2
+
3
+ Your role is to coordinate a team of specialized robots to handle complex tasks.
4
+ Think like a project director: delegate what can be parallelized, direct what
5
+ requires deep specialization, and synthesize what needs holistic judgment.
6
+
7
+ When you receive a request, apply this decision framework:
8
+
9
+ - **Direct response**: Simple, self-contained questions — answer yourself.
10
+ - **/spawn specialist-type**: Questions requiring narrow domain expertise.
11
+ This creates a specialist lead agent on the fly and routes the task to it.
12
+ Example: `/spawn security-expert` before asking about API hardening.
13
+ - **/decompose**: Complex requests with multiple independent dimensions.
14
+ This splits the task into parallel workstreams across your robot team,
15
+ then synthesizes the results into a unified response.
16
+ - **/delegate**: Structured multi-step tasks where steps have dependencies.
17
+ This creates a tracked execution plan with ordered hand-offs between robots.
18
+
19
+ Always be explicit about which coordination strategy you are using and why.
20
+ When you spawn or decompose, briefly state the workstreams or specialist roles
21
+ before execution so the user understands the plan.
@@ -0,0 +1,139 @@
1
+ # TaskFlow — Project & Task Management Web Application
2
+
3
+ ## Overview
4
+
5
+ Build a standalone Ruby application using Sinatra as the web framework.
6
+ TaskFlow is a multi-user project and task management system similar to a
7
+ simplified Trello or Asana. The application must be a single deployable
8
+ Ruby process with no external service dependencies beyond SQLite.
9
+
10
+ ## Technology Stack
11
+
12
+ - Ruby 3.x
13
+ - Sinatra 4.x (classic style, single `app.rb` entry point)
14
+ - SQLite via the Sequel ORM (migrations in `db/migrations/`)
15
+ - ERB templates in `views/` (layouts, partials)
16
+ - Bootstrap 5 via CDN for styling
17
+ - bcrypt for password hashing
18
+ - jwt gem for API token issuance
19
+ - Rack::Session::Cookie for browser sessions
20
+ - WEBrick for development; Puma for production
21
+
22
+ ## Authentication & Authorization
23
+
24
+ - Users register with email, password (bcrypt), and display name
25
+ - Login issues a session cookie (browser) and a JWT (API clients)
26
+ - Passwords must be at least 8 characters; emails must be unique
27
+ - Sessions expire after 24 hours; JWT tokens after 1 hour
28
+ - Route protection: unauthenticated requests redirect to /login (browser)
29
+ or return 401 JSON (API requests with Accept: application/json)
30
+ - Role system: each project has members with role `owner`, `editor`, or `viewer`
31
+ - Only owners may delete a project or change member roles
32
+ - Editors may create/update/delete tasks within the project
33
+ - Viewers may only read
34
+
35
+ ## Data Model
36
+
37
+ ### User
38
+ - id (integer, primary key)
39
+ - email (string, unique, not null)
40
+ - password_digest (string, not null)
41
+ - display_name (string, not null)
42
+ - created_at, updated_at (timestamps)
43
+
44
+ ### Project
45
+ - id (integer, primary key)
46
+ - name (string, not null)
47
+ - description (text)
48
+ - owner_id (integer, FK → users.id)
49
+ - created_at, updated_at
50
+
51
+ ### ProjectMember
52
+ - id (integer, primary key)
53
+ - project_id (integer, FK → projects.id)
54
+ - user_id (integer, FK → users.id)
55
+ - role (string: 'owner' | 'editor' | 'viewer')
56
+ - joined_at (timestamp)
57
+ - UNIQUE constraint on (project_id, user_id)
58
+
59
+ ### Task
60
+ - id (integer, primary key)
61
+ - project_id (integer, FK → projects.id)
62
+ - title (string, not null)
63
+ - description (text)
64
+ - status (string: 'todo' | 'in_progress' | 'done', default 'todo')
65
+ - priority (string: 'low' | 'medium' | 'high', default 'medium')
66
+ - assigned_to (integer, FK → users.id, nullable)
67
+ - due_date (date, nullable)
68
+ - created_by (integer, FK → users.id)
69
+ - created_at, updated_at
70
+
71
+ ## Web Routes (Browser)
72
+
73
+ - GET / → redirect to /dashboard if logged in, else /login
74
+ - GET /login → login form
75
+ - POST /login → authenticate, set session, redirect to /dashboard
76
+ - GET /register → registration form
77
+ - POST /register → create user, auto-login, redirect to /dashboard
78
+ - POST /logout → clear session, redirect to /login
79
+ - GET /dashboard → list all projects the current user is a member of
80
+ - GET /projects/new → form to create a project
81
+ - POST /projects → create project, add user as owner
82
+ - GET /projects/:id → project detail: task list, member list
83
+ - GET /projects/:id/edit → edit project name/description
84
+ - PUT /projects/:id → update project
85
+ - DELETE /projects/:id → delete project (owner only)
86
+ - GET /projects/:id/tasks/new → new task form
87
+ - POST /projects/:id/tasks → create task
88
+ - GET /projects/:id/tasks/:tid/edit → edit task form
89
+ - PUT /projects/:id/tasks/:tid → update task
90
+ - DELETE /projects/:id/tasks/:tid → delete task
91
+
92
+ ## REST API Routes (JSON, prefix /api/v1)
93
+
94
+ - POST /api/v1/auth/login → { token: "jwt..." }
95
+ - GET /api/v1/projects → array of projects for current user
96
+ - POST /api/v1/projects → create project
97
+ - GET /api/v1/projects/:id → project + tasks
98
+ - PUT /api/v1/projects/:id → update project
99
+ - DELETE /api/v1/projects/:id → delete project
100
+ - GET /api/v1/projects/:id/tasks → task list
101
+ - POST /api/v1/projects/:id/tasks → create task
102
+ - PUT /api/v1/projects/:id/tasks/:tid → update task
103
+ - DELETE /api/v1/projects/:id/tasks/:tid → delete task
104
+ - GET /api/v1/projects/:id/members → member list
105
+ - POST /api/v1/projects/:id/members → add member
106
+ - DELETE /api/v1/projects/:id/members/:uid → remove member
107
+
108
+ ## Views (ERB)
109
+
110
+ - layout.erb → HTML shell, nav bar, flash messages, Bootstrap 5
111
+ - dashboard.erb → project cards grid, "New Project" button
112
+ - projects/show.erb → task board (Kanban columns by status), member sidebar
113
+ - projects/form.erb → shared create/edit form
114
+ - tasks/form.erb → shared create/edit form
115
+ - auth/login.erb → login form
116
+ - auth/register.erb → registration form
117
+ - partials/_flash.erb → flash message component
118
+ - partials/_task_card.erb → task card with status badge and priority indicator
119
+
120
+ ## Infrastructure & Configuration
121
+
122
+ - `app.rb` → Sinatra application class, requires all components
123
+ - `config.ru` → Rack entry point, mounts the app
124
+ - `Gemfile` → all dependencies with locked versions
125
+ - `db/migrations/` → numbered Sequel migration files
126
+ - `db/schema.rb` → auto-generated schema dump
127
+ - `lib/auth_helpers.rb` → session/JWT helpers, `current_user`, `require_login`
128
+ - `lib/api_helpers.rb` → JSON response helpers, token verification
129
+ - `config/database.rb` → Sequel connection setup (dev/test/prod environments)
130
+ - `config/settings.rb` → app constants: session secret, JWT secret, token TTL
131
+ - `.env.example` → template for required environment variables
132
+
133
+ ## Testing
134
+
135
+ - RSpec with rack-test for request specs
136
+ - Factory pattern for test data (no FactoryBot, plain Ruby factory methods)
137
+ - Tests in `spec/`: `spec/auth_spec.rb`, `spec/projects_spec.rb`,
138
+ `spec/tasks_spec.rb`, `spec/api_spec.rb`
139
+ - Test database: separate SQLite file, schema reset before each suite
@@ -0,0 +1,16 @@
1
+ # examples/rules/01_classify_ruby.rb
2
+ #
3
+ # Custom classification rule: detect Ruby-specific requests.
4
+ #
5
+ # Install: cp examples/rules/01_classify_ruby.rb ~/.config/aia/rules/
6
+
7
+ AIA.rules_for(:classify) do
8
+ rule "ruby_request" do
9
+ on :turn_input do
10
+ text matches(/\b(ruby|rails|gem|bundler|rake|rspec|minitest|rubocop|sorbet)\b/i)
11
+ end
12
+ perform do |_facts|
13
+ AIA.decisions.add(:classification, domain: "code", subdomain: "ruby", source: "user_ruby_request")
14
+ end
15
+ end
16
+ end
@@ -0,0 +1,19 @@
1
+ # examples/rules/02_prefer_claude_for_code.rb
2
+ #
3
+ # Model selection rule: prefer Claude for code tasks.
4
+ # When the classify KB tags a prompt as "code" domain,
5
+ # suggest Claude as the preferred model.
6
+ #
7
+ # Install: cp examples/rules/02_prefer_claude_for_code.rb ~/.config/aia/rules/
8
+
9
+ AIA.rules_for(:model_select) do
10
+ rule "prefer_claude_for_code" do
11
+ on :classification_decision, domain: "code"
12
+ on :model, name: satisfies { |n| n.to_s.include?("claude") }
13
+ perform do |facts|
14
+ AIA.decisions.add(:model_decision,
15
+ model: facts[1][:name],
16
+ reason: "user rule: prefer Claude for code tasks")
17
+ end
18
+ end
19
+ end
@@ -0,0 +1,19 @@
1
+ # examples/rules/03_gate_prompt_length.rb
2
+ #
3
+ # Quality gate rule: warn when context files are very large.
4
+ # This supplements the built-in 100KB warning with a custom threshold.
5
+ #
6
+ # Install: cp examples/rules/03_gate_prompt_length.rb ~/.config/aia/rules/
7
+
8
+ AIA.rules_for(:gate) do
9
+ rule "very_large_context_warning" do
10
+ on :context_stats, large: true
11
+ perform do |facts|
12
+ size = facts[0][:total_size]
13
+ if size && size > 500_000
14
+ AIA.decisions.add(:gate, action: "warn",
15
+ message: "Context exceeds 500KB (#{size / 1024}KB). This may be slow and expensive.")
16
+ end
17
+ end
18
+ end
19
+ end
@@ -0,0 +1,41 @@
1
+ # examples/rules/04_tool_selection.rb
2
+ #
3
+ # Custom tool selection rules (optional).
4
+ #
5
+ # AIA automatically builds tool routing rules at startup by analyzing
6
+ # each loaded tool's name and description against domain keyword patterns.
7
+ # This file shows how to ADD custom rules on top of the automatic ones.
8
+ #
9
+ # Use cases for custom rules:
10
+ # - Override automatic classification for a specific tool
11
+ # - Add domain routing for a custom tool with unusual naming
12
+ # - Activate tools based on input text patterns instead of domain
13
+ #
14
+ # Install: cp examples/rules/04_tool_selection.rb ~/.config/aia/rules/
15
+ #
16
+ # Facts available in the :route KB:
17
+ # :tool — name: String, description: String, active: true
18
+ # :classification_decision — domain: String, ...
19
+ # :turn_input — text: String, length: Integer
20
+
21
+ AIA.rules_for(:route) do
22
+ # Example: activate a custom tool for a specific keyword
23
+ # rule "activate_my_custom_tool" do
24
+ # on :turn_input do
25
+ # text matches(/\b(my_keyword)\b/i)
26
+ # end
27
+ # on :tool, name: "my_custom_tool"
28
+ # perform do |facts|
29
+ # AIA.decisions.add(:tool_activate, tool: facts[1][:name], reason: "custom keyword match")
30
+ # end
31
+ # end
32
+
33
+ # Example: force a tool into a domain it wasn't auto-classified into
34
+ # rule "activate_special_tool_for_data" do
35
+ # on :classification_decision, domain: "data"
36
+ # on :tool, name: "my_special_tool"
37
+ # perform do |facts|
38
+ # AIA.decisions.add(:tool_activate, tool: facts[1][:name], reason: "custom data domain override")
39
+ # end
40
+ # end
41
+ end
@@ -0,0 +1,30 @@
1
+ # AIA User Rules
2
+
3
+ > **NOTE (2026-03-28)**: The KBS rule engine has been removed from AIA (see ADR-010).
4
+ > This directory is preserved for historical reference only. The `.rb` example files
5
+ > here no longer work. User-defined rule hooks (`~/.config/aia/rules/*.rb`) are no
6
+ > longer supported. Tool filtering is now handled exclusively by the TF-IDF/vector
7
+ > strategies (A=TF-IDF, B=Zvec, C=SqliteVec, D=LSI).
8
+
9
+ User rules extended the built-in KBS rule engine with custom classification,
10
+ model selection, MCP routing, quality gate, and learning rules.
11
+
12
+ ## Setup
13
+
14
+ Place `.rb` files in `~/.config/aia/rules/`. AIA loads them alphabetically
15
+ at startup. Each file calls `AIA.rules_for(:kb_name)` to register rules
16
+ targeting a specific knowledge base.
17
+
18
+ ## Available Knowledge Bases
19
+
20
+ | KB Name | Purpose | Facts Available |
21
+ |----------------|--------------------------------------|--------------------------------|
22
+ | `:classify` | Categorize prompts by domain/intent | `:turn_input`, `:context_file` |
23
+ | `:model_select`| Choose model based on classification | `:classification_decision`, `:model` |
24
+ | `:route` | Activate MCP servers | `:classification_decision`, `:mcp_server` |
25
+ | `:gate` | Pre-send quality checks | `:turn_input`, `:context_stats`, `:session_stats` |
26
+ | `:learn` | Post-response tracking | `:response_outcome`, `:session_stats` |
27
+
28
+ ## Examples
29
+
30
+ See the `.rb` files in this directory for working examples.