pikuri-core 0.0.6 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +6 -4
  3. data/lib/pikuri/agent/chat_transport.rb +128 -24
  4. data/lib/pikuri/agent/configurator.rb +47 -107
  5. data/lib/pikuri/agent/context_window_detector.rb +80 -70
  6. data/lib/pikuri/agent/control/cancellable.rb +87 -66
  7. data/lib/pikuri/agent/control/interloper.rb +127 -105
  8. data/lib/pikuri/agent/control/step_limit.rb +46 -30
  9. data/lib/pikuri/agent/control.rb +14 -34
  10. data/lib/pikuri/agent/event.rb +125 -163
  11. data/lib/pikuri/agent/extension.rb +121 -83
  12. data/lib/pikuri/agent/extension_context.rb +120 -0
  13. data/lib/pikuri/agent/history.rb +653 -0
  14. data/lib/pikuri/agent/listener/rate_limited.rb +40 -66
  15. data/lib/pikuri/agent/listener/terminal.rb +147 -110
  16. data/lib/pikuri/agent/listener/token_log.rb +116 -108
  17. data/lib/pikuri/agent/listener.rb +23 -36
  18. data/lib/pikuri/agent/listener_list.rb +26 -57
  19. data/lib/pikuri/agent/synthesizer.rb +76 -92
  20. data/lib/pikuri/agent.rb +939 -642
  21. data/lib/pikuri/bundler_env.rb +68 -0
  22. data/lib/pikuri/extractor/html.rb +63 -110
  23. data/lib/pikuri/extractor/passthrough.rb +20 -30
  24. data/lib/pikuri/extractor.rb +93 -154
  25. data/lib/pikuri/file_type.rb +63 -135
  26. data/lib/pikuri/finalizers.rb +32 -47
  27. data/lib/pikuri/paths.rb +104 -13
  28. data/lib/pikuri/ruby_llm_patches.rb +106 -0
  29. data/lib/pikuri/sanitizer.rb +157 -0
  30. data/lib/pikuri/subprocess.rb +75 -119
  31. data/lib/pikuri/testing.rb +296 -0
  32. data/lib/pikuri/tool/calculator.rb +56 -66
  33. data/lib/pikuri/tool/execute_context.rb +42 -0
  34. data/lib/pikuri/tool/fetch.rb +51 -77
  35. data/lib/pikuri/tool/parameters.rb +70 -15
  36. data/lib/pikuri/tool/scraper.rb +55 -97
  37. data/lib/pikuri/tool/search/brave.rb +70 -84
  38. data/lib/pikuri/tool/search/duckduckgo.rb +65 -86
  39. data/lib/pikuri/tool/search/engines.rb +249 -93
  40. data/lib/pikuri/tool/search/exa.rb +75 -97
  41. data/lib/pikuri/tool/search/rate_limiter.rb +61 -38
  42. data/lib/pikuri/tool/search/result.rb +10 -15
  43. data/lib/pikuri/tool/trifecta_legs.rb +217 -0
  44. data/lib/pikuri/tool/web_scrape.rb +38 -54
  45. data/lib/pikuri/tool/web_search.rb +121 -26
  46. data/lib/pikuri/tool.rb +140 -65
  47. data/lib/pikuri/trifecta/contribution.rb +43 -0
  48. data/lib/pikuri/trifecta/node.rb +47 -0
  49. data/lib/pikuri/trifecta/report.rb +230 -0
  50. data/lib/pikuri/trifecta.rb +127 -0
  51. data/lib/pikuri/url_cache.rb +33 -49
  52. data/lib/pikuri/version.rb +1 -1
  53. data/lib/pikuri-core.rb +72 -86
  54. data/prompts/agent-loop.txt +5 -0
  55. data/prompts/pikuri-chat.txt +3 -12
  56. metadata +18 -8
@@ -1,59 +1,49 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require 'ruby_llm'
3
4
  require 'faraday'
4
5
  require 'json'
5
6
  require 'cgi'
6
7
 
7
8
  module Pikuri
8
9
  class Agent
9
- # Resolves the model's context-window cap from three sources, in order:
10
- # an explicit override, the value ruby_llm reports for the model, or a
11
- # llama.cpp +/props+ probe. Returns +nil+ if none of those produce a
12
- # value.
10
+ # Resolves the model's context-window cap by asking the server that serves
11
+ # it. The only authoritative runtime source is llama.cpp's non-standard
12
+ # +/props+ endpoint, reporting the server's *launched* +n_ctx+ (the real
13
+ # window — possibly smaller than the model's max, e.g. +llama-server -c
14
+ # 8192+ on a 128k model). Returns +nil+ — an honest "we don't know" — for
15
+ # anything else. {Agent#detect_and_emit_context_cap!} calls it only when
16
+ # the transport carries no explicit {ChatTransport#context_window}.
13
17
  #
14
- # Used by {Agent#initialize} at construction time to feed
15
- # {Listener::TokenLog} a cap it can render alongside the running
16
- # context size (so the +ctx=12.2k/32.0k+ line tells the operator how
17
- # close the conversation is to the limit).
18
+ # This probe and an explicit {ChatTransport#context_window} are the *only*
19
+ # sources: +RubyLLM::Model::Info#context_window+ is deliberately never
20
+ # consulted (why: +DECISIONS.md+ +D_context_window_sources+).
18
21
  #
19
- # == Precedence
22
+ # == The openai-provider gate + auto-derived URL
20
23
  #
21
- # 1. +override+ — the +Agent.new(context_window:)+ kwarg. Wins over
22
- # everything; an explicit value is the operator's statement of
23
- # truth.
24
- # 2. +ruby_llm_reported+ — +RubyLLM::Model::Info#context_window+ from
25
- # {Agent#chat}'s resolved model. Populated for models in ruby_llm's
26
- # bundled registry (OpenAI, Anthropic, Gemini, …); +nil+ for custom
27
- # local model ids that fall through to +Model::Info.default+.
28
- # 3. +llama_probe_url+ — HTTP GET against llama.cpp's non-standard
29
- # +/props+ endpoint. The server exposes the launched +n_ctx+ at
30
- # +default_generation_settings.n_ctx+ there. Probed only when the
31
- # first two are +nil+. Provider-specific to llama.cpp; the caller
32
- # (typically +bin/pikuri-chat+) derives the right URL from its configured
33
- # base.
24
+ # The probe only makes sense against an OpenAI-compatible local server
25
+ # (llama.cpp) via ruby_llm's +:openai+ provider, so {.detect} runs only for
26
+ # +transport.provider == :openai+ and derives the URL from the *same*
27
+ # +RubyLLM.config.openai_api_base+ the chat uses (+/props+ is at the host
28
+ # root, so the +/v1+ suffix is stripped) — so it can't target a different
29
+ # server than the chat. A bare +:openai+ at real +api.openai.com+ gets one
30
+ # fast +/props+ 404 that degrades to +nil+.
34
31
  #
35
32
  # == llama.cpp router mode
36
33
  #
37
- # A llama.cpp *router* (the multi-instance front that proxies to N
38
- # on-demand model servers) answers a bare +/props+ with
39
- # +{"role":"router", ..., "n_ctx":0}+ — there is no single loaded
40
- # model at the router itself, so its top-level +n_ctx+ is +0+. The
41
- # real per-model cap is one proxied hop away: +GET /props?model=<id>+
42
- # routes the probe to that model's instance, whose +/props+ carries
43
- # the launched +n_ctx+. So when the bare probe reports +role: router+
44
- # and a +model_id+ is known, this re-probes with the model id before
45
- # giving up. A plain single-model server is untouched: its bare
46
- # +/props+ already carries a positive +n_ctx+, so the router branch
47
- # never runs.
34
+ # A llama.cpp *router* (the multi-instance front) answers a bare +/props+
35
+ # with +{"role":"router", ..., "n_ctx":0}+ — no single loaded model. The
36
+ # real per-model cap is one hop away: +GET /props?model=<id>+ routes to that
37
+ # model's instance. So when the bare probe reports +role: router+ and a
38
+ # +model_id+ is known, this re-probes with the id. A plain single-model
39
+ # server (bare +/props+ already positive) never hits this branch.
48
40
  #
49
41
  # == Failure handling
50
42
  #
51
- # The probe is best-effort. HTTP error, timeout, non-JSON body, or a
52
- # missing/invalid +n_ctx+ field all return +nil+ and log one +warn+
53
- # line via +Pikuri.logger_for('ContextWindowDetector')+. This is the
54
- # CLAUDE.md "secondary to the loop" carve-out — a wedged or
55
- # non-llama.cpp server should not abort agent construction over a
56
- # cosmetic readout.
43
+ # Best-effort: HTTP error, timeout, non-JSON body, or a missing/invalid
44
+ # +n_ctx+ all return +nil+ with one +warn+ line. The CLAUDE.md
45
+ # "secondary to the loop" carve-out — a wedged or non-llama.cpp server must
46
+ # not abort agent construction over a cosmetic readout.
57
47
  class ContextWindowDetector
58
48
  # Subsystem logger; set its level with
59
49
  # +PIKURI_LOG_CONTEXTWINDOWDETECTOR+ or the global +PIKURI_LOG+.
@@ -61,49 +51,69 @@ module Pikuri
61
51
  # @return [Logger]
62
52
  LOGGER = Pikuri.logger_for('ContextWindowDetector')
63
53
 
64
- # Connect timeout in seconds for the llama.cpp +/props+ probe.
65
- # Short on purpose: this runs synchronously during +Agent.new+ and
66
- # a wedged server should not stall startup noticeably.
54
+ # Connect timeout (s) for the +/props+ probe. Short: a server that isn't
55
+ # listening should fail fast rather than stall {Agent} construction.
67
56
  #
68
57
  # @return [Integer]
69
58
  OPEN_TIMEOUT = 2
70
- # Read timeout in seconds for the llama.cpp +/props+ probe; matches
71
- # {OPEN_TIMEOUT} for the same reason.
59
+ # Read timeout (s) for the +/props+ probe — generous, unlike
60
+ # {OPEN_TIMEOUT}: a router answers +/props?model=<id>+ only after spinning
61
+ # up that model's instance, and a cold load can take 10+s (which the next
62
+ # chat turn must wait for anyway). A shorter timeout would lose the cap
63
+ # exactly when switching to a cold model.
72
64
  #
73
65
  # @return [Integer]
74
- READ_TIMEOUT = 2
75
-
76
- # @param override [Integer, nil] explicit cap from the caller; wins if
77
- # non-+nil+
78
- # @param ruby_llm_reported [Integer, nil] value off
79
- # +RubyLLM::Chat#model.context_window+
80
- # @param llama_probe_url [String, nil] full URL to llama.cpp +/props+;
81
- # +nil+ or empty string skips the probe
82
- # @param model_id [String, nil] the chat model id, used only to
83
- # follow a llama.cpp router via +/props?model=<id>+ when the bare
84
- # probe reports +role: router+. +nil+ or empty disables that
85
- # second hop.
86
- def initialize(override:, ruby_llm_reported:, llama_probe_url:, model_id: nil)
87
- @override = override
88
- @ruby_llm_reported = ruby_llm_reported
89
- @llama_probe_url = llama_probe_url
90
- @model_id = model_id
66
+ READ_TIMEOUT = 30
67
+
68
+ # Resolve the context-window cap for +transport+ by probing its server.
69
+ #
70
+ # @param transport [Agent::ChatTransport] +provider+ gates the probe,
71
+ # +model+ drives the router +?model=+ hop
72
+ # @param openai_base [String, nil] base URL the probe URL derives from;
73
+ # defaults to the live +RubyLLM.config.openai_api_base+ (passed
74
+ # explicitly only by tests, to avoid mutating global config)
75
+ # @return [Integer, nil] the launched +n_ctx+, or +nil+ for a
76
+ # non-+:openai+ transport, unconfigured base, or any probe failure
77
+ def self.detect(transport, openai_base: RubyLLM.config.openai_api_base)
78
+ return nil unless transport.provider == :openai
79
+
80
+ url = props_url(openai_base)
81
+ return nil if url.nil?
82
+
83
+ new(probe_url: url, model_id: transport.model).probe
91
84
  end
92
85
 
93
- # @return [Integer, nil] resolved cap, or +nil+ if no source produced
94
- # one
95
- def detect
96
- return @override if @override
97
- return @ruby_llm_reported if @ruby_llm_reported
98
- return nil if @llama_probe_url.nil? || @llama_probe_url.empty?
86
+ # Derive the +/props+ URL from the OpenAI-compatible base (+/props+ is at
87
+ # the host root, so a trailing +/v1+ is stripped).
88
+ #
89
+ # @param openai_base [String, nil]
90
+ # @return [String, nil] the +/props+ URL, or +nil+ when the base is blank
91
+ def self.props_url(openai_base)
92
+ base = openai_base.to_s.strip.chomp('/')
93
+ return nil if base.empty?
94
+
95
+ "#{base.delete_suffix('/v1')}/props"
96
+ end
97
+
98
+ # @param probe_url [String] full URL to llama.cpp +/props+
99
+ # @param model_id [String, nil] the chat model id, used to follow a
100
+ # llama.cpp router via +/props?model=<id>+ when the bare probe
101
+ # reports +role: router+. +nil+ or empty disables that second hop.
102
+ def initialize(probe_url:, model_id:)
103
+ @probe_url = probe_url
104
+ @model_id = model_id
105
+ end
99
106
 
107
+ # @return [Integer, nil] resolved cap, or +nil+ if the probe
108
+ # produced none
109
+ def probe
100
110
  probe_llama_cpp
101
111
  end
102
112
 
103
113
  private
104
114
 
105
115
  def probe_llama_cpp
106
- data = fetch_props(@llama_probe_url)
116
+ data = fetch_props(@probe_url)
107
117
  return nil if data.nil?
108
118
 
109
119
  n_ctx = positive_n_ctx(data)
@@ -114,12 +124,12 @@ module Pikuri
114
124
  return probe_router_model if data['role'] == 'router' && model_id_present?
115
125
 
116
126
  warn_and_nil(
117
- "no positive integer at default_generation_settings.n_ctx in #{@llama_probe_url} response"
127
+ "no positive integer at default_generation_settings.n_ctx in #{@probe_url} response"
118
128
  )
119
129
  end
120
130
 
121
131
  def probe_router_model
122
- url = "#{@llama_probe_url}?model=#{CGI.escape(@model_id)}"
132
+ url = "#{@probe_url}?model=#{CGI.escape(@model_id)}"
123
133
  data = fetch_props(url)
124
134
  return nil if data.nil?
125
135
 
@@ -3,112 +3,133 @@
3
3
  module Pikuri
4
4
  class Agent
5
5
  module Control
6
- # Cooperative cancellation token. The instance is normally
7
- # constructed on the main thread and handed to
8
- # {Agent#initialize} via the +cancellable:+ kwarg; an
9
- # out-of-band caller (a SIGINT trap, a TUI key binding, an
10
- # IPC handler) calls {#cancel!} to flip the flag, and the
11
- # next {#check!} on the run thread raises {Cancelled} —
12
- # which {Agent#run_loop} catches, normalizes into an
13
- # {Event::Cancelled} on the listener stream, and re-raises
14
- # so the caller's REPL can return control to the user.
6
+ # Cooperative cancellation token. Built on the main thread and handed to
7
+ # {Agent#initialize} via +cancellable:+; an out-of-band caller (a SIGINT
8
+ # trap, a TUI key, an IPC handler) calls {#cancel!} to flip the flag, and
9
+ # the next {#check!} on the run thread raises {Cancelled} — which
10
+ # {Agent#run_loop} catches, emits as {Event::Cancelled}, and re-raises so
11
+ # the REPL returns control to the user.
15
12
  #
16
13
  # == Cancellation boundary
17
14
  #
18
- # The +Agent+ calls {#check!} from its +before_tool_call+
19
- # wiring — between an LLM response that requested a tool
20
- # and the actual tool invocation — which is the only point
21
- # at which the conversation state is consistent (no
22
- # in-flight subprocess, no half-applied write). An in-flight
23
- # LLM HTTP call is *not* interrupted; the response lands,
24
- # then the next tool-call boundary trips. An in-flight tool
25
- # (notably +Bash+) is also not interrupted — cancellation
26
- # lands after the tool returns. Both are intentional v1
27
- # scope: the "gentle cancel" semantic that pikuri promises.
15
+ # The +Agent+ calls {#check!} from +before_tool_call+ — between an LLM
16
+ # response requesting a tool and the tool invocation — the only point
17
+ # where conversation state is consistent (no in-flight subprocess, no
18
+ # half-applied write). An in-flight LLM HTTP call is *not* interrupted
19
+ # (the response lands, then the next boundary trips), nor is an in-flight
20
+ # tool (notably +Bash+). Both are the intentional "gentle cancel" v1 scope.
28
21
  #
29
- # == Thread safety
30
- #
31
- # {#cancel!} is intended to be called from a thread other
32
- # than the one running {Agent#run_loop} (the typical case
33
- # is a SIGINT trap handler on the main thread while the
34
- # agent runs on a worker, or vice versa). A plain boolean
35
- # ivar is sufficient under MRI: writes and reads of a
36
- # single reference are atomic with respect to the GVL, and
37
- # the only state transition we care about is +false → true+
38
- # before the next {#check!} fires. There is no double-cancel
39
- # hazard; repeated {#cancel!} calls are idempotent.
22
+ # {#sleep} widens that boundary by one case: code *waiting* inside a tool
23
+ # can wait through this instead of +Kernel#sleep+ and become
24
+ # interruptible. It reaches waits only — a thread blocked in IO (a Faraday
25
+ # read, +Subprocess#wait+) is still uninterruptible, so most of a tool's
26
+ # blocking is untouched by design.
40
27
  #
41
28
  # == Sub-agent semantics
42
29
  #
43
- # Cancellation is a global "stop the whole tree" signal —
44
- # the +agent+ tool from +pikuri-subagents+ shares the
45
- # parent's +Cancellable+ by reference when spawning a child,
46
- # so one {#cancel!} call stops the parent, every running
47
- # sub-agent, and the synthesizer rescue. The sharing rule
48
- # lives at the spawn site (sub-agent code), not on this
49
- # class.
30
+ # Cancellation is a global "stop the whole tree" signal — the +agent+ tool
31
+ # shares the parent's +Cancellable+ by reference with each child, so one
32
+ # {#cancel!} stops the parent, every sub-agent, and the synthesizer. The
33
+ # sharing rule lives at the spawn site, not here.
34
+ #
35
+ # Thread-safe: {#cancel!} is meant to run on a thread other than the loop's
36
+ # (a SIGINT trap while the agent runs on a worker). A plain boolean ivar
37
+ # suffices under MRI — the only transition that matters is +false → true+
38
+ # before the next {#check!}, and repeated {#cancel!} is idempotent.
50
39
  class Cancellable
51
- # Raised by {#check!} once {#cancel!} has been called.
52
- # Carries no fields; the cancellation reason ("the user
53
- # asked us to stop") is implicit in the exception class.
54
- # {Agent#run_loop} catches this, emits {Event::Cancelled},
55
- # and re-raises so the caller (typically a REPL) can
56
- # return control to the user.
40
+ # Raised by {#check!} once {#cancel!} has been called; carries no fields.
41
+ # {Agent#run_loop} catches it, emits {Event::Cancelled}, and re-raises.
57
42
  class Cancelled < StandardError
58
43
  def initialize
59
44
  super('Agent loop cancelled')
60
45
  end
61
46
  end
62
47
 
48
+ # Longest a {#sleep} stays unresponsive to a {#cancel!}: it wakes on
49
+ # this cadence to re-check the flag. Polling rather than a
50
+ # +ConditionVariable+ because {#cancel!} must stay callable from a
51
+ # SIGINT trap, where taking a +Mutex+ raises +ThreadError+ — so the
52
+ # waking side pays, and the signalling side stays a bare flag write.
53
+ #
54
+ # @return [Float] seconds
55
+ SLICE_SECONDS = 0.1
56
+
63
57
  def initialize
64
58
  @cancelled = false
65
59
  end
66
60
 
67
- # Flip the flag. Safe to call from a thread other than
68
- # the one running the agent loop, and safe to call
69
- # multiple times (idempotent). Takes effect at the next
70
- # {#check!} on the run thread — see the class header for
71
- # the "gentle cancel" caveats.
61
+ # A token that never trips — the null object for code that structurally
62
+ # needs one (a {#sleep} to wait through, a +cancellable:+ parameter) when
63
+ # the host wired no cancellation. Frozen, so a stray {#cancel!} raises
64
+ # +FrozenError+ rather than silently cancelling every agent that
65
+ # defaulted to it. Defined below {#initialize} so +@cancelled+ is set.
66
+ #
67
+ # @return [Cancellable]
68
+ NEVER = new.freeze
69
+
70
+ # Flip the flag. Safe off the loop thread and idempotent. Takes effect
71
+ # at the next {#check!} — see the class header for the gentle-cancel
72
+ # caveats.
72
73
  #
73
74
  # @return [void]
74
75
  def cancel!
75
76
  @cancelled = true
76
77
  end
77
78
 
78
- # @return [Boolean] whether {#cancel!} has been called
79
- # since the last {#reset!}; observable from any thread.
79
+ # @return [Boolean] whether {#cancel!} has been called since the last
80
+ # {#reset!}; observable from any thread.
80
81
  def cancelled?
81
82
  @cancelled
82
83
  end
83
84
 
84
- # Raise {Cancelled} when the flag is set; otherwise no-op.
85
- # Called by {Agent} from its +before_tool_call+ wiring.
85
+ # Raise {Cancelled} when the flag is set, else no-op. Called from
86
+ # {Agent}'s +before_tool_call+ wiring.
86
87
  #
87
88
  # @return [void]
88
- # @raise [Cancelled] when {#cancel!} has been called since
89
- # the last {#reset!}
89
+ # @raise [Cancelled] when {#cancel!} has been called since {#reset!}
90
90
  def check!
91
91
  raise Cancelled if @cancelled
92
92
  end
93
93
 
94
- # Reset the flag back to armed. Called by {Agent} at the
95
- # start of each turn so a stale cancellation from a prior
96
- # turn does not poison the next one. Mid-loop
97
- # {Control::Interloper} injections deliberately do *not*
98
- # trigger a reset — otherwise the cancel-then-inject
99
- # ordering would lose the cancellation: +cancel!+ sets
100
- # the flag, the injection lands and resets it, and the
101
- # next +before_tool_call+ no longer raises.
94
+ # Wait +seconds+, cutting the wait short with {Cancelled} if {#cancel!}
95
+ # lands while waiting — the drop-in for +Kernel#sleep+ anywhere a tool
96
+ # paces or polls:
97
+ #
98
+ # cancellable.sleep(2.0) # returns after 2s, or raises on a cancel
99
+ #
100
+ # Checks the flag before waiting at all, so an already-cancelled token
101
+ # raises without a delay, and a non-positive +seconds+ is still a
102
+ # cancellation check rather than a no-op. Wakes every {SLICE_SECONDS} to
103
+ # look, so a cancel is honoured within that.
104
+ #
105
+ # @param seconds [Float] how long to wait; +<= 0+ returns at once.
106
+ # @return [void]
107
+ # @raise [Cancelled] if {#cancel!} has been called, before or during
108
+ # the wait.
109
+ def sleep(seconds)
110
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds
111
+ loop do
112
+ check!
113
+ remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)
114
+ return if remaining <= 0
115
+
116
+ # Kernel., or this recurses into itself.
117
+ Kernel.sleep([remaining, SLICE_SECONDS].min)
118
+ end
119
+ end
120
+
121
+ # Reset the flag to armed. Called by {Agent} at each turn start so a
122
+ # stale cancel doesn't poison the next turn. Mid-loop
123
+ # {Control::Interloper} injections deliberately do *not* reset —
124
+ # otherwise cancel-then-inject would lose the cancel (the injection
125
+ # resets the flag before the next +before_tool_call+ reads it).
102
126
  #
103
127
  # @return [void]
104
128
  def reset!
105
129
  @cancelled = false
106
130
  end
107
131
 
108
- # @return [String] short label for {Agent#to_s}; reflects
109
- # the current flag state so a startup banner or debug
110
- # print can tell an armed token apart from one that has
111
- # already tripped.
132
+ # @return [String] short label for {Agent#to_s}, reflecting the flag state.
112
133
  def to_s
113
134
  "Cancellable(#{@cancelled ? 'cancelled' : 'armed'})"
114
135
  end