pikuri-core 0.0.7 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -1
  3. data/lib/pikuri/agent/chat_transport.rb +73 -93
  4. data/lib/pikuri/agent/configurator.rb +46 -106
  5. data/lib/pikuri/agent/context_window_detector.rb +44 -85
  6. data/lib/pikuri/agent/control/cancellable.rb +87 -66
  7. data/lib/pikuri/agent/control/interloper.rb +127 -105
  8. data/lib/pikuri/agent/control/step_limit.rb +25 -41
  9. data/lib/pikuri/agent/control.rb +14 -34
  10. data/lib/pikuri/agent/event.rb +123 -188
  11. data/lib/pikuri/agent/extension.rb +118 -94
  12. data/lib/pikuri/agent/extension_context.rb +50 -77
  13. data/lib/pikuri/agent/history.rb +653 -0
  14. data/lib/pikuri/agent/listener/rate_limited.rb +40 -66
  15. data/lib/pikuri/agent/listener/terminal.rb +143 -117
  16. data/lib/pikuri/agent/listener/token_log.rb +101 -140
  17. data/lib/pikuri/agent/listener.rb +23 -43
  18. data/lib/pikuri/agent/listener_list.rb +26 -47
  19. data/lib/pikuri/agent/synthesizer.rb +45 -87
  20. data/lib/pikuri/agent.rb +816 -474
  21. data/lib/pikuri/bundler_env.rb +68 -0
  22. data/lib/pikuri/extractor/html.rb +63 -110
  23. data/lib/pikuri/extractor/passthrough.rb +20 -30
  24. data/lib/pikuri/extractor.rb +93 -154
  25. data/lib/pikuri/file_type.rb +63 -135
  26. data/lib/pikuri/finalizers.rb +32 -47
  27. data/lib/pikuri/paths.rb +104 -13
  28. data/lib/pikuri/ruby_llm_patches.rb +106 -0
  29. data/lib/pikuri/sanitizer.rb +45 -67
  30. data/lib/pikuri/subprocess.rb +75 -119
  31. data/lib/pikuri/testing.rb +296 -0
  32. data/lib/pikuri/tool/calculator.rb +56 -66
  33. data/lib/pikuri/tool/execute_context.rb +42 -0
  34. data/lib/pikuri/tool/fetch.rb +51 -77
  35. data/lib/pikuri/tool/parameters.rb +21 -29
  36. data/lib/pikuri/tool/scraper.rb +55 -97
  37. data/lib/pikuri/tool/search/brave.rb +52 -80
  38. data/lib/pikuri/tool/search/duckduckgo.rb +59 -91
  39. data/lib/pikuri/tool/search/engines.rb +230 -97
  40. data/lib/pikuri/tool/search/exa.rb +56 -90
  41. data/lib/pikuri/tool/search/rate_limiter.rb +61 -38
  42. data/lib/pikuri/tool/search/result.rb +10 -15
  43. data/lib/pikuri/tool/trifecta_legs.rb +217 -0
  44. data/lib/pikuri/tool/web_scrape.rb +38 -54
  45. data/lib/pikuri/tool/web_search.rb +100 -24
  46. data/lib/pikuri/tool.rb +140 -65
  47. data/lib/pikuri/trifecta/contribution.rb +43 -0
  48. data/lib/pikuri/trifecta/node.rb +47 -0
  49. data/lib/pikuri/trifecta/report.rb +230 -0
  50. data/lib/pikuri/trifecta.rb +127 -0
  51. data/lib/pikuri/url_cache.rb +33 -49
  52. data/lib/pikuri/version.rb +1 -1
  53. data/lib/pikuri-core.rb +72 -88
  54. data/prompts/agent-loop.txt +5 -0
  55. data/prompts/pikuri-chat.txt +3 -12
  56. metadata +14 -3
@@ -7,72 +7,43 @@ require 'cgi'
7
7
 
8
8
  module Pikuri
9
9
  class Agent
10
- # Resolves the model's context-window cap by asking the server that
11
- # actually serves it. The only authoritative runtime source pikuri
12
- # has is llama.cpp's non-standard +/props+ endpoint, which reports
13
- # the server's *launched* +n_ctx+ (the real window — possibly
14
- # smaller than the model's theoretical max, e.g. +llama-server -c
15
- # 8192+ on a 128k model). Returns +nil+ — an honest "we don't
16
- # know" — for anything else.
10
+ # Resolves the model's context-window cap by asking the server that serves
11
+ # it. The only authoritative runtime source is llama.cpp's non-standard
12
+ # +/props+ endpoint, reporting the server's *launched* +n_ctx+ (the real
13
+ # window — possibly smaller than the model's max, e.g. +llama-server -c
14
+ # 8192+ on a 128k model). Returns +nil+ — an honest "we don't know" — for
15
+ # anything else. {Agent#detect_and_emit_context_cap!} calls it only when
16
+ # the transport carries no explicit {ChatTransport#context_window}.
17
17
  #
18
- # Used by {Agent#detect_and_emit_context_cap!} at construction and
19
- # after every model switch to feed {Listener::TokenLog} a cap it can
20
- # render alongside the running context size (so the +ctx=12.2k/32.0k+
21
- # line tells the operator how close the conversation is to the limit).
22
- # The caller prefers an explicit/inherited
23
- # {ChatTransport#context_window} over this probe; this runs only when
24
- # the transport carries none.
25
- #
26
- # == Why no ruby_llm registry source
27
- #
28
- # +RubyLLM::Model::Info#context_window+ is a static lookup in a
29
- # bundled +models.json+ snapshot: +nil+ for every +assume_exists+
30
- # local model id, +nil+ for anything newer than the snapshot, and —
31
- # worst — a *frozen* value for known models, so a window the provider
32
- # later bumped (256k → 1M) still reports the old number. A cap you
33
- # have to caveat defeats the cap's only job (a number trustworthy
34
- # enough to act on before +RubyLLM::ContextLengthExceededError+), so
35
- # pikuri deliberately does not consult it. The probe (server truth)
36
- # and an explicit {ChatTransport#context_window} (operator/parent
37
- # truth) are the only two sources; absent both, the cap is +nil+.
18
+ # This probe and an explicit {ChatTransport#context_window} are the *only*
19
+ # sources: +RubyLLM::Model::Info#context_window+ is deliberately never
20
+ # consulted (why: +DECISIONS.md+ +D_context_window_sources+).
38
21
  #
39
22
  # == The openai-provider gate + auto-derived URL
40
23
  #
41
- # The probe only makes sense against an OpenAI-compatible local
42
- # server (llama.cpp), reached through ruby_llm's +:openai+ provider
43
- # with a custom base. So {.detect} runs only when
44
- # +transport.provider == :openai+ and derives the probe URL from the
45
- # *same* +RubyLLM.config.openai_api_base+ the chat itself uses —
46
- # +/props+ lives at the host root, NOT under +/v1+, so the +/v1+
47
- # suffix is stripped. Deriving from the live config (rather than a
48
- # URL passed in) means the probe can't target a different server than
49
- # the chat. A bare +:openai+ pointed at real +api.openai.com+ gets
50
- # one fast +/props+ 404 that degrades to +nil+ (the simple gate; not
51
- # worth narrowing — you're already sending that server the whole
52
- # conversation).
24
+ # The probe only makes sense against an OpenAI-compatible local server
25
+ # (llama.cpp) via ruby_llm's +:openai+ provider, so {.detect} runs only for
26
+ # +transport.provider == :openai+ and derives the URL from the *same*
27
+ # +RubyLLM.config.openai_api_base+ the chat uses (+/props+ is at the host
28
+ # root, so the +/v1+ suffix is stripped) — so it can't target a different
29
+ # server than the chat. A bare +:openai+ at real +api.openai.com+ gets one
30
+ # fast +/props+ 404 that degrades to +nil+.
53
31
  #
54
32
  # == llama.cpp router mode
55
33
  #
56
- # A llama.cpp *router* (the multi-instance front that proxies to N
57
- # on-demand model servers) answers a bare +/props+ with
58
- # +{"role":"router", ..., "n_ctx":0}+ — there is no single loaded
59
- # model at the router itself, so its top-level +n_ctx+ is +0+. The
60
- # real per-model cap is one proxied hop away: +GET /props?model=<id>+
61
- # routes the probe to that model's instance, whose +/props+ carries
62
- # the launched +n_ctx+. So when the bare probe reports +role: router+
63
- # and a +model_id+ is known, this re-probes with the model id before
64
- # giving up. A plain single-model server is untouched: its bare
65
- # +/props+ already carries a positive +n_ctx+, so the router branch
66
- # never runs.
34
+ # A llama.cpp *router* (the multi-instance front) answers a bare +/props+
35
+ # with +{"role":"router", ..., "n_ctx":0}+ — no single loaded model. The
36
+ # real per-model cap is one hop away: +GET /props?model=<id>+ routes to that
37
+ # model's instance. So when the bare probe reports +role: router+ and a
38
+ # +model_id+ is known, this re-probes with the id. A plain single-model
39
+ # server (bare +/props+ already positive) never hits this branch.
67
40
  #
68
41
  # == Failure handling
69
42
  #
70
- # The probe is best-effort. HTTP error, timeout, non-JSON body, or a
71
- # missing/invalid +n_ctx+ field all return +nil+ and log one +warn+
72
- # line via +Pikuri.logger_for('ContextWindowDetector')+. This is the
73
- # CLAUDE.md "secondary to the loop" carve-out — a wedged or
74
- # non-llama.cpp server should not abort agent construction over a
75
- # cosmetic readout.
43
+ # Best-effort: HTTP error, timeout, non-JSON body, or a missing/invalid
44
+ # +n_ctx+ all return +nil+ with one +warn+ line. The CLAUDE.md
45
+ # "secondary to the loop" carve-out — a wedged or non-llama.cpp server must
46
+ # not abort agent construction over a cosmetic readout.
76
47
  class ContextWindowDetector
77
48
  # Subsystem logger; set its level with
78
49
  # +PIKURI_LOG_CONTEXTWINDOWDETECTOR+ or the global +PIKURI_LOG+.
@@ -80,39 +51,29 @@ module Pikuri
80
51
  # @return [Logger]
81
52
  LOGGER = Pikuri.logger_for('ContextWindowDetector')
82
53
 
83
- # Connect timeout in seconds for the llama.cpp +/props+ probe.
84
- # Short on purpose: a server that isn't even listening should fail
85
- # fast rather than stall +Agent+ construction.
54
+ # Connect timeout (s) for the +/props+ probe. Short: a server that isn't
55
+ # listening should fail fast rather than stall {Agent} construction.
86
56
  #
87
57
  # @return [Integer]
88
58
  OPEN_TIMEOUT = 2
89
- # Read timeout in seconds for the llama.cpp +/props+ probe.
90
- # Generous on purpose, and the reason it differs from
91
- # {OPEN_TIMEOUT}: a llama.cpp router answers +/props?model=<id>+
92
- # only *after* spinning up that model's instance, and a cold model
93
- # load can take 10+ seconds — which the next chat turn must wait
94
- # for anyway. A read timeout shorter than the load would abandon
95
- # the probe (and lose the cap) precisely when switching to a
96
- # cold model. A server that accepts the connection but then hangs
97
- # would stall the actual chat identically, so tolerating the wait
98
- # here costs nothing extra.
59
+ # Read timeout (s) for the +/props+ probe — generous, unlike
60
+ # {OPEN_TIMEOUT}: a router answers +/props?model=<id>+ only after spinning
61
+ # up that model's instance, and a cold load can take 10+s (which the next
62
+ # chat turn must wait for anyway). A shorter timeout would lose the cap
63
+ # exactly when switching to a cold model.
99
64
  #
100
65
  # @return [Integer]
101
66
  READ_TIMEOUT = 30
102
67
 
103
- # Resolve the context-window cap for +transport+ by probing the
104
- # server that serves it.
68
+ # Resolve the context-window cap for +transport+ by probing its server.
105
69
  #
106
- # @param transport [Agent::ChatTransport] the model-resolution
107
- # triple; +provider+ gates the probe and +model+ drives the
108
- # router +?model=+ hop
109
- # @param openai_base [String, nil] the configured OpenAI-compatible
110
- # base URL the probe URL is derived from; defaults to the live
111
- # +RubyLLM.config.openai_api_base+. Passed explicitly only by
112
- # tests, which don't want to mutate global config.
70
+ # @param transport [Agent::ChatTransport] +provider+ gates the probe,
71
+ # +model+ drives the router +?model=+ hop
72
+ # @param openai_base [String, nil] base URL the probe URL derives from;
73
+ # defaults to the live +RubyLLM.config.openai_api_base+ (passed
74
+ # explicitly only by tests, to avoid mutating global config)
113
75
  # @return [Integer, nil] the launched +n_ctx+, or +nil+ for a
114
- # non-+:openai+ transport, an unconfigured base, or any probe
115
- # failure
76
+ # non-+:openai+ transport, unconfigured base, or any probe failure
116
77
  def self.detect(transport, openai_base: RubyLLM.config.openai_api_base)
117
78
  return nil unless transport.provider == :openai
118
79
 
@@ -122,13 +83,11 @@ module Pikuri
122
83
  new(probe_url: url, model_id: transport.model).probe
123
84
  end
124
85
 
125
- # Derive the llama.cpp +/props+ URL from the OpenAI-compatible
126
- # base. +/props+ sits at the host root, so a trailing +/v1+ is
127
- # stripped before appending.
86
+ # Derive the +/props+ URL from the OpenAI-compatible base (+/props+ is at
87
+ # the host root, so a trailing +/v1+ is stripped).
128
88
  #
129
89
  # @param openai_base [String, nil]
130
- # @return [String, nil] the +/props+ URL, or +nil+ when the base
131
- # is blank
90
+ # @return [String, nil] the +/props+ URL, or +nil+ when the base is blank
132
91
  def self.props_url(openai_base)
133
92
  base = openai_base.to_s.strip.chomp('/')
134
93
  return nil if base.empty?
@@ -3,112 +3,133 @@
3
3
  module Pikuri
4
4
  class Agent
5
5
  module Control
6
- # Cooperative cancellation token. The instance is normally
7
- # constructed on the main thread and handed to
8
- # {Agent#initialize} via the +cancellable:+ kwarg; an
9
- # out-of-band caller (a SIGINT trap, a TUI key binding, an
10
- # IPC handler) calls {#cancel!} to flip the flag, and the
11
- # next {#check!} on the run thread raises {Cancelled} —
12
- # which {Agent#run_loop} catches, normalizes into an
13
- # {Event::Cancelled} on the listener stream, and re-raises
14
- # so the caller's REPL can return control to the user.
6
+ # Cooperative cancellation token. Built on the main thread and handed to
7
+ # {Agent#initialize} via +cancellable:+; an out-of-band caller (a SIGINT
8
+ # trap, a TUI key, an IPC handler) calls {#cancel!} to flip the flag, and
9
+ # the next {#check!} on the run thread raises {Cancelled} — which
10
+ # {Agent#run_loop} catches, emits as {Event::Cancelled}, and re-raises so
11
+ # the REPL returns control to the user.
15
12
  #
16
13
  # == Cancellation boundary
17
14
  #
18
- # The +Agent+ calls {#check!} from its +before_tool_call+
19
- # wiring — between an LLM response that requested a tool
20
- # and the actual tool invocation — which is the only point
21
- # at which the conversation state is consistent (no
22
- # in-flight subprocess, no half-applied write). An in-flight
23
- # LLM HTTP call is *not* interrupted; the response lands,
24
- # then the next tool-call boundary trips. An in-flight tool
25
- # (notably +Bash+) is also not interrupted — cancellation
26
- # lands after the tool returns. Both are intentional v1
27
- # scope: the "gentle cancel" semantic that pikuri promises.
15
+ # The +Agent+ calls {#check!} from +before_tool_call+ — between an LLM
16
+ # response requesting a tool and the tool invocation — the only point
17
+ # where conversation state is consistent (no in-flight subprocess, no
18
+ # half-applied write). An in-flight LLM HTTP call is *not* interrupted
19
+ # (the response lands, then the next boundary trips), nor is an in-flight
20
+ # tool (notably +Bash+). Both are the intentional "gentle cancel" v1 scope.
28
21
  #
29
- # == Thread safety
30
- #
31
- # {#cancel!} is intended to be called from a thread other
32
- # than the one running {Agent#run_loop} (the typical case
33
- # is a SIGINT trap handler on the main thread while the
34
- # agent runs on a worker, or vice versa). A plain boolean
35
- # ivar is sufficient under MRI: writes and reads of a
36
- # single reference are atomic with respect to the GVL, and
37
- # the only state transition we care about is +false → true+
38
- # before the next {#check!} fires. There is no double-cancel
39
- # hazard; repeated {#cancel!} calls are idempotent.
22
+ # {#sleep} widens that boundary by one case: code *waiting* inside a tool
23
+ # can wait through this instead of +Kernel#sleep+ and become
24
+ # interruptible. It reaches waits only — a thread blocked in IO (a Faraday
25
+ # read, +Subprocess#wait+) is still uninterruptible, so most of a tool's
26
+ # blocking is untouched by design.
40
27
  #
41
28
  # == Sub-agent semantics
42
29
  #
43
- # Cancellation is a global "stop the whole tree" signal —
44
- # the +agent+ tool from +pikuri-subagents+ shares the
45
- # parent's +Cancellable+ by reference when spawning a child,
46
- # so one {#cancel!} call stops the parent, every running
47
- # sub-agent, and the synthesizer rescue. The sharing rule
48
- # lives at the spawn site (sub-agent code), not on this
49
- # class.
30
+ # Cancellation is a global "stop the whole tree" signal — the +agent+ tool
31
+ # shares the parent's +Cancellable+ by reference with each child, so one
32
+ # {#cancel!} stops the parent, every sub-agent, and the synthesizer. The
33
+ # sharing rule lives at the spawn site, not here.
34
+ #
35
+ # Thread-safe: {#cancel!} is meant to run on a thread other than the loop's
36
+ # (a SIGINT trap while the agent runs on a worker). A plain boolean ivar
37
+ # suffices under MRI — the only transition that matters is +false → true+
38
+ # before the next {#check!}, and repeated {#cancel!} is idempotent.
50
39
  class Cancellable
51
- # Raised by {#check!} once {#cancel!} has been called.
52
- # Carries no fields; the cancellation reason ("the user
53
- # asked us to stop") is implicit in the exception class.
54
- # {Agent#run_loop} catches this, emits {Event::Cancelled},
55
- # and re-raises so the caller (typically a REPL) can
56
- # return control to the user.
40
+ # Raised by {#check!} once {#cancel!} has been called; carries no fields.
41
+ # {Agent#run_loop} catches it, emits {Event::Cancelled}, and re-raises.
57
42
  class Cancelled < StandardError
58
43
  def initialize
59
44
  super('Agent loop cancelled')
60
45
  end
61
46
  end
62
47
 
48
+ # Longest a {#sleep} stays unresponsive to a {#cancel!}: it wakes on
49
+ # this cadence to re-check the flag. Polling rather than a
50
+ # +ConditionVariable+ because {#cancel!} must stay callable from a
51
+ # SIGINT trap, where taking a +Mutex+ raises +ThreadError+ — so the
52
+ # waking side pays, and the signalling side stays a bare flag write.
53
+ #
54
+ # @return [Float] seconds
55
+ SLICE_SECONDS = 0.1
56
+
63
57
  def initialize
64
58
  @cancelled = false
65
59
  end
66
60
 
67
- # Flip the flag. Safe to call from a thread other than
68
- # the one running the agent loop, and safe to call
69
- # multiple times (idempotent). Takes effect at the next
70
- # {#check!} on the run thread — see the class header for
71
- # the "gentle cancel" caveats.
61
+ # A token that never trips — the null object for code that structurally
62
+ # needs one (a {#sleep} to wait through, a +cancellable:+ parameter) when
63
+ # the host wired no cancellation. Frozen, so a stray {#cancel!} raises
64
+ # +FrozenError+ rather than silently cancelling every agent that
65
+ # defaulted to it. Defined below {#initialize} so +@cancelled+ is set.
66
+ #
67
+ # @return [Cancellable]
68
+ NEVER = new.freeze
69
+
70
+ # Flip the flag. Safe off the loop thread and idempotent. Takes effect
71
+ # at the next {#check!} — see the class header for the gentle-cancel
72
+ # caveats.
72
73
  #
73
74
  # @return [void]
74
75
  def cancel!
75
76
  @cancelled = true
76
77
  end
77
78
 
78
- # @return [Boolean] whether {#cancel!} has been called
79
- # since the last {#reset!}; observable from any thread.
79
+ # @return [Boolean] whether {#cancel!} has been called since the last
80
+ # {#reset!}; observable from any thread.
80
81
  def cancelled?
81
82
  @cancelled
82
83
  end
83
84
 
84
- # Raise {Cancelled} when the flag is set; otherwise no-op.
85
- # Called by {Agent} from its +before_tool_call+ wiring.
85
+ # Raise {Cancelled} when the flag is set, else no-op. Called from
86
+ # {Agent}'s +before_tool_call+ wiring.
86
87
  #
87
88
  # @return [void]
88
- # @raise [Cancelled] when {#cancel!} has been called since
89
- # the last {#reset!}
89
+ # @raise [Cancelled] when {#cancel!} has been called since {#reset!}
90
90
  def check!
91
91
  raise Cancelled if @cancelled
92
92
  end
93
93
 
94
- # Reset the flag back to armed. Called by {Agent} at the
95
- # start of each turn so a stale cancellation from a prior
96
- # turn does not poison the next one. Mid-loop
97
- # {Control::Interloper} injections deliberately do *not*
98
- # trigger a reset — otherwise the cancel-then-inject
99
- # ordering would lose the cancellation: +cancel!+ sets
100
- # the flag, the injection lands and resets it, and the
101
- # next +before_tool_call+ no longer raises.
94
+ # Wait +seconds+, cutting the wait short with {Cancelled} if {#cancel!}
95
+ # lands while waiting — the drop-in for +Kernel#sleep+ anywhere a tool
96
+ # paces or polls:
97
+ #
98
+ # cancellable.sleep(2.0) # returns after 2s, or raises on a cancel
99
+ #
100
+ # Checks the flag before waiting at all, so an already-cancelled token
101
+ # raises without a delay, and a non-positive +seconds+ is still a
102
+ # cancellation check rather than a no-op. Wakes every {SLICE_SECONDS} to
103
+ # look, so a cancel is honoured within that.
104
+ #
105
+ # @param seconds [Float] how long to wait; +<= 0+ returns at once.
106
+ # @return [void]
107
+ # @raise [Cancelled] if {#cancel!} has been called, before or during
108
+ # the wait.
109
+ def sleep(seconds)
110
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds
111
+ loop do
112
+ check!
113
+ remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)
114
+ return if remaining <= 0
115
+
116
+ # Kernel., or this recurses into itself.
117
+ Kernel.sleep([remaining, SLICE_SECONDS].min)
118
+ end
119
+ end
120
+
121
+ # Reset the flag to armed. Called by {Agent} at each turn start so a
122
+ # stale cancel doesn't poison the next turn. Mid-loop
123
+ # {Control::Interloper} injections deliberately do *not* reset —
124
+ # otherwise cancel-then-inject would lose the cancel (the injection
125
+ # resets the flag before the next +before_tool_call+ reads it).
102
126
  #
103
127
  # @return [void]
104
128
  def reset!
105
129
  @cancelled = false
106
130
  end
107
131
 
108
- # @return [String] short label for {Agent#to_s}; reflects
109
- # the current flag state so a startup banner or debug
110
- # print can tell an armed token apart from one that has
111
- # already tripped.
132
+ # @return [String] short label for {Agent#to_s}, reflecting the flag state.
112
133
  def to_s
113
134
  "Cancellable(#{@cancelled ? 'cancelled' : 'armed'})"
114
135
  end