pcli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pcli/__init__.py +1 -0
- pcli/__main__.py +4 -0
- pcli/agent/__init__.py +0 -0
- pcli/agent/activity.py +116 -0
- pcli/agent/compaction.py +205 -0
- pcli/agent/context_pruning.py +88 -0
- pcli/agent/headless.py +209 -0
- pcli/agent/loop.py +442 -0
- pcli/agent/prompt.py +371 -0
- pcli/agent/runtime.py +240 -0
- pcli/browser/__init__.py +0 -0
- pcli/browser/session.py +135 -0
- pcli/cli.py +757 -0
- pcli/config/__init__.py +0 -0
- pcli/config/paths.py +95 -0
- pcli/config/settings.py +435 -0
- pcli/cost/__init__.py +0 -0
- pcli/cost/context.py +275 -0
- pcli/cost/context_detect.py +183 -0
- pcli/cost/pricing_table.py +141 -0
- pcli/cost/tracker.py +126 -0
- pcli/llm/__init__.py +0 -0
- pcli/llm/client.py +285 -0
- pcli/llm/errors.py +37 -0
- pcli/llm/models.py +100 -0
- pcli/llm/streaming.py +108 -0
- pcli/memory/__init__.py +0 -0
- pcli/memory/extraction.py +106 -0
- pcli/memory/models.py +103 -0
- pcli/memory/store.py +88 -0
- pcli/permissions/__init__.py +0 -0
- pcli/permissions/guardrails.py +219 -0
- pcli/permissions/manager.py +215 -0
- pcli/permissions/policy.py +70 -0
- pcli/sandbox/__init__.py +0 -0
- pcli/sandbox/base.py +50 -0
- pcli/sandbox/docker_backend.py +107 -0
- pcli/sandbox/limits.py +63 -0
- pcli/sandbox/null_backend.py +92 -0
- pcli/sandbox/selector.py +75 -0
- pcli/sandbox/subprocess_backend.py +376 -0
- pcli/scheduler/__init__.py +0 -0
- pcli/scheduler/daemon.py +194 -0
- pcli/scheduler/models.py +97 -0
- pcli/scheduler/runner.py +84 -0
- pcli/scheduler/store.py +75 -0
- pcli/scheduler/triggers.py +84 -0
- pcli/session/__init__.py +0 -0
- pcli/session/audit.py +122 -0
- pcli/session/directory_check.py +28 -0
- pcli/session/export.py +57 -0
- pcli/session/importer.py +92 -0
- pcli/session/models.py +168 -0
- pcli/session/store.py +127 -0
- pcli/telegram/__init__.py +0 -0
- pcli/telegram/bot.py +266 -0
- pcli/telegram/daemon.py +1197 -0
- pcli/telegram/permissions.py +131 -0
- pcli/telegram/sender.py +58 -0
- pcli/tools/__init__.py +0 -0
- pcli/tools/_nested_agent.py +204 -0
- pcli/tools/agent_tools.py +264 -0
- pcli/tools/agent_tools_store.py +69 -0
- pcli/tools/artifacts.py +47 -0
- pcli/tools/base.py +185 -0
- pcli/tools/builtin/__init__.py +0 -0
- pcli/tools/builtin/agent_tool_register_tool.py +100 -0
- pcli/tools/builtin/artifact_tool.py +212 -0
- pcli/tools/builtin/ask_tool.py +77 -0
- pcli/tools/builtin/browser_tool.py +253 -0
- pcli/tools/builtin/decision_tool.py +73 -0
- pcli/tools/builtin/describe_tool.py +389 -0
- pcli/tools/builtin/diff_tools.py +225 -0
- pcli/tools/builtin/fs_tools.py +371 -0
- pcli/tools/builtin/grep_tool.py +88 -0
- pcli/tools/builtin/memory_tool.py +108 -0
- pcli/tools/builtin/network_tools.py +107 -0
- pcli/tools/builtin/pip_tool.py +106 -0
- pcli/tools/builtin/shell_tool.py +240 -0
- pcli/tools/builtin/subagent_tool.py +146 -0
- pcli/tools/builtin/todo_tool.py +122 -0
- pcli/tools/builtin/toolbox_register_tool.py +76 -0
- pcli/tools/builtin/web_tools.py +322 -0
- pcli/tools/pydiscovery/__init__.py +0 -0
- pcli/tools/pydiscovery/cache.py +51 -0
- pcli/tools/pydiscovery/index.py +48 -0
- pcli/tools/pydiscovery/invoke.py +181 -0
- pcli/tools/pydiscovery/search.py +117 -0
- pcli/tools/registry.py +138 -0
- pcli/tools/toolbox/__init__.py +0 -0
- pcli/tools/toolbox/introspect.py +48 -0
- pcli/tools/toolbox/manager.py +336 -0
- pcli/tools/toolbox/plugin_base.py +51 -0
- pcli/tools/toolbox/plugins/__init__.py +6 -0
- pcli/tools/toolbox/plugins/httpd.py +99 -0
- pcli/tools/toolbox/plugins/kafka.py +162 -0
- pcli/tools/toolbox/plugins/kubectl.py +211 -0
- pcli/tools/toolbox/plugins/sge.py +146 -0
- pcli/tools/toolbox/store.py +65 -0
- pcli/tools/toolbox/synthesize.py +100 -0
- pcli/tui/__init__.py +0 -0
- pcli/tui/app.py +37 -0
- pcli/tui/screens/__init__.py +0 -0
- pcli/tui/screens/ask_question_modal.py +54 -0
- pcli/tui/screens/chat.py +2070 -0
- pcli/tui/screens/confirm_modal.py +39 -0
- pcli/tui/screens/models.py +43 -0
- pcli/tui/screens/permission_modal.py +71 -0
- pcli/tui/screens/sessions.py +162 -0
- pcli/tui/screens/subagent_activity_modal.py +71 -0
- pcli/tui/shell_passthrough.py +56 -0
- pcli/tui/styles/pcli.tcss +241 -0
- pcli/tui/themes.py +84 -0
- pcli/tui/widgets/__init__.py +0 -0
- pcli/tui/widgets/chat_input.py +240 -0
- pcli/tui/widgets/command_suggestions.py +33 -0
- pcli/tui/widgets/message_view.py +328 -0
- pcli/tui/widgets/paste_input.py +99 -0
- pcli/tui/widgets/paste_marker.py +69 -0
- pcli/tui/widgets/status_bar.py +133 -0
- pcli/tui/widgets/status_pane.py +58 -0
- pcli/util/__init__.py +0 -0
- pcli/util/ids.py +15 -0
- pcli/util/logging.py +18 -0
- pcli/util/text.py +10 -0
- pcli_agent-0.1.0.dist-info/METADATA +259 -0
- pcli_agent-0.1.0.dist-info/RECORD +130 -0
- pcli_agent-0.1.0.dist-info/WHEEL +4 -0
- pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
pcli/__main__.py
ADDED
pcli/agent/__init__.py
ADDED
|
File without changes
|
pcli/agent/activity.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Ephemeral live-activity state for the TUI's status pane.
|
|
2
|
+
|
|
3
|
+
Deliberately NOT part of Session (not persisted, not exported/imported): it
|
|
4
|
+
exists only so a running subagent can report its progress up to the screen
|
|
5
|
+
without cluttering the main conversation with its intermediate tool calls
|
|
6
|
+
(see tools/builtin/subagent_tool.py's own docstring on that design choice).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
|
|
14
|
+
from pcli.util.text import truncate
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class SubagentToolCall:
|
|
19
|
+
name: str
|
|
20
|
+
arguments: str
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class SubagentActivity:
|
|
25
|
+
task: str
|
|
26
|
+
tool_calls: int = 0
|
|
27
|
+
last_tool: str | None = None
|
|
28
|
+
call_log: list[SubagentToolCall] = field(default_factory=list)
|
|
29
|
+
"""Full history of tool calls made so far (name + raw arguments JSON),
|
|
30
|
+
for the /subagent inspect command and for giving the user something to
|
|
31
|
+
relate an in-flight ask_user_question to - tool_calls/last_tool above
|
|
32
|
+
stay as plain counters/names for the existing status-bar line."""
|
|
33
|
+
pending_question: tuple[str, list[str] | None] | None = None
|
|
34
|
+
"""Set while the subagent is blocked inside ask_user_question, so /subagent
|
|
35
|
+
can show what it's currently waiting on."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ActivityTracker:
|
|
39
|
+
"""One shared instance per chat screen. A single subagent slot is enough
|
|
40
|
+
since the agent loop dispatches tool calls sequentially, never
|
|
41
|
+
concurrently — at most one subagent is ever running at a time."""
|
|
42
|
+
|
|
43
|
+
def __init__(self) -> None:
|
|
44
|
+
self._subagent: SubagentActivity | None = None
|
|
45
|
+
self._subscribers: list[Callable[[], None]] = []
|
|
46
|
+
|
|
47
|
+
def subscribe(self, callback: Callable[[], None]) -> None:
|
|
48
|
+
self._subscribers.append(callback)
|
|
49
|
+
|
|
50
|
+
def unsubscribe(self, callback: Callable[[], None]) -> None:
|
|
51
|
+
"""No-op if callback isn't currently subscribed (e.g. a modal that
|
|
52
|
+
never mounted, or an already-cleaned-up double dismiss) - cleanup
|
|
53
|
+
code should never have to guard this call itself."""
|
|
54
|
+
if callback in self._subscribers:
|
|
55
|
+
self._subscribers.remove(callback)
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def subagent(self) -> SubagentActivity | None:
|
|
59
|
+
return self._subagent
|
|
60
|
+
|
|
61
|
+
def start_subagent(self, task: str) -> None:
|
|
62
|
+
self._subagent = SubagentActivity(task=task)
|
|
63
|
+
self._notify()
|
|
64
|
+
|
|
65
|
+
def record_subagent_tool_call(self, tool_name: str, arguments: str = "") -> None:
|
|
66
|
+
if self._subagent is None:
|
|
67
|
+
return
|
|
68
|
+
self._subagent.tool_calls += 1
|
|
69
|
+
self._subagent.last_tool = tool_name
|
|
70
|
+
self._subagent.call_log.append(SubagentToolCall(name=tool_name, arguments=arguments))
|
|
71
|
+
self._notify()
|
|
72
|
+
|
|
73
|
+
def set_subagent_pending_question(self, question: str, options: list[str] | None) -> None:
|
|
74
|
+
if self._subagent is None:
|
|
75
|
+
return
|
|
76
|
+
self._subagent.pending_question = (question, options)
|
|
77
|
+
self._notify()
|
|
78
|
+
|
|
79
|
+
def clear_subagent_pending_question(self) -> None:
|
|
80
|
+
if self._subagent is None:
|
|
81
|
+
return
|
|
82
|
+
self._subagent.pending_question = None
|
|
83
|
+
self._notify()
|
|
84
|
+
|
|
85
|
+
def finish_subagent(self) -> None:
|
|
86
|
+
self._subagent = None
|
|
87
|
+
self._notify()
|
|
88
|
+
|
|
89
|
+
def _notify(self) -> None:
|
|
90
|
+
for callback in self._subscribers:
|
|
91
|
+
callback()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def format_subagent_activity(sub: SubagentActivity) -> str:
|
|
95
|
+
"""Shared rendering of a SubagentActivity snapshot - used by both the
|
|
96
|
+
/subagent command (a one-off system message, rendered as Markdown - see
|
|
97
|
+
message_view.py's _render_message) and SubagentActivityModal (a live
|
|
98
|
+
view wrapped in Text instead, precisely to sidestep Rich console markup
|
|
99
|
+
parsing of untrusted tool arguments - see that widget's own docstring).
|
|
100
|
+
Real Markdown here (blank lines between blocks) so the /subagent
|
|
101
|
+
rendering doesn't collapse into one run-on paragraph the way plain
|
|
102
|
+
"\\n"-joined lines would; "N. " tool-call entries were already valid
|
|
103
|
+
Markdown ordered-list syntax on their own, so those are unchanged."""
|
|
104
|
+
lines = [f"**Subagent task:** {sub.task}", "", f"**Tool calls so far:** {len(sub.call_log)}"]
|
|
105
|
+
if sub.pending_question is not None:
|
|
106
|
+
question, options = sub.pending_question
|
|
107
|
+
lines.append("")
|
|
108
|
+
lines.append(f"**Currently waiting on your answer to:**\n{question}")
|
|
109
|
+
if options:
|
|
110
|
+
lines.append("")
|
|
111
|
+
lines.append("Options: " + ", ".join(options))
|
|
112
|
+
if sub.call_log:
|
|
113
|
+
lines.append("")
|
|
114
|
+
for i, call in enumerate(sub.call_log, start=1):
|
|
115
|
+
lines.append(f"{i}. {call.name}({truncate(call.arguments, 200)})")
|
|
116
|
+
return "\n".join(lines)
|
pcli/agent/compaction.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Auto-compaction: when a session's conversation grows large relative to its
|
|
2
|
+
model's context window, summarizes the oldest turns via a single dedicated
|
|
3
|
+
LLM call and replaces them with the summary — archiving the original, full
|
|
4
|
+
transcript to the artifact library first (the same ArtifactStore/
|
|
5
|
+
fetch_artifact mechanism that already archives oversized tool results in
|
|
6
|
+
agent/loop.py's _archive_if_large), so nothing is silently lost, just moved
|
|
7
|
+
out of the live context.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
from pcli.llm.client import GatewayClient
|
|
15
|
+
from pcli.llm.models import ChatMessage, Usage
|
|
16
|
+
from pcli.session.models import Message, Session
|
|
17
|
+
from pcli.tools.artifacts import ArtifactStore
|
|
18
|
+
from pcli.tools.builtin.todo_tool import render_todos
|
|
19
|
+
|
|
20
|
+
_COMPACTION_SYSTEM_PROMPT = (
|
|
21
|
+
"You are summarizing an in-progress coding-agent conversation so it can continue with "
|
|
22
|
+
"much less context. Write a concise but complete summary covering: what the user asked "
|
|
23
|
+
"for, what has been done so far (files changed, commands run, key outcomes), and any "
|
|
24
|
+
"assumptions you made along the way and why — state them explicitly, since the "
|
|
25
|
+
"continuation must not silently re-litigate or contradict something already assumed and "
|
|
26
|
+
"acted on. The session's own recorded decisions and current todo list are appended "
|
|
27
|
+
"separately, verbatim, after your summary — don't restate them, but do mention anything "
|
|
28
|
+
"still pending or unresolved that isn't already reflected there. Do not include "
|
|
29
|
+
"pleasantries or restate this instruction — write plain prose/bullets the assistant can "
|
|
30
|
+
"use to pick up exactly where it left off."
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class CompactionResult:
|
|
36
|
+
summary_message: Message
|
|
37
|
+
artifact_id: str
|
|
38
|
+
messages_compacted: int
|
|
39
|
+
usage: Usage
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def turn_boundaries(messages: list[Message]) -> list[int]:
|
|
43
|
+
"""Indices of role=='user' messages — the only safe places to cut, since
|
|
44
|
+
a full turn's tool_calls/tool-result pairs always live entirely between
|
|
45
|
+
one user message and the next. Cutting anywhere else risks splitting an
|
|
46
|
+
assistant(tool_calls=...) from its matching tool-role response, which
|
|
47
|
+
breaks the chat-completions wire format.
|
|
48
|
+
|
|
49
|
+
Used (via compaction_cutoff) by both this module and
|
|
50
|
+
agent/context_pruning.py — the same turn-boundary-safety reasoning
|
|
51
|
+
applies to pruning old tool results, not just full-turn compaction."""
|
|
52
|
+
return [i for i, m in enumerate(messages) if m.role == "user"]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _round_boundaries(messages: list[Message]) -> list[int]:
|
|
56
|
+
"""Finer-grained fallback for turn_boundaries: every index safe to cut
|
|
57
|
+
at (message.role != "tool", so a tool result is never separated from the
|
|
58
|
+
assistant tool_calls message that requested it) - not just role=='user'.
|
|
59
|
+
|
|
60
|
+
A single long-running task (one user message, then dozens of internal
|
|
61
|
+
assistant/tool round trips as the model works through it) has exactly
|
|
62
|
+
one 'user' boundary for its entire life. Gating on turn_boundaries alone
|
|
63
|
+
means that session's eligibility check (len(boundaries) <= keep_recent_
|
|
64
|
+
turns) never becomes false, no matter how much context that one turn's
|
|
65
|
+
own history has accumulated - real debugged case: a session hit its
|
|
66
|
+
model's context limit and stalled with no visible error (see
|
|
67
|
+
cost/context.py's looks_like_context_ceiling), and /compact reported
|
|
68
|
+
"nothing to compact" even though the conversation was enormous, because
|
|
69
|
+
it had only ever received a single user message. Used only as a
|
|
70
|
+
fallback, when turn_boundaries alone doesn't yield enough boundaries to
|
|
71
|
+
safely keep keep_recent_turns of them - a normal multi-turn conversation
|
|
72
|
+
is unaffected and keeps using turn_boundaries exactly as before."""
|
|
73
|
+
return [i for i, m in enumerate(messages) if m.role != "tool"]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
_MIN_ROUNDS_FOR_FALLBACK = 20
|
|
77
|
+
"""How many round trips a single (or few) user turn must accumulate before
|
|
78
|
+
the _round_boundaries fallback below kicks in at all. Deliberately much
|
|
79
|
+
bigger than any ordinary keep_recent_turns value (2 by default): an
|
|
80
|
+
ordinary short turn - a handful of tool calls, still the single most recent
|
|
81
|
+
thing the user is looking at - has nowhere near this many round-boundary
|
|
82
|
+
entries, so it's left completely alone, exactly as if the fallback didn't
|
|
83
|
+
exist. Only once a single turn has genuinely ballooned well past that (a
|
|
84
|
+
long autonomous task making dozens of tool calls) does the fallback treat
|
|
85
|
+
it as something worth compacting/pruning into."""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def compaction_cutoff(messages: list[Message], *, keep_recent_turns: int) -> int | None:
|
|
89
|
+
"""The index up to which messages are eligible to be summarized/pruned,
|
|
90
|
+
keeping the most recent turns' worth of exchanges verbatim - None if
|
|
91
|
+
there isn't enough history yet to safely do anything.
|
|
92
|
+
|
|
93
|
+
Prefers turn_boundaries (real user-typed turns): if there are more of
|
|
94
|
+
those than keep_recent_turns, behavior is exactly what it always was.
|
|
95
|
+
Otherwise falls back to the finer _round_boundaries - but only once the
|
|
96
|
+
single (or few) turn(s) have grown past _MIN_ROUNDS_FOR_FALLBACK, and
|
|
97
|
+
even then keeping at least that many of the most recent rounds verbatim
|
|
98
|
+
(not the much smaller keep_recent_turns itself, which is calibrated for
|
|
99
|
+
whole turns, not individual round trips) - see both docstrings for why."""
|
|
100
|
+
boundaries = turn_boundaries(messages)
|
|
101
|
+
if len(boundaries) > keep_recent_turns:
|
|
102
|
+
return boundaries[-keep_recent_turns] if keep_recent_turns > 0 else len(messages)
|
|
103
|
+
|
|
104
|
+
round_boundaries = _round_boundaries(messages)
|
|
105
|
+
fallback_keep = max(keep_recent_turns, _MIN_ROUNDS_FOR_FALLBACK)
|
|
106
|
+
if len(round_boundaries) <= fallback_keep:
|
|
107
|
+
return None
|
|
108
|
+
return round_boundaries[-fallback_keep]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _system_prompt_prefix_len(messages: list[Message]) -> int:
|
|
112
|
+
"""1 if the very first message is the session's leading system prompt
|
|
113
|
+
(role=='system'), else 0. Deliberately checks only messages[0], not a
|
|
114
|
+
run of every leading system-role message: a previous compaction's own
|
|
115
|
+
summary message also has role=='system' and sits right after the real
|
|
116
|
+
prompt, but must stay eligible to be folded into a *later* compaction —
|
|
117
|
+
counting a run would permanently protect it instead."""
|
|
118
|
+
return 1 if messages and messages[0].role == "system" else 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _render_transcript(messages: list[Message]) -> str:
|
|
122
|
+
"""Plain-text rendering of a message range for archiving — a human/LLM
|
|
123
|
+
-readable record retrievable via fetch_artifact, not sent to any model
|
|
124
|
+
as-is."""
|
|
125
|
+
lines: list[str] = []
|
|
126
|
+
for m in messages:
|
|
127
|
+
header = f"--- {m.role} ---"
|
|
128
|
+
if m.tool_calls:
|
|
129
|
+
calls = ", ".join(f"{c.function.name}({c.function.arguments})" for c in m.tool_calls)
|
|
130
|
+
lines.append(f"{header}\n[tool_calls: {calls}]")
|
|
131
|
+
elif m.content:
|
|
132
|
+
lines.append(f"{header}\n{m.content}")
|
|
133
|
+
return "\n\n".join(lines)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _render_ground_truth_state(session: Session) -> str:
|
|
137
|
+
"""Renders the session's own recorded decisions and current todo list —
|
|
138
|
+
both live on Session directly (record_decision/write_todos), not inside
|
|
139
|
+
session.messages, so this source data is never itself at risk from
|
|
140
|
+
compaction. Spliced verbatim into the compaction summary rather than
|
|
141
|
+
trusted to the summarizer's own prose reconstruction of the raw
|
|
142
|
+
transcript: a decision's rationale (the "why") is exactly the kind of
|
|
143
|
+
detail easy to lose in a lossy re-summarization pass, and the todo
|
|
144
|
+
list needs to reflect its actual current state, not whatever it
|
|
145
|
+
happened to look like at some earlier point in the now-compacted
|
|
146
|
+
history. Returns "" (nothing to append) if there are no decisions and
|
|
147
|
+
no todos yet."""
|
|
148
|
+
parts: list[str] = []
|
|
149
|
+
if session.decisions:
|
|
150
|
+
lines = [f"- {d.decision} — {d.rationale}" for d in session.decisions]
|
|
151
|
+
parts.append("**Decisions recorded so far:**\n" + "\n".join(lines))
|
|
152
|
+
if session.todos:
|
|
153
|
+
parts.append("**Current todo list:**\n" + render_todos(session.todos))
|
|
154
|
+
return "\n\n".join(parts)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
async def maybe_compact(
|
|
158
|
+
session: Session,
|
|
159
|
+
*,
|
|
160
|
+
gateway_client: GatewayClient,
|
|
161
|
+
model: str | None,
|
|
162
|
+
artifact_store: ArtifactStore,
|
|
163
|
+
keep_recent_turns: int = 2,
|
|
164
|
+
) -> CompactionResult | None:
|
|
165
|
+
"""Compacts the oldest turns of session.messages in place, returning
|
|
166
|
+
None if there isn't enough history to safely compact yet (fewer than
|
|
167
|
+
keep_recent_turns+1 user turns, or - the fallback compaction_cutoff
|
|
168
|
+
applies for a session dominated by one long, tool-call-heavy turn -
|
|
169
|
+
round trips)."""
|
|
170
|
+
cut_index = compaction_cutoff(session.messages, keep_recent_turns=keep_recent_turns)
|
|
171
|
+
if cut_index is None:
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
prefix_len = _system_prompt_prefix_len(session.messages)
|
|
175
|
+
to_compact = session.messages[prefix_len:cut_index]
|
|
176
|
+
if not to_compact:
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
transcript = _render_transcript(to_compact)
|
|
180
|
+
artifact_id = artifact_store.put(transcript)
|
|
181
|
+
|
|
182
|
+
summary_messages = [
|
|
183
|
+
ChatMessage(role="system", content=_COMPACTION_SYSTEM_PROMPT),
|
|
184
|
+
ChatMessage(role="user", content=transcript),
|
|
185
|
+
]
|
|
186
|
+
assistant_message, usage = await gateway_client.collect(summary_messages, model=model)
|
|
187
|
+
summary_text = assistant_message.content or "(no summary produced)"
|
|
188
|
+
|
|
189
|
+
note = (
|
|
190
|
+
f"\n\n[Compacted {len(to_compact)} earlier message(s) to reduce context usage. "
|
|
191
|
+
f"Archived as artifact_id='{artifact_id}'. Call fetch_artifact(artifact_id="
|
|
192
|
+
f"'{artifact_id}') if you need something specific from the original conversation.]"
|
|
193
|
+
)
|
|
194
|
+
ground_truth = _render_ground_truth_state(session)
|
|
195
|
+
content = summary_text + (f"\n\n{ground_truth}" if ground_truth else "") + note
|
|
196
|
+
summary_message = Message(role="system", content=content)
|
|
197
|
+
|
|
198
|
+
session.messages[prefix_len:cut_index] = [summary_message]
|
|
199
|
+
|
|
200
|
+
return CompactionResult(
|
|
201
|
+
summary_message=summary_message,
|
|
202
|
+
artifact_id=artifact_id,
|
|
203
|
+
messages_compacted=len(to_compact),
|
|
204
|
+
usage=usage,
|
|
205
|
+
)
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Tool-result pruning: a lightweight, mechanical (no LLM call) pass that
|
|
2
|
+
shrinks the content of old, already-resolved tool-role messages down to a
|
|
3
|
+
short placeholder — archiving the original first (retrievable via
|
|
4
|
+
fetch_artifact, same ArtifactStore mechanism agent/loop.py's
|
|
5
|
+
_archive_if_large and agent/compaction.py's maybe_compact already use).
|
|
6
|
+
|
|
7
|
+
Distinct from maybe_compact: compaction is coarse and late (only fires once
|
|
8
|
+
context usage crosses a threshold, replaces entire old turns with one LLM
|
|
9
|
+
summary). This pass runs every turn, touches only tool-role message content
|
|
10
|
+
(never user/assistant text or the assistant's own tool_calls), and needs no
|
|
11
|
+
model call — small, already-spent tool results (a `which foo` probe, a
|
|
12
|
+
one-line `ls`) are exactly what this targets, well before compaction's
|
|
13
|
+
threshold would ever notice them.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
|
|
20
|
+
from pcli.agent.compaction import compaction_cutoff
|
|
21
|
+
from pcli.session.models import Message, Session
|
|
22
|
+
from pcli.tools.artifacts import ArtifactStore
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def extract_purpose(arguments_json: str) -> str | None:
|
|
26
|
+
"""Best-effort: given a tool call's raw JSON arguments string, returns
|
|
27
|
+
the model-supplied "purpose" value (see ToolSpec.to_openai_tool) if
|
|
28
|
+
present and well-formed, else None. Never raises — a malformed or
|
|
29
|
+
missing purpose is just not shown, not an error."""
|
|
30
|
+
try:
|
|
31
|
+
arguments = json.loads(arguments_json or "{}")
|
|
32
|
+
except json.JSONDecodeError:
|
|
33
|
+
return None
|
|
34
|
+
if not isinstance(arguments, dict):
|
|
35
|
+
return None
|
|
36
|
+
purpose = arguments.get("purpose")
|
|
37
|
+
return purpose if isinstance(purpose, str) and purpose.strip() else None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _find_purpose(messages: list[Message], tool_call_id: str | None) -> str | None:
|
|
41
|
+
"""Scans every assistant tool_calls entry for one matching tool_call_id
|
|
42
|
+
and extracts its purpose — tool_call ids are unique per call, so a full
|
|
43
|
+
scan (not just the immediately preceding message) is correct and, for
|
|
44
|
+
real session sizes, cheap."""
|
|
45
|
+
if tool_call_id is None:
|
|
46
|
+
return None
|
|
47
|
+
for message in messages:
|
|
48
|
+
if message.role != "assistant" or not message.tool_calls:
|
|
49
|
+
continue
|
|
50
|
+
for call in message.tool_calls:
|
|
51
|
+
if call.id == tool_call_id:
|
|
52
|
+
return extract_purpose(call.function.arguments)
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def prune_old_tool_results(
|
|
57
|
+
session: Session, *, keep_recent_turns: int, artifact_store: ArtifactStore
|
|
58
|
+
) -> int:
|
|
59
|
+
"""Replaces old tool-role message content with a compact placeholder,
|
|
60
|
+
mutating session.messages in place — the most recent keep_recent_turns
|
|
61
|
+
turns' tool results stay verbatim. Idempotent: a message already pruned
|
|
62
|
+
(pruned_artifact_id set) is skipped, so this is safe to call every turn
|
|
63
|
+
without re-archiving or duplicating artifacts. Returns how many messages
|
|
64
|
+
were pruned this call (0 if nothing was old enough yet)."""
|
|
65
|
+
cutoff_index = compaction_cutoff(session.messages, keep_recent_turns=keep_recent_turns)
|
|
66
|
+
if cutoff_index is None:
|
|
67
|
+
return 0 # nothing old enough to prune yet
|
|
68
|
+
|
|
69
|
+
pruned_count = 0
|
|
70
|
+
for i in range(cutoff_index):
|
|
71
|
+
message = session.messages[i]
|
|
72
|
+
if message.role != "tool" or message.pruned_artifact_id is not None or not message.content:
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
original_content = message.content
|
|
76
|
+
artifact_id = artifact_store.put(original_content)
|
|
77
|
+
purpose = _find_purpose(session.messages, message.tool_call_id)
|
|
78
|
+
|
|
79
|
+
note = f"[Pruned tool result ({len(original_content):,} chars) to save context."
|
|
80
|
+
if purpose:
|
|
81
|
+
note += f" Purpose: {purpose}."
|
|
82
|
+
note += f" Call fetch_artifact(artifact_id='{artifact_id}') if you need it.]"
|
|
83
|
+
|
|
84
|
+
message.content = note
|
|
85
|
+
message.pruned_artifact_id = artifact_id
|
|
86
|
+
pruned_count += 1
|
|
87
|
+
|
|
88
|
+
return pruned_count
|
pcli/agent/headless.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Runs a single task non-interactively - the machinery behind `pcli run`
|
|
2
|
+
(cli.py) and each incoming message `pcli telegram` handles (telegram/
|
|
3
|
+
daemon.py). No MessageView/StatusBar: progress is reported through a plain
|
|
4
|
+
callback instead. `ask`/`ask_question` both default to None - no UI to ask
|
|
5
|
+
through - which the permission/ask_user_question machinery already handles
|
|
6
|
+
safely on its own (see agent/runtime.py's make_tool_context); `pcli
|
|
7
|
+
telegram` is the one caller that actually supplies them, wired to Telegram
|
|
8
|
+
inline-keyboard prompts instead of None.
|
|
9
|
+
|
|
10
|
+
Deliberately narrower than ChatScreen._run_one_turn: no tool-result
|
|
11
|
+
pruning, no auto-compaction, and no memory-extraction pass (all three are
|
|
12
|
+
tied to ChatScreen's own StatusBar/MessageView plumbing and are aimed at a
|
|
13
|
+
long-lived interactive session accumulating history over hours - a
|
|
14
|
+
headless run is normally short-lived per invocation). It does replicate
|
|
15
|
+
one piece: auto-continuing a response truncated by the token limit
|
|
16
|
+
(TurnCompleteEvent.response_truncated, see agent/loop.py) - skipping that
|
|
17
|
+
would be a *worse* silent-stall bug here than in the TUI, since there's no
|
|
18
|
+
user present to notice and type "continue" themselves.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from collections.abc import Callable
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from pcli.agent.loop import AgentLoop
|
|
28
|
+
from pcli.agent.prompt import PLAN_MODE_REINFORCEMENT, build_system_prompt
|
|
29
|
+
from pcli.agent.runtime import (
|
|
30
|
+
AgentRuntime,
|
|
31
|
+
effective_max_tool_iterations,
|
|
32
|
+
make_tool_context,
|
|
33
|
+
record_tool_invocation,
|
|
34
|
+
)
|
|
35
|
+
from pcli.config.settings import Settings
|
|
36
|
+
from pcli.cost.tracker import CostTracker, cost_budget_reason
|
|
37
|
+
from pcli.llm.models import ChatMessage
|
|
38
|
+
from pcli.memory.models import render_memory_section
|
|
39
|
+
from pcli.memory.store import read_memory
|
|
40
|
+
from pcli.permissions.manager import AskCallback, PermissionManager
|
|
41
|
+
from pcli.session.models import Message, Session
|
|
42
|
+
from pcli.session.store import SessionStore
|
|
43
|
+
from pcli.tools.artifacts import SessionArtifactStore
|
|
44
|
+
from pcli.tools.base import AskQuestionCallback
|
|
45
|
+
from pcli.tools.registry import ToolRegistry
|
|
46
|
+
|
|
47
|
+
_MAX_CONSECUTIVE_AUTO_CONTINUES = 3
|
|
48
|
+
"""Same cap and reasoning as ChatScreen's own _MAX_CONSECUTIVE_AUTO_CONTINUES
|
|
49
|
+
(tui/screens/chat.py) - not shared as a single constant, since the two
|
|
50
|
+
call sites' own docstrings are the more useful place to read the reasoning
|
|
51
|
+
from, and a numeric constant this small isn't worth an import just to avoid
|
|
52
|
+
repeating "3"."""
|
|
53
|
+
_AUTO_CONTINUE_MESSAGE = "Continue."
|
|
54
|
+
|
|
55
|
+
ProgressCallback = Callable[[str], None]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class HeadlessTurnResult:
|
|
60
|
+
session: Session
|
|
61
|
+
final_text: str
|
|
62
|
+
terminated_early: bool
|
|
63
|
+
truncations_exhausted: bool = False
|
|
64
|
+
"""True if the response was still being cut off by the token limit
|
|
65
|
+
after _MAX_CONSECUTIVE_AUTO_CONTINUES automatic retries - the caller
|
|
66
|
+
may want to flag this distinctly from an ordinary finish (e.g. a
|
|
67
|
+
non-zero exit code from `pcli run`, since the work is likely genuinely
|
|
68
|
+
incomplete)."""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def new_headless_session(
|
|
72
|
+
store: SessionStore, settings: Settings, cwd: Path
|
|
73
|
+
) -> Session:
|
|
74
|
+
"""A fresh Session with the same system prompt (incl. global user
|
|
75
|
+
memory, if enabled) a brand-new TUI session gets - see ChatScreen.
|
|
76
|
+
__init__'s identical construction in tui/screens/chat.py."""
|
|
77
|
+
extra_sections: list[str] = []
|
|
78
|
+
if settings.memory_enabled:
|
|
79
|
+
memory_section = render_memory_section(read_memory().entries)
|
|
80
|
+
if memory_section:
|
|
81
|
+
extra_sections.append(memory_section)
|
|
82
|
+
session = store.new_session(
|
|
83
|
+
model=settings.default_model,
|
|
84
|
+
gateway_base_url=settings.gateway_base_url,
|
|
85
|
+
working_dir=str(cwd),
|
|
86
|
+
)
|
|
87
|
+
session.messages.append(
|
|
88
|
+
Message(role="system", content=build_system_prompt(extra_sections=extra_sections or None))
|
|
89
|
+
)
|
|
90
|
+
return session
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
async def run_headless_task(
|
|
94
|
+
task: str,
|
|
95
|
+
*,
|
|
96
|
+
session: Session,
|
|
97
|
+
runtime: AgentRuntime,
|
|
98
|
+
settings: Settings,
|
|
99
|
+
permission_manager: PermissionManager,
|
|
100
|
+
cwd: Path,
|
|
101
|
+
store: SessionStore,
|
|
102
|
+
on_progress: ProgressCallback = lambda _line: None,
|
|
103
|
+
ask: AskCallback | None = None,
|
|
104
|
+
ask_question: AskQuestionCallback | None = None,
|
|
105
|
+
plan_mode: bool = False,
|
|
106
|
+
tool_registry: ToolRegistry | None = None,
|
|
107
|
+
) -> HeadlessTurnResult:
|
|
108
|
+
"""plan_mode/tool_registry mirror ChatScreen._set_plan_mode's own two
|
|
109
|
+
layers (see agent/prompt.py's PLAN_MODE_REINFORCEMENT for the third):
|
|
110
|
+
tool_registry is the primary mechanism (the model never even sees a
|
|
111
|
+
disallowed tool - the caller is expected to pass runtime.tool_registry
|
|
112
|
+
pre-filtered via ToolRegistry.filtered(lambda t: t.plan_mode_safe) when
|
|
113
|
+
plan_mode is True, the same way ChatScreen._effective_tool_registry
|
|
114
|
+
does; this function has no opinion of its own on what "plan mode"
|
|
115
|
+
means, it just wires through whatever registry it's given), defaulting
|
|
116
|
+
to runtime.tool_registry when not overridden; plan_mode itself both
|
|
117
|
+
activates AgentLoop's dispatch-time backstop (via ctx.plan_mode, below)
|
|
118
|
+
and injects the per-turn reinforcement message further down."""
|
|
119
|
+
artifact_store = SessionArtifactStore(store, session.id)
|
|
120
|
+
cost_tracker = CostTracker(session)
|
|
121
|
+
effective_tool_registry = tool_registry if tool_registry is not None else runtime.tool_registry
|
|
122
|
+
|
|
123
|
+
def tool_context_factory():
|
|
124
|
+
return make_tool_context(
|
|
125
|
+
runtime,
|
|
126
|
+
settings,
|
|
127
|
+
cwd,
|
|
128
|
+
session=session,
|
|
129
|
+
permission_manager=permission_manager,
|
|
130
|
+
artifact_store=artifact_store,
|
|
131
|
+
ask=ask,
|
|
132
|
+
ask_question=ask_question,
|
|
133
|
+
plan_mode=plan_mode,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
agent_loop = AgentLoop(
|
|
137
|
+
runtime.client,
|
|
138
|
+
model=settings.default_model or None,
|
|
139
|
+
tool_registry=effective_tool_registry,
|
|
140
|
+
permission_manager=permission_manager,
|
|
141
|
+
tool_context_factory=tool_context_factory,
|
|
142
|
+
max_tool_iterations=effective_max_tool_iterations(settings),
|
|
143
|
+
artifact_threshold_chars=settings.artifact_threshold_chars,
|
|
144
|
+
temperature=settings.default_temperature,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
session.messages.append(Message(role="user", content=task))
|
|
148
|
+
on_progress(f"> {task}")
|
|
149
|
+
store.save(session)
|
|
150
|
+
|
|
151
|
+
final_text = ""
|
|
152
|
+
terminated_early = False
|
|
153
|
+
truncations_exhausted = False
|
|
154
|
+
consecutive_truncations = 0
|
|
155
|
+
run_again = True
|
|
156
|
+
while run_again:
|
|
157
|
+
run_again = False
|
|
158
|
+
chat_messages = [m.to_chat_message() for m in session.messages]
|
|
159
|
+
if plan_mode:
|
|
160
|
+
chat_messages.append(ChatMessage(role="system", content=PLAN_MODE_REINFORCEMENT))
|
|
161
|
+
text_parts: list[str] = []
|
|
162
|
+
async for event in agent_loop.run_turn(
|
|
163
|
+
chat_messages,
|
|
164
|
+
ask=ask,
|
|
165
|
+
budget_check=lambda: cost_budget_reason(session, settings.max_session_cost_usd),
|
|
166
|
+
):
|
|
167
|
+
if event.kind == "text_delta":
|
|
168
|
+
text_parts.append(event.text)
|
|
169
|
+
elif event.kind == "usage":
|
|
170
|
+
cost_tracker.record_turn(settings.default_model or session.model, event.usage)
|
|
171
|
+
elif event.kind == "tool_start":
|
|
172
|
+
on_progress(
|
|
173
|
+
f" -> {event.tool_call.function.name}({event.tool_call.function.arguments})"
|
|
174
|
+
)
|
|
175
|
+
elif event.kind == "tool_result":
|
|
176
|
+
for extra in event.extra_usage:
|
|
177
|
+
cost_tracker.record_turn(
|
|
178
|
+
settings.default_model or session.model, extra, source="subagent"
|
|
179
|
+
)
|
|
180
|
+
preview = event.output if len(event.output) <= 200 else event.output[:200] + "..."
|
|
181
|
+
on_progress(f" <- {preview}")
|
|
182
|
+
record_tool_invocation(
|
|
183
|
+
session, event, audit_enabled=settings.audit_mode_enabled
|
|
184
|
+
)
|
|
185
|
+
elif event.kind == "turn_complete":
|
|
186
|
+
session.messages.extend(Message.from_chat_message(m) for m in event.new_messages)
|
|
187
|
+
terminated_early = event.terminated_early
|
|
188
|
+
if event.response_truncated:
|
|
189
|
+
if consecutive_truncations < _MAX_CONSECUTIVE_AUTO_CONTINUES:
|
|
190
|
+
consecutive_truncations += 1
|
|
191
|
+
on_progress(
|
|
192
|
+
" [pcli] Response was cut off by the token limit - continuing "
|
|
193
|
+
f"automatically ({consecutive_truncations}/{_MAX_CONSECUTIVE_AUTO_CONTINUES})."
|
|
194
|
+
)
|
|
195
|
+
session.messages.append(Message(role="user", content=_AUTO_CONTINUE_MESSAGE))
|
|
196
|
+
run_again = True
|
|
197
|
+
else:
|
|
198
|
+
truncations_exhausted = True
|
|
199
|
+
else:
|
|
200
|
+
consecutive_truncations = 0
|
|
201
|
+
store.save(session)
|
|
202
|
+
final_text = "".join(text_parts) or final_text
|
|
203
|
+
|
|
204
|
+
return HeadlessTurnResult(
|
|
205
|
+
session=session,
|
|
206
|
+
final_text=final_text,
|
|
207
|
+
terminated_early=terminated_early,
|
|
208
|
+
truncations_exhausted=truncations_exhausted,
|
|
209
|
+
)
|