pi-agent-cli-lc 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pi_agent_cli_lc-0.3.0/.gitignore +33 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/PKG-INFO +4 -3
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/README.md +1 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/agent.example.toml +7 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/__main__.py +151 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/agent.py +93 -7
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/__init__.py +56 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/collector.py +130 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/docker_runner.py +474 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/egress_policy.py +129 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/evaluator.py +582 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/golden_seed.py +204 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/models.py +188 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/task_loader.py +123 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/config.py +24 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/context_files.py +111 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/create_harness.py +64 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/events.py +261 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/factory.py +163 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/headless.py +44 -3
- pi_agent_cli_lc-0.3.0/pi_agent_cli/prompt_options.py +55 -0
- pi_agent_cli_lc-0.3.0/pi_agent_cli/system_prompt.py +203 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pyproject.toml +3 -3
- pi_agent_cli_lc-0.3.0/tests/snapshots/default_coding_tools_prompt.txt +31 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_acp_agent.py +126 -0
- pi_agent_cli_lc-0.3.0/tests/test_benchmark_loader.py +25 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_config.py +26 -0
- pi_agent_cli_lc-0.3.0/tests/test_context_files.py +75 -0
- pi_agent_cli_lc-0.3.0/tests/test_evaluator_cache.py +64 -0
- pi_agent_cli_lc-0.3.0/tests/test_factory_skills.py +77 -0
- pi_agent_cli_lc-0.3.0/tests/test_golden_checkpoint.py +191 -0
- pi_agent_cli_lc-0.3.0/tests/test_headless.py +141 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_pelican_real_llm.py +1 -1
- pi_agent_cli_lc-0.3.0/tests/test_system_prompt.py +156 -0
- pi_agent_cli_lc-0.2.0/.gitignore +0 -16
- pi_agent_cli_lc-0.2.0/pi_agent_cli/__main__.py +0 -81
- pi_agent_cli_lc-0.2.0/pi_agent_cli/benchmarks/__init__.py +0 -15
- pi_agent_cli_lc-0.2.0/pi_agent_cli/events.py +0 -141
- pi_agent_cli_lc-0.2.0/pi_agent_cli/factory.py +0 -70
- pi_agent_cli_lc-0.2.0/pi_agent_cli/prompt.py +0 -24
- pi_agent_cli_lc-0.2.0/tests/test_factory_skills.py +0 -43
- pi_agent_cli_lc-0.2.0/tests/test_headless.py +0 -74
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/config.toml.example +0 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/local.env.example +0 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/__init__.py +0 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/benchmarks/pelican.py +0 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/permissions.py +0 -0
- {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_pelican_benchmark.py +0 -0
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*.egg-info/
|
|
4
|
+
.eggs/
|
|
5
|
+
dist/
|
|
6
|
+
build/
|
|
7
|
+
.pytest_cache/
|
|
8
|
+
.pytest-audit/
|
|
9
|
+
.mypy_cache/
|
|
10
|
+
.ruff_cache/
|
|
11
|
+
.venv/
|
|
12
|
+
.venv-*/
|
|
13
|
+
venv/
|
|
14
|
+
.env
|
|
15
|
+
# Rust TUI workspace (Apache-2.0 fork under tui/)
|
|
16
|
+
tui/target/
|
|
17
|
+
**/*.rs.bk
|
|
18
|
+
|
|
19
|
+
# Benchmark evaluation cache and trial artifacts
|
|
20
|
+
.cache/
|
|
21
|
+
.pi-eval/
|
|
22
|
+
|
|
23
|
+
# Temporary files, coverage and logs
|
|
24
|
+
*.log
|
|
25
|
+
*.tmp
|
|
26
|
+
*.bak
|
|
27
|
+
*.swp
|
|
28
|
+
*.orig
|
|
29
|
+
.coverage
|
|
30
|
+
.coverage.*
|
|
31
|
+
coverage.xml
|
|
32
|
+
htmlcov/
|
|
33
|
+
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pi-agent-cli-lc
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Standard ACP agent over AgentHarness (no x.ai extensions)
|
|
5
5
|
Project-URL: Homepage, https://github.com/zy1233/pi-python
|
|
6
6
|
Project-URL: Repository, https://github.com/zy1233/pi-python
|
|
@@ -20,8 +20,8 @@ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
|
20
20
|
Classifier: Typing :: Typed
|
|
21
21
|
Requires-Python: >=3.11
|
|
22
22
|
Requires-Dist: agent-client-protocol>=0.12.0
|
|
23
|
-
Requires-Dist: pi-agent-core-lc==0.
|
|
24
|
-
Requires-Dist: pi-agent-harness-lc==0.
|
|
23
|
+
Requires-Dist: pi-agent-core-lc==0.3.0
|
|
24
|
+
Requires-Dist: pi-agent-harness-lc==0.3.0
|
|
25
25
|
Requires-Dist: pydantic>=2.0
|
|
26
26
|
Description-Content-Type: text/markdown
|
|
27
27
|
|
|
@@ -31,6 +31,7 @@ Standard [Agent Client Protocol (ACP)](https://agentclientprotocol.com/) agent o
|
|
|
31
31
|
|
|
32
32
|
- stdio entry: `python -m pi_agent_cli` (or console script `pi-agent-cli`)
|
|
33
33
|
- headless one-shot: `python -m pi_agent_cli -p "..."`
|
|
34
|
+
- Headless prompt overrides (override `agent.toml` `[prompt]`): `--system-prompt`, `--system-prompt-file`, `--append-system-prompt` / `--rules`, `--append-system-prompt-file`, `--no-context-files`
|
|
34
35
|
- Config: `~/.pi-python/agent.toml` (see `agent.example.toml` in this directory)
|
|
35
36
|
- No `x.ai/*` vendor RPCs — core + harness + ACP only
|
|
36
37
|
|
|
@@ -4,6 +4,7 @@ Standard [Agent Client Protocol (ACP)](https://agentclientprotocol.com/) agent o
|
|
|
4
4
|
|
|
5
5
|
- stdio entry: `python -m pi_agent_cli` (or console script `pi-agent-cli`)
|
|
6
6
|
- headless one-shot: `python -m pi_agent_cli -p "..."`
|
|
7
|
+
- Headless prompt overrides (override `agent.toml` `[prompt]`): `--system-prompt`, `--system-prompt-file`, `--append-system-prompt` / `--rules`, `--append-system-prompt-file`, `--no-context-files`
|
|
7
8
|
- Config: `~/.pi-python/agent.toml` (see `agent.example.toml` in this directory)
|
|
8
9
|
- No `x.ai/*` vendor RPCs — core + harness + ACP only
|
|
9
10
|
|
|
@@ -20,3 +20,10 @@ paths = ["~/.pi-python/skills", ".pi/skills"]
|
|
|
20
20
|
[agent]
|
|
21
21
|
# Used by the Rust TUI when PI_AGENT_COMMAND is unset (Windows: prefer a venv python.exe).
|
|
22
22
|
command = "python -m pi_agent_cli"
|
|
23
|
+
|
|
24
|
+
[prompt]
|
|
25
|
+
# no_context_files = false
|
|
26
|
+
# custom_system_prompt = "You are a helpful assistant."
|
|
27
|
+
# custom_system_prompt_file = "~/.pi-python/agent/SYSTEM.md"
|
|
28
|
+
# append_system_prompt = "Always run tests after edits."
|
|
29
|
+
# append_system_prompt_file = "~/.pi-python/agent/APPEND_SYSTEM.md"
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""stdio entry: python -m pi_agent_cli
|
|
2
|
+
|
|
3
|
+
Default: ACP agent on stdio.
|
|
4
|
+
`python -m pi_agent_cli -p "..."`: one-shot headless turn (no TUI).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import asyncio
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from acp import run_agent
|
|
16
|
+
|
|
17
|
+
from pi_agent_cli.agent import PiAcpAgent
|
|
18
|
+
from pi_agent_cli.config import load_local_env
|
|
19
|
+
from pi_agent_cli.headless import HeadlessPromptOverrides, resolve_print_prompt, run_print
|
|
20
|
+
|
|
21
|
+
_PROMPT_CLI_FLAG_NAMES = (
|
|
22
|
+
"system_prompt",
|
|
23
|
+
"system_prompt_file",
|
|
24
|
+
"append_system_prompt",
|
|
25
|
+
"append_system_prompt_file",
|
|
26
|
+
"no_context_files",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
async def _amain() -> None:
|
|
31
|
+
await run_agent(PiAcpAgent())
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
35
|
+
parser = argparse.ArgumentParser(
|
|
36
|
+
prog="pi-agent-cli",
|
|
37
|
+
description="Standard ACP agent over AgentHarness (stdio), or one-shot -p print.",
|
|
38
|
+
)
|
|
39
|
+
src = parser.add_mutually_exclusive_group()
|
|
40
|
+
src.add_argument(
|
|
41
|
+
"-p",
|
|
42
|
+
"--print",
|
|
43
|
+
dest="print_prompt",
|
|
44
|
+
metavar="PROMPT",
|
|
45
|
+
help="Run one prompt, print the assistant text, and exit (no TUI, no ACP stdio).",
|
|
46
|
+
)
|
|
47
|
+
src.add_argument(
|
|
48
|
+
"--prompt-json",
|
|
49
|
+
metavar="JSON",
|
|
50
|
+
help="Single-turn prompt as a JSON string or list of content blocks.",
|
|
51
|
+
)
|
|
52
|
+
src.add_argument(
|
|
53
|
+
"--prompt-file",
|
|
54
|
+
metavar="PATH",
|
|
55
|
+
type=Path,
|
|
56
|
+
help="Read the single-turn prompt from a file.",
|
|
57
|
+
)
|
|
58
|
+
parser.add_argument(
|
|
59
|
+
"--cwd",
|
|
60
|
+
metavar="PATH",
|
|
61
|
+
type=Path,
|
|
62
|
+
help="Working directory for the headless session (default: process cwd).",
|
|
63
|
+
)
|
|
64
|
+
system = parser.add_mutually_exclusive_group()
|
|
65
|
+
system.add_argument(
|
|
66
|
+
"--system-prompt",
|
|
67
|
+
"--system-prompt-override",
|
|
68
|
+
dest="system_prompt",
|
|
69
|
+
metavar="TEXT",
|
|
70
|
+
help="Replace the default system prompt (headless only).",
|
|
71
|
+
)
|
|
72
|
+
system.add_argument(
|
|
73
|
+
"--system-prompt-file",
|
|
74
|
+
metavar="PATH",
|
|
75
|
+
type=Path,
|
|
76
|
+
help="Read the system prompt override from a file (headless only).",
|
|
77
|
+
)
|
|
78
|
+
append = parser.add_mutually_exclusive_group()
|
|
79
|
+
append.add_argument(
|
|
80
|
+
"--append-system-prompt",
|
|
81
|
+
"--rules",
|
|
82
|
+
dest="append_system_prompt",
|
|
83
|
+
metavar="TEXT",
|
|
84
|
+
help="Append text to the system prompt (headless only).",
|
|
85
|
+
)
|
|
86
|
+
append.add_argument(
|
|
87
|
+
"--append-system-prompt-file",
|
|
88
|
+
metavar="PATH",
|
|
89
|
+
type=Path,
|
|
90
|
+
help="Read append text from a file (headless only).",
|
|
91
|
+
)
|
|
92
|
+
parser.add_argument(
|
|
93
|
+
"--no-context-files",
|
|
94
|
+
action="store_true",
|
|
95
|
+
help="Skip AGENTS.md / CLAUDE.md discovery (headless only).",
|
|
96
|
+
)
|
|
97
|
+
return parser
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _prompt_overrides_from_args(args: argparse.Namespace) -> HeadlessPromptOverrides:
|
|
101
|
+
return HeadlessPromptOverrides(
|
|
102
|
+
system_prompt=args.system_prompt,
|
|
103
|
+
system_prompt_file=args.system_prompt_file,
|
|
104
|
+
append_system_prompt=args.append_system_prompt,
|
|
105
|
+
append_system_prompt_file=args.append_system_prompt_file,
|
|
106
|
+
no_context_files=True if args.no_context_files else None,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _has_prompt_cli_flags(args: argparse.Namespace) -> bool:
|
|
111
|
+
return any(getattr(args, name) for name in _PROMPT_CLI_FLAG_NAMES)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def main() -> None:
|
|
115
|
+
load_local_env()
|
|
116
|
+
parser = _build_parser()
|
|
117
|
+
args = parser.parse_args()
|
|
118
|
+
headless = any(
|
|
119
|
+
value is not None for value in (args.print_prompt, args.prompt_json, args.prompt_file)
|
|
120
|
+
)
|
|
121
|
+
if not headless and _has_prompt_cli_flags(args):
|
|
122
|
+
print(
|
|
123
|
+
"error: --system-prompt, --append-system-prompt, and --no-context-files "
|
|
124
|
+
"require headless mode (-p, --prompt-json, or --prompt-file)",
|
|
125
|
+
file=sys.stderr,
|
|
126
|
+
)
|
|
127
|
+
raise SystemExit(2)
|
|
128
|
+
if headless:
|
|
129
|
+
try:
|
|
130
|
+
prompt = resolve_print_prompt(
|
|
131
|
+
print_prompt=args.print_prompt,
|
|
132
|
+
prompt_json=args.prompt_json,
|
|
133
|
+
prompt_file=args.prompt_file,
|
|
134
|
+
)
|
|
135
|
+
except (OSError, ValueError, json.JSONDecodeError) as exc:
|
|
136
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
137
|
+
raise SystemExit(2) from exc
|
|
138
|
+
raise SystemExit(
|
|
139
|
+
asyncio.run(
|
|
140
|
+
run_print(
|
|
141
|
+
prompt,
|
|
142
|
+
cwd=args.cwd,
|
|
143
|
+
prompt_overrides=_prompt_overrides_from_args(args),
|
|
144
|
+
)
|
|
145
|
+
)
|
|
146
|
+
)
|
|
147
|
+
asyncio.run(_amain())
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
if __name__ == "__main__":
|
|
151
|
+
main()
|
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import asyncio
|
|
6
|
+
from dataclasses import replace
|
|
6
7
|
from pathlib import Path
|
|
7
|
-
from typing import Any
|
|
8
|
+
from typing import Any, cast
|
|
8
9
|
|
|
9
10
|
from acp import PROTOCOL_VERSION, RequestError
|
|
10
11
|
from acp.interfaces import Agent, Client
|
|
@@ -25,6 +26,7 @@ from acp.schema import (
|
|
|
25
26
|
PromptCapabilities,
|
|
26
27
|
PromptResponse,
|
|
27
28
|
ResourceContentBlock,
|
|
29
|
+
ResumeSessionResponse,
|
|
28
30
|
SessionCapabilities,
|
|
29
31
|
SessionCloseCapabilities,
|
|
30
32
|
SessionInfo,
|
|
@@ -34,8 +36,8 @@ from acp.schema import (
|
|
|
34
36
|
TextContentBlock,
|
|
35
37
|
)
|
|
36
38
|
|
|
37
|
-
from pi_agent_cli.config import CliConfig, load_config, pi_home
|
|
38
|
-
from pi_agent_cli.events import project_event
|
|
39
|
+
from pi_agent_cli.config import CliConfig, PermissionMode, load_config, pi_home
|
|
40
|
+
from pi_agent_cli.events import project_event, project_message_replay
|
|
39
41
|
from pi_agent_cli.factory import create_session_harness, default_stream_fn, load_session_resources
|
|
40
42
|
from pi_agent_cli.permissions import (
|
|
41
43
|
PERMISSION_OPTIONS,
|
|
@@ -111,7 +113,7 @@ class PiAcpAgent(Agent):
|
|
|
111
113
|
session = await self._repo.create({"cwd": cwd})
|
|
112
114
|
session_id = (await session.get_metadata()).id
|
|
113
115
|
await self._bind_session(session_id, session, cwd)
|
|
114
|
-
return NewSessionResponse(session_id=session_id)
|
|
116
|
+
return NewSessionResponse(session_id=session_id, field_meta=self._session_response_meta())
|
|
115
117
|
|
|
116
118
|
async def load_session(
|
|
117
119
|
self,
|
|
@@ -126,7 +128,12 @@ class PiAcpAgent(Agent):
|
|
|
126
128
|
raise RequestError.resource_not_found(session_id)
|
|
127
129
|
session = await self._repo.open(metadata)
|
|
128
130
|
await self._bind_session(session_id, session, metadata.cwd or cwd)
|
|
129
|
-
|
|
131
|
+
if self._conn is not None:
|
|
132
|
+
context = await session.build_context()
|
|
133
|
+
for msg in context.messages:
|
|
134
|
+
for update in project_message_replay(msg):
|
|
135
|
+
await self._conn.session_update(session_id=session_id, update=update)
|
|
136
|
+
return LoadSessionResponse(field_meta=self._session_response_meta())
|
|
130
137
|
|
|
131
138
|
async def list_sessions(
|
|
132
139
|
self, cwd: str | None = None, cursor: str | None = None, **kwargs: Any
|
|
@@ -143,6 +150,22 @@ class PiAcpAgent(Agent):
|
|
|
143
150
|
]
|
|
144
151
|
return ListSessionsResponse(sessions=sessions)
|
|
145
152
|
|
|
153
|
+
async def resume_session(
|
|
154
|
+
self,
|
|
155
|
+
session_id: str,
|
|
156
|
+
cwd: str,
|
|
157
|
+
additional_directories: list[str] | None = None,
|
|
158
|
+
mcp_servers: list[HttpMcpServer | SseMcpServer | McpServerStdio] | None = None,
|
|
159
|
+
**kwargs: Any,
|
|
160
|
+
) -> ResumeSessionResponse:
|
|
161
|
+
metadata = await self._find_metadata(session_id)
|
|
162
|
+
if metadata is None:
|
|
163
|
+
raise RequestError.resource_not_found(session_id)
|
|
164
|
+
session = await self._repo.open(metadata)
|
|
165
|
+
await self._bind_session(session_id, session, metadata.cwd or cwd)
|
|
166
|
+
# ACP session/resume intentionally does not replay history.
|
|
167
|
+
return ResumeSessionResponse(field_meta=self._session_response_meta())
|
|
168
|
+
|
|
146
169
|
async def close_session(self, session_id: str, **kwargs: Any) -> CloseSessionResponse | None:
|
|
147
170
|
self._harnesses.pop(session_id, None)
|
|
148
171
|
return CloseSessionResponse()
|
|
@@ -178,11 +201,42 @@ class PiAcpAgent(Agent):
|
|
|
178
201
|
task.add_done_callback(self._abort_tasks.discard)
|
|
179
202
|
|
|
180
203
|
async def ext_method(self, method: str, params: dict[str, Any]) -> dict[str, Any]:
|
|
204
|
+
if method == "pi/session/delete":
|
|
205
|
+
session_id_raw = params.get("sessionId", params.get("session_id"))
|
|
206
|
+
if not isinstance(session_id_raw, str) or not session_id_raw.strip():
|
|
207
|
+
raise RequestError.invalid_params(
|
|
208
|
+
{"reason": "sessionId is required", "method": method}
|
|
209
|
+
)
|
|
210
|
+
session_id = session_id_raw.strip()
|
|
211
|
+
metadata = await self._find_metadata(session_id)
|
|
212
|
+
harness = self._harnesses.pop(session_id, None)
|
|
213
|
+
if harness is not None:
|
|
214
|
+
await harness.abort()
|
|
215
|
+
# Idempotent delete: missing session still returns success.
|
|
216
|
+
if metadata is not None:
|
|
217
|
+
await self._repo.delete(metadata)
|
|
218
|
+
return {"sessionId": session_id, "deleted": True}
|
|
181
219
|
raise RequestError.method_not_found(method)
|
|
182
220
|
|
|
183
221
|
async def ext_notification(self, method: str, params: dict[str, Any]) -> None:
|
|
222
|
+
mode = _permission_mode_from_notification(method, params)
|
|
223
|
+
if mode is not None:
|
|
224
|
+
self._config = replace(self._config, permission=mode)
|
|
184
225
|
return None
|
|
185
226
|
|
|
227
|
+
def _session_response_meta(self) -> dict[str, Any] | None:
|
|
228
|
+
model_id = self._config.model_id.strip()
|
|
229
|
+
if not model_id:
|
|
230
|
+
return None
|
|
231
|
+
meta: dict[str, Any] = {
|
|
232
|
+
"pi/currentModelId": model_id,
|
|
233
|
+
"pi/currentModelDisplayName": model_id,
|
|
234
|
+
}
|
|
235
|
+
provider = self._config.provider.strip()
|
|
236
|
+
if provider:
|
|
237
|
+
meta["pi/provider"] = provider
|
|
238
|
+
return meta
|
|
239
|
+
|
|
186
240
|
def _require_harness(self, session_id: str) -> AgentHarness:
|
|
187
241
|
harness = self._harnesses.get(session_id)
|
|
188
242
|
if harness is None:
|
|
@@ -204,7 +258,7 @@ class PiAcpAgent(Agent):
|
|
|
204
258
|
return await self._handle_tool_call(session_id, event)
|
|
205
259
|
|
|
206
260
|
resources = await load_session_resources(cwd=cwd, config=self._config)
|
|
207
|
-
harness = create_session_harness(
|
|
261
|
+
harness = await create_session_harness(
|
|
208
262
|
session=session,
|
|
209
263
|
cwd=cwd,
|
|
210
264
|
config=self._config,
|
|
@@ -287,11 +341,43 @@ def _prompt_to_text_images(
|
|
|
287
341
|
|
|
288
342
|
|
|
289
343
|
def _stop_reason(message: Any) -> str:
|
|
344
|
+
"""Map internal AssistantMessage stopReason to ACP PromptResponse stop_reason.
|
|
345
|
+
|
|
346
|
+
ACP stopReason enum: 'end_turn' | 'max_tokens' | 'max_turn_requests' | 'refusal' | 'cancelled'.
|
|
347
|
+
Internal stopReason values: 'stop' | 'length' | 'toolUse' | 'error' | 'aborted'.
|
|
348
|
+
"""
|
|
290
349
|
reason = getattr(message, "stopReason", None) or "stop"
|
|
291
350
|
if reason == "aborted":
|
|
292
351
|
return "cancelled"
|
|
293
352
|
if reason == "length":
|
|
294
353
|
return "max_tokens"
|
|
295
354
|
if reason == "error":
|
|
296
|
-
|
|
355
|
+
err = str(getattr(message, "errorMessage", "") or "")
|
|
356
|
+
if any(w in err.lower() for w in ("refus", "policy", "filter", "safety")):
|
|
357
|
+
return "refusal"
|
|
358
|
+
return "end_turn"
|
|
297
359
|
return "end_turn"
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _permission_mode_from_notification(
|
|
363
|
+
method: str, params: dict[str, Any] | None
|
|
364
|
+
) -> PermissionMode | None:
|
|
365
|
+
"""Best-effort permission mode sync for live sessions.
|
|
366
|
+
|
|
367
|
+
TUI settings changes are sent as extension notifications; we accept known
|
|
368
|
+
suffixes and update in-memory mode so the next tool call in this process
|
|
369
|
+
uses the new policy immediately.
|
|
370
|
+
"""
|
|
371
|
+
suffix = method.rsplit("/", 1)[-1].strip().lower()
|
|
372
|
+
if suffix not in {"yolo_mode_changed", "permission_mode_changed"}:
|
|
373
|
+
return None
|
|
374
|
+
payload = params or {}
|
|
375
|
+
raw = payload.get("permission_mode")
|
|
376
|
+
if raw is None:
|
|
377
|
+
raw = payload.get("permissionMode")
|
|
378
|
+
if not isinstance(raw, str):
|
|
379
|
+
return None
|
|
380
|
+
normalized = raw.strip().lower()
|
|
381
|
+
if normalized not in {"ask", "auto", "always-approve"}:
|
|
382
|
+
return None
|
|
383
|
+
return cast(PermissionMode, normalized)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Benchmark evaluation subsystem for pi-agent-cli and pi-python harness."""
|
|
2
|
+
|
|
3
|
+
from pi_agent_cli.benchmarks.collector import (
|
|
4
|
+
format_markdown_report,
|
|
5
|
+
print_summary_table,
|
|
6
|
+
save_summary_artifacts,
|
|
7
|
+
)
|
|
8
|
+
from pi_agent_cli.benchmarks.egress_policy import (
|
|
9
|
+
OFFICIAL_ALLOWED_HOSTS,
|
|
10
|
+
EgressPolicy,
|
|
11
|
+
resolve_task_egress_policy,
|
|
12
|
+
)
|
|
13
|
+
from pi_agent_cli.benchmarks.evaluator import run_trial
|
|
14
|
+
from pi_agent_cli.benchmarks.golden_seed import (
|
|
15
|
+
get_seed_dir,
|
|
16
|
+
is_seed_ready,
|
|
17
|
+
load_golden_manifest,
|
|
18
|
+
provision_golden_seeds,
|
|
19
|
+
)
|
|
20
|
+
from pi_agent_cli.benchmarks.models import (
|
|
21
|
+
BenchmarkSummary,
|
|
22
|
+
BenchmarkTask,
|
|
23
|
+
TrialResult,
|
|
24
|
+
)
|
|
25
|
+
from pi_agent_cli.benchmarks.pelican import (
|
|
26
|
+
PELICAN_PROMPT,
|
|
27
|
+
PelicanSvgReport,
|
|
28
|
+
extract_svg,
|
|
29
|
+
save_pelican_artifact,
|
|
30
|
+
validate_pelican_svg,
|
|
31
|
+
)
|
|
32
|
+
from pi_agent_cli.benchmarks.task_loader import discover_tasks, load_task_from_dir
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"OFFICIAL_ALLOWED_HOSTS",
|
|
36
|
+
"PELICAN_PROMPT",
|
|
37
|
+
"BenchmarkSummary",
|
|
38
|
+
"BenchmarkTask",
|
|
39
|
+
"EgressPolicy",
|
|
40
|
+
"PelicanSvgReport",
|
|
41
|
+
"TrialResult",
|
|
42
|
+
"discover_tasks",
|
|
43
|
+
"extract_svg",
|
|
44
|
+
"format_markdown_report",
|
|
45
|
+
"get_seed_dir",
|
|
46
|
+
"is_seed_ready",
|
|
47
|
+
"load_golden_manifest",
|
|
48
|
+
"load_task_from_dir",
|
|
49
|
+
"print_summary_table",
|
|
50
|
+
"provision_golden_seeds",
|
|
51
|
+
"resolve_task_egress_policy",
|
|
52
|
+
"run_trial",
|
|
53
|
+
"save_pelican_artifact",
|
|
54
|
+
"save_summary_artifacts",
|
|
55
|
+
"validate_pelican_svg",
|
|
56
|
+
]
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Report collector and summarizer for benchmark evaluations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from pi_agent_cli.benchmarks.models import BenchmarkSummary
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def format_markdown_report(summary: BenchmarkSummary) -> str:
|
|
11
|
+
"""Generate a clean Markdown evaluation report."""
|
|
12
|
+
pass_pct = f"{summary.pass_rate * 100:.1f}%"
|
|
13
|
+
cache_pct = f"{summary.typical_cache_hit_rate * 100:.1f}%"
|
|
14
|
+
cost_per_pass = (
|
|
15
|
+
f"${summary.effective_cost_per_pass:.4f}"
|
|
16
|
+
if summary.effective_cost_per_pass is not None
|
|
17
|
+
else "N/A"
|
|
18
|
+
)
|
|
19
|
+
tokens_per_pass = (
|
|
20
|
+
f"{summary.tokens_per_solved:,.0f}" if summary.tokens_per_solved is not None else "N/A"
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
md = [
|
|
24
|
+
f"# Benchmark Evaluation Report: {summary.run_id}",
|
|
25
|
+
"",
|
|
26
|
+
f"- **Harness**: `{summary.harness}`",
|
|
27
|
+
f"- **Model**: `{summary.model}` ({summary.provider})",
|
|
28
|
+
f"- **Time**: `{summary.started_at}` to `{summary.finished_at}`",
|
|
29
|
+
f"- **Tasks**: {summary.successful_tasks} / {summary.completed_tasks} passed ({pass_pct})",
|
|
30
|
+
"",
|
|
31
|
+
"## Key Metrics",
|
|
32
|
+
"",
|
|
33
|
+
"| Metric | Value | Meaning |",
|
|
34
|
+
"| :--- | :--- | :--- |",
|
|
35
|
+
f"| **Pass Rate** | **{pass_pct}** | Solved and verified tasks |",
|
|
36
|
+
f"| **Effective Cost / Pass** | **{cost_per_pass}** | Amortized cost per passed task |",
|
|
37
|
+
]
|
|
38
|
+
if summary.effective_cost_per_pass_kimi_k3 is not None:
|
|
39
|
+
md.append(
|
|
40
|
+
f"| **Normalized Cost / Pass (Kimi K3)** | "
|
|
41
|
+
f"**${summary.effective_cost_per_pass_kimi_k3:.4f}** | "
|
|
42
|
+
"Amortized cost normalized to official benchmark pricing |"
|
|
43
|
+
)
|
|
44
|
+
md.extend(
|
|
45
|
+
[
|
|
46
|
+
f"| **Tokens / Solved** | **{tokens_per_pass}** | Amortized tokens per passed task |",
|
|
47
|
+
f"| **Total Cost** | ${summary.total_cost_usd:.4f} | Total API spend for the run |",
|
|
48
|
+
f"| **Typical Cache Hit Rate** | {cache_pct} | Prompt caching efficiency |",
|
|
49
|
+
f"| **Mean Turns** | {summary.mean_turns:.1f} | Average interaction turns |",
|
|
50
|
+
(
|
|
51
|
+
f"| **Mean No-Action Turns** | {summary.mean_no_action_turns:.1f} | "
|
|
52
|
+
"Turns without file/shell operations (overhead) |"
|
|
53
|
+
),
|
|
54
|
+
(
|
|
55
|
+
f"| **Median Duration** | {summary.median_duration_seconds:.1f}s | "
|
|
56
|
+
"Median elapsed time |"
|
|
57
|
+
),
|
|
58
|
+
"",
|
|
59
|
+
"## Task Results",
|
|
60
|
+
"",
|
|
61
|
+
(
|
|
62
|
+
"| Status | Task ID | Duration | Turns (No-Act) | "
|
|
63
|
+
"Tokens | Cache | Cost | Norm Cost (K3) |"
|
|
64
|
+
),
|
|
65
|
+
"| :---: | :--- | :---: | :---: | :---: | :---: | :---: | :---: |",
|
|
66
|
+
]
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
for t in summary.trials:
|
|
70
|
+
icon = "✅" if t.success else "❌"
|
|
71
|
+
cache_str = f"{t.cache_hit_rate_normalized * 100:.0f}%"
|
|
72
|
+
k3_cost = (
|
|
73
|
+
f"${t.cost_kimi_k3_normalized_usd:.4f}"
|
|
74
|
+
if t.cost_kimi_k3_normalized_usd is not None
|
|
75
|
+
else "N/A"
|
|
76
|
+
)
|
|
77
|
+
md.append(
|
|
78
|
+
f"| {icon} | `{t.id}` | {t.duration_seconds:.1f}s | "
|
|
79
|
+
f"{t.turns} ({t.no_action_turns}) | {t.total_tokens:,} | "
|
|
80
|
+
f"{cache_str} | ${t.cost_first_cold_usd:.4f} | {k3_cost} |"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
md.append("")
|
|
84
|
+
return "\n".join(md)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def print_summary_table(summary: BenchmarkSummary) -> None:
|
|
88
|
+
"""Print an ASCII table of results to standard output."""
|
|
89
|
+
pass_pct = f"{summary.pass_rate * 100:.1f}%"
|
|
90
|
+
cost_per_pass = (
|
|
91
|
+
f"${summary.effective_cost_per_pass:.4f}"
|
|
92
|
+
if summary.effective_cost_per_pass is not None
|
|
93
|
+
else "N/A"
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
print("\n" + "=" * 70)
|
|
97
|
+
print(f" BENCHMARK RUN SUMMARY: {summary.run_id}")
|
|
98
|
+
print(f" Model: {summary.provider}:{summary.model}")
|
|
99
|
+
print("=" * 70)
|
|
100
|
+
print(f" Passed: {summary.successful_tasks}/{summary.completed_tasks} ({pass_pct})")
|
|
101
|
+
print(f" Effective Cost: {cost_per_pass}")
|
|
102
|
+
if summary.effective_cost_per_pass_kimi_k3 is not None:
|
|
103
|
+
print(f" Norm Cost (K3): ${summary.effective_cost_per_pass_kimi_k3:.4f}")
|
|
104
|
+
print(f" Total Cost: ${summary.total_cost_usd:.4f}")
|
|
105
|
+
print(f" Typical Cache: {summary.typical_cache_hit_rate * 100:.1f}%")
|
|
106
|
+
print(f" Mean Turns (No-Act): {summary.mean_turns:.1f} ({summary.mean_no_action_turns:.1f})")
|
|
107
|
+
print("-" * 70)
|
|
108
|
+
print(f" {'Status':<8} {'Task ID':<30} {'Turns':<8} {'Tokens':<10} {'Cost':<10}")
|
|
109
|
+
print("-" * 70)
|
|
110
|
+
for t in summary.trials:
|
|
111
|
+
stat = "PASS" if t.success else "FAIL"
|
|
112
|
+
cost_str = f"${t.cost_first_cold_usd:<9.4f}"
|
|
113
|
+
print(f" {stat:<8} {t.title:<30} {t.turns:<8} {t.total_tokens:<10} {cost_str}")
|
|
114
|
+
print("=" * 70 + "\n")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def save_summary_artifacts(
|
|
118
|
+
summary: BenchmarkSummary,
|
|
119
|
+
output_dir: Path,
|
|
120
|
+
) -> Path:
|
|
121
|
+
"""Save summary JSON and Markdown report into output directory."""
|
|
122
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
123
|
+
|
|
124
|
+
summary_json_path = output_dir / "eval-summary.json"
|
|
125
|
+
summary_json_path.write_text(summary.to_json(indent=2), encoding="utf-8")
|
|
126
|
+
|
|
127
|
+
report_md_path = output_dir / "REPORT.md"
|
|
128
|
+
report_md_path.write_text(format_markdown_report(summary), encoding="utf-8")
|
|
129
|
+
|
|
130
|
+
return report_md_path
|