pi-agent-cli-lc 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. pi_agent_cli_lc-0.3.0/.gitignore +33 -0
  2. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/PKG-INFO +4 -3
  3. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/README.md +1 -0
  4. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/agent.example.toml +7 -0
  5. pi_agent_cli_lc-0.3.0/pi_agent_cli/__main__.py +151 -0
  6. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/agent.py +93 -7
  7. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/__init__.py +56 -0
  8. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/collector.py +130 -0
  9. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/docker_runner.py +474 -0
  10. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/egress_policy.py +129 -0
  11. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/evaluator.py +582 -0
  12. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/golden_seed.py +204 -0
  13. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/models.py +188 -0
  14. pi_agent_cli_lc-0.3.0/pi_agent_cli/benchmarks/task_loader.py +123 -0
  15. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/config.py +24 -0
  16. pi_agent_cli_lc-0.3.0/pi_agent_cli/context_files.py +111 -0
  17. pi_agent_cli_lc-0.3.0/pi_agent_cli/create_harness.py +64 -0
  18. pi_agent_cli_lc-0.3.0/pi_agent_cli/events.py +261 -0
  19. pi_agent_cli_lc-0.3.0/pi_agent_cli/factory.py +163 -0
  20. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/headless.py +44 -3
  21. pi_agent_cli_lc-0.3.0/pi_agent_cli/prompt_options.py +55 -0
  22. pi_agent_cli_lc-0.3.0/pi_agent_cli/system_prompt.py +203 -0
  23. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pyproject.toml +3 -3
  24. pi_agent_cli_lc-0.3.0/tests/snapshots/default_coding_tools_prompt.txt +31 -0
  25. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_acp_agent.py +126 -0
  26. pi_agent_cli_lc-0.3.0/tests/test_benchmark_loader.py +25 -0
  27. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_config.py +26 -0
  28. pi_agent_cli_lc-0.3.0/tests/test_context_files.py +75 -0
  29. pi_agent_cli_lc-0.3.0/tests/test_evaluator_cache.py +64 -0
  30. pi_agent_cli_lc-0.3.0/tests/test_factory_skills.py +77 -0
  31. pi_agent_cli_lc-0.3.0/tests/test_golden_checkpoint.py +191 -0
  32. pi_agent_cli_lc-0.3.0/tests/test_headless.py +141 -0
  33. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_pelican_real_llm.py +1 -1
  34. pi_agent_cli_lc-0.3.0/tests/test_system_prompt.py +156 -0
  35. pi_agent_cli_lc-0.2.0/.gitignore +0 -16
  36. pi_agent_cli_lc-0.2.0/pi_agent_cli/__main__.py +0 -81
  37. pi_agent_cli_lc-0.2.0/pi_agent_cli/benchmarks/__init__.py +0 -15
  38. pi_agent_cli_lc-0.2.0/pi_agent_cli/events.py +0 -141
  39. pi_agent_cli_lc-0.2.0/pi_agent_cli/factory.py +0 -70
  40. pi_agent_cli_lc-0.2.0/pi_agent_cli/prompt.py +0 -24
  41. pi_agent_cli_lc-0.2.0/tests/test_factory_skills.py +0 -43
  42. pi_agent_cli_lc-0.2.0/tests/test_headless.py +0 -74
  43. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/config.toml.example +0 -0
  44. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/local.env.example +0 -0
  45. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/__init__.py +0 -0
  46. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/benchmarks/pelican.py +0 -0
  47. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/pi_agent_cli/permissions.py +0 -0
  48. {pi_agent_cli_lc-0.2.0 → pi_agent_cli_lc-0.3.0}/tests/test_pelican_benchmark.py +0 -0
@@ -0,0 +1,33 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .eggs/
5
+ dist/
6
+ build/
7
+ .pytest_cache/
8
+ .pytest-audit/
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .venv/
12
+ .venv-*/
13
+ venv/
14
+ .env
15
+ # Rust TUI workspace (Apache-2.0 fork under tui/)
16
+ tui/target/
17
+ **/*.rs.bk
18
+
19
+ # Benchmark evaluation cache and trial artifacts
20
+ .cache/
21
+ .pi-eval/
22
+
23
+ # Temporary files, coverage and logs
24
+ *.log
25
+ *.tmp
26
+ *.bak
27
+ *.swp
28
+ *.orig
29
+ .coverage
30
+ .coverage.*
31
+ coverage.xml
32
+ htmlcov/
33
+
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pi-agent-cli-lc
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Standard ACP agent over AgentHarness (no x.ai extensions)
5
5
  Project-URL: Homepage, https://github.com/zy1233/pi-python
6
6
  Project-URL: Repository, https://github.com/zy1233/pi-python
@@ -20,8 +20,8 @@ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
20
  Classifier: Typing :: Typed
21
21
  Requires-Python: >=3.11
22
22
  Requires-Dist: agent-client-protocol>=0.12.0
23
- Requires-Dist: pi-agent-core-lc==0.2.0
24
- Requires-Dist: pi-agent-harness-lc==0.2.0
23
+ Requires-Dist: pi-agent-core-lc==0.3.0
24
+ Requires-Dist: pi-agent-harness-lc==0.3.0
25
25
  Requires-Dist: pydantic>=2.0
26
26
  Description-Content-Type: text/markdown
27
27
 
@@ -31,6 +31,7 @@ Standard [Agent Client Protocol (ACP)](https://agentclientprotocol.com/) agent o
31
31
 
32
32
  - stdio entry: `python -m pi_agent_cli` (or console script `pi-agent-cli`)
33
33
  - headless one-shot: `python -m pi_agent_cli -p "..."`
34
+ - Headless prompt overrides (override `agent.toml` `[prompt]`): `--system-prompt`, `--system-prompt-file`, `--append-system-prompt` / `--rules`, `--append-system-prompt-file`, `--no-context-files`
34
35
  - Config: `~/.pi-python/agent.toml` (see `agent.example.toml` in this directory)
35
36
  - No `x.ai/*` vendor RPCs — core + harness + ACP only
36
37
 
@@ -4,6 +4,7 @@ Standard [Agent Client Protocol (ACP)](https://agentclientprotocol.com/) agent o
4
4
 
5
5
  - stdio entry: `python -m pi_agent_cli` (or console script `pi-agent-cli`)
6
6
  - headless one-shot: `python -m pi_agent_cli -p "..."`
7
+ - Headless prompt overrides (override `agent.toml` `[prompt]`): `--system-prompt`, `--system-prompt-file`, `--append-system-prompt` / `--rules`, `--append-system-prompt-file`, `--no-context-files`
7
8
  - Config: `~/.pi-python/agent.toml` (see `agent.example.toml` in this directory)
8
9
  - No `x.ai/*` vendor RPCs — core + harness + ACP only
9
10
 
@@ -20,3 +20,10 @@ paths = ["~/.pi-python/skills", ".pi/skills"]
20
20
  [agent]
21
21
  # Used by the Rust TUI when PI_AGENT_COMMAND is unset (Windows: prefer a venv python.exe).
22
22
  command = "python -m pi_agent_cli"
23
+
24
+ [prompt]
25
+ # no_context_files = false
26
+ # custom_system_prompt = "You are a helpful assistant."
27
+ # custom_system_prompt_file = "~/.pi-python/agent/SYSTEM.md"
28
+ # append_system_prompt = "Always run tests after edits."
29
+ # append_system_prompt_file = "~/.pi-python/agent/APPEND_SYSTEM.md"
@@ -0,0 +1,151 @@
1
+ """stdio entry: python -m pi_agent_cli
2
+
3
+ Default: ACP agent on stdio.
4
+ `python -m pi_agent_cli -p "..."`: one-shot headless turn (no TUI).
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import asyncio
11
+ import json
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ from acp import run_agent
16
+
17
+ from pi_agent_cli.agent import PiAcpAgent
18
+ from pi_agent_cli.config import load_local_env
19
+ from pi_agent_cli.headless import HeadlessPromptOverrides, resolve_print_prompt, run_print
20
+
21
+ _PROMPT_CLI_FLAG_NAMES = (
22
+ "system_prompt",
23
+ "system_prompt_file",
24
+ "append_system_prompt",
25
+ "append_system_prompt_file",
26
+ "no_context_files",
27
+ )
28
+
29
+
30
+ async def _amain() -> None:
31
+ await run_agent(PiAcpAgent())
32
+
33
+
34
+ def _build_parser() -> argparse.ArgumentParser:
35
+ parser = argparse.ArgumentParser(
36
+ prog="pi-agent-cli",
37
+ description="Standard ACP agent over AgentHarness (stdio), or one-shot -p print.",
38
+ )
39
+ src = parser.add_mutually_exclusive_group()
40
+ src.add_argument(
41
+ "-p",
42
+ "--print",
43
+ dest="print_prompt",
44
+ metavar="PROMPT",
45
+ help="Run one prompt, print the assistant text, and exit (no TUI, no ACP stdio).",
46
+ )
47
+ src.add_argument(
48
+ "--prompt-json",
49
+ metavar="JSON",
50
+ help="Single-turn prompt as a JSON string or list of content blocks.",
51
+ )
52
+ src.add_argument(
53
+ "--prompt-file",
54
+ metavar="PATH",
55
+ type=Path,
56
+ help="Read the single-turn prompt from a file.",
57
+ )
58
+ parser.add_argument(
59
+ "--cwd",
60
+ metavar="PATH",
61
+ type=Path,
62
+ help="Working directory for the headless session (default: process cwd).",
63
+ )
64
+ system = parser.add_mutually_exclusive_group()
65
+ system.add_argument(
66
+ "--system-prompt",
67
+ "--system-prompt-override",
68
+ dest="system_prompt",
69
+ metavar="TEXT",
70
+ help="Replace the default system prompt (headless only).",
71
+ )
72
+ system.add_argument(
73
+ "--system-prompt-file",
74
+ metavar="PATH",
75
+ type=Path,
76
+ help="Read the system prompt override from a file (headless only).",
77
+ )
78
+ append = parser.add_mutually_exclusive_group()
79
+ append.add_argument(
80
+ "--append-system-prompt",
81
+ "--rules",
82
+ dest="append_system_prompt",
83
+ metavar="TEXT",
84
+ help="Append text to the system prompt (headless only).",
85
+ )
86
+ append.add_argument(
87
+ "--append-system-prompt-file",
88
+ metavar="PATH",
89
+ type=Path,
90
+ help="Read append text from a file (headless only).",
91
+ )
92
+ parser.add_argument(
93
+ "--no-context-files",
94
+ action="store_true",
95
+ help="Skip AGENTS.md / CLAUDE.md discovery (headless only).",
96
+ )
97
+ return parser
98
+
99
+
100
+ def _prompt_overrides_from_args(args: argparse.Namespace) -> HeadlessPromptOverrides:
101
+ return HeadlessPromptOverrides(
102
+ system_prompt=args.system_prompt,
103
+ system_prompt_file=args.system_prompt_file,
104
+ append_system_prompt=args.append_system_prompt,
105
+ append_system_prompt_file=args.append_system_prompt_file,
106
+ no_context_files=True if args.no_context_files else None,
107
+ )
108
+
109
+
110
+ def _has_prompt_cli_flags(args: argparse.Namespace) -> bool:
111
+ return any(getattr(args, name) for name in _PROMPT_CLI_FLAG_NAMES)
112
+
113
+
114
+ def main() -> None:
115
+ load_local_env()
116
+ parser = _build_parser()
117
+ args = parser.parse_args()
118
+ headless = any(
119
+ value is not None for value in (args.print_prompt, args.prompt_json, args.prompt_file)
120
+ )
121
+ if not headless and _has_prompt_cli_flags(args):
122
+ print(
123
+ "error: --system-prompt, --append-system-prompt, and --no-context-files "
124
+ "require headless mode (-p, --prompt-json, or --prompt-file)",
125
+ file=sys.stderr,
126
+ )
127
+ raise SystemExit(2)
128
+ if headless:
129
+ try:
130
+ prompt = resolve_print_prompt(
131
+ print_prompt=args.print_prompt,
132
+ prompt_json=args.prompt_json,
133
+ prompt_file=args.prompt_file,
134
+ )
135
+ except (OSError, ValueError, json.JSONDecodeError) as exc:
136
+ print(f"error: {exc}", file=sys.stderr)
137
+ raise SystemExit(2) from exc
138
+ raise SystemExit(
139
+ asyncio.run(
140
+ run_print(
141
+ prompt,
142
+ cwd=args.cwd,
143
+ prompt_overrides=_prompt_overrides_from_args(args),
144
+ )
145
+ )
146
+ )
147
+ asyncio.run(_amain())
148
+
149
+
150
+ if __name__ == "__main__":
151
+ main()
@@ -3,8 +3,9 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import asyncio
6
+ from dataclasses import replace
6
7
  from pathlib import Path
7
- from typing import Any
8
+ from typing import Any, cast
8
9
 
9
10
  from acp import PROTOCOL_VERSION, RequestError
10
11
  from acp.interfaces import Agent, Client
@@ -25,6 +26,7 @@ from acp.schema import (
25
26
  PromptCapabilities,
26
27
  PromptResponse,
27
28
  ResourceContentBlock,
29
+ ResumeSessionResponse,
28
30
  SessionCapabilities,
29
31
  SessionCloseCapabilities,
30
32
  SessionInfo,
@@ -34,8 +36,8 @@ from acp.schema import (
34
36
  TextContentBlock,
35
37
  )
36
38
 
37
- from pi_agent_cli.config import CliConfig, load_config, pi_home
38
- from pi_agent_cli.events import project_event
39
+ from pi_agent_cli.config import CliConfig, PermissionMode, load_config, pi_home
40
+ from pi_agent_cli.events import project_event, project_message_replay
39
41
  from pi_agent_cli.factory import create_session_harness, default_stream_fn, load_session_resources
40
42
  from pi_agent_cli.permissions import (
41
43
  PERMISSION_OPTIONS,
@@ -111,7 +113,7 @@ class PiAcpAgent(Agent):
111
113
  session = await self._repo.create({"cwd": cwd})
112
114
  session_id = (await session.get_metadata()).id
113
115
  await self._bind_session(session_id, session, cwd)
114
- return NewSessionResponse(session_id=session_id)
116
+ return NewSessionResponse(session_id=session_id, field_meta=self._session_response_meta())
115
117
 
116
118
  async def load_session(
117
119
  self,
@@ -126,7 +128,12 @@ class PiAcpAgent(Agent):
126
128
  raise RequestError.resource_not_found(session_id)
127
129
  session = await self._repo.open(metadata)
128
130
  await self._bind_session(session_id, session, metadata.cwd or cwd)
129
- return LoadSessionResponse()
131
+ if self._conn is not None:
132
+ context = await session.build_context()
133
+ for msg in context.messages:
134
+ for update in project_message_replay(msg):
135
+ await self._conn.session_update(session_id=session_id, update=update)
136
+ return LoadSessionResponse(field_meta=self._session_response_meta())
130
137
 
131
138
  async def list_sessions(
132
139
  self, cwd: str | None = None, cursor: str | None = None, **kwargs: Any
@@ -143,6 +150,22 @@ class PiAcpAgent(Agent):
143
150
  ]
144
151
  return ListSessionsResponse(sessions=sessions)
145
152
 
153
+ async def resume_session(
154
+ self,
155
+ session_id: str,
156
+ cwd: str,
157
+ additional_directories: list[str] | None = None,
158
+ mcp_servers: list[HttpMcpServer | SseMcpServer | McpServerStdio] | None = None,
159
+ **kwargs: Any,
160
+ ) -> ResumeSessionResponse:
161
+ metadata = await self._find_metadata(session_id)
162
+ if metadata is None:
163
+ raise RequestError.resource_not_found(session_id)
164
+ session = await self._repo.open(metadata)
165
+ await self._bind_session(session_id, session, metadata.cwd or cwd)
166
+ # ACP session/resume intentionally does not replay history.
167
+ return ResumeSessionResponse(field_meta=self._session_response_meta())
168
+
146
169
  async def close_session(self, session_id: str, **kwargs: Any) -> CloseSessionResponse | None:
147
170
  self._harnesses.pop(session_id, None)
148
171
  return CloseSessionResponse()
@@ -178,11 +201,42 @@ class PiAcpAgent(Agent):
178
201
  task.add_done_callback(self._abort_tasks.discard)
179
202
 
180
203
  async def ext_method(self, method: str, params: dict[str, Any]) -> dict[str, Any]:
204
+ if method == "pi/session/delete":
205
+ session_id_raw = params.get("sessionId", params.get("session_id"))
206
+ if not isinstance(session_id_raw, str) or not session_id_raw.strip():
207
+ raise RequestError.invalid_params(
208
+ {"reason": "sessionId is required", "method": method}
209
+ )
210
+ session_id = session_id_raw.strip()
211
+ metadata = await self._find_metadata(session_id)
212
+ harness = self._harnesses.pop(session_id, None)
213
+ if harness is not None:
214
+ await harness.abort()
215
+ # Idempotent delete: missing session still returns success.
216
+ if metadata is not None:
217
+ await self._repo.delete(metadata)
218
+ return {"sessionId": session_id, "deleted": True}
181
219
  raise RequestError.method_not_found(method)
182
220
 
183
221
  async def ext_notification(self, method: str, params: dict[str, Any]) -> None:
222
+ mode = _permission_mode_from_notification(method, params)
223
+ if mode is not None:
224
+ self._config = replace(self._config, permission=mode)
184
225
  return None
185
226
 
227
+ def _session_response_meta(self) -> dict[str, Any] | None:
228
+ model_id = self._config.model_id.strip()
229
+ if not model_id:
230
+ return None
231
+ meta: dict[str, Any] = {
232
+ "pi/currentModelId": model_id,
233
+ "pi/currentModelDisplayName": model_id,
234
+ }
235
+ provider = self._config.provider.strip()
236
+ if provider:
237
+ meta["pi/provider"] = provider
238
+ return meta
239
+
186
240
  def _require_harness(self, session_id: str) -> AgentHarness:
187
241
  harness = self._harnesses.get(session_id)
188
242
  if harness is None:
@@ -204,7 +258,7 @@ class PiAcpAgent(Agent):
204
258
  return await self._handle_tool_call(session_id, event)
205
259
 
206
260
  resources = await load_session_resources(cwd=cwd, config=self._config)
207
- harness = create_session_harness(
261
+ harness = await create_session_harness(
208
262
  session=session,
209
263
  cwd=cwd,
210
264
  config=self._config,
@@ -287,11 +341,43 @@ def _prompt_to_text_images(
287
341
 
288
342
 
289
343
  def _stop_reason(message: Any) -> str:
344
+ """Map internal AssistantMessage stopReason to ACP PromptResponse stop_reason.
345
+
346
+ ACP stopReason enum: 'end_turn' | 'max_tokens' | 'max_turn_requests' | 'refusal' | 'cancelled'.
347
+ Internal stopReason values: 'stop' | 'length' | 'toolUse' | 'error' | 'aborted'.
348
+ """
290
349
  reason = getattr(message, "stopReason", None) or "stop"
291
350
  if reason == "aborted":
292
351
  return "cancelled"
293
352
  if reason == "length":
294
353
  return "max_tokens"
295
354
  if reason == "error":
296
- return "refusal"
355
+ err = str(getattr(message, "errorMessage", "") or "")
356
+ if any(w in err.lower() for w in ("refus", "policy", "filter", "safety")):
357
+ return "refusal"
358
+ return "end_turn"
297
359
  return "end_turn"
360
+
361
+
362
+ def _permission_mode_from_notification(
363
+ method: str, params: dict[str, Any] | None
364
+ ) -> PermissionMode | None:
365
+ """Best-effort permission mode sync for live sessions.
366
+
367
+ TUI settings changes are sent as extension notifications; we accept known
368
+ suffixes and update in-memory mode so the next tool call in this process
369
+ uses the new policy immediately.
370
+ """
371
+ suffix = method.rsplit("/", 1)[-1].strip().lower()
372
+ if suffix not in {"yolo_mode_changed", "permission_mode_changed"}:
373
+ return None
374
+ payload = params or {}
375
+ raw = payload.get("permission_mode")
376
+ if raw is None:
377
+ raw = payload.get("permissionMode")
378
+ if not isinstance(raw, str):
379
+ return None
380
+ normalized = raw.strip().lower()
381
+ if normalized not in {"ask", "auto", "always-approve"}:
382
+ return None
383
+ return cast(PermissionMode, normalized)
@@ -0,0 +1,56 @@
1
+ """Benchmark evaluation subsystem for pi-agent-cli and pi-python harness."""
2
+
3
+ from pi_agent_cli.benchmarks.collector import (
4
+ format_markdown_report,
5
+ print_summary_table,
6
+ save_summary_artifacts,
7
+ )
8
+ from pi_agent_cli.benchmarks.egress_policy import (
9
+ OFFICIAL_ALLOWED_HOSTS,
10
+ EgressPolicy,
11
+ resolve_task_egress_policy,
12
+ )
13
+ from pi_agent_cli.benchmarks.evaluator import run_trial
14
+ from pi_agent_cli.benchmarks.golden_seed import (
15
+ get_seed_dir,
16
+ is_seed_ready,
17
+ load_golden_manifest,
18
+ provision_golden_seeds,
19
+ )
20
+ from pi_agent_cli.benchmarks.models import (
21
+ BenchmarkSummary,
22
+ BenchmarkTask,
23
+ TrialResult,
24
+ )
25
+ from pi_agent_cli.benchmarks.pelican import (
26
+ PELICAN_PROMPT,
27
+ PelicanSvgReport,
28
+ extract_svg,
29
+ save_pelican_artifact,
30
+ validate_pelican_svg,
31
+ )
32
+ from pi_agent_cli.benchmarks.task_loader import discover_tasks, load_task_from_dir
33
+
34
+ __all__ = [
35
+ "OFFICIAL_ALLOWED_HOSTS",
36
+ "PELICAN_PROMPT",
37
+ "BenchmarkSummary",
38
+ "BenchmarkTask",
39
+ "EgressPolicy",
40
+ "PelicanSvgReport",
41
+ "TrialResult",
42
+ "discover_tasks",
43
+ "extract_svg",
44
+ "format_markdown_report",
45
+ "get_seed_dir",
46
+ "is_seed_ready",
47
+ "load_golden_manifest",
48
+ "load_task_from_dir",
49
+ "print_summary_table",
50
+ "provision_golden_seeds",
51
+ "resolve_task_egress_policy",
52
+ "run_trial",
53
+ "save_pelican_artifact",
54
+ "save_summary_artifacts",
55
+ "validate_pelican_svg",
56
+ ]
@@ -0,0 +1,130 @@
1
+ """Report collector and summarizer for benchmark evaluations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+
7
+ from pi_agent_cli.benchmarks.models import BenchmarkSummary
8
+
9
+
10
+ def format_markdown_report(summary: BenchmarkSummary) -> str:
11
+ """Generate a clean Markdown evaluation report."""
12
+ pass_pct = f"{summary.pass_rate * 100:.1f}%"
13
+ cache_pct = f"{summary.typical_cache_hit_rate * 100:.1f}%"
14
+ cost_per_pass = (
15
+ f"${summary.effective_cost_per_pass:.4f}"
16
+ if summary.effective_cost_per_pass is not None
17
+ else "N/A"
18
+ )
19
+ tokens_per_pass = (
20
+ f"{summary.tokens_per_solved:,.0f}" if summary.tokens_per_solved is not None else "N/A"
21
+ )
22
+
23
+ md = [
24
+ f"# Benchmark Evaluation Report: {summary.run_id}",
25
+ "",
26
+ f"- **Harness**: `{summary.harness}`",
27
+ f"- **Model**: `{summary.model}` ({summary.provider})",
28
+ f"- **Time**: `{summary.started_at}` to `{summary.finished_at}`",
29
+ f"- **Tasks**: {summary.successful_tasks} / {summary.completed_tasks} passed ({pass_pct})",
30
+ "",
31
+ "## Key Metrics",
32
+ "",
33
+ "| Metric | Value | Meaning |",
34
+ "| :--- | :--- | :--- |",
35
+ f"| **Pass Rate** | **{pass_pct}** | Solved and verified tasks |",
36
+ f"| **Effective Cost / Pass** | **{cost_per_pass}** | Amortized cost per passed task |",
37
+ ]
38
+ if summary.effective_cost_per_pass_kimi_k3 is not None:
39
+ md.append(
40
+ f"| **Normalized Cost / Pass (Kimi K3)** | "
41
+ f"**${summary.effective_cost_per_pass_kimi_k3:.4f}** | "
42
+ "Amortized cost normalized to official benchmark pricing |"
43
+ )
44
+ md.extend(
45
+ [
46
+ f"| **Tokens / Solved** | **{tokens_per_pass}** | Amortized tokens per passed task |",
47
+ f"| **Total Cost** | ${summary.total_cost_usd:.4f} | Total API spend for the run |",
48
+ f"| **Typical Cache Hit Rate** | {cache_pct} | Prompt caching efficiency |",
49
+ f"| **Mean Turns** | {summary.mean_turns:.1f} | Average interaction turns |",
50
+ (
51
+ f"| **Mean No-Action Turns** | {summary.mean_no_action_turns:.1f} | "
52
+ "Turns without file/shell operations (overhead) |"
53
+ ),
54
+ (
55
+ f"| **Median Duration** | {summary.median_duration_seconds:.1f}s | "
56
+ "Median elapsed time |"
57
+ ),
58
+ "",
59
+ "## Task Results",
60
+ "",
61
+ (
62
+ "| Status | Task ID | Duration | Turns (No-Act) | "
63
+ "Tokens | Cache | Cost | Norm Cost (K3) |"
64
+ ),
65
+ "| :---: | :--- | :---: | :---: | :---: | :---: | :---: | :---: |",
66
+ ]
67
+ )
68
+
69
+ for t in summary.trials:
70
+ icon = "✅" if t.success else "❌"
71
+ cache_str = f"{t.cache_hit_rate_normalized * 100:.0f}%"
72
+ k3_cost = (
73
+ f"${t.cost_kimi_k3_normalized_usd:.4f}"
74
+ if t.cost_kimi_k3_normalized_usd is not None
75
+ else "N/A"
76
+ )
77
+ md.append(
78
+ f"| {icon} | `{t.id}` | {t.duration_seconds:.1f}s | "
79
+ f"{t.turns} ({t.no_action_turns}) | {t.total_tokens:,} | "
80
+ f"{cache_str} | ${t.cost_first_cold_usd:.4f} | {k3_cost} |"
81
+ )
82
+
83
+ md.append("")
84
+ return "\n".join(md)
85
+
86
+
87
+ def print_summary_table(summary: BenchmarkSummary) -> None:
88
+ """Print an ASCII table of results to standard output."""
89
+ pass_pct = f"{summary.pass_rate * 100:.1f}%"
90
+ cost_per_pass = (
91
+ f"${summary.effective_cost_per_pass:.4f}"
92
+ if summary.effective_cost_per_pass is not None
93
+ else "N/A"
94
+ )
95
+
96
+ print("\n" + "=" * 70)
97
+ print(f" BENCHMARK RUN SUMMARY: {summary.run_id}")
98
+ print(f" Model: {summary.provider}:{summary.model}")
99
+ print("=" * 70)
100
+ print(f" Passed: {summary.successful_tasks}/{summary.completed_tasks} ({pass_pct})")
101
+ print(f" Effective Cost: {cost_per_pass}")
102
+ if summary.effective_cost_per_pass_kimi_k3 is not None:
103
+ print(f" Norm Cost (K3): ${summary.effective_cost_per_pass_kimi_k3:.4f}")
104
+ print(f" Total Cost: ${summary.total_cost_usd:.4f}")
105
+ print(f" Typical Cache: {summary.typical_cache_hit_rate * 100:.1f}%")
106
+ print(f" Mean Turns (No-Act): {summary.mean_turns:.1f} ({summary.mean_no_action_turns:.1f})")
107
+ print("-" * 70)
108
+ print(f" {'Status':<8} {'Task ID':<30} {'Turns':<8} {'Tokens':<10} {'Cost':<10}")
109
+ print("-" * 70)
110
+ for t in summary.trials:
111
+ stat = "PASS" if t.success else "FAIL"
112
+ cost_str = f"${t.cost_first_cold_usd:<9.4f}"
113
+ print(f" {stat:<8} {t.title:<30} {t.turns:<8} {t.total_tokens:<10} {cost_str}")
114
+ print("=" * 70 + "\n")
115
+
116
+
117
+ def save_summary_artifacts(
118
+ summary: BenchmarkSummary,
119
+ output_dir: Path,
120
+ ) -> Path:
121
+ """Save summary JSON and Markdown report into output directory."""
122
+ output_dir.mkdir(parents=True, exist_ok=True)
123
+
124
+ summary_json_path = output_dir / "eval-summary.json"
125
+ summary_json_path.write_text(summary.to_json(indent=2), encoding="utf-8")
126
+
127
+ report_md_path = output_dir / "REPORT.md"
128
+ report_md_path.write_text(format_markdown_report(summary), encoding="utf-8")
129
+
130
+ return report_md_path