panagent 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- panagent/__init__.py +30 -0
- panagent/__main__.py +3 -0
- panagent/api.py +171 -0
- panagent/browser.py +210 -0
- panagent/cli.py +228 -0
- panagent/detect.py +154 -0
- panagent/errors.py +18 -0
- panagent/model.py +154 -0
- panagent/py.typed +0 -0
- panagent/readers.py +484 -0
- panagent/web.py +774 -0
- panagent/writers.py +533 -0
- panagent-0.3.0.dist-info/METADATA +167 -0
- panagent-0.3.0.dist-info/RECORD +18 -0
- panagent-0.3.0.dist-info/WHEEL +5 -0
- panagent-0.3.0.dist-info/entry_points.txt +2 -0
- panagent-0.3.0.dist-info/licenses/LICENSE +21 -0
- panagent-0.3.0.dist-info/top_level.txt +1 -0
panagent/__init__.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Move agent conversations between Claude Code, Codex, ChatGPT, Claude and tavya.
|
|
2
|
+
|
|
3
|
+
import panagent
|
|
4
|
+
conversation = panagent.load("https://chatgpt.com/share/...")
|
|
5
|
+
print(panagent.render(conversation, "markdown").text)
|
|
6
|
+
panagent.install(conversation, "claude-code").command # 'cd ... && claude --resume ...'
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
__version__ = "0.3.0"
|
|
10
|
+
|
|
11
|
+
from .api import Installed, install, load, parse, render
|
|
12
|
+
from .errors import AcquisitionError, BrowserRequired, FormatError, PanagentError
|
|
13
|
+
from .model import SCHEMA, new_conversation, validate_conversation
|
|
14
|
+
from .writers import Rendered
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"SCHEMA",
|
|
18
|
+
"AcquisitionError",
|
|
19
|
+
"BrowserRequired",
|
|
20
|
+
"FormatError",
|
|
21
|
+
"Installed",
|
|
22
|
+
"PanagentError",
|
|
23
|
+
"Rendered",
|
|
24
|
+
"install",
|
|
25
|
+
"load",
|
|
26
|
+
"new_conversation",
|
|
27
|
+
"parse",
|
|
28
|
+
"render",
|
|
29
|
+
"validate_conversation",
|
|
30
|
+
]
|
panagent/__main__.py
ADDED
panagent/api.py
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""The library interface: load any supported source, render or install it as any target."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from shlex import quote
|
|
12
|
+
from typing import Any
|
|
13
|
+
from uuid import uuid4
|
|
14
|
+
|
|
15
|
+
from .browser import fetch_share_browser
|
|
16
|
+
from .detect import canonical_format, detect_text, url_format
|
|
17
|
+
from .errors import AcquisitionError, BrowserRequired, PanagentError
|
|
18
|
+
from .model import default_mode, validate_conversation
|
|
19
|
+
from .readers import READERS, read_file
|
|
20
|
+
from .web import WEB_READERS, fetch_share
|
|
21
|
+
from .writers import WRITERS, Rendered
|
|
22
|
+
|
|
23
|
+
# Shares whose pages may need a real browser (a challenge or a client-rendered app).
|
|
24
|
+
BROWSER_FORMATS = {"chatgpt-share", "claude-share"}
|
|
25
|
+
NATIVE_TARGETS = {"claude-code", "codex"}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def parse(
|
|
29
|
+
text: str,
|
|
30
|
+
source_format: str | None = None,
|
|
31
|
+
*,
|
|
32
|
+
source_uri: str | None = None,
|
|
33
|
+
conversation: str | None = None,
|
|
34
|
+
) -> dict[str, Any]:
|
|
35
|
+
"""Read a conversation from text in any supported input format (detected when omitted)."""
|
|
36
|
+
fmt = canonical_format(source_format) if source_format else detect_text(text)
|
|
37
|
+
reader = READERS.get(fmt) or WEB_READERS.get(fmt)
|
|
38
|
+
if reader is None:
|
|
39
|
+
raise PanagentError(f"{fmt} is output-only")
|
|
40
|
+
return validate_conversation(reader(text, source_uri=source_uri, conversation=conversation))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def load(
|
|
44
|
+
source: str | os.PathLike[str],
|
|
45
|
+
source_format: str | None = None,
|
|
46
|
+
*,
|
|
47
|
+
conversation: str | None = None,
|
|
48
|
+
timeout: float = 30.0,
|
|
49
|
+
browser: str = "never",
|
|
50
|
+
browser_timeout: float = 120.0,
|
|
51
|
+
cdp_url: str | None = None,
|
|
52
|
+
browser_profile: str | None = None,
|
|
53
|
+
) -> dict[str, Any]:
|
|
54
|
+
"""Read a conversation from a file or a public share URL.
|
|
55
|
+
|
|
56
|
+
browser is "never" (the library default), "auto", "headless" or "headed";
|
|
57
|
+
it only applies to ChatGPT and Claude shares that plain HTTPS cannot read.
|
|
58
|
+
"""
|
|
59
|
+
source = os.fspath(source)
|
|
60
|
+
share = url_format(source) if "://" in source else None
|
|
61
|
+
if not share:
|
|
62
|
+
return parse(read_file(Path(source)), source_format, source_uri=source, conversation=conversation)
|
|
63
|
+
fmt = canonical_format(source_format) if source_format else share
|
|
64
|
+
if fmt != share:
|
|
65
|
+
raise PanagentError(f"URL does not match source format {fmt}")
|
|
66
|
+
reader = WEB_READERS[fmt]
|
|
67
|
+
|
|
68
|
+
def through_browser() -> dict[str, Any]:
|
|
69
|
+
text = fetch_share_browser(source, timeout=browser_timeout, mode=browser, cdp_url=cdp_url, profile=browser_profile)
|
|
70
|
+
return validate_conversation(reader(text, source_uri=source))
|
|
71
|
+
|
|
72
|
+
if fmt in BROWSER_FORMATS and (browser in {"headless", "headed"} or cdp_url):
|
|
73
|
+
return through_browser()
|
|
74
|
+
try:
|
|
75
|
+
return validate_conversation(reader(fetch_share(source, timeout=timeout), source_uri=source))
|
|
76
|
+
except AcquisitionError as exc:
|
|
77
|
+
retryable = isinstance(exc, BrowserRequired) or exc.status in {403, 429}
|
|
78
|
+
if fmt not in BROWSER_FORMATS or browser == "never" or not retryable:
|
|
79
|
+
raise
|
|
80
|
+
return through_browser()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def render(
|
|
84
|
+
conv: dict[str, Any],
|
|
85
|
+
target: str,
|
|
86
|
+
*,
|
|
87
|
+
mode: str = "auto",
|
|
88
|
+
cwd: str | None = None,
|
|
89
|
+
session_id: str | None = None,
|
|
90
|
+
) -> Rendered:
|
|
91
|
+
"""Write a conversation as target ("ir", "markdown", "claude-code" or "codex").
|
|
92
|
+
|
|
93
|
+
mode "auto" gives native targets a guarded context message for web snapshots
|
|
94
|
+
and a turn-by-turn transcript for agent sessions; IR and Markdown ignore it.
|
|
95
|
+
"""
|
|
96
|
+
fmt = canonical_format(target)
|
|
97
|
+
writer = WRITERS.get(fmt)
|
|
98
|
+
if writer is None:
|
|
99
|
+
raise PanagentError(f"{fmt} is input-only")
|
|
100
|
+
validate_conversation(conv)
|
|
101
|
+
if mode == "auto":
|
|
102
|
+
mode = default_mode(conv) if fmt in NATIVE_TARGETS else "transcript"
|
|
103
|
+
if mode not in {"context", "transcript"}:
|
|
104
|
+
raise PanagentError(f"unknown mode: {mode}")
|
|
105
|
+
rendered = writer(conv, mode=mode, cwd=cwd, session_id=session_id)
|
|
106
|
+
rendered.format, rendered.mode = fmt, mode
|
|
107
|
+
return rendered
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass
|
|
111
|
+
class Installed:
|
|
112
|
+
path: Path
|
|
113
|
+
session_id: str
|
|
114
|
+
command: str
|
|
115
|
+
rendered: Rendered
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def install(
|
|
119
|
+
conv: dict[str, Any],
|
|
120
|
+
target: str,
|
|
121
|
+
*,
|
|
122
|
+
cwd: str | os.PathLike[str] | None = None,
|
|
123
|
+
session_id: str | None = None,
|
|
124
|
+
mode: str = "auto",
|
|
125
|
+
home: str | os.PathLike[str] | None = None,
|
|
126
|
+
) -> Installed:
|
|
127
|
+
"""Add a conversation to Claude Code's or Codex's own history as a new session.
|
|
128
|
+
|
|
129
|
+
Claude Code finds sessions per project, so cwd (default: the current
|
|
130
|
+
directory) is where `claude --resume` must run. home overrides the CLI's
|
|
131
|
+
data directory ($CLAUDE_CONFIG_DIR or ~/.claude; $CODEX_HOME or ~/.codex).
|
|
132
|
+
An existing session file is never overwritten.
|
|
133
|
+
"""
|
|
134
|
+
fmt = canonical_format(target)
|
|
135
|
+
if fmt not in NATIVE_TARGETS:
|
|
136
|
+
raise PanagentError("only claude-code and codex sessions can be installed")
|
|
137
|
+
directory = str(Path(cwd or os.getcwd()).expanduser().resolve())
|
|
138
|
+
# The installed copy is a new session of the target CLI, never the source's identity.
|
|
139
|
+
session_id = session_id or str(uuid4())
|
|
140
|
+
rendered = render(conv, fmt, mode=mode, cwd=directory, session_id=session_id)
|
|
141
|
+
if fmt == "claude-code":
|
|
142
|
+
root = Path(home or os.environ.get("CLAUDE_CONFIG_DIR") or Path.home() / ".claude").expanduser()
|
|
143
|
+
path = root / "projects" / re.sub(r"[^A-Za-z0-9]", "-", directory) / f"{session_id}.jsonl"
|
|
144
|
+
command = f"cd {quote(directory)} && claude --resume {session_id}"
|
|
145
|
+
else:
|
|
146
|
+
root = Path(home or os.environ.get("CODEX_HOME") or Path.home() / ".codex").expanduser()
|
|
147
|
+
started = _utc(rendered.text.split("\n", 1)[0])
|
|
148
|
+
path = (root / "sessions" / started.strftime("%Y/%m/%d")
|
|
149
|
+
/ f"rollout-{started.strftime('%Y-%m-%dT%H-%M-%S')}-{session_id}.jsonl")
|
|
150
|
+
command = f"codex resume {session_id}"
|
|
151
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
152
|
+
try:
|
|
153
|
+
descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
|
|
154
|
+
except FileExistsError as exc:
|
|
155
|
+
raise PanagentError(f"session {session_id} already exists at {path}") from exc
|
|
156
|
+
try:
|
|
157
|
+
with os.fdopen(descriptor, "w", encoding="utf-8", newline="") as handle:
|
|
158
|
+
handle.write(rendered.text)
|
|
159
|
+
except BaseException:
|
|
160
|
+
path.unlink(missing_ok=True)
|
|
161
|
+
raise
|
|
162
|
+
return Installed(path, session_id, command, rendered)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _utc(first_record: str) -> datetime:
|
|
166
|
+
"""The session start of a Codex rollout (from its session_meta record), else now."""
|
|
167
|
+
try:
|
|
168
|
+
value = json.loads(first_record)["timestamp"]
|
|
169
|
+
return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(timezone.utc)
|
|
170
|
+
except (ValueError, KeyError, TypeError, AttributeError):
|
|
171
|
+
return datetime.now(timezone.utc)
|
panagent/browser.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Optional real-browser acquisition for share pages that resist plain HTTP."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
import tempfile
|
|
9
|
+
import time
|
|
10
|
+
from contextlib import ExitStack
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
from urllib.parse import urlparse
|
|
14
|
+
|
|
15
|
+
from .errors import AcquisitionError
|
|
16
|
+
from .web import is_challenge_page
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
CLAUDE_MESSAGE_SELECTOR = (
|
|
20
|
+
'[data-message-author-role], [data-author-role], '
|
|
21
|
+
'[data-testid*="user-message"], [data-testid*="human-message"], '
|
|
22
|
+
'[data-testid*="assistant-message"], [data-testid*="ai-message"]'
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def fetch_share_browser(
|
|
27
|
+
url: str,
|
|
28
|
+
*,
|
|
29
|
+
timeout: float = 120.0,
|
|
30
|
+
mode: str = "auto",
|
|
31
|
+
cdp_url: str | None = None,
|
|
32
|
+
profile: str | None = None,
|
|
33
|
+
) -> str:
|
|
34
|
+
"""Render a share in Chromium and return structured JSON or final HTML.
|
|
35
|
+
|
|
36
|
+
Playwright is deliberately optional. A CDP connection is the safest path for
|
|
37
|
+
an already-open, user-controlled Chrome because authentication and challenge
|
|
38
|
+
cookies remain in that browser. Otherwise a dedicated persistent profile can
|
|
39
|
+
be used with a headed browser so the user can complete a challenge directly.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
if cdp_url and urlparse(cdp_url).hostname not in {"127.0.0.1", "::1", "localhost"}:
|
|
43
|
+
raise AcquisitionError("--cdp-url is restricted to loopback because Chrome debugging grants full browser control")
|
|
44
|
+
|
|
45
|
+
try:
|
|
46
|
+
from playwright.sync_api import Error as PlaywrightError
|
|
47
|
+
from playwright.sync_api import TimeoutError as PlaywrightTimeoutError
|
|
48
|
+
from playwright.sync_api import sync_playwright
|
|
49
|
+
except ImportError as exc:
|
|
50
|
+
raise AcquisitionError(
|
|
51
|
+
"browser acquisition requires the optional Playwright support. Install it with "
|
|
52
|
+
"`python -m pip install 'panagent[browser]'`, then run `playwright install chromium`, "
|
|
53
|
+
"or pass --cdp-url for an existing Chrome debugging endpoint."
|
|
54
|
+
) from exc
|
|
55
|
+
|
|
56
|
+
deadline = time.monotonic() + max(timeout, 1.0)
|
|
57
|
+
captured_json: list[str] = []
|
|
58
|
+
browser: Any = None
|
|
59
|
+
context: Any = None
|
|
60
|
+
page: Any = None
|
|
61
|
+
owns_context = False
|
|
62
|
+
try:
|
|
63
|
+
# ExitStack closes the tab/context before Playwright disconnects. This
|
|
64
|
+
# matters for CDP: disconnecting first would leave our tab in the user's
|
|
65
|
+
# already-running browser.
|
|
66
|
+
with sync_playwright() as playwright, ExitStack() as cleanup:
|
|
67
|
+
if cdp_url:
|
|
68
|
+
try:
|
|
69
|
+
browser = playwright.chromium.connect_over_cdp(cdp_url)
|
|
70
|
+
except PlaywrightError as exc:
|
|
71
|
+
raise AcquisitionError(f"could not connect to Chrome at {cdp_url}: {exc}") from exc
|
|
72
|
+
had_context = bool(browser.contexts)
|
|
73
|
+
context = browser.contexts[0] if had_context else browser.new_context()
|
|
74
|
+
owns_context = not had_context
|
|
75
|
+
else:
|
|
76
|
+
headed = _headed(mode)
|
|
77
|
+
if mode in {"auto", "headed"} and not headed:
|
|
78
|
+
raise AcquisitionError(
|
|
79
|
+
"Claude requires an interactive browser challenge in this environment. Run from a desktop "
|
|
80
|
+
"terminal with --browser headed, or expose an existing Chrome with remote debugging and "
|
|
81
|
+
"pass --cdp-url."
|
|
82
|
+
)
|
|
83
|
+
user_data_dir = profile
|
|
84
|
+
if user_data_dir is None:
|
|
85
|
+
temporary_profile = tempfile.TemporaryDirectory(prefix="panagent-browser-")
|
|
86
|
+
cleanup.callback(temporary_profile.cleanup)
|
|
87
|
+
user_data_dir = temporary_profile.name
|
|
88
|
+
Path(user_data_dir).expanduser().mkdir(parents=True, exist_ok=True)
|
|
89
|
+
context = _launch_persistent(playwright, str(Path(user_data_dir).expanduser()), headless=not headed)
|
|
90
|
+
owns_context = True
|
|
91
|
+
|
|
92
|
+
if owns_context:
|
|
93
|
+
cleanup.callback(_safe_close, context)
|
|
94
|
+
page = context.new_page()
|
|
95
|
+
cleanup.callback(_safe_close, page)
|
|
96
|
+
|
|
97
|
+
def capture(response: Any) -> None:
|
|
98
|
+
if "claude.ai" not in url or "json" not in (response.headers.get("content-type") or "").lower():
|
|
99
|
+
return
|
|
100
|
+
try:
|
|
101
|
+
body = response.text()
|
|
102
|
+
except PlaywrightError:
|
|
103
|
+
return
|
|
104
|
+
if "chat_messages" in body or '"messages"' in body and ('"uuid"' in body or '"title"' in body):
|
|
105
|
+
captured_json.append(body)
|
|
106
|
+
|
|
107
|
+
page.on("response", capture)
|
|
108
|
+
try:
|
|
109
|
+
page.goto(url, wait_until="domcontentloaded", timeout=min(timeout, 30.0) * 1000)
|
|
110
|
+
except PlaywrightTimeoutError:
|
|
111
|
+
# A challenge or client-rendered app can keep navigation busy. The
|
|
112
|
+
# polling below inspects the page until the full acquisition timeout.
|
|
113
|
+
pass
|
|
114
|
+
|
|
115
|
+
while time.monotonic() < deadline:
|
|
116
|
+
if captured_json:
|
|
117
|
+
return captured_json[-1]
|
|
118
|
+
if "claude.ai" in url:
|
|
119
|
+
exported = _claude_dom_export(page)
|
|
120
|
+
if exported is not None:
|
|
121
|
+
return json.dumps(exported, ensure_ascii=False)
|
|
122
|
+
else:
|
|
123
|
+
html = page.content()
|
|
124
|
+
if "streamController.enqueue" in html or page.locator("[data-message-author-role]").count() >= 2:
|
|
125
|
+
return html
|
|
126
|
+
if mode == "headless" and not cdp_url and is_challenge_page(page.content(), page.url):
|
|
127
|
+
raise AcquisitionError(
|
|
128
|
+
"Claude returned a challenge that cannot be completed in a headless browser. "
|
|
129
|
+
"Use --browser headed or --cdp-url with a user-controlled Chrome."
|
|
130
|
+
)
|
|
131
|
+
page.wait_for_timeout(500)
|
|
132
|
+
|
|
133
|
+
html = page.content()
|
|
134
|
+
if is_challenge_page(html, page.url):
|
|
135
|
+
raise AcquisitionError(
|
|
136
|
+
"the browser challenge was not completed before the timeout. Complete it in the opened browser "
|
|
137
|
+
"and retry with a longer --browser-timeout, or use --cdp-url with that browser."
|
|
138
|
+
)
|
|
139
|
+
if "claude.ai" in url:
|
|
140
|
+
raise AcquisitionError(
|
|
141
|
+
"Claude rendered no recognizable conversation messages. Keep the share open until it fully "
|
|
142
|
+
"loads, or use the documented browser JSON/HTML export fallback."
|
|
143
|
+
)
|
|
144
|
+
return html
|
|
145
|
+
except PlaywrightError as exc:
|
|
146
|
+
raise AcquisitionError(
|
|
147
|
+
f"browser acquisition failed: {exc}. Install a Playwright browser with `playwright install chromium`, "
|
|
148
|
+
"or pass --cdp-url for an existing Chrome instance."
|
|
149
|
+
) from exc
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _safe_close(resource: Any) -> None:
|
|
153
|
+
try:
|
|
154
|
+
resource.close()
|
|
155
|
+
except Exception:
|
|
156
|
+
pass
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _headed(mode: str) -> bool:
|
|
160
|
+
if mode == "headed":
|
|
161
|
+
return True
|
|
162
|
+
if mode == "headless":
|
|
163
|
+
return False
|
|
164
|
+
if not sys.stdin.isatty():
|
|
165
|
+
return False
|
|
166
|
+
if sys.platform.startswith("linux"):
|
|
167
|
+
return bool(os.environ.get("DISPLAY") or os.environ.get("WAYLAND_DISPLAY"))
|
|
168
|
+
return True
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _launch_persistent(playwright: Any, profile: str, *, headless: bool) -> Any:
|
|
172
|
+
errors: list[str] = []
|
|
173
|
+
for options in ({"channel": "chrome"}, {}):
|
|
174
|
+
try:
|
|
175
|
+
return playwright.chromium.launch_persistent_context(profile, headless=headless, **options)
|
|
176
|
+
except Exception as exc:
|
|
177
|
+
errors.append(str(exc))
|
|
178
|
+
raise AcquisitionError(
|
|
179
|
+
"could not launch Chrome/Chromium. Install Google Chrome or run `playwright install chromium`. "
|
|
180
|
+
+ " | ".join(errors)
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _claude_dom_export(page: Any) -> dict[str, Any] | None:
|
|
185
|
+
value = page.evaluate(
|
|
186
|
+
"""(selector) => {
|
|
187
|
+
const all = [...document.querySelectorAll(selector)];
|
|
188
|
+
const nodes = all.filter((node) => !all.some((other) => other !== node && node.contains(other)));
|
|
189
|
+
const chat_messages = nodes.map((element, index) => {
|
|
190
|
+
const testid = (element.getAttribute('data-testid') || '').toLowerCase();
|
|
191
|
+
const nativeRole = element.getAttribute('data-message-author-role') ||
|
|
192
|
+
element.getAttribute('data-author-role') ||
|
|
193
|
+
(testid.includes('user') || testid.includes('human') ? 'human' : 'assistant');
|
|
194
|
+
return {
|
|
195
|
+
uuid: element.getAttribute('data-message-id') || element.id || `browser-${index}`,
|
|
196
|
+
sender: nativeRole === 'user' ? 'human' : nativeRole,
|
|
197
|
+
text: element.innerText
|
|
198
|
+
};
|
|
199
|
+
}).filter((item) => item.text && ['human', 'assistant', 'system'].includes(item.sender));
|
|
200
|
+
if (chat_messages.length < 2) return null;
|
|
201
|
+
return {
|
|
202
|
+
name: document.title.replace(/^Claude\\s*[-–]\\s*/, ''),
|
|
203
|
+
source_url: location.href,
|
|
204
|
+
chat_messages
|
|
205
|
+
};
|
|
206
|
+
}""",
|
|
207
|
+
CLAUDE_MESSAGE_SELECTOR,
|
|
208
|
+
)
|
|
209
|
+
return value if isinstance(value, dict) else None
|
|
210
|
+
|
panagent/cli.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any, Sequence
|
|
10
|
+
from uuid import UUID
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .api import NATIVE_TARGETS, install, load, parse, render
|
|
14
|
+
from .detect import FORMAT_ALIASES, canonical_format, detect_text
|
|
15
|
+
from .errors import PanagentError
|
|
16
|
+
from .readers import READERS, read_file
|
|
17
|
+
from .web import WEB_READERS, list_conversations
|
|
18
|
+
from .writers import WRITERS, Rendered
|
|
19
|
+
|
|
20
|
+
INPUTS = sorted(name for name, fmt in FORMAT_ALIASES.items() if fmt in READERS or fmt in WEB_READERS)
|
|
21
|
+
OUTPUTS = sorted(name for name, fmt in FORMAT_ALIASES.items() if fmt in WRITERS)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _session_id(value: str) -> str:
|
|
25
|
+
try:
|
|
26
|
+
return str(UUID(value))
|
|
27
|
+
except ValueError as exc:
|
|
28
|
+
raise argparse.ArgumentTypeError("session id must be a UUID") from exc
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def parser() -> argparse.ArgumentParser:
|
|
32
|
+
result = argparse.ArgumentParser(
|
|
33
|
+
prog="panagent",
|
|
34
|
+
description="Move conversations between Claude Code, Codex, ChatGPT, Claude and tavya through a provenance-preserving neutral format.",
|
|
35
|
+
)
|
|
36
|
+
result.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
37
|
+
subparsers = result.add_subparsers(dest="command", required=True)
|
|
38
|
+
|
|
39
|
+
convert = subparsers.add_parser("convert", help="convert a conversation or public share")
|
|
40
|
+
convert.add_argument("source", help="input file, '-' for stdin, or a public share URL")
|
|
41
|
+
convert.add_argument("--from", dest="source_format", choices=INPUTS, help="source format (default: detect)")
|
|
42
|
+
convert.add_argument("--to", dest="target_format", required=True, choices=OUTPUTS, help="destination format")
|
|
43
|
+
destination = convert.add_mutually_exclusive_group()
|
|
44
|
+
destination.add_argument("-o", "--output", default="-", help="output file (default: stdout)")
|
|
45
|
+
destination.add_argument(
|
|
46
|
+
"--install",
|
|
47
|
+
action="store_true",
|
|
48
|
+
help="add the result to the destination CLI's own history as a new session and print its resume command",
|
|
49
|
+
)
|
|
50
|
+
convert.add_argument("--conversation", help="conversation id or exact title, for exports holding several")
|
|
51
|
+
convert.add_argument(
|
|
52
|
+
"--mode",
|
|
53
|
+
choices=["auto", "context", "transcript"],
|
|
54
|
+
default="auto",
|
|
55
|
+
help="native output strategy; auto uses guarded context for web chats and transcript for agent sessions",
|
|
56
|
+
)
|
|
57
|
+
convert.add_argument("--cwd", help="working directory recorded in a generated native session (--install: default current)")
|
|
58
|
+
convert.add_argument(
|
|
59
|
+
"--session-id",
|
|
60
|
+
type=_session_id,
|
|
61
|
+
help="destination native session id (default: preserve a UUID source id; --install: a new id)",
|
|
62
|
+
)
|
|
63
|
+
convert.add_argument("--timeout", type=float, default=30.0, help="share request timeout in seconds (default: 30)")
|
|
64
|
+
convert.add_argument(
|
|
65
|
+
"--browser",
|
|
66
|
+
choices=["auto", "never", "headless", "headed"],
|
|
67
|
+
default="auto",
|
|
68
|
+
help="browser fallback for challenged ChatGPT/Claude shares (default: auto; Playwright is optional)",
|
|
69
|
+
)
|
|
70
|
+
convert.add_argument(
|
|
71
|
+
"--browser-timeout",
|
|
72
|
+
type=float,
|
|
73
|
+
default=120.0,
|
|
74
|
+
help="seconds to wait for browser rendering or a user-completed challenge (default: 120)",
|
|
75
|
+
)
|
|
76
|
+
convert.add_argument("--cdp-url", help="connect to an existing Chrome debugging endpoint, for example http://127.0.0.1:9222")
|
|
77
|
+
convert.add_argument("--browser-profile", help="dedicated Chrome/Chromium profile directory used by browser acquisition")
|
|
78
|
+
convert.add_argument("--report", help="write a machine-readable conversion/loss report")
|
|
79
|
+
convert.add_argument("--quiet", action="store_true", help="do not print warnings or summary to stderr")
|
|
80
|
+
convert.add_argument("--fail-on-warning", action="store_true", help="return exit status 3 when any warning is emitted")
|
|
81
|
+
convert.set_defaults(handler=_convert)
|
|
82
|
+
|
|
83
|
+
validate = subparsers.add_parser("validate", help="parse and validate an input without converting it")
|
|
84
|
+
validate.add_argument("source", help="input file or '-' for stdin")
|
|
85
|
+
validate.add_argument("--from", dest="source_format", choices=INPUTS, help="source format (default: detect)")
|
|
86
|
+
validate.add_argument("--conversation", help="conversation id or exact title, for exports holding several")
|
|
87
|
+
validate.add_argument("--quiet", action="store_true")
|
|
88
|
+
validate.set_defaults(handler=_validate)
|
|
89
|
+
|
|
90
|
+
listing = subparsers.add_parser("list", help="list the conversations in a ChatGPT or Claude data export")
|
|
91
|
+
listing.add_argument("source", help="conversations.json, or '-' for stdin")
|
|
92
|
+
listing.add_argument("--from", dest="source_format", choices=INPUTS, help="source format (default: detect)")
|
|
93
|
+
listing.set_defaults(handler=_list)
|
|
94
|
+
return result
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _read_source(source: str) -> str:
|
|
98
|
+
if source == "-":
|
|
99
|
+
return sys.stdin.read()
|
|
100
|
+
if "://" in source:
|
|
101
|
+
raise PanagentError("this command reads local files or stdin; use convert for a share URL")
|
|
102
|
+
return read_file(Path(source))
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _load(args: argparse.Namespace, *, allow_url: bool) -> dict[str, Any]:
|
|
106
|
+
if args.source == "-" or not allow_url:
|
|
107
|
+
return parse(_read_source(args.source), args.source_format, conversation=args.conversation)
|
|
108
|
+
return load(
|
|
109
|
+
args.source,
|
|
110
|
+
args.source_format,
|
|
111
|
+
conversation=args.conversation,
|
|
112
|
+
timeout=args.timeout,
|
|
113
|
+
browser=args.browser,
|
|
114
|
+
browser_timeout=args.browser_timeout,
|
|
115
|
+
cdp_url=args.cdp_url,
|
|
116
|
+
browser_profile=args.browser_profile,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _convert(args: argparse.Namespace) -> int:
|
|
121
|
+
conv = _load(args, allow_url=True)
|
|
122
|
+
options = {"mode": args.mode, "cwd": args.cwd, "session_id": args.session_id}
|
|
123
|
+
if args.install:
|
|
124
|
+
if canonical_format(args.target_format) not in NATIVE_TARGETS:
|
|
125
|
+
raise PanagentError("--install needs --to claude-code or --to codex")
|
|
126
|
+
installed = install(conv, args.target_format, **options)
|
|
127
|
+
rendered, destination = installed.rendered, str(installed.path)
|
|
128
|
+
print(installed.command)
|
|
129
|
+
else:
|
|
130
|
+
rendered, destination = render(conv, args.target_format, **options), args.output
|
|
131
|
+
_write_output(args.output, rendered.text)
|
|
132
|
+
warnings = [*conv.get("warnings", []), *rendered.warnings]
|
|
133
|
+
report = _report(conv, rendered, warnings)
|
|
134
|
+
if args.report:
|
|
135
|
+
_write_output(args.report, json.dumps(report, ensure_ascii=False, indent=2) + "\n")
|
|
136
|
+
if not args.quiet:
|
|
137
|
+
_print_report(report, destination)
|
|
138
|
+
return 3 if args.fail_on_warning and any(item.get("severity", "warning") != "info" for item in warnings) else 0
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _validate(args: argparse.Namespace) -> int:
|
|
142
|
+
conv = _load(args, allow_url=False)
|
|
143
|
+
if not args.quiet:
|
|
144
|
+
print(
|
|
145
|
+
f"valid {conv['source']['format']}: {len(conv['messages'])} messages, {len(conv.get('warnings', []))} warnings",
|
|
146
|
+
file=sys.stderr,
|
|
147
|
+
)
|
|
148
|
+
return 0
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _list(args: argparse.Namespace) -> int:
|
|
152
|
+
text = _read_source(args.source)
|
|
153
|
+
fmt = canonical_format(args.source_format) if args.source_format else detect_text(text)
|
|
154
|
+
if fmt in {"chatgpt-share", "claude-share"} and text.lstrip("\ufeff \t\r\n").startswith(("[", "{")):
|
|
155
|
+
rows = list_conversations(text, fmt)
|
|
156
|
+
else:
|
|
157
|
+
conv = parse(text, fmt)
|
|
158
|
+
rows = [{"id": conv["id"], "title": conv.get("title") or "", "updated_at": conv.get("updated_at"),
|
|
159
|
+
"messages": len(conv["messages"])}]
|
|
160
|
+
for row in rows:
|
|
161
|
+
print("\t".join([str(row["id"] or ""), (row["updated_at"] or "")[:10], str(row["messages"]), row["title"]]))
|
|
162
|
+
return 0
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _report(conv: dict[str, Any], rendered: Rendered, warnings: list[dict[str, Any]]) -> dict[str, Any]:
|
|
166
|
+
blocks: dict[str, int] = {}
|
|
167
|
+
roles: dict[str, int] = {}
|
|
168
|
+
for item in conv["messages"]:
|
|
169
|
+
roles[item["role"]] = roles.get(item["role"], 0) + 1
|
|
170
|
+
for block in item["content"]:
|
|
171
|
+
kind = block["type"]
|
|
172
|
+
blocks[kind] = blocks.get(kind, 0) + 1
|
|
173
|
+
return {
|
|
174
|
+
"source_format": conv["source"]["format"],
|
|
175
|
+
"target_format": rendered.format,
|
|
176
|
+
"mode": rendered.mode,
|
|
177
|
+
"conversation_id": conv.get("id"),
|
|
178
|
+
"messages": len(conv["messages"]),
|
|
179
|
+
"roles": roles,
|
|
180
|
+
"content_blocks": blocks,
|
|
181
|
+
"capabilities": conv.get("capabilities", {}),
|
|
182
|
+
"warnings": warnings,
|
|
183
|
+
"output_bytes": len(rendered.text.encode("utf-8")),
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _print_report(report: dict[str, Any], output: str) -> None:
|
|
188
|
+
destination = "stdout" if output == "-" else output
|
|
189
|
+
print(
|
|
190
|
+
f"panagent: {report['source_format']} -> {report['target_format']} ({report['mode']}): "
|
|
191
|
+
f"{report['messages']} messages written to {destination}",
|
|
192
|
+
file=sys.stderr,
|
|
193
|
+
)
|
|
194
|
+
# Readers warn once per affected record; one line per code is enough on a terminal.
|
|
195
|
+
grouped: dict[str, list[dict[str, Any]]] = {}
|
|
196
|
+
for item in report["warnings"]:
|
|
197
|
+
grouped.setdefault(item.get("code", "warning"), []).append(item)
|
|
198
|
+
for code, items in grouped.items():
|
|
199
|
+
count = f" ({len(items)}x)" if len(items) > 1 else ""
|
|
200
|
+
print(f"panagent: {items[0].get('severity', 'warning')}: {code}{count}: {items[0].get('message', '')}", file=sys.stderr)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _write_output(target: str, content: str) -> None:
|
|
204
|
+
if target == "-":
|
|
205
|
+
sys.stdout.write(content)
|
|
206
|
+
return
|
|
207
|
+
path = Path(target)
|
|
208
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
209
|
+
descriptor, temporary = tempfile.mkstemp(prefix=f".{path.name}.", dir=path.parent, text=True)
|
|
210
|
+
try:
|
|
211
|
+
with os.fdopen(descriptor, "w", encoding="utf-8", newline="") as handle:
|
|
212
|
+
handle.write(content)
|
|
213
|
+
os.replace(temporary, path)
|
|
214
|
+
except BaseException:
|
|
215
|
+
try:
|
|
216
|
+
os.unlink(temporary)
|
|
217
|
+
except OSError:
|
|
218
|
+
pass
|
|
219
|
+
raise
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
223
|
+
args = parser().parse_args(argv)
|
|
224
|
+
try:
|
|
225
|
+
return int(args.handler(args))
|
|
226
|
+
except (PanagentError, OSError, UnicodeError, ValueError, TypeError, KeyError) as exc:
|
|
227
|
+
print(f"panagent: error: {exc}", file=sys.stderr)
|
|
228
|
+
return 2
|