froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/probe.py
ADDED
|
@@ -0,0 +1,1044 @@
|
|
|
1
|
+
"""`froid-loop probe-adapter`: collect + sanitize adapter-finalization data.
|
|
2
|
+
|
|
3
|
+
Finalizing a generic-adapter CLI profile needs facts that live in no doc: the
|
|
4
|
+
CLI's exact hook payload shape (field names/casing, whether transcript_path /
|
|
5
|
+
session_id / cwd are present), where its transcript lives and in what format,
|
|
6
|
+
and the token-usage schema a `usage_parser` must read. This command pulls all of
|
|
7
|
+
that and runs it through the audited :mod:`froid_loop.sanitize` chokepoint, so a
|
|
8
|
+
user of any coding CLI can run one command and paste back a clean, content-free
|
|
9
|
+
report.
|
|
10
|
+
|
|
11
|
+
Two strategies, one report shape:
|
|
12
|
+
|
|
13
|
+
- SCAN (default, zero process launch beyond ``--version``/``--help``): locate the
|
|
14
|
+
newest already-existing transcript by convention, read the declared hook config,
|
|
15
|
+
infer the token schema. Works whenever the user has used the CLI before.
|
|
16
|
+
- PROBE (``--probe``, opt-in): in an ephemeral ``mkdtemp`` workspace, register the
|
|
17
|
+
full-payload capture hook for every native event, launch one trivial content-free
|
|
18
|
+
turn in a tmux window, capture each event's complete payload, then tear down. The
|
|
19
|
+
raw capture exists only transiently inside the temp dir, which is ``rmtree``'d in a
|
|
20
|
+
``finally`` (even on exception / Ctrl-C).
|
|
21
|
+
|
|
22
|
+
One finding, two render targets: :func:`render_markdown` for the human report
|
|
23
|
+
(the CLI default) and :func:`render_json` for the machine-readable document that
|
|
24
|
+
``--json`` emits instead (the :mod:`froid_loop.machine` contract — one object on
|
|
25
|
+
stdout, nothing else). The document carries :data:`SCHEMA_VERSION` as a top-level
|
|
26
|
+
``schema_version``; do not confuse it with the document's ``version`` key, which
|
|
27
|
+
holds the *probed CLI's* own ``--version`` output.
|
|
28
|
+
|
|
29
|
+
Safety model — the same two layers as :mod:`froid_loop.diagnostics` (closed by
|
|
30
|
+
#199). Captured data is scrubbed/reduced at COLLECTION time (captured payloads
|
|
31
|
+
ship as key-path:type *schema*, never values; paths are per-component redacted
|
|
32
|
+
with the project basename routed through a :class:`sanitize.Pseudonymizer`
|
|
33
|
+
alias), and both renderers run :func:`sanitize.guard` over their own rendered
|
|
34
|
+
bytes before returning — a hard-rule hit (email/secret/home-path/url-creds/
|
|
35
|
+
username) refuses to emit, while a stray occurrence of a registered alias
|
|
36
|
+
original is repaired, re-verified, and disclosed. The residual is stated
|
|
37
|
+
honestly: identifier-shaped proprietary values the probe cannot know about
|
|
38
|
+
(an arbitrary slug in a dynamic key, say) match no hard rule and no registered
|
|
39
|
+
extra, so the guard cannot see them — which is why collection reduces payloads
|
|
40
|
+
to shape instead of trying to enumerate what might be sensitive.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import glob
|
|
46
|
+
import os
|
|
47
|
+
import re
|
|
48
|
+
import shlex
|
|
49
|
+
import shutil
|
|
50
|
+
import subprocess
|
|
51
|
+
import tempfile
|
|
52
|
+
import time
|
|
53
|
+
from dataclasses import dataclass, field
|
|
54
|
+
from datetime import datetime, timezone
|
|
55
|
+
from importlib import resources
|
|
56
|
+
from pathlib import Path
|
|
57
|
+
|
|
58
|
+
from . import runs, sanitize
|
|
59
|
+
from .adapters.multiplexer import MultiplexerError, get_multiplexer
|
|
60
|
+
from .adapters.profile import CLIProfile
|
|
61
|
+
from .install import merge_hooks, relay_registered
|
|
62
|
+
from .process_host import get_process_host
|
|
63
|
+
|
|
64
|
+
# cmd_probe catches `probe.LeakDetected` around the renderers, mirroring
|
|
65
|
+
# diagnostics — the noqa keeps ruff's F401 autofix from deleting the re-export.
|
|
66
|
+
from .sanitize import LeakDetected # noqa: F401 — re-export
|
|
67
|
+
from .signals import SignalWatcher
|
|
68
|
+
from .tokens import _jsonl_entries, read_usage
|
|
69
|
+
|
|
70
|
+
# Version of the `--json` document (machine.py contract). Distinct from the
|
|
71
|
+
# document's `version` key, which holds the *probed CLI's* `--version` output.
|
|
72
|
+
# v2: `captured_events[].payload` (scrubbed values) was removed in favor of
|
|
73
|
+
# `payload_schema` (key paths + leaf types, never values) — a field removal,
|
|
74
|
+
# which the additive-only contract says must bump the version (#199).
|
|
75
|
+
SCHEMA_VERSION = 2
|
|
76
|
+
|
|
77
|
+
# Per-parser transcript-location conventions (from tokens.py docstrings).
|
|
78
|
+
TRANSCRIPT_GLOBS = {
|
|
79
|
+
"claude-jsonl": "~/.claude/projects/*/*.jsonl",
|
|
80
|
+
"codex-rollout": "~/.codex/sessions/*/*/*/rollout-*.jsonl",
|
|
81
|
+
"gemini-chat": "~/.gemini/tmp/*/chats/session-*.jsonl",
|
|
82
|
+
"copilot-events": "~/.copilot/session-state/*/events.jsonl",
|
|
83
|
+
}
|
|
84
|
+
# Fallback family glob keyed by the `cli` name, so a CLI whose usage_parser is
|
|
85
|
+
# still "none" (e.g. antigravity, freshly added) still gets transcript discovery.
|
|
86
|
+
FAMILY_GLOBS = {
|
|
87
|
+
"claude": "~/.claude/projects/*/*.jsonl",
|
|
88
|
+
"codex": "~/.codex/sessions/*/*/*/rollout-*.jsonl",
|
|
89
|
+
"gemini": "~/.gemini/tmp/*/chats/session-*.jsonl",
|
|
90
|
+
"copilot": "~/.copilot/session-state/*/events.jsonl",
|
|
91
|
+
# agy (Antigravity CLI) writes one transcript per conversation, keyed by
|
|
92
|
+
# conversationId. Verified against agy 1.1.3 by capturing a live Stop hook,
|
|
93
|
+
# whose transcriptPath is exactly this shape. Note `transcript_full.jsonl`,
|
|
94
|
+
# not `transcript.jsonl` — agy's own hooks.md shows a WORKSPACE-relative
|
|
95
|
+
# `<ws>/.gemini/antigravity/transcript.jsonl` in its payload example, but
|
|
96
|
+
# that is illustrative: the real path is home-rooted, under brain/.
|
|
97
|
+
"antigravity": (
|
|
98
|
+
"~/.gemini/antigravity-cli/brain/*/.system_generated/logs/transcript_full.jsonl"
|
|
99
|
+
),
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
_TOKEN_KEY_RE = re.compile(
|
|
103
|
+
r"(token|tokens|cached|input|output|prompt|completion|thoughts|usage)", re.I
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
PROBE_HOOK_NAME = "froid_loop_probe_hook.py"
|
|
107
|
+
PROBE_PROMPT = "Reply with exactly: OK"
|
|
108
|
+
PROBE_TASK_ID = "probe"
|
|
109
|
+
PROBE_GRACE_S = 3.0
|
|
110
|
+
MAX_SCHEMA_ENTRIES = 200
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# --------------------------------------------------------------- dataclasses
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class FlagFinding:
|
|
118
|
+
binary: str
|
|
119
|
+
found: bool
|
|
120
|
+
version: str | None = None # scrubbed
|
|
121
|
+
help: str | None = None # scrubbed
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass
|
|
125
|
+
class TranscriptFinding:
|
|
126
|
+
glob: str | None = None # the convention glob used (already ~-relative)
|
|
127
|
+
location: str | None = None # redacted path of the chosen transcript
|
|
128
|
+
fmt: str | None = None # "jsonl" | "json"
|
|
129
|
+
size_bytes: int | None = None
|
|
130
|
+
line_count: int | None = None
|
|
131
|
+
mtime_date: str | None = None # date only (no time), UTC
|
|
132
|
+
multiple: bool = False
|
|
133
|
+
note: str | None = None
|
|
134
|
+
real_path: Path | None = None # NOT rendered; used for schema inference
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass
|
|
138
|
+
class TokenSchema:
|
|
139
|
+
parser: str
|
|
140
|
+
entries_scanned: int = 0
|
|
141
|
+
parsed_usage: dict | None = None # only when parser != "none"
|
|
142
|
+
key_paths: list[str] = field(default_factory=list) # "a.b.c:int", TYPE only
|
|
143
|
+
token_field_candidates: list[str] = field(default_factory=list)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@dataclass
|
|
147
|
+
class EventCapture:
|
|
148
|
+
native_event: str
|
|
149
|
+
canonical_event: str | None
|
|
150
|
+
payload_keys: list[str] # top-level field names, identifier-gated
|
|
151
|
+
payload_schema: list[str] # dotted key paths + leaf TYPES only, never values
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass
|
|
155
|
+
class ProfileFinding:
|
|
156
|
+
cli: str
|
|
157
|
+
mode: str # "scan" | "probe"
|
|
158
|
+
known_profile: bool
|
|
159
|
+
binary: str
|
|
160
|
+
parser: str
|
|
161
|
+
dialect: str | None = None
|
|
162
|
+
flags: FlagFinding | None = None
|
|
163
|
+
declared_events: dict = field(default_factory=dict) # native -> canonical
|
|
164
|
+
registered: bool | None = None # scan: hooks present in the CLI's config?
|
|
165
|
+
captured_events: list[EventCapture] = field(default_factory=list) # probe
|
|
166
|
+
transcript: TranscriptFinding | None = None
|
|
167
|
+
tokens: TokenSchema | None = None
|
|
168
|
+
warnings: list[str] = field(default_factory=list)
|
|
169
|
+
next_steps: list[str] = field(default_factory=list)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@dataclass
|
|
173
|
+
class Hints:
|
|
174
|
+
binary: str | None = None
|
|
175
|
+
transcript: str | None = None
|
|
176
|
+
session_dir: str | None = None
|
|
177
|
+
model: str | None = None
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
# ------------------------------------------------------------ version / help
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _run_capture(argv: list[str], timeout_s: float) -> str | None:
|
|
184
|
+
try:
|
|
185
|
+
# errors="replace" is what keeps run_version_help's documented "Never
|
|
186
|
+
# raises" true (#383): a banner byte the locale codec cannot decode is a
|
|
187
|
+
# UnicodeDecodeError — a ValueError, outside the guard below.
|
|
188
|
+
proc = subprocess.run(
|
|
189
|
+
argv, capture_output=True, text=True, errors="replace", timeout=timeout_s
|
|
190
|
+
)
|
|
191
|
+
except (OSError, subprocess.SubprocessError):
|
|
192
|
+
return None
|
|
193
|
+
out = (proc.stdout or "") + (proc.stderr or "")
|
|
194
|
+
return out.strip() or None
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def binary_runs(binary: str, timeout_s: float = 10) -> int | None:
|
|
198
|
+
"""Return the exit code of ``binary --version``, or None if it never ran.
|
|
199
|
+
|
|
200
|
+
The liveness half of a PATH check. ``shutil.which`` answers "a file with that
|
|
201
|
+
name is on PATH and has the execute bit", which a dead WSL/npm shim satisfies
|
|
202
|
+
while every launch of it fails (#294) — so ``validate`` reported OK on an
|
|
203
|
+
install that could not start a session. Running the binary once is the only
|
|
204
|
+
thing that separates the two.
|
|
205
|
+
|
|
206
|
+
Never raises, and that is load-bearing rather than defensive style: machine.py
|
|
207
|
+
records that every gate in ``cmd_validate`` runs inside a ``try`` so "the
|
|
208
|
+
command has no error path of its own — its rc is purely the verdict". A probe
|
|
209
|
+
that raised would give it one. The guard is ``_run_capture``'s exactly, and
|
|
210
|
+
the return is deliberately left as bytes (no ``text=True``): nothing here reads
|
|
211
|
+
the output, so the locale decode that forced ``errors="replace"`` on that
|
|
212
|
+
function never happens and cannot raise the ``UnicodeDecodeError`` the guard
|
|
213
|
+
does not name.
|
|
214
|
+
|
|
215
|
+
None (could not launch, or timed out) and a nonzero code are separate answers
|
|
216
|
+
to the caller, not one sentinel: the first has no return code to report.
|
|
217
|
+
|
|
218
|
+
``stdin=DEVNULL`` is required, not cosmetic. With the caller's tty inherited, a
|
|
219
|
+
shim that prompts blocks on the read for the whole timeout — measured 4.00s
|
|
220
|
+
against 0.00s — inside an interactive command.
|
|
221
|
+
|
|
222
|
+
Not folded into :func:`run_version_help`, which discards the return code by
|
|
223
|
+
design and spawns TWO children (``--version`` then ``--help``) at ``timeout_s``
|
|
224
|
+
each: reusing it would cost up to 20s per profile here.
|
|
225
|
+
"""
|
|
226
|
+
try:
|
|
227
|
+
proc = subprocess.run(
|
|
228
|
+
[binary, "--version"],
|
|
229
|
+
capture_output=True,
|
|
230
|
+
check=False,
|
|
231
|
+
stdin=subprocess.DEVNULL,
|
|
232
|
+
timeout=timeout_s,
|
|
233
|
+
)
|
|
234
|
+
except (OSError, subprocess.SubprocessError):
|
|
235
|
+
return None
|
|
236
|
+
return proc.returncode
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def run_version_help(binary: str, timeout_s: float = 10) -> FlagFinding:
|
|
240
|
+
"""Scrubbed ``--version``/``--help`` for a binary. Never raises."""
|
|
241
|
+
if not shutil.which(binary):
|
|
242
|
+
return FlagFinding(binary=binary, found=False)
|
|
243
|
+
version = _run_capture([binary, "--version"], timeout_s)
|
|
244
|
+
help_txt = _run_capture([binary, "--help"], timeout_s)
|
|
245
|
+
return FlagFinding(
|
|
246
|
+
binary=binary,
|
|
247
|
+
found=True,
|
|
248
|
+
version=(
|
|
249
|
+
sanitize.scrub_text(version, max_lines=5, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS)
|
|
250
|
+
if version
|
|
251
|
+
else None
|
|
252
|
+
),
|
|
253
|
+
help=(
|
|
254
|
+
sanitize.scrub_text(help_txt, max_lines=80, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS)
|
|
255
|
+
if help_txt
|
|
256
|
+
else None
|
|
257
|
+
),
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
# ------------------------------------------------------ transcript discovery
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _redact_location(path: Path, aliases: dict[str, str] | None = None) -> str:
|
|
265
|
+
"""Redact a path to a paste-safe form: home -> ``~``, any known-sensitive
|
|
266
|
+
component (e.g. the project directory name, which is identifier-shaped and
|
|
267
|
+
would pass the gate below verbatim) -> its pseudonymizer alias, and any
|
|
268
|
+
other component that isn't a plain machine identifier (e.g. a munged-cwd
|
|
269
|
+
dir that embeds a username) -> ``<redacted>``. The session-id filename
|
|
270
|
+
usually survives."""
|
|
271
|
+
|
|
272
|
+
def comp(c: str) -> str:
|
|
273
|
+
if aliases and c in aliases:
|
|
274
|
+
return aliases[c]
|
|
275
|
+
# The username check closes a hole the identifier gate leaves open:
|
|
276
|
+
# a component like `pytest-of-alice` (or a $TMPDIR under /var/folders
|
|
277
|
+
# named after the user) is identifier-shaped yet names the user — and
|
|
278
|
+
# the egress guard's username hard rule would refuse the whole report.
|
|
279
|
+
if not sanitize.looks_like_identifier(c) or sanitize.embeds_current_username(c):
|
|
280
|
+
return "<redacted>"
|
|
281
|
+
return c
|
|
282
|
+
|
|
283
|
+
home = Path(os.path.expanduser("~"))
|
|
284
|
+
try:
|
|
285
|
+
rel = path.relative_to(home)
|
|
286
|
+
return "/".join(["~", *(comp(c) for c in rel.parts)])
|
|
287
|
+
except ValueError:
|
|
288
|
+
# ``parts[0]`` is the anchor on an absolute path. A bare root separator
|
|
289
|
+
# ("/" on POSIX, "\\" for a rooted path on Windows) is structure, not a
|
|
290
|
+
# component: drop it, or the identifier gate turns the separator itself
|
|
291
|
+
# into a phantom leading ``<redacted>``. A drive or UNC anchor ("C:\\",
|
|
292
|
+
# "\\\\server\\share\\") does carry content, so it stays and is judged.
|
|
293
|
+
parts = list(path.parts)
|
|
294
|
+
if parts and parts[0] in ("/", "\\"):
|
|
295
|
+
parts.pop(0)
|
|
296
|
+
return "/" + "/".join(comp(c) for c in parts if c)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _describe_transcript(
|
|
300
|
+
path: Path,
|
|
301
|
+
*,
|
|
302
|
+
glob_pat: str | None,
|
|
303
|
+
multiple: bool,
|
|
304
|
+
aliases: dict[str, str] | None = None,
|
|
305
|
+
) -> TranscriptFinding:
|
|
306
|
+
try:
|
|
307
|
+
stat = path.stat()
|
|
308
|
+
size = stat.st_size
|
|
309
|
+
mtime_date = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).strftime("%Y-%m-%d")
|
|
310
|
+
except OSError:
|
|
311
|
+
size, mtime_date = None, None
|
|
312
|
+
line_count = None
|
|
313
|
+
try:
|
|
314
|
+
with path.open(encoding="utf-8", errors="replace") as f:
|
|
315
|
+
line_count = sum(1 for _ in f)
|
|
316
|
+
except OSError:
|
|
317
|
+
pass
|
|
318
|
+
return TranscriptFinding(
|
|
319
|
+
glob=glob_pat,
|
|
320
|
+
location=_redact_location(path, aliases),
|
|
321
|
+
fmt="jsonl" if path.suffix == ".jsonl" else (path.suffix.lstrip(".") or "unknown"),
|
|
322
|
+
size_bytes=size,
|
|
323
|
+
line_count=line_count,
|
|
324
|
+
mtime_date=mtime_date,
|
|
325
|
+
multiple=multiple,
|
|
326
|
+
real_path=path,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _newest(paths: list[Path]) -> Path:
|
|
331
|
+
return max(paths, key=lambda p: p.stat().st_mtime if p.exists() else 0)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def discover_transcript(
|
|
335
|
+
parser: str,
|
|
336
|
+
*,
|
|
337
|
+
cli: str,
|
|
338
|
+
hints: Hints,
|
|
339
|
+
aliases: dict[str, str] | None = None,
|
|
340
|
+
) -> TranscriptFinding | None:
|
|
341
|
+
"""Locate the newest existing transcript via override or convention glob."""
|
|
342
|
+
if hints.transcript:
|
|
343
|
+
path = Path(hints.transcript).expanduser()
|
|
344
|
+
if not path.is_file():
|
|
345
|
+
return TranscriptFinding(note=f"--transcript path does not exist: {path.name}")
|
|
346
|
+
return _describe_transcript(path, glob_pat=None, multiple=False, aliases=aliases)
|
|
347
|
+
|
|
348
|
+
if hints.session_dir:
|
|
349
|
+
base = Path(hints.session_dir).expanduser()
|
|
350
|
+
matches = sorted(base.glob("**/*.jsonl")) or sorted(base.glob("**/*.json"))
|
|
351
|
+
if not matches:
|
|
352
|
+
return TranscriptFinding(note=f"no *.jsonl/*.json under --session-dir {base.name}")
|
|
353
|
+
return _describe_transcript(
|
|
354
|
+
_newest(matches), glob_pat=None, multiple=len(matches) > 1, aliases=aliases
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
pattern = TRANSCRIPT_GLOBS.get(parser) or FAMILY_GLOBS.get(cli)
|
|
358
|
+
if not pattern:
|
|
359
|
+
return TranscriptFinding(
|
|
360
|
+
note="no transcript-location convention for this CLI; "
|
|
361
|
+
"pass --transcript PATH or --session-dir DIR"
|
|
362
|
+
)
|
|
363
|
+
matches = [Path(p) for p in glob.glob(os.path.expanduser(pattern))]
|
|
364
|
+
matches = [p for p in matches if p.is_file()]
|
|
365
|
+
if not matches:
|
|
366
|
+
return TranscriptFinding(
|
|
367
|
+
glob=pattern,
|
|
368
|
+
note="no existing transcript matched the convention glob; "
|
|
369
|
+
"use --transcript / --session-dir, or run --probe",
|
|
370
|
+
)
|
|
371
|
+
return _describe_transcript(
|
|
372
|
+
_newest(matches), glob_pat=pattern, multiple=len(matches) > 1, aliases=aliases
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
# ---------------------------------------------------------- schema inference
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _type_name(value) -> str:
|
|
380
|
+
if value is None:
|
|
381
|
+
return "null"
|
|
382
|
+
if isinstance(value, bool):
|
|
383
|
+
return "bool"
|
|
384
|
+
if isinstance(value, int):
|
|
385
|
+
return "int"
|
|
386
|
+
if isinstance(value, float):
|
|
387
|
+
return "float"
|
|
388
|
+
if isinstance(value, str):
|
|
389
|
+
return "str"
|
|
390
|
+
return "other"
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _walk_paths(obj, prefix: str, out: set[str]) -> None:
|
|
394
|
+
"""Collect dotted key paths with the LEAF TYPE only (never values); list
|
|
395
|
+
indices collapse to ``[]`` so ``messages[].tokens.input:int`` is one path.
|
|
396
|
+
|
|
397
|
+
A dict key that isn't a plain identifier (e.g. a transcript that keys by
|
|
398
|
+
relative file path or a per-file backup id) is collapsed to ``<key>`` —
|
|
399
|
+
static field names (the ones a parser keys on, like ``input_tokens``) survive
|
|
400
|
+
untouched, but dynamic keys can't leak paths/content into the summary. A
|
|
401
|
+
credential-shaped key (``ghp_…`` as a map key) is identifier-shaped yet must
|
|
402
|
+
not ship, so it collapses too — the same hole :func:`sanitize.scrub_json`
|
|
403
|
+
closes for values."""
|
|
404
|
+
if isinstance(obj, dict):
|
|
405
|
+
for key, value in obj.items():
|
|
406
|
+
key = str(key)
|
|
407
|
+
if not sanitize.looks_like_identifier(key) or sanitize.looks_like_secret(key):
|
|
408
|
+
key = "<key>"
|
|
409
|
+
child = f"{prefix}.{key}" if prefix else key
|
|
410
|
+
_walk_paths(value, child, out)
|
|
411
|
+
elif isinstance(obj, list):
|
|
412
|
+
child = f"{prefix}[]"
|
|
413
|
+
for value in obj:
|
|
414
|
+
_walk_paths(value, child, out)
|
|
415
|
+
else:
|
|
416
|
+
out.add(f"{prefix}:{_type_name(obj)}")
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _is_token_candidate(path: str) -> bool:
|
|
420
|
+
name, _, typ = path.rpartition(":")
|
|
421
|
+
if typ != "int":
|
|
422
|
+
return False
|
|
423
|
+
last = name.split(".")[-1].replace("[]", "")
|
|
424
|
+
return bool(_TOKEN_KEY_RE.search(last))
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def infer_token_schema(
|
|
428
|
+
parser: str, path: Path, *, max_entries: int = MAX_SCHEMA_ENTRIES
|
|
429
|
+
) -> TokenSchema:
|
|
430
|
+
"""Structural key-path summary (types only) + token-field candidates.
|
|
431
|
+
|
|
432
|
+
Works even when ``parser == "none"``: the candidates are exactly what a
|
|
433
|
+
maintainer needs to write a parser for a brand-new CLI. When a real parser
|
|
434
|
+
exists, its parsed integer counts are included as a self-check.
|
|
435
|
+
"""
|
|
436
|
+
paths: set[str] = set()
|
|
437
|
+
scanned = 0
|
|
438
|
+
for entry in _jsonl_entries(path):
|
|
439
|
+
if scanned >= max_entries:
|
|
440
|
+
break
|
|
441
|
+
scanned += 1
|
|
442
|
+
_walk_paths(entry, "", paths)
|
|
443
|
+
candidates = sorted(p for p in paths if _is_token_candidate(p))
|
|
444
|
+
parsed = None
|
|
445
|
+
if parser != "none":
|
|
446
|
+
usage = read_usage(parser, path)
|
|
447
|
+
if usage is not None:
|
|
448
|
+
parsed = usage.to_dict()
|
|
449
|
+
return TokenSchema(
|
|
450
|
+
parser=parser,
|
|
451
|
+
entries_scanned=scanned,
|
|
452
|
+
parsed_usage=parsed,
|
|
453
|
+
key_paths=sorted(paths),
|
|
454
|
+
token_field_candidates=candidates,
|
|
455
|
+
)
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
# --------------------------------------------------------------- hook config
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _hooks_registered(project: Path, profile: CLIProfile) -> bool:
|
|
462
|
+
config_path = project / profile.hooks.config_path
|
|
463
|
+
if not config_path.is_file():
|
|
464
|
+
return False
|
|
465
|
+
import json
|
|
466
|
+
|
|
467
|
+
try:
|
|
468
|
+
config = json.loads(config_path.read_text(encoding="utf-8"))
|
|
469
|
+
except (json.JSONDecodeError, OSError):
|
|
470
|
+
return False
|
|
471
|
+
if not isinstance(config, dict):
|
|
472
|
+
return False
|
|
473
|
+
return relay_registered(config, profile.hooks.dialect, profile.hooks.events)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
# ----------------------------------------------------------------- SCAN mode
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def scan(
|
|
480
|
+
*,
|
|
481
|
+
cli: str,
|
|
482
|
+
profile: CLIProfile | None,
|
|
483
|
+
project: Path,
|
|
484
|
+
hints: Hints,
|
|
485
|
+
pseudo: sanitize.Pseudonymizer | None = None,
|
|
486
|
+
) -> ProfileFinding:
|
|
487
|
+
# Aliases registered up front (the project basename) are routed at
|
|
488
|
+
# collection time so a normal run needs zero egress repairs.
|
|
489
|
+
aliases = {orig: alias for _ns, orig, alias in pseudo.entries()} if pseudo else None
|
|
490
|
+
binary = hints.binary or (profile.binary if profile else cli)
|
|
491
|
+
parser = profile.usage_parser if profile else "none"
|
|
492
|
+
finding = ProfileFinding(
|
|
493
|
+
cli=cli,
|
|
494
|
+
mode="scan",
|
|
495
|
+
known_profile=profile is not None,
|
|
496
|
+
# The finding is render-only; a --binary hint may be an absolute path
|
|
497
|
+
# under $HOME, which must not reach the report (the raw local keeps
|
|
498
|
+
# driving which/--version below).
|
|
499
|
+
binary=sanitize.redact_home(binary),
|
|
500
|
+
parser=parser,
|
|
501
|
+
dialect=profile.hooks.dialect if profile else None,
|
|
502
|
+
declared_events=dict(profile.hooks.events) if profile else {},
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
finding.flags = run_version_help(binary)
|
|
506
|
+
if not finding.flags.found:
|
|
507
|
+
# finding.binary, not the raw local: a home-rooted --binary hint in a
|
|
508
|
+
# rendered warning would trip the egress guard's home-path rule.
|
|
509
|
+
finding.warnings.append(
|
|
510
|
+
f"binary {finding.binary!r} not found on PATH — version/help unavailable "
|
|
511
|
+
"(scan continues from on-disk conventions)"
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
if profile is not None:
|
|
515
|
+
finding.registered = _hooks_registered(project, profile)
|
|
516
|
+
if not finding.registered:
|
|
517
|
+
finding.next_steps.append(
|
|
518
|
+
f"hooks not registered in {profile.hooks.config_path}; "
|
|
519
|
+
f"`froid-loop init --cli {cli}` to validate the dialect end-to-end, "
|
|
520
|
+
"or re-run with --probe"
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
finding.transcript = discover_transcript(parser, cli=cli, hints=hints, aliases=aliases)
|
|
524
|
+
if finding.transcript and finding.transcript.note:
|
|
525
|
+
finding.warnings.append(finding.transcript.note)
|
|
526
|
+
if finding.transcript and finding.transcript.real_path is not None:
|
|
527
|
+
finding.tokens = infer_token_schema(parser, finding.transcript.real_path)
|
|
528
|
+
if finding.transcript.multiple:
|
|
529
|
+
finding.next_steps.append(
|
|
530
|
+
"multiple fresh transcripts matched; pass --transcript to pin the right one"
|
|
531
|
+
)
|
|
532
|
+
return finding
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
# ---------------------------------------------------------- PROBE tmux launcher
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
class _ProbeLauncher:
|
|
539
|
+
"""The few multiplexer primitives PROBE needs — deliberately NOT a
|
|
540
|
+
GenericAdapter, which mandates a Policy and story-completion logic irrelevant
|
|
541
|
+
here. Drives the shared backend so PROBE shells out to no multiplexer
|
|
542
|
+
directly."""
|
|
543
|
+
|
|
544
|
+
def __init__(self, session_name: str):
|
|
545
|
+
self.session_name = session_name
|
|
546
|
+
self.mux = get_multiplexer()
|
|
547
|
+
|
|
548
|
+
def start(self, argv: list[str], env: dict[str, str], cwd: Path, log_file: Path) -> str | None:
|
|
549
|
+
try:
|
|
550
|
+
self.mux.new_session(self.session_name, cwd, 220, 50)
|
|
551
|
+
command = " ".join(shlex.quote(a) for a in argv)
|
|
552
|
+
# `env` carries the profile's own `[env]` table verbatim, and a
|
|
553
|
+
# profile declaring FROID_LOOP_STATE_DIR would aim a froid-loop
|
|
554
|
+
# wrapper in that window at a different state root — and so a
|
|
555
|
+
# different registry, where this very session reads as gone. The
|
|
556
|
+
# pin chokepoint forces the entry to this process's own answer in
|
|
557
|
+
# both arms, the underivable one included (runs.pin_state_root);
|
|
558
|
+
# the engine's window merge applies the same rule.
|
|
559
|
+
window_env = runs.pin_state_root(env)
|
|
560
|
+
window_id = self.mux.new_window(
|
|
561
|
+
self.session_name, PROBE_TASK_ID, cwd, window_env, command
|
|
562
|
+
)
|
|
563
|
+
except MultiplexerError:
|
|
564
|
+
return None
|
|
565
|
+
# pipe-pane may race a window that dies instantly; tolerate failure.
|
|
566
|
+
self.mux.pipe_pane(window_id, log_file)
|
|
567
|
+
return window_id
|
|
568
|
+
|
|
569
|
+
def window_alive(self, window_id: str) -> bool:
|
|
570
|
+
return self.mux.window_alive(self.session_name, window_id)
|
|
571
|
+
|
|
572
|
+
def kill(self) -> None:
|
|
573
|
+
self.mux.kill_session(self.session_name)
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _probe_argv(profile: CLIProfile, binary: str, hints: Hints) -> list[str]:
|
|
577
|
+
argv = [
|
|
578
|
+
binary,
|
|
579
|
+
*profile.launch_args,
|
|
580
|
+
# Send the probe prompt verbatim, NOT through profile.render_prompt: a
|
|
581
|
+
# content-free turn has no skill name, so a skill-templating prompt_template
|
|
582
|
+
# (copilot, codex) would render a nonexistent .../skills//SKILL.md path the
|
|
583
|
+
# agent hunts for, and the turn never ends within the probe timeout.
|
|
584
|
+
PROBE_PROMPT,
|
|
585
|
+
*profile.bypass_args,
|
|
586
|
+
]
|
|
587
|
+
if hints.model:
|
|
588
|
+
argv += [profile.model_flag, hints.model]
|
|
589
|
+
return argv
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
def _captured_transcript_path(capture_dir: Path) -> Path | None:
|
|
593
|
+
"""The transcript path the CLI handed a hook on stdin, newest signal first.
|
|
594
|
+
|
|
595
|
+
Ground truth beats convention: a CLI that reports its own transcript (agy's
|
|
596
|
+
`transcriptPath`, Claude's `transcript_path`) names the exact file the turn
|
|
597
|
+
was recorded to, including the session id a convention glob can only wildcard.
|
|
598
|
+
"""
|
|
599
|
+
import json
|
|
600
|
+
|
|
601
|
+
for signal_file in sorted(capture_dir.glob("*.signal.json"), reverse=True):
|
|
602
|
+
try:
|
|
603
|
+
raw = json.loads(signal_file.read_text(encoding="utf-8"))
|
|
604
|
+
except (json.JSONDecodeError, OSError):
|
|
605
|
+
continue
|
|
606
|
+
path = raw.get("transcript_path") if isinstance(raw, dict) else None
|
|
607
|
+
if isinstance(path, str) and path:
|
|
608
|
+
return Path(path).expanduser()
|
|
609
|
+
return None
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
def _collect_captures(capture_dir: Path, events_map: dict[str, str]) -> list[EventCapture]:
|
|
613
|
+
"""Reduce each raw captured payload to its SHAPE: top-level keys and dotted
|
|
614
|
+
key-path:type paths. The payload's diagnostic value for profile authoring is
|
|
615
|
+
where the fields live and what type they are — so no payload *value* of any
|
|
616
|
+
kind ships, which removes the widest identifier-shaped egress surface
|
|
617
|
+
outright (#199). The walk sees the raw dict so leaf types are faithful;
|
|
618
|
+
dynamic (non-identifier) keys collapse to ``<key>`` in both projections."""
|
|
619
|
+
captures: list[EventCapture] = []
|
|
620
|
+
for payload_file in sorted(capture_dir.glob("*.payload.json")):
|
|
621
|
+
import json
|
|
622
|
+
|
|
623
|
+
try:
|
|
624
|
+
raw = json.loads(payload_file.read_text(encoding="utf-8"))
|
|
625
|
+
except (json.JSONDecodeError, OSError):
|
|
626
|
+
continue
|
|
627
|
+
if not isinstance(raw, dict):
|
|
628
|
+
continue
|
|
629
|
+
native = str(raw.pop("argv_event", "Unknown"))
|
|
630
|
+
paths: set[str] = set()
|
|
631
|
+
_walk_paths(raw, "", paths)
|
|
632
|
+
captures.append(
|
|
633
|
+
EventCapture(
|
|
634
|
+
native_event=native,
|
|
635
|
+
canonical_event=events_map.get(native),
|
|
636
|
+
payload_keys=sorted(
|
|
637
|
+
str(k) if sanitize.looks_like_identifier(str(k)) else "<key>" for k in raw
|
|
638
|
+
),
|
|
639
|
+
payload_schema=sorted(paths),
|
|
640
|
+
)
|
|
641
|
+
)
|
|
642
|
+
return captures
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
def probe(
|
|
646
|
+
*,
|
|
647
|
+
cli: str,
|
|
648
|
+
profile: CLIProfile,
|
|
649
|
+
project: Path,
|
|
650
|
+
hints: Hints,
|
|
651
|
+
timeout_s: float = 90,
|
|
652
|
+
keep_temp: bool = False,
|
|
653
|
+
pseudo: sanitize.Pseudonymizer | None = None,
|
|
654
|
+
) -> ProfileFinding:
|
|
655
|
+
import json
|
|
656
|
+
|
|
657
|
+
aliases = {orig: alias for _ns, orig, alias in pseudo.entries()} if pseudo else None
|
|
658
|
+
binary = hints.binary or profile.binary
|
|
659
|
+
finding = ProfileFinding(
|
|
660
|
+
cli=cli,
|
|
661
|
+
mode="probe",
|
|
662
|
+
known_profile=True,
|
|
663
|
+
# render-only; see the identical redaction in scan()
|
|
664
|
+
binary=sanitize.redact_home(binary),
|
|
665
|
+
parser=profile.usage_parser,
|
|
666
|
+
dialect=profile.hooks.dialect,
|
|
667
|
+
declared_events=dict(profile.hooks.events),
|
|
668
|
+
)
|
|
669
|
+
finding.flags = run_version_help(binary)
|
|
670
|
+
|
|
671
|
+
# The live probe launches through the selected multiplexer backend (see
|
|
672
|
+
# _ProbeLauncher), so gate on THAT backend's availability rather than a
|
|
673
|
+
# hardcoded `which("tmux")` — a Windows host running herdr (or a future psmux)
|
|
674
|
+
# must still probe. `available()` is guarded so a backend whose host probe
|
|
675
|
+
# raises reads as unavailable, exactly like selection's _usable().
|
|
676
|
+
mux = get_multiplexer()
|
|
677
|
+
try:
|
|
678
|
+
mux_ready = bool(mux.available())
|
|
679
|
+
except Exception: # a raising host probe means "cannot probe", not a crash
|
|
680
|
+
mux_ready = False
|
|
681
|
+
if not mux_ready or not shutil.which(binary):
|
|
682
|
+
# finding.binary, not the raw local — see the identical note in scan()
|
|
683
|
+
missing = f"multiplexer backend {type(mux).__name__}" if not mux_ready else finding.binary
|
|
684
|
+
finding.warnings.append(f"{missing} not on PATH — cannot probe; falling back to scan")
|
|
685
|
+
scanned = scan(cli=cli, profile=profile, project=project, hints=hints, pseudo=pseudo)
|
|
686
|
+
scanned.mode = "probe"
|
|
687
|
+
return scanned
|
|
688
|
+
|
|
689
|
+
tmpdir = Path(tempfile.mkdtemp(prefix="froid-loop-probe-"))
|
|
690
|
+
launcher = _ProbeLauncher(session_name=f"froid-loop-probe-{tmpdir.name}")
|
|
691
|
+
try:
|
|
692
|
+
capture_dir = tmpdir / "capture"
|
|
693
|
+
capture_dir.mkdir(parents=True, exist_ok=True)
|
|
694
|
+
|
|
695
|
+
# 1. lay down the capture hook + a hook config registered through the very
|
|
696
|
+
# same merge_hooks `froid-loop init` uses — so a bad dialect surfaces live.
|
|
697
|
+
hook_src = resources.files("froid_loop.data").joinpath(PROBE_HOOK_NAME)
|
|
698
|
+
hook_path = tmpdir / PROBE_HOOK_NAME
|
|
699
|
+
hook_path.write_text(hook_src.read_text(encoding="utf-8"), encoding="utf-8")
|
|
700
|
+
host = get_process_host()
|
|
701
|
+
interp = host.hook_interpreter()
|
|
702
|
+
registrations = {
|
|
703
|
+
native: f"{interp} {host.shell_quote(str(hook_path))} {canonical}"
|
|
704
|
+
for native, canonical in profile.hooks.events.items()
|
|
705
|
+
}
|
|
706
|
+
config, _ = merge_hooks({}, registrations, profile.hooks.dialect)
|
|
707
|
+
config_path = tmpdir / profile.hooks.config_path
|
|
708
|
+
config_path.parent.mkdir(parents=True, exist_ok=True)
|
|
709
|
+
config_path.write_text(json.dumps(config, indent=2) + "\n", encoding="utf-8")
|
|
710
|
+
|
|
711
|
+
# 2. launch one trivial content-free turn in a fresh tmux window
|
|
712
|
+
argv = _probe_argv(profile, binary, hints)
|
|
713
|
+
env = {
|
|
714
|
+
**profile.env,
|
|
715
|
+
"FROID_LOOP_RUN_DIR": str(tmpdir),
|
|
716
|
+
"FROID_LOOP_TASK_ID": PROBE_TASK_ID,
|
|
717
|
+
"FROID_LOOP_PROBE_CAPTURE_DIR": str(capture_dir),
|
|
718
|
+
}
|
|
719
|
+
log_file = tmpdir / "probe.log"
|
|
720
|
+
watcher = SignalWatcher(capture_dir)
|
|
721
|
+
launched_ns = time.time_ns()
|
|
722
|
+
window_id = launcher.start(argv, env, tmpdir, log_file)
|
|
723
|
+
if window_id is None:
|
|
724
|
+
finding.warnings.append("could not launch the CLI in tmux; no events captured")
|
|
725
|
+
return finding
|
|
726
|
+
|
|
727
|
+
# 3. completion: first of — canonical Stop for `probe`; any capture file
|
|
728
|
+
# appeared and the window died; window died; deadline.
|
|
729
|
+
deadline = time.monotonic() + timeout_s
|
|
730
|
+
while True:
|
|
731
|
+
remaining = deadline - time.monotonic()
|
|
732
|
+
if remaining <= 0:
|
|
733
|
+
finding.warnings.append(
|
|
734
|
+
"no Stop event before --timeout; the CLI may need first-run auth "
|
|
735
|
+
"(a pending login dialog reads as a timeout). See the log tail below."
|
|
736
|
+
)
|
|
737
|
+
break
|
|
738
|
+
event = watcher.wait_for(
|
|
739
|
+
PROBE_TASK_ID,
|
|
740
|
+
{"Stop"},
|
|
741
|
+
timeout_s=min(remaining, 5.0),
|
|
742
|
+
since_ns=launched_ns,
|
|
743
|
+
)
|
|
744
|
+
if event is not None:
|
|
745
|
+
break
|
|
746
|
+
try:
|
|
747
|
+
alive = launcher.window_alive(window_id)
|
|
748
|
+
except MultiplexerError:
|
|
749
|
+
# transient transport hang is not proof the window died; retry
|
|
750
|
+
# on the next tick rather than mis-reporting a dead CLI window.
|
|
751
|
+
continue
|
|
752
|
+
captured_any = any(capture_dir.glob("*.payload.json"))
|
|
753
|
+
if not alive:
|
|
754
|
+
if not captured_any:
|
|
755
|
+
finding.warnings.append(
|
|
756
|
+
"the CLI window died before any hook fired — the dialect may be "
|
|
757
|
+
f"rejected for {profile.hooks.dialect}, or launch/auth failed. "
|
|
758
|
+
"See the log tail below."
|
|
759
|
+
)
|
|
760
|
+
break
|
|
761
|
+
|
|
762
|
+
# 4. one short grace poll so a Stop's sibling files all land, then collect.
|
|
763
|
+
time.sleep(PROBE_GRACE_S)
|
|
764
|
+
finding.captured_events = _collect_captures(capture_dir, profile.hooks.events)
|
|
765
|
+
if not finding.captured_events:
|
|
766
|
+
finding.next_steps.append(
|
|
767
|
+
"no hook payloads captured — confirm the CLI is authenticated and that "
|
|
768
|
+
f"the {profile.hooks.dialect} hook config is accepted, then re-run --probe"
|
|
769
|
+
)
|
|
770
|
+
tail = _log_tail(log_file)
|
|
771
|
+
if tail:
|
|
772
|
+
finding.warnings.append("log tail (scrubbed):\n" + tail)
|
|
773
|
+
|
|
774
|
+
# 5. transcript: prefer the exact path the CLI handed the hook on stdin
|
|
775
|
+
# over the convention glob — the payload names this turn's file, while
|
|
776
|
+
# the glob can only pick the newest match and may land on an unrelated
|
|
777
|
+
# session. Falls back to the glob when the CLI reports no path.
|
|
778
|
+
live = _captured_transcript_path(capture_dir)
|
|
779
|
+
if live is not None and live.is_file():
|
|
780
|
+
finding.transcript = _describe_transcript(
|
|
781
|
+
live, glob_pat=None, multiple=False, aliases=aliases
|
|
782
|
+
)
|
|
783
|
+
else:
|
|
784
|
+
finding.transcript = discover_transcript(
|
|
785
|
+
profile.usage_parser, cli=cli, hints=hints, aliases=aliases
|
|
786
|
+
)
|
|
787
|
+
if finding.transcript and finding.transcript.note:
|
|
788
|
+
finding.warnings.append(finding.transcript.note)
|
|
789
|
+
if finding.transcript and finding.transcript.real_path is not None:
|
|
790
|
+
finding.tokens = infer_token_schema(profile.usage_parser, finding.transcript.real_path)
|
|
791
|
+
return finding
|
|
792
|
+
finally:
|
|
793
|
+
launcher.kill()
|
|
794
|
+
if keep_temp:
|
|
795
|
+
# ~-relative so a $TMPDIR under $HOME can't trip the egress guard;
|
|
796
|
+
# the shell expands ~ so the printed path stays inspectable.
|
|
797
|
+
finding.warnings.append(
|
|
798
|
+
f"--keep-temp: RAW probe data retained at {sanitize.redact_home(str(tmpdir))} "
|
|
799
|
+
"— DO NOT SHARE; delete it after inspection"
|
|
800
|
+
)
|
|
801
|
+
else:
|
|
802
|
+
shutil.rmtree(tmpdir, ignore_errors=True)
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
def _log_tail(log_file: Path, max_lines: int = 20) -> str | None:
|
|
806
|
+
try:
|
|
807
|
+
text = log_file.read_text(encoding="utf-8", errors="replace")
|
|
808
|
+
except OSError:
|
|
809
|
+
return None
|
|
810
|
+
if not text.strip():
|
|
811
|
+
return None
|
|
812
|
+
lines = text.splitlines()[-max_lines:]
|
|
813
|
+
return sanitize.scrub_text(
|
|
814
|
+
"\n".join(lines), max_lines=max_lines, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS
|
|
815
|
+
)
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
# ------------------------------------------------------------------ rendering
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def _fmt_kv(label: str, value) -> str:
|
|
822
|
+
return f"- **{label}:** {value}"
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def render_markdown(
|
|
826
|
+
f: ProfileFinding,
|
|
827
|
+
*,
|
|
828
|
+
pseudo: sanitize.Pseudonymizer | None = None,
|
|
829
|
+
repairs: list[tuple[str, int]] | None = None,
|
|
830
|
+
) -> str:
|
|
831
|
+
out: list[str] = []
|
|
832
|
+
out.append(f"# Profile finalize report — {f.cli} ({f.mode})")
|
|
833
|
+
out.append("")
|
|
834
|
+
|
|
835
|
+
# Summary
|
|
836
|
+
out.append("## Summary")
|
|
837
|
+
out.append(_fmt_kv("CLI", f.cli))
|
|
838
|
+
out.append(
|
|
839
|
+
_fmt_kv("binary", f"{f.binary} ({'found' if f.flags and f.flags.found else 'NOT found'})")
|
|
840
|
+
)
|
|
841
|
+
out.append(_fmt_kv("known profile", "yes" if f.known_profile else "no (reduced report)"))
|
|
842
|
+
out.append(_fmt_kv("hook dialect", f.dialect or "—"))
|
|
843
|
+
out.append(_fmt_kv("usage_parser", f.parser))
|
|
844
|
+
if f.registered is not None:
|
|
845
|
+
out.append(_fmt_kv("hooks registered", "yes" if f.registered else "no"))
|
|
846
|
+
out.append(_fmt_kv("warnings", str(len(f.warnings))))
|
|
847
|
+
out.append("")
|
|
848
|
+
|
|
849
|
+
# CLI flags
|
|
850
|
+
out.append("## CLI flags")
|
|
851
|
+
out.append(_fmt_kv("launch_args / bypass_args", "see profile (rendered verbatim below)"))
|
|
852
|
+
if f.flags and f.flags.version:
|
|
853
|
+
out.append("\n```\n" + f.flags.version + "\n```")
|
|
854
|
+
if f.flags and f.flags.help:
|
|
855
|
+
out.append("\n<details><summary>--help (scrubbed)</summary>\n")
|
|
856
|
+
out.append("```\n" + f.flags.help + "\n```")
|
|
857
|
+
out.append("</details>")
|
|
858
|
+
if not f.flags or not f.flags.found:
|
|
859
|
+
out.append("_binary not available; flags/help not captured._")
|
|
860
|
+
out.append("")
|
|
861
|
+
|
|
862
|
+
# Hook payload shape
|
|
863
|
+
out.append("## Hook payload shape")
|
|
864
|
+
if f.mode == "scan":
|
|
865
|
+
if f.declared_events:
|
|
866
|
+
out.append(
|
|
867
|
+
"Declared native → canonical events (registered = "
|
|
868
|
+
f"{'yes' if f.registered else 'no'}):"
|
|
869
|
+
)
|
|
870
|
+
for native, canonical in f.declared_events.items():
|
|
871
|
+
out.append(f"- `{native}` → `{canonical}`")
|
|
872
|
+
else:
|
|
873
|
+
out.append("_no profile; events unknown. Re-run with --probe to capture payloads._")
|
|
874
|
+
else:
|
|
875
|
+
if f.captured_events:
|
|
876
|
+
for ev in f.captured_events:
|
|
877
|
+
out.append(f"### `{ev.native_event}` → `{ev.canonical_event or '?'}`")
|
|
878
|
+
out.append(
|
|
879
|
+
_fmt_kv("payload keys", ", ".join(f"`{k}`" for k in ev.payload_keys) or "—")
|
|
880
|
+
)
|
|
881
|
+
out.append("\n**Payload schema** (key paths + leaf types, never values):")
|
|
882
|
+
if ev.payload_schema:
|
|
883
|
+
out.append("\n```\n" + "\n".join(ev.payload_schema) + "\n```")
|
|
884
|
+
else:
|
|
885
|
+
out.append("\n- _empty payload._")
|
|
886
|
+
else:
|
|
887
|
+
out.append("_no hook payloads captured (see warnings)._")
|
|
888
|
+
out.append("")
|
|
889
|
+
|
|
890
|
+
# Transcript
|
|
891
|
+
out.append("## Transcript")
|
|
892
|
+
t = f.transcript
|
|
893
|
+
if t and t.real_path is not None:
|
|
894
|
+
out.append(_fmt_kv("location", f"`{t.location}`"))
|
|
895
|
+
if t.glob:
|
|
896
|
+
out.append(_fmt_kv("matched glob", f"`{t.glob}`"))
|
|
897
|
+
out.append(_fmt_kv("format", t.fmt))
|
|
898
|
+
out.append(_fmt_kv("size", f"{t.size_bytes} bytes"))
|
|
899
|
+
out.append(_fmt_kv("lines", t.line_count))
|
|
900
|
+
out.append(_fmt_kv("mtime", t.mtime_date))
|
|
901
|
+
if t.multiple:
|
|
902
|
+
out.append("- _multiple candidates matched; newest shown — pass --transcript to pin._")
|
|
903
|
+
else:
|
|
904
|
+
out.append("_no transcript located._" + (f" ({t.note})" if t and t.note else ""))
|
|
905
|
+
out.append("")
|
|
906
|
+
|
|
907
|
+
# Token usage schema
|
|
908
|
+
out.append("## Token usage schema")
|
|
909
|
+
tk = f.tokens
|
|
910
|
+
if tk:
|
|
911
|
+
out.append(_fmt_kv("declared parser", tk.parser))
|
|
912
|
+
out.append(_fmt_kv("entries scanned", tk.entries_scanned))
|
|
913
|
+
if tk.parsed_usage is not None:
|
|
914
|
+
out.append(_fmt_kv("parsed counts (self-check)", f"`{tk.parsed_usage}`"))
|
|
915
|
+
out.append(
|
|
916
|
+
"\n**Token-field candidates** (int leaves; per-call-vs-cumulative is a human call):"
|
|
917
|
+
)
|
|
918
|
+
if tk.token_field_candidates:
|
|
919
|
+
for cand in tk.token_field_candidates:
|
|
920
|
+
out.append(f"- `{cand}`")
|
|
921
|
+
else:
|
|
922
|
+
out.append("- _none matched the token-name heuristic._")
|
|
923
|
+
out.append("\n<details><summary>All key paths (types only, no values)</summary>\n")
|
|
924
|
+
out.append("```\n" + "\n".join(tk.key_paths) + "\n```")
|
|
925
|
+
out.append("</details>")
|
|
926
|
+
else:
|
|
927
|
+
out.append("_no transcript to infer from._")
|
|
928
|
+
out.append("")
|
|
929
|
+
|
|
930
|
+
# Warnings / next steps
|
|
931
|
+
out.append("## Warnings / next steps")
|
|
932
|
+
if not f.warnings and not f.next_steps:
|
|
933
|
+
out.append("_none._")
|
|
934
|
+
for w in f.warnings:
|
|
935
|
+
out.append(f"- ⚠️ {w}")
|
|
936
|
+
for s in f.next_steps:
|
|
937
|
+
out.append(f"- → {s}")
|
|
938
|
+
out.append("")
|
|
939
|
+
|
|
940
|
+
rendered = "\n".join(out)
|
|
941
|
+
rendered, reps = sanitize.guard(rendered, pseudo)
|
|
942
|
+
if reps:
|
|
943
|
+
note = [
|
|
944
|
+
"",
|
|
945
|
+
"### Backstop repairs",
|
|
946
|
+
"",
|
|
947
|
+
"_The leak self-check caught stray occurrences of pseudonymized "
|
|
948
|
+
"identifiers that the per-field routing missed, and substituted "
|
|
949
|
+
"their aliases — a froid-loop routing gap; please report it._",
|
|
950
|
+
"",
|
|
951
|
+
]
|
|
952
|
+
for label, count in reps:
|
|
953
|
+
note.append(f"- `{label}`: {count} stray occurrence(s) pseudonymized")
|
|
954
|
+
note.append("")
|
|
955
|
+
rendered += "\n".join(note)
|
|
956
|
+
# The note is appended after the repair loop verified the body, so
|
|
957
|
+
# re-check the whole thing: the note must sit inside the verified bytes.
|
|
958
|
+
sanitize.assert_clean(rendered, pseudo)
|
|
959
|
+
if repairs is not None:
|
|
960
|
+
repairs.extend(reps)
|
|
961
|
+
return rendered
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
def render_json(
|
|
965
|
+
f: ProfileFinding,
|
|
966
|
+
*,
|
|
967
|
+
pseudo: sanitize.Pseudonymizer | None = None,
|
|
968
|
+
repairs: list[tuple[str, int]] | None = None,
|
|
969
|
+
) -> str:
|
|
970
|
+
import json
|
|
971
|
+
|
|
972
|
+
def transcript_dict(t: TranscriptFinding | None):
|
|
973
|
+
if t is None:
|
|
974
|
+
return None
|
|
975
|
+
return {
|
|
976
|
+
"glob": t.glob,
|
|
977
|
+
"location": t.location,
|
|
978
|
+
"format": t.fmt,
|
|
979
|
+
"size_bytes": t.size_bytes,
|
|
980
|
+
"line_count": t.line_count,
|
|
981
|
+
"mtime_date": t.mtime_date,
|
|
982
|
+
"multiple": t.multiple,
|
|
983
|
+
"note": t.note,
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
data = {
|
|
987
|
+
"schema_version": SCHEMA_VERSION,
|
|
988
|
+
"cli": f.cli,
|
|
989
|
+
"mode": f.mode,
|
|
990
|
+
"known_profile": f.known_profile,
|
|
991
|
+
"binary": f.binary,
|
|
992
|
+
"binary_found": bool(f.flags and f.flags.found),
|
|
993
|
+
"dialect": f.dialect,
|
|
994
|
+
"usage_parser": f.parser,
|
|
995
|
+
"hooks_registered": f.registered,
|
|
996
|
+
"declared_events": f.declared_events,
|
|
997
|
+
"version": f.flags.version if f.flags else None,
|
|
998
|
+
"help": f.flags.help if f.flags else None,
|
|
999
|
+
"captured_events": [
|
|
1000
|
+
{
|
|
1001
|
+
"native_event": ev.native_event,
|
|
1002
|
+
"canonical_event": ev.canonical_event,
|
|
1003
|
+
"payload_keys": ev.payload_keys,
|
|
1004
|
+
"payload_schema": ev.payload_schema,
|
|
1005
|
+
}
|
|
1006
|
+
for ev in f.captured_events
|
|
1007
|
+
],
|
|
1008
|
+
"transcript": transcript_dict(f.transcript),
|
|
1009
|
+
"tokens": (
|
|
1010
|
+
{
|
|
1011
|
+
"parser": f.tokens.parser,
|
|
1012
|
+
"entries_scanned": f.tokens.entries_scanned,
|
|
1013
|
+
"parsed_usage": f.tokens.parsed_usage,
|
|
1014
|
+
"key_paths": f.tokens.key_paths,
|
|
1015
|
+
"token_field_candidates": f.tokens.token_field_candidates,
|
|
1016
|
+
}
|
|
1017
|
+
if f.tokens
|
|
1018
|
+
else None
|
|
1019
|
+
),
|
|
1020
|
+
"warnings": f.warnings,
|
|
1021
|
+
"next_steps": f.next_steps,
|
|
1022
|
+
}
|
|
1023
|
+
# sort_keys so two probes of the same CLI diff cleanly — the document has
|
|
1024
|
+
# consumers now, and dict-literal order is an implementation detail.
|
|
1025
|
+
# ensure_ascii=False is a SAFETY requirement, not cosmetics (see
|
|
1026
|
+
# diagnostics.render_json): with the default, a non-ASCII sensitive value
|
|
1027
|
+
# reaches the guard as \uXXXX escapes and matches nothing, yet json.loads
|
|
1028
|
+
# hands the consumer back the original. machine.emit/write_document emit
|
|
1029
|
+
# this string verbatim, so the guarded bytes ARE the emitted bytes.
|
|
1030
|
+
rendered = json.dumps(data, indent=2, sort_keys=True, ensure_ascii=False)
|
|
1031
|
+
rendered, reps = sanitize.guard(rendered, pseudo)
|
|
1032
|
+
if reps:
|
|
1033
|
+
# Disclose the repair in the document itself so the routing gap surfaces
|
|
1034
|
+
# as a reportable bug. Substitution preserved JSON validity — a leaked
|
|
1035
|
+
# original is identifier-shaped and its alias is [A-Za-z0-9-], neither
|
|
1036
|
+
# side carries quotes or backslashes — so reload-and-extend is safe.
|
|
1037
|
+
# backstop_repairs is an optional additive key: absent on a clean report.
|
|
1038
|
+
loaded = json.loads(rendered)
|
|
1039
|
+
loaded["backstop_repairs"] = dict(reps)
|
|
1040
|
+
rendered = json.dumps(loaded, indent=2, sort_keys=True, ensure_ascii=False)
|
|
1041
|
+
sanitize.assert_clean(rendered, pseudo)
|
|
1042
|
+
if repairs is not None:
|
|
1043
|
+
repairs.extend(reps)
|
|
1044
|
+
return rendered
|