froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/probe.py ADDED
@@ -0,0 +1,1044 @@
1
+ """`froid-loop probe-adapter`: collect + sanitize adapter-finalization data.
2
+
3
+ Finalizing a generic-adapter CLI profile needs facts that live in no doc: the
4
+ CLI's exact hook payload shape (field names/casing, whether transcript_path /
5
+ session_id / cwd are present), where its transcript lives and in what format,
6
+ and the token-usage schema a `usage_parser` must read. This command pulls all of
7
+ that and runs it through the audited :mod:`froid_loop.sanitize` chokepoint, so a
8
+ user of any coding CLI can run one command and paste back a clean, content-free
9
+ report.
10
+
11
+ Two strategies, one report shape:
12
+
13
+ - SCAN (default, zero process launch beyond ``--version``/``--help``): locate the
14
+ newest already-existing transcript by convention, read the declared hook config,
15
+ infer the token schema. Works whenever the user has used the CLI before.
16
+ - PROBE (``--probe``, opt-in): in an ephemeral ``mkdtemp`` workspace, register the
17
+ full-payload capture hook for every native event, launch one trivial content-free
18
+ turn in a tmux window, capture each event's complete payload, then tear down. The
19
+ raw capture exists only transiently inside the temp dir, which is ``rmtree``'d in a
20
+ ``finally`` (even on exception / Ctrl-C).
21
+
22
+ One finding, two render targets: :func:`render_markdown` for the human report
23
+ (the CLI default) and :func:`render_json` for the machine-readable document that
24
+ ``--json`` emits instead (the :mod:`froid_loop.machine` contract — one object on
25
+ stdout, nothing else). The document carries :data:`SCHEMA_VERSION` as a top-level
26
+ ``schema_version``; do not confuse it with the document's ``version`` key, which
27
+ holds the *probed CLI's* own ``--version`` output.
28
+
29
+ Safety model — the same two layers as :mod:`froid_loop.diagnostics` (closed by
30
+ #199). Captured data is scrubbed/reduced at COLLECTION time (captured payloads
31
+ ship as key-path:type *schema*, never values; paths are per-component redacted
32
+ with the project basename routed through a :class:`sanitize.Pseudonymizer`
33
+ alias), and both renderers run :func:`sanitize.guard` over their own rendered
34
+ bytes before returning — a hard-rule hit (email/secret/home-path/url-creds/
35
+ username) refuses to emit, while a stray occurrence of a registered alias
36
+ original is repaired, re-verified, and disclosed. The residual is stated
37
+ honestly: identifier-shaped proprietary values the probe cannot know about
38
+ (an arbitrary slug in a dynamic key, say) match no hard rule and no registered
39
+ extra, so the guard cannot see them — which is why collection reduces payloads
40
+ to shape instead of trying to enumerate what might be sensitive.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import glob
46
+ import os
47
+ import re
48
+ import shlex
49
+ import shutil
50
+ import subprocess
51
+ import tempfile
52
+ import time
53
+ from dataclasses import dataclass, field
54
+ from datetime import datetime, timezone
55
+ from importlib import resources
56
+ from pathlib import Path
57
+
58
+ from . import runs, sanitize
59
+ from .adapters.multiplexer import MultiplexerError, get_multiplexer
60
+ from .adapters.profile import CLIProfile
61
+ from .install import merge_hooks, relay_registered
62
+ from .process_host import get_process_host
63
+
64
+ # cmd_probe catches `probe.LeakDetected` around the renderers, mirroring
65
+ # diagnostics — the noqa keeps ruff's F401 autofix from deleting the re-export.
66
+ from .sanitize import LeakDetected # noqa: F401 — re-export
67
+ from .signals import SignalWatcher
68
+ from .tokens import _jsonl_entries, read_usage
69
+
70
+ # Version of the `--json` document (machine.py contract). Distinct from the
71
+ # document's `version` key, which holds the *probed CLI's* `--version` output.
72
+ # v2: `captured_events[].payload` (scrubbed values) was removed in favor of
73
+ # `payload_schema` (key paths + leaf types, never values) — a field removal,
74
+ # which the additive-only contract says must bump the version (#199).
75
+ SCHEMA_VERSION = 2
76
+
77
+ # Per-parser transcript-location conventions (from tokens.py docstrings).
78
+ TRANSCRIPT_GLOBS = {
79
+ "claude-jsonl": "~/.claude/projects/*/*.jsonl",
80
+ "codex-rollout": "~/.codex/sessions/*/*/*/rollout-*.jsonl",
81
+ "gemini-chat": "~/.gemini/tmp/*/chats/session-*.jsonl",
82
+ "copilot-events": "~/.copilot/session-state/*/events.jsonl",
83
+ }
84
+ # Fallback family glob keyed by the `cli` name, so a CLI whose usage_parser is
85
+ # still "none" (e.g. antigravity, freshly added) still gets transcript discovery.
86
+ FAMILY_GLOBS = {
87
+ "claude": "~/.claude/projects/*/*.jsonl",
88
+ "codex": "~/.codex/sessions/*/*/*/rollout-*.jsonl",
89
+ "gemini": "~/.gemini/tmp/*/chats/session-*.jsonl",
90
+ "copilot": "~/.copilot/session-state/*/events.jsonl",
91
+ # agy (Antigravity CLI) writes one transcript per conversation, keyed by
92
+ # conversationId. Verified against agy 1.1.3 by capturing a live Stop hook,
93
+ # whose transcriptPath is exactly this shape. Note `transcript_full.jsonl`,
94
+ # not `transcript.jsonl` — agy's own hooks.md shows a WORKSPACE-relative
95
+ # `<ws>/.gemini/antigravity/transcript.jsonl` in its payload example, but
96
+ # that is illustrative: the real path is home-rooted, under brain/.
97
+ "antigravity": (
98
+ "~/.gemini/antigravity-cli/brain/*/.system_generated/logs/transcript_full.jsonl"
99
+ ),
100
+ }
101
+
102
+ _TOKEN_KEY_RE = re.compile(
103
+ r"(token|tokens|cached|input|output|prompt|completion|thoughts|usage)", re.I
104
+ )
105
+
106
+ PROBE_HOOK_NAME = "froid_loop_probe_hook.py"
107
+ PROBE_PROMPT = "Reply with exactly: OK"
108
+ PROBE_TASK_ID = "probe"
109
+ PROBE_GRACE_S = 3.0
110
+ MAX_SCHEMA_ENTRIES = 200
111
+
112
+
113
+ # --------------------------------------------------------------- dataclasses
114
+
115
+
116
+ @dataclass
117
+ class FlagFinding:
118
+ binary: str
119
+ found: bool
120
+ version: str | None = None # scrubbed
121
+ help: str | None = None # scrubbed
122
+
123
+
124
+ @dataclass
125
+ class TranscriptFinding:
126
+ glob: str | None = None # the convention glob used (already ~-relative)
127
+ location: str | None = None # redacted path of the chosen transcript
128
+ fmt: str | None = None # "jsonl" | "json"
129
+ size_bytes: int | None = None
130
+ line_count: int | None = None
131
+ mtime_date: str | None = None # date only (no time), UTC
132
+ multiple: bool = False
133
+ note: str | None = None
134
+ real_path: Path | None = None # NOT rendered; used for schema inference
135
+
136
+
137
+ @dataclass
138
+ class TokenSchema:
139
+ parser: str
140
+ entries_scanned: int = 0
141
+ parsed_usage: dict | None = None # only when parser != "none"
142
+ key_paths: list[str] = field(default_factory=list) # "a.b.c:int", TYPE only
143
+ token_field_candidates: list[str] = field(default_factory=list)
144
+
145
+
146
+ @dataclass
147
+ class EventCapture:
148
+ native_event: str
149
+ canonical_event: str | None
150
+ payload_keys: list[str] # top-level field names, identifier-gated
151
+ payload_schema: list[str] # dotted key paths + leaf TYPES only, never values
152
+
153
+
154
+ @dataclass
155
+ class ProfileFinding:
156
+ cli: str
157
+ mode: str # "scan" | "probe"
158
+ known_profile: bool
159
+ binary: str
160
+ parser: str
161
+ dialect: str | None = None
162
+ flags: FlagFinding | None = None
163
+ declared_events: dict = field(default_factory=dict) # native -> canonical
164
+ registered: bool | None = None # scan: hooks present in the CLI's config?
165
+ captured_events: list[EventCapture] = field(default_factory=list) # probe
166
+ transcript: TranscriptFinding | None = None
167
+ tokens: TokenSchema | None = None
168
+ warnings: list[str] = field(default_factory=list)
169
+ next_steps: list[str] = field(default_factory=list)
170
+
171
+
172
+ @dataclass
173
+ class Hints:
174
+ binary: str | None = None
175
+ transcript: str | None = None
176
+ session_dir: str | None = None
177
+ model: str | None = None
178
+
179
+
180
+ # ------------------------------------------------------------ version / help
181
+
182
+
183
+ def _run_capture(argv: list[str], timeout_s: float) -> str | None:
184
+ try:
185
+ # errors="replace" is what keeps run_version_help's documented "Never
186
+ # raises" true (#383): a banner byte the locale codec cannot decode is a
187
+ # UnicodeDecodeError — a ValueError, outside the guard below.
188
+ proc = subprocess.run(
189
+ argv, capture_output=True, text=True, errors="replace", timeout=timeout_s
190
+ )
191
+ except (OSError, subprocess.SubprocessError):
192
+ return None
193
+ out = (proc.stdout or "") + (proc.stderr or "")
194
+ return out.strip() or None
195
+
196
+
197
+ def binary_runs(binary: str, timeout_s: float = 10) -> int | None:
198
+ """Return the exit code of ``binary --version``, or None if it never ran.
199
+
200
+ The liveness half of a PATH check. ``shutil.which`` answers "a file with that
201
+ name is on PATH and has the execute bit", which a dead WSL/npm shim satisfies
202
+ while every launch of it fails (#294) — so ``validate`` reported OK on an
203
+ install that could not start a session. Running the binary once is the only
204
+ thing that separates the two.
205
+
206
+ Never raises, and that is load-bearing rather than defensive style: machine.py
207
+ records that every gate in ``cmd_validate`` runs inside a ``try`` so "the
208
+ command has no error path of its own — its rc is purely the verdict". A probe
209
+ that raised would give it one. The guard is ``_run_capture``'s exactly, and
210
+ the return is deliberately left as bytes (no ``text=True``): nothing here reads
211
+ the output, so the locale decode that forced ``errors="replace"`` on that
212
+ function never happens and cannot raise the ``UnicodeDecodeError`` the guard
213
+ does not name.
214
+
215
+ None (could not launch, or timed out) and a nonzero code are separate answers
216
+ to the caller, not one sentinel: the first has no return code to report.
217
+
218
+ ``stdin=DEVNULL`` is required, not cosmetic. With the caller's tty inherited, a
219
+ shim that prompts blocks on the read for the whole timeout — measured 4.00s
220
+ against 0.00s — inside an interactive command.
221
+
222
+ Not folded into :func:`run_version_help`, which discards the return code by
223
+ design and spawns TWO children (``--version`` then ``--help``) at ``timeout_s``
224
+ each: reusing it would cost up to 20s per profile here.
225
+ """
226
+ try:
227
+ proc = subprocess.run(
228
+ [binary, "--version"],
229
+ capture_output=True,
230
+ check=False,
231
+ stdin=subprocess.DEVNULL,
232
+ timeout=timeout_s,
233
+ )
234
+ except (OSError, subprocess.SubprocessError):
235
+ return None
236
+ return proc.returncode
237
+
238
+
239
+ def run_version_help(binary: str, timeout_s: float = 10) -> FlagFinding:
240
+ """Scrubbed ``--version``/``--help`` for a binary. Never raises."""
241
+ if not shutil.which(binary):
242
+ return FlagFinding(binary=binary, found=False)
243
+ version = _run_capture([binary, "--version"], timeout_s)
244
+ help_txt = _run_capture([binary, "--help"], timeout_s)
245
+ return FlagFinding(
246
+ binary=binary,
247
+ found=True,
248
+ version=(
249
+ sanitize.scrub_text(version, max_lines=5, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS)
250
+ if version
251
+ else None
252
+ ),
253
+ help=(
254
+ sanitize.scrub_text(help_txt, max_lines=80, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS)
255
+ if help_txt
256
+ else None
257
+ ),
258
+ )
259
+
260
+
261
+ # ------------------------------------------------------ transcript discovery
262
+
263
+
264
+ def _redact_location(path: Path, aliases: dict[str, str] | None = None) -> str:
265
+ """Redact a path to a paste-safe form: home -> ``~``, any known-sensitive
266
+ component (e.g. the project directory name, which is identifier-shaped and
267
+ would pass the gate below verbatim) -> its pseudonymizer alias, and any
268
+ other component that isn't a plain machine identifier (e.g. a munged-cwd
269
+ dir that embeds a username) -> ``<redacted>``. The session-id filename
270
+ usually survives."""
271
+
272
+ def comp(c: str) -> str:
273
+ if aliases and c in aliases:
274
+ return aliases[c]
275
+ # The username check closes a hole the identifier gate leaves open:
276
+ # a component like `pytest-of-alice` (or a $TMPDIR under /var/folders
277
+ # named after the user) is identifier-shaped yet names the user — and
278
+ # the egress guard's username hard rule would refuse the whole report.
279
+ if not sanitize.looks_like_identifier(c) or sanitize.embeds_current_username(c):
280
+ return "<redacted>"
281
+ return c
282
+
283
+ home = Path(os.path.expanduser("~"))
284
+ try:
285
+ rel = path.relative_to(home)
286
+ return "/".join(["~", *(comp(c) for c in rel.parts)])
287
+ except ValueError:
288
+ # ``parts[0]`` is the anchor on an absolute path. A bare root separator
289
+ # ("/" on POSIX, "\\" for a rooted path on Windows) is structure, not a
290
+ # component: drop it, or the identifier gate turns the separator itself
291
+ # into a phantom leading ``<redacted>``. A drive or UNC anchor ("C:\\",
292
+ # "\\\\server\\share\\") does carry content, so it stays and is judged.
293
+ parts = list(path.parts)
294
+ if parts and parts[0] in ("/", "\\"):
295
+ parts.pop(0)
296
+ return "/" + "/".join(comp(c) for c in parts if c)
297
+
298
+
299
+ def _describe_transcript(
300
+ path: Path,
301
+ *,
302
+ glob_pat: str | None,
303
+ multiple: bool,
304
+ aliases: dict[str, str] | None = None,
305
+ ) -> TranscriptFinding:
306
+ try:
307
+ stat = path.stat()
308
+ size = stat.st_size
309
+ mtime_date = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).strftime("%Y-%m-%d")
310
+ except OSError:
311
+ size, mtime_date = None, None
312
+ line_count = None
313
+ try:
314
+ with path.open(encoding="utf-8", errors="replace") as f:
315
+ line_count = sum(1 for _ in f)
316
+ except OSError:
317
+ pass
318
+ return TranscriptFinding(
319
+ glob=glob_pat,
320
+ location=_redact_location(path, aliases),
321
+ fmt="jsonl" if path.suffix == ".jsonl" else (path.suffix.lstrip(".") or "unknown"),
322
+ size_bytes=size,
323
+ line_count=line_count,
324
+ mtime_date=mtime_date,
325
+ multiple=multiple,
326
+ real_path=path,
327
+ )
328
+
329
+
330
+ def _newest(paths: list[Path]) -> Path:
331
+ return max(paths, key=lambda p: p.stat().st_mtime if p.exists() else 0)
332
+
333
+
334
+ def discover_transcript(
335
+ parser: str,
336
+ *,
337
+ cli: str,
338
+ hints: Hints,
339
+ aliases: dict[str, str] | None = None,
340
+ ) -> TranscriptFinding | None:
341
+ """Locate the newest existing transcript via override or convention glob."""
342
+ if hints.transcript:
343
+ path = Path(hints.transcript).expanduser()
344
+ if not path.is_file():
345
+ return TranscriptFinding(note=f"--transcript path does not exist: {path.name}")
346
+ return _describe_transcript(path, glob_pat=None, multiple=False, aliases=aliases)
347
+
348
+ if hints.session_dir:
349
+ base = Path(hints.session_dir).expanduser()
350
+ matches = sorted(base.glob("**/*.jsonl")) or sorted(base.glob("**/*.json"))
351
+ if not matches:
352
+ return TranscriptFinding(note=f"no *.jsonl/*.json under --session-dir {base.name}")
353
+ return _describe_transcript(
354
+ _newest(matches), glob_pat=None, multiple=len(matches) > 1, aliases=aliases
355
+ )
356
+
357
+ pattern = TRANSCRIPT_GLOBS.get(parser) or FAMILY_GLOBS.get(cli)
358
+ if not pattern:
359
+ return TranscriptFinding(
360
+ note="no transcript-location convention for this CLI; "
361
+ "pass --transcript PATH or --session-dir DIR"
362
+ )
363
+ matches = [Path(p) for p in glob.glob(os.path.expanduser(pattern))]
364
+ matches = [p for p in matches if p.is_file()]
365
+ if not matches:
366
+ return TranscriptFinding(
367
+ glob=pattern,
368
+ note="no existing transcript matched the convention glob; "
369
+ "use --transcript / --session-dir, or run --probe",
370
+ )
371
+ return _describe_transcript(
372
+ _newest(matches), glob_pat=pattern, multiple=len(matches) > 1, aliases=aliases
373
+ )
374
+
375
+
376
+ # ---------------------------------------------------------- schema inference
377
+
378
+
379
+ def _type_name(value) -> str:
380
+ if value is None:
381
+ return "null"
382
+ if isinstance(value, bool):
383
+ return "bool"
384
+ if isinstance(value, int):
385
+ return "int"
386
+ if isinstance(value, float):
387
+ return "float"
388
+ if isinstance(value, str):
389
+ return "str"
390
+ return "other"
391
+
392
+
393
+ def _walk_paths(obj, prefix: str, out: set[str]) -> None:
394
+ """Collect dotted key paths with the LEAF TYPE only (never values); list
395
+ indices collapse to ``[]`` so ``messages[].tokens.input:int`` is one path.
396
+
397
+ A dict key that isn't a plain identifier (e.g. a transcript that keys by
398
+ relative file path or a per-file backup id) is collapsed to ``<key>`` —
399
+ static field names (the ones a parser keys on, like ``input_tokens``) survive
400
+ untouched, but dynamic keys can't leak paths/content into the summary. A
401
+ credential-shaped key (``ghp_…`` as a map key) is identifier-shaped yet must
402
+ not ship, so it collapses too — the same hole :func:`sanitize.scrub_json`
403
+ closes for values."""
404
+ if isinstance(obj, dict):
405
+ for key, value in obj.items():
406
+ key = str(key)
407
+ if not sanitize.looks_like_identifier(key) or sanitize.looks_like_secret(key):
408
+ key = "<key>"
409
+ child = f"{prefix}.{key}" if prefix else key
410
+ _walk_paths(value, child, out)
411
+ elif isinstance(obj, list):
412
+ child = f"{prefix}[]"
413
+ for value in obj:
414
+ _walk_paths(value, child, out)
415
+ else:
416
+ out.add(f"{prefix}:{_type_name(obj)}")
417
+
418
+
419
+ def _is_token_candidate(path: str) -> bool:
420
+ name, _, typ = path.rpartition(":")
421
+ if typ != "int":
422
+ return False
423
+ last = name.split(".")[-1].replace("[]", "")
424
+ return bool(_TOKEN_KEY_RE.search(last))
425
+
426
+
427
+ def infer_token_schema(
428
+ parser: str, path: Path, *, max_entries: int = MAX_SCHEMA_ENTRIES
429
+ ) -> TokenSchema:
430
+ """Structural key-path summary (types only) + token-field candidates.
431
+
432
+ Works even when ``parser == "none"``: the candidates are exactly what a
433
+ maintainer needs to write a parser for a brand-new CLI. When a real parser
434
+ exists, its parsed integer counts are included as a self-check.
435
+ """
436
+ paths: set[str] = set()
437
+ scanned = 0
438
+ for entry in _jsonl_entries(path):
439
+ if scanned >= max_entries:
440
+ break
441
+ scanned += 1
442
+ _walk_paths(entry, "", paths)
443
+ candidates = sorted(p for p in paths if _is_token_candidate(p))
444
+ parsed = None
445
+ if parser != "none":
446
+ usage = read_usage(parser, path)
447
+ if usage is not None:
448
+ parsed = usage.to_dict()
449
+ return TokenSchema(
450
+ parser=parser,
451
+ entries_scanned=scanned,
452
+ parsed_usage=parsed,
453
+ key_paths=sorted(paths),
454
+ token_field_candidates=candidates,
455
+ )
456
+
457
+
458
+ # --------------------------------------------------------------- hook config
459
+
460
+
461
+ def _hooks_registered(project: Path, profile: CLIProfile) -> bool:
462
+ config_path = project / profile.hooks.config_path
463
+ if not config_path.is_file():
464
+ return False
465
+ import json
466
+
467
+ try:
468
+ config = json.loads(config_path.read_text(encoding="utf-8"))
469
+ except (json.JSONDecodeError, OSError):
470
+ return False
471
+ if not isinstance(config, dict):
472
+ return False
473
+ return relay_registered(config, profile.hooks.dialect, profile.hooks.events)
474
+
475
+
476
+ # ----------------------------------------------------------------- SCAN mode
477
+
478
+
479
+ def scan(
480
+ *,
481
+ cli: str,
482
+ profile: CLIProfile | None,
483
+ project: Path,
484
+ hints: Hints,
485
+ pseudo: sanitize.Pseudonymizer | None = None,
486
+ ) -> ProfileFinding:
487
+ # Aliases registered up front (the project basename) are routed at
488
+ # collection time so a normal run needs zero egress repairs.
489
+ aliases = {orig: alias for _ns, orig, alias in pseudo.entries()} if pseudo else None
490
+ binary = hints.binary or (profile.binary if profile else cli)
491
+ parser = profile.usage_parser if profile else "none"
492
+ finding = ProfileFinding(
493
+ cli=cli,
494
+ mode="scan",
495
+ known_profile=profile is not None,
496
+ # The finding is render-only; a --binary hint may be an absolute path
497
+ # under $HOME, which must not reach the report (the raw local keeps
498
+ # driving which/--version below).
499
+ binary=sanitize.redact_home(binary),
500
+ parser=parser,
501
+ dialect=profile.hooks.dialect if profile else None,
502
+ declared_events=dict(profile.hooks.events) if profile else {},
503
+ )
504
+
505
+ finding.flags = run_version_help(binary)
506
+ if not finding.flags.found:
507
+ # finding.binary, not the raw local: a home-rooted --binary hint in a
508
+ # rendered warning would trip the egress guard's home-path rule.
509
+ finding.warnings.append(
510
+ f"binary {finding.binary!r} not found on PATH — version/help unavailable "
511
+ "(scan continues from on-disk conventions)"
512
+ )
513
+
514
+ if profile is not None:
515
+ finding.registered = _hooks_registered(project, profile)
516
+ if not finding.registered:
517
+ finding.next_steps.append(
518
+ f"hooks not registered in {profile.hooks.config_path}; "
519
+ f"`froid-loop init --cli {cli}` to validate the dialect end-to-end, "
520
+ "or re-run with --probe"
521
+ )
522
+
523
+ finding.transcript = discover_transcript(parser, cli=cli, hints=hints, aliases=aliases)
524
+ if finding.transcript and finding.transcript.note:
525
+ finding.warnings.append(finding.transcript.note)
526
+ if finding.transcript and finding.transcript.real_path is not None:
527
+ finding.tokens = infer_token_schema(parser, finding.transcript.real_path)
528
+ if finding.transcript.multiple:
529
+ finding.next_steps.append(
530
+ "multiple fresh transcripts matched; pass --transcript to pin the right one"
531
+ )
532
+ return finding
533
+
534
+
535
+ # ---------------------------------------------------------- PROBE tmux launcher
536
+
537
+
538
+ class _ProbeLauncher:
539
+ """The few multiplexer primitives PROBE needs — deliberately NOT a
540
+ GenericAdapter, which mandates a Policy and story-completion logic irrelevant
541
+ here. Drives the shared backend so PROBE shells out to no multiplexer
542
+ directly."""
543
+
544
+ def __init__(self, session_name: str):
545
+ self.session_name = session_name
546
+ self.mux = get_multiplexer()
547
+
548
+ def start(self, argv: list[str], env: dict[str, str], cwd: Path, log_file: Path) -> str | None:
549
+ try:
550
+ self.mux.new_session(self.session_name, cwd, 220, 50)
551
+ command = " ".join(shlex.quote(a) for a in argv)
552
+ # `env` carries the profile's own `[env]` table verbatim, and a
553
+ # profile declaring FROID_LOOP_STATE_DIR would aim a froid-loop
554
+ # wrapper in that window at a different state root — and so a
555
+ # different registry, where this very session reads as gone. The
556
+ # pin chokepoint forces the entry to this process's own answer in
557
+ # both arms, the underivable one included (runs.pin_state_root);
558
+ # the engine's window merge applies the same rule.
559
+ window_env = runs.pin_state_root(env)
560
+ window_id = self.mux.new_window(
561
+ self.session_name, PROBE_TASK_ID, cwd, window_env, command
562
+ )
563
+ except MultiplexerError:
564
+ return None
565
+ # pipe-pane may race a window that dies instantly; tolerate failure.
566
+ self.mux.pipe_pane(window_id, log_file)
567
+ return window_id
568
+
569
+ def window_alive(self, window_id: str) -> bool:
570
+ return self.mux.window_alive(self.session_name, window_id)
571
+
572
+ def kill(self) -> None:
573
+ self.mux.kill_session(self.session_name)
574
+
575
+
576
+ def _probe_argv(profile: CLIProfile, binary: str, hints: Hints) -> list[str]:
577
+ argv = [
578
+ binary,
579
+ *profile.launch_args,
580
+ # Send the probe prompt verbatim, NOT through profile.render_prompt: a
581
+ # content-free turn has no skill name, so a skill-templating prompt_template
582
+ # (copilot, codex) would render a nonexistent .../skills//SKILL.md path the
583
+ # agent hunts for, and the turn never ends within the probe timeout.
584
+ PROBE_PROMPT,
585
+ *profile.bypass_args,
586
+ ]
587
+ if hints.model:
588
+ argv += [profile.model_flag, hints.model]
589
+ return argv
590
+
591
+
592
+ def _captured_transcript_path(capture_dir: Path) -> Path | None:
593
+ """The transcript path the CLI handed a hook on stdin, newest signal first.
594
+
595
+ Ground truth beats convention: a CLI that reports its own transcript (agy's
596
+ `transcriptPath`, Claude's `transcript_path`) names the exact file the turn
597
+ was recorded to, including the session id a convention glob can only wildcard.
598
+ """
599
+ import json
600
+
601
+ for signal_file in sorted(capture_dir.glob("*.signal.json"), reverse=True):
602
+ try:
603
+ raw = json.loads(signal_file.read_text(encoding="utf-8"))
604
+ except (json.JSONDecodeError, OSError):
605
+ continue
606
+ path = raw.get("transcript_path") if isinstance(raw, dict) else None
607
+ if isinstance(path, str) and path:
608
+ return Path(path).expanduser()
609
+ return None
610
+
611
+
612
+ def _collect_captures(capture_dir: Path, events_map: dict[str, str]) -> list[EventCapture]:
613
+ """Reduce each raw captured payload to its SHAPE: top-level keys and dotted
614
+ key-path:type paths. The payload's diagnostic value for profile authoring is
615
+ where the fields live and what type they are — so no payload *value* of any
616
+ kind ships, which removes the widest identifier-shaped egress surface
617
+ outright (#199). The walk sees the raw dict so leaf types are faithful;
618
+ dynamic (non-identifier) keys collapse to ``<key>`` in both projections."""
619
+ captures: list[EventCapture] = []
620
+ for payload_file in sorted(capture_dir.glob("*.payload.json")):
621
+ import json
622
+
623
+ try:
624
+ raw = json.loads(payload_file.read_text(encoding="utf-8"))
625
+ except (json.JSONDecodeError, OSError):
626
+ continue
627
+ if not isinstance(raw, dict):
628
+ continue
629
+ native = str(raw.pop("argv_event", "Unknown"))
630
+ paths: set[str] = set()
631
+ _walk_paths(raw, "", paths)
632
+ captures.append(
633
+ EventCapture(
634
+ native_event=native,
635
+ canonical_event=events_map.get(native),
636
+ payload_keys=sorted(
637
+ str(k) if sanitize.looks_like_identifier(str(k)) else "<key>" for k in raw
638
+ ),
639
+ payload_schema=sorted(paths),
640
+ )
641
+ )
642
+ return captures
643
+
644
+
645
+ def probe(
646
+ *,
647
+ cli: str,
648
+ profile: CLIProfile,
649
+ project: Path,
650
+ hints: Hints,
651
+ timeout_s: float = 90,
652
+ keep_temp: bool = False,
653
+ pseudo: sanitize.Pseudonymizer | None = None,
654
+ ) -> ProfileFinding:
655
+ import json
656
+
657
+ aliases = {orig: alias for _ns, orig, alias in pseudo.entries()} if pseudo else None
658
+ binary = hints.binary or profile.binary
659
+ finding = ProfileFinding(
660
+ cli=cli,
661
+ mode="probe",
662
+ known_profile=True,
663
+ # render-only; see the identical redaction in scan()
664
+ binary=sanitize.redact_home(binary),
665
+ parser=profile.usage_parser,
666
+ dialect=profile.hooks.dialect,
667
+ declared_events=dict(profile.hooks.events),
668
+ )
669
+ finding.flags = run_version_help(binary)
670
+
671
+ # The live probe launches through the selected multiplexer backend (see
672
+ # _ProbeLauncher), so gate on THAT backend's availability rather than a
673
+ # hardcoded `which("tmux")` — a Windows host running herdr (or a future psmux)
674
+ # must still probe. `available()` is guarded so a backend whose host probe
675
+ # raises reads as unavailable, exactly like selection's _usable().
676
+ mux = get_multiplexer()
677
+ try:
678
+ mux_ready = bool(mux.available())
679
+ except Exception: # a raising host probe means "cannot probe", not a crash
680
+ mux_ready = False
681
+ if not mux_ready or not shutil.which(binary):
682
+ # finding.binary, not the raw local — see the identical note in scan()
683
+ missing = f"multiplexer backend {type(mux).__name__}" if not mux_ready else finding.binary
684
+ finding.warnings.append(f"{missing} not on PATH — cannot probe; falling back to scan")
685
+ scanned = scan(cli=cli, profile=profile, project=project, hints=hints, pseudo=pseudo)
686
+ scanned.mode = "probe"
687
+ return scanned
688
+
689
+ tmpdir = Path(tempfile.mkdtemp(prefix="froid-loop-probe-"))
690
+ launcher = _ProbeLauncher(session_name=f"froid-loop-probe-{tmpdir.name}")
691
+ try:
692
+ capture_dir = tmpdir / "capture"
693
+ capture_dir.mkdir(parents=True, exist_ok=True)
694
+
695
+ # 1. lay down the capture hook + a hook config registered through the very
696
+ # same merge_hooks `froid-loop init` uses — so a bad dialect surfaces live.
697
+ hook_src = resources.files("froid_loop.data").joinpath(PROBE_HOOK_NAME)
698
+ hook_path = tmpdir / PROBE_HOOK_NAME
699
+ hook_path.write_text(hook_src.read_text(encoding="utf-8"), encoding="utf-8")
700
+ host = get_process_host()
701
+ interp = host.hook_interpreter()
702
+ registrations = {
703
+ native: f"{interp} {host.shell_quote(str(hook_path))} {canonical}"
704
+ for native, canonical in profile.hooks.events.items()
705
+ }
706
+ config, _ = merge_hooks({}, registrations, profile.hooks.dialect)
707
+ config_path = tmpdir / profile.hooks.config_path
708
+ config_path.parent.mkdir(parents=True, exist_ok=True)
709
+ config_path.write_text(json.dumps(config, indent=2) + "\n", encoding="utf-8")
710
+
711
+ # 2. launch one trivial content-free turn in a fresh tmux window
712
+ argv = _probe_argv(profile, binary, hints)
713
+ env = {
714
+ **profile.env,
715
+ "FROID_LOOP_RUN_DIR": str(tmpdir),
716
+ "FROID_LOOP_TASK_ID": PROBE_TASK_ID,
717
+ "FROID_LOOP_PROBE_CAPTURE_DIR": str(capture_dir),
718
+ }
719
+ log_file = tmpdir / "probe.log"
720
+ watcher = SignalWatcher(capture_dir)
721
+ launched_ns = time.time_ns()
722
+ window_id = launcher.start(argv, env, tmpdir, log_file)
723
+ if window_id is None:
724
+ finding.warnings.append("could not launch the CLI in tmux; no events captured")
725
+ return finding
726
+
727
+ # 3. completion: first of — canonical Stop for `probe`; any capture file
728
+ # appeared and the window died; window died; deadline.
729
+ deadline = time.monotonic() + timeout_s
730
+ while True:
731
+ remaining = deadline - time.monotonic()
732
+ if remaining <= 0:
733
+ finding.warnings.append(
734
+ "no Stop event before --timeout; the CLI may need first-run auth "
735
+ "(a pending login dialog reads as a timeout). See the log tail below."
736
+ )
737
+ break
738
+ event = watcher.wait_for(
739
+ PROBE_TASK_ID,
740
+ {"Stop"},
741
+ timeout_s=min(remaining, 5.0),
742
+ since_ns=launched_ns,
743
+ )
744
+ if event is not None:
745
+ break
746
+ try:
747
+ alive = launcher.window_alive(window_id)
748
+ except MultiplexerError:
749
+ # transient transport hang is not proof the window died; retry
750
+ # on the next tick rather than mis-reporting a dead CLI window.
751
+ continue
752
+ captured_any = any(capture_dir.glob("*.payload.json"))
753
+ if not alive:
754
+ if not captured_any:
755
+ finding.warnings.append(
756
+ "the CLI window died before any hook fired — the dialect may be "
757
+ f"rejected for {profile.hooks.dialect}, or launch/auth failed. "
758
+ "See the log tail below."
759
+ )
760
+ break
761
+
762
+ # 4. one short grace poll so a Stop's sibling files all land, then collect.
763
+ time.sleep(PROBE_GRACE_S)
764
+ finding.captured_events = _collect_captures(capture_dir, profile.hooks.events)
765
+ if not finding.captured_events:
766
+ finding.next_steps.append(
767
+ "no hook payloads captured — confirm the CLI is authenticated and that "
768
+ f"the {profile.hooks.dialect} hook config is accepted, then re-run --probe"
769
+ )
770
+ tail = _log_tail(log_file)
771
+ if tail:
772
+ finding.warnings.append("log tail (scrubbed):\n" + tail)
773
+
774
+ # 5. transcript: prefer the exact path the CLI handed the hook on stdin
775
+ # over the convention glob — the payload names this turn's file, while
776
+ # the glob can only pick the newest match and may land on an unrelated
777
+ # session. Falls back to the glob when the CLI reports no path.
778
+ live = _captured_transcript_path(capture_dir)
779
+ if live is not None and live.is_file():
780
+ finding.transcript = _describe_transcript(
781
+ live, glob_pat=None, multiple=False, aliases=aliases
782
+ )
783
+ else:
784
+ finding.transcript = discover_transcript(
785
+ profile.usage_parser, cli=cli, hints=hints, aliases=aliases
786
+ )
787
+ if finding.transcript and finding.transcript.note:
788
+ finding.warnings.append(finding.transcript.note)
789
+ if finding.transcript and finding.transcript.real_path is not None:
790
+ finding.tokens = infer_token_schema(profile.usage_parser, finding.transcript.real_path)
791
+ return finding
792
+ finally:
793
+ launcher.kill()
794
+ if keep_temp:
795
+ # ~-relative so a $TMPDIR under $HOME can't trip the egress guard;
796
+ # the shell expands ~ so the printed path stays inspectable.
797
+ finding.warnings.append(
798
+ f"--keep-temp: RAW probe data retained at {sanitize.redact_home(str(tmpdir))} "
799
+ "— DO NOT SHARE; delete it after inspection"
800
+ )
801
+ else:
802
+ shutil.rmtree(tmpdir, ignore_errors=True)
803
+
804
+
805
+ def _log_tail(log_file: Path, max_lines: int = 20) -> str | None:
806
+ try:
807
+ text = log_file.read_text(encoding="utf-8", errors="replace")
808
+ except OSError:
809
+ return None
810
+ if not text.strip():
811
+ return None
812
+ lines = text.splitlines()[-max_lines:]
813
+ return sanitize.scrub_text(
814
+ "\n".join(lines), max_lines=max_lines, max_chars=sanitize.SCRUB_TEXT_MAX_CHARS
815
+ )
816
+
817
+
818
+ # ------------------------------------------------------------------ rendering
819
+
820
+
821
+ def _fmt_kv(label: str, value) -> str:
822
+ return f"- **{label}:** {value}"
823
+
824
+
825
+ def render_markdown(
826
+ f: ProfileFinding,
827
+ *,
828
+ pseudo: sanitize.Pseudonymizer | None = None,
829
+ repairs: list[tuple[str, int]] | None = None,
830
+ ) -> str:
831
+ out: list[str] = []
832
+ out.append(f"# Profile finalize report — {f.cli} ({f.mode})")
833
+ out.append("")
834
+
835
+ # Summary
836
+ out.append("## Summary")
837
+ out.append(_fmt_kv("CLI", f.cli))
838
+ out.append(
839
+ _fmt_kv("binary", f"{f.binary} ({'found' if f.flags and f.flags.found else 'NOT found'})")
840
+ )
841
+ out.append(_fmt_kv("known profile", "yes" if f.known_profile else "no (reduced report)"))
842
+ out.append(_fmt_kv("hook dialect", f.dialect or "—"))
843
+ out.append(_fmt_kv("usage_parser", f.parser))
844
+ if f.registered is not None:
845
+ out.append(_fmt_kv("hooks registered", "yes" if f.registered else "no"))
846
+ out.append(_fmt_kv("warnings", str(len(f.warnings))))
847
+ out.append("")
848
+
849
+ # CLI flags
850
+ out.append("## CLI flags")
851
+ out.append(_fmt_kv("launch_args / bypass_args", "see profile (rendered verbatim below)"))
852
+ if f.flags and f.flags.version:
853
+ out.append("\n```\n" + f.flags.version + "\n```")
854
+ if f.flags and f.flags.help:
855
+ out.append("\n<details><summary>--help (scrubbed)</summary>\n")
856
+ out.append("```\n" + f.flags.help + "\n```")
857
+ out.append("</details>")
858
+ if not f.flags or not f.flags.found:
859
+ out.append("_binary not available; flags/help not captured._")
860
+ out.append("")
861
+
862
+ # Hook payload shape
863
+ out.append("## Hook payload shape")
864
+ if f.mode == "scan":
865
+ if f.declared_events:
866
+ out.append(
867
+ "Declared native → canonical events (registered = "
868
+ f"{'yes' if f.registered else 'no'}):"
869
+ )
870
+ for native, canonical in f.declared_events.items():
871
+ out.append(f"- `{native}` → `{canonical}`")
872
+ else:
873
+ out.append("_no profile; events unknown. Re-run with --probe to capture payloads._")
874
+ else:
875
+ if f.captured_events:
876
+ for ev in f.captured_events:
877
+ out.append(f"### `{ev.native_event}` → `{ev.canonical_event or '?'}`")
878
+ out.append(
879
+ _fmt_kv("payload keys", ", ".join(f"`{k}`" for k in ev.payload_keys) or "—")
880
+ )
881
+ out.append("\n**Payload schema** (key paths + leaf types, never values):")
882
+ if ev.payload_schema:
883
+ out.append("\n```\n" + "\n".join(ev.payload_schema) + "\n```")
884
+ else:
885
+ out.append("\n- _empty payload._")
886
+ else:
887
+ out.append("_no hook payloads captured (see warnings)._")
888
+ out.append("")
889
+
890
+ # Transcript
891
+ out.append("## Transcript")
892
+ t = f.transcript
893
+ if t and t.real_path is not None:
894
+ out.append(_fmt_kv("location", f"`{t.location}`"))
895
+ if t.glob:
896
+ out.append(_fmt_kv("matched glob", f"`{t.glob}`"))
897
+ out.append(_fmt_kv("format", t.fmt))
898
+ out.append(_fmt_kv("size", f"{t.size_bytes} bytes"))
899
+ out.append(_fmt_kv("lines", t.line_count))
900
+ out.append(_fmt_kv("mtime", t.mtime_date))
901
+ if t.multiple:
902
+ out.append("- _multiple candidates matched; newest shown — pass --transcript to pin._")
903
+ else:
904
+ out.append("_no transcript located._" + (f" ({t.note})" if t and t.note else ""))
905
+ out.append("")
906
+
907
+ # Token usage schema
908
+ out.append("## Token usage schema")
909
+ tk = f.tokens
910
+ if tk:
911
+ out.append(_fmt_kv("declared parser", tk.parser))
912
+ out.append(_fmt_kv("entries scanned", tk.entries_scanned))
913
+ if tk.parsed_usage is not None:
914
+ out.append(_fmt_kv("parsed counts (self-check)", f"`{tk.parsed_usage}`"))
915
+ out.append(
916
+ "\n**Token-field candidates** (int leaves; per-call-vs-cumulative is a human call):"
917
+ )
918
+ if tk.token_field_candidates:
919
+ for cand in tk.token_field_candidates:
920
+ out.append(f"- `{cand}`")
921
+ else:
922
+ out.append("- _none matched the token-name heuristic._")
923
+ out.append("\n<details><summary>All key paths (types only, no values)</summary>\n")
924
+ out.append("```\n" + "\n".join(tk.key_paths) + "\n```")
925
+ out.append("</details>")
926
+ else:
927
+ out.append("_no transcript to infer from._")
928
+ out.append("")
929
+
930
+ # Warnings / next steps
931
+ out.append("## Warnings / next steps")
932
+ if not f.warnings and not f.next_steps:
933
+ out.append("_none._")
934
+ for w in f.warnings:
935
+ out.append(f"- ⚠️ {w}")
936
+ for s in f.next_steps:
937
+ out.append(f"- → {s}")
938
+ out.append("")
939
+
940
+ rendered = "\n".join(out)
941
+ rendered, reps = sanitize.guard(rendered, pseudo)
942
+ if reps:
943
+ note = [
944
+ "",
945
+ "### Backstop repairs",
946
+ "",
947
+ "_The leak self-check caught stray occurrences of pseudonymized "
948
+ "identifiers that the per-field routing missed, and substituted "
949
+ "their aliases — a froid-loop routing gap; please report it._",
950
+ "",
951
+ ]
952
+ for label, count in reps:
953
+ note.append(f"- `{label}`: {count} stray occurrence(s) pseudonymized")
954
+ note.append("")
955
+ rendered += "\n".join(note)
956
+ # The note is appended after the repair loop verified the body, so
957
+ # re-check the whole thing: the note must sit inside the verified bytes.
958
+ sanitize.assert_clean(rendered, pseudo)
959
+ if repairs is not None:
960
+ repairs.extend(reps)
961
+ return rendered
962
+
963
+
964
+ def render_json(
965
+ f: ProfileFinding,
966
+ *,
967
+ pseudo: sanitize.Pseudonymizer | None = None,
968
+ repairs: list[tuple[str, int]] | None = None,
969
+ ) -> str:
970
+ import json
971
+
972
+ def transcript_dict(t: TranscriptFinding | None):
973
+ if t is None:
974
+ return None
975
+ return {
976
+ "glob": t.glob,
977
+ "location": t.location,
978
+ "format": t.fmt,
979
+ "size_bytes": t.size_bytes,
980
+ "line_count": t.line_count,
981
+ "mtime_date": t.mtime_date,
982
+ "multiple": t.multiple,
983
+ "note": t.note,
984
+ }
985
+
986
+ data = {
987
+ "schema_version": SCHEMA_VERSION,
988
+ "cli": f.cli,
989
+ "mode": f.mode,
990
+ "known_profile": f.known_profile,
991
+ "binary": f.binary,
992
+ "binary_found": bool(f.flags and f.flags.found),
993
+ "dialect": f.dialect,
994
+ "usage_parser": f.parser,
995
+ "hooks_registered": f.registered,
996
+ "declared_events": f.declared_events,
997
+ "version": f.flags.version if f.flags else None,
998
+ "help": f.flags.help if f.flags else None,
999
+ "captured_events": [
1000
+ {
1001
+ "native_event": ev.native_event,
1002
+ "canonical_event": ev.canonical_event,
1003
+ "payload_keys": ev.payload_keys,
1004
+ "payload_schema": ev.payload_schema,
1005
+ }
1006
+ for ev in f.captured_events
1007
+ ],
1008
+ "transcript": transcript_dict(f.transcript),
1009
+ "tokens": (
1010
+ {
1011
+ "parser": f.tokens.parser,
1012
+ "entries_scanned": f.tokens.entries_scanned,
1013
+ "parsed_usage": f.tokens.parsed_usage,
1014
+ "key_paths": f.tokens.key_paths,
1015
+ "token_field_candidates": f.tokens.token_field_candidates,
1016
+ }
1017
+ if f.tokens
1018
+ else None
1019
+ ),
1020
+ "warnings": f.warnings,
1021
+ "next_steps": f.next_steps,
1022
+ }
1023
+ # sort_keys so two probes of the same CLI diff cleanly — the document has
1024
+ # consumers now, and dict-literal order is an implementation detail.
1025
+ # ensure_ascii=False is a SAFETY requirement, not cosmetics (see
1026
+ # diagnostics.render_json): with the default, a non-ASCII sensitive value
1027
+ # reaches the guard as \uXXXX escapes and matches nothing, yet json.loads
1028
+ # hands the consumer back the original. machine.emit/write_document emit
1029
+ # this string verbatim, so the guarded bytes ARE the emitted bytes.
1030
+ rendered = json.dumps(data, indent=2, sort_keys=True, ensure_ascii=False)
1031
+ rendered, reps = sanitize.guard(rendered, pseudo)
1032
+ if reps:
1033
+ # Disclose the repair in the document itself so the routing gap surfaces
1034
+ # as a reportable bug. Substitution preserved JSON validity — a leaked
1035
+ # original is identifier-shaped and its alias is [A-Za-z0-9-], neither
1036
+ # side carries quotes or backslashes — so reload-and-extend is safe.
1037
+ # backstop_repairs is an optional additive key: absent on a clean report.
1038
+ loaded = json.loads(rendered)
1039
+ loaded["backstop_repairs"] = dict(reps)
1040
+ rendered = json.dumps(loaded, indent=2, sort_keys=True, ensure_ascii=False)
1041
+ sanitize.assert_clean(rendered, pseudo)
1042
+ if repairs is not None:
1043
+ repairs.extend(reps)
1044
+ return rendered