syncade 0.6.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncade/__init__.py +3 -0
- syncade/__main__.py +6 -0
- syncade/adapters/__init__.py +0 -0
- syncade/adapters/anthropic.py +457 -0
- syncade/adapters/base.py +221 -0
- syncade/adapters/fake.py +73 -0
- syncade/adapters/fake_common.py +29 -0
- syncade/adapters/fake_producer_audit_draft.py +460 -0
- syncade/adapters/fake_reviewer_synth.py +310 -0
- syncade/adapters/openai.py +484 -0
- syncade/adapters/openai_parsing.py +119 -0
- syncade/adapters/producer.py +221 -0
- syncade/adapters/producer_anthropic.py +300 -0
- syncade/adapters/producer_openai.py +226 -0
- syncade/adapters/registry.py +81 -0
- syncade/auth_check.py +554 -0
- syncade/auth_preflight.py +342 -0
- syncade/base_resolution.py +214 -0
- syncade/billing.py +141 -0
- syncade/checks_config.py +113 -0
- syncade/cli/__init__.py +546 -0
- syncade/cli/auth_gate.py +59 -0
- syncade/cli/config_keys.py +135 -0
- syncade/cli/config_list.py +82 -0
- syncade/cli/config_menu_rows.py +166 -0
- syncade/cli/config_mode.py +609 -0
- syncade/cli/config_overrides.py +122 -0
- syncade/cli/config_tui.py +476 -0
- syncade/cli/doctor_mode.py +72 -0
- syncade/cli/gc_mode.py +109 -0
- syncade/cli/install_skill.py +514 -0
- syncade/cli/metrics_mode.py +363 -0
- syncade/cli/modes.py +573 -0
- syncade/cli/parser.py +450 -0
- syncade/cli/parser_types.py +137 -0
- syncade/cli/paths.py +38 -0
- syncade/cli/preflight_paths.py +90 -0
- syncade/cli/resolve.py +116 -0
- syncade/cli/resume_mode.py +324 -0
- syncade/cli/toml_writer.py +410 -0
- syncade/cli/validate.py +421 -0
- syncade/config.py +478 -0
- syncade/config_auth.py +310 -0
- syncade/config_cold.py +209 -0
- syncade/config_gc.py +55 -0
- syncade/config_loader.py +182 -0
- syncade/config_loop.py +282 -0
- syncade/config_producer.py +222 -0
- syncade/config_retry.py +49 -0
- syncade/config_types.py +59 -0
- syncade/diff_filter.py +437 -0
- syncade/dispatcher.py +571 -0
- syncade/doctor.py +425 -0
- syncade/doctor_env.py +218 -0
- syncade/doctor_preview.py +524 -0
- syncade/doctor_types.py +28 -0
- syncade/exit_codes.py +82 -0
- syncade/findings.py +242 -0
- syncade/findings_json.py +456 -0
- syncade/gc.py +211 -0
- syncade/gc_execute.py +372 -0
- syncade/gc_protection.py +129 -0
- syncade/gc_types.py +50 -0
- syncade/gc_worktrees.py +200 -0
- syncade/git_object_id.py +12 -0
- syncade/git_preconditions.py +389 -0
- syncade/logging.py +289 -0
- syncade/metrics/__init__.py +32 -0
- syncade/metrics/aggregate.py +550 -0
- syncade/metrics/schema.py +221 -0
- syncade/orchestrator/__init__.py +61 -0
- syncade/orchestrator/_runs_dir.py +24 -0
- syncade/orchestrator/branch_advance.py +165 -0
- syncade/orchestrator/branch_guard.py +98 -0
- syncade/orchestrator/budget.py +107 -0
- syncade/orchestrator/escalation_coverage.py +81 -0
- syncade/orchestrator/loop.py +611 -0
- syncade/orchestrator/loop_dispatch_check.py +112 -0
- syncade/orchestrator/loop_finalize.py +404 -0
- syncade/orchestrator/loop_preflight.py +131 -0
- syncade/orchestrator/loop_resume.py +91 -0
- syncade/orchestrator/loop_rmtree.py +70 -0
- syncade/orchestrator/loop_round_step.py +599 -0
- syncade/orchestrator/prior_round.py +336 -0
- syncade/orchestrator/producer_phase.py +169 -0
- syncade/orchestrator/results.py +306 -0
- syncade/orchestrator/resume.py +96 -0
- syncade/orchestrator/resume_load.py +483 -0
- syncade/orchestrator/resume_plan.py +554 -0
- syncade/orchestrator/resume_target.py +215 -0
- syncade/orchestrator/resume_types.py +182 -0
- syncade/orchestrator/reviewer_template_failure.py +99 -0
- syncade/orchestrator/round.py +573 -0
- syncade/orchestrator/round_checks.py +91 -0
- syncade/orchestrator/round_no_changes.py +369 -0
- syncade/orchestrator/round_predispatch.py +212 -0
- syncade/orchestrator/verdict.py +279 -0
- syncade/persistence/__init__.py +189 -0
- syncade/persistence/_atomic.py +33 -0
- syncade/persistence/_clusters.py +70 -0
- syncade/persistence/_findings_verdict.py +201 -0
- syncade/persistence/_markdown.py +286 -0
- syncade/persistence/_validation.py +37 -0
- syncade/persistence/checks.py +249 -0
- syncade/persistence/decision_needed.py +289 -0
- syncade/persistence/findings_md.py +389 -0
- syncade/persistence/handoff.py +389 -0
- syncade/persistence/handoff_classify.py +196 -0
- syncade/persistence/last_reviewed.py +67 -0
- syncade/persistence/loop_manifest.py +165 -0
- syncade/persistence/loop_summary.py +352 -0
- syncade/persistence/loop_summary_text.py +428 -0
- syncade/persistence/producer.py +250 -0
- syncade/persistence/reviewer.py +198 -0
- syncade/persistence/round_manifest.py +238 -0
- syncade/persistence/run_init.py +153 -0
- syncade/persistence/run_summary.py +585 -0
- syncade/persistence/run_summary_next_steps.py +443 -0
- syncade/persistence/synth.py +242 -0
- syncade/persistence/test_run.py +152 -0
- syncade/presets.py +36 -0
- syncade/pricing_config.py +72 -0
- syncade/process.py +600 -0
- syncade/producer.py +189 -0
- syncade/producer_attempt.py +463 -0
- syncade/producer_escalation.py +146 -0
- syncade/producer_git.py +199 -0
- syncade/producer_result.py +205 -0
- syncade/prompts.py +448 -0
- syncade/prompts_loader.py +238 -0
- syncade/retry.py +159 -0
- syncade/run_inputs.py +40 -0
- syncade/run_status.py +198 -0
- syncade/selfcheck.py +471 -0
- syncade/skills/claude/README.md +221 -0
- syncade/skills/claude/SKILL.md +625 -0
- syncade/skills/codex/README.md +116 -0
- syncade/skills/codex/SKILL.md +574 -0
- syncade/snapshot.py +598 -0
- syncade/spec_audit.py +437 -0
- syncade/spec_audit_schema.py +190 -0
- syncade/spec_draft.py +423 -0
- syncade/spec_source.py +135 -0
- syncade/synthesis.py +428 -0
- syncade/synthesis_clusters.py +203 -0
- syncade/synthesis_repair.py +230 -0
- syncade/synthesis_schema.py +65 -0
- syncade/synthesizer/__init__.py +38 -0
- syncade/synthesizer/constants.py +33 -0
- syncade/synthesizer/driver.py +531 -0
- syncade/synthesizer/rendering.py +63 -0
- syncade/synthesizer/result.py +73 -0
- syncade/synthesizer/validation.py +421 -0
- syncade/synthesizer/workspace.py +208 -0
- syncade/templates/presets/balanced.toml +13 -0
- syncade/templates/presets/cheap.toml +12 -0
- syncade/templates/presets/thorough.toml +9 -0
- syncade/templates/producer.md +231 -0
- syncade/templates/reviewer.md +279 -0
- syncade/templates/reviewer_adversarial.md +164 -0
- syncade/templates/reviewer_codex.md +165 -0
- syncade/templates/spec_audit.md +168 -0
- syncade/templates/spec_draft.md +62 -0
- syncade/templates/synthesizer.md +204 -0
- syncade/test_runner.py +476 -0
- syncade/test_runner_classify.py +98 -0
- syncade/transcript.py +150 -0
- syncade/usage.py +407 -0
- syncade/worktree.py +497 -0
- syncade/worktree_env.py +133 -0
- syncade/worktree_paths.py +139 -0
- syncade-0.6.2.dist-info/METADATA +314 -0
- syncade-0.6.2.dist-info/RECORD +177 -0
- syncade-0.6.2.dist-info/WHEEL +5 -0
- syncade-0.6.2.dist-info/entry_points.txt +2 -0
- syncade-0.6.2.dist-info/licenses/LICENSE +202 -0
- syncade-0.6.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Missing-command stderr extraction.
|
|
2
|
+
|
|
3
|
+
The shell-variant "command not found" patterns support
|
|
4
|
+
``_extract_missing_binary``, which the test runner uses to name the missing
|
|
5
|
+
binary when bash reports rc 127. Pure ``re``-only logic.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
# shell-variant "command not found" patterns. Each
|
|
13
|
+
# regex captures the missing-command name from one shell family's
|
|
14
|
+
# stderr format. Lines must start with a shell name prefix (or a
|
|
15
|
+
# common shell binary path) to avoid false-positive matches
|
|
16
|
+
# against ordinary test output that happens to contain the
|
|
17
|
+
# substring "command not found".
|
|
18
|
+
#
|
|
19
|
+
# Variants covered:
|
|
20
|
+
# - bash: ``bash: line 1: foo: command not found``
|
|
21
|
+
# - sh: ``sh: foo: command not found``
|
|
22
|
+
# - dash: ``sh: 1: foo: not found`` / ``dash: 1: foo: not found``
|
|
23
|
+
# - zsh: ``zsh: command not found: foo``
|
|
24
|
+
# - busybox: ``/bin/sh: foo: not found`` / ``ash: foo: not found``
|
|
25
|
+
# - generic: ``<any-shell-path>: <prefix>: <cmd>: (command not found|not found)``
|
|
26
|
+
#
|
|
27
|
+
# A few defenses against false matches:
|
|
28
|
+
# - Each pattern anchors at start-of-line (``re.MULTILINE``).
|
|
29
|
+
# - The leading shell-name field accepts only word-class /
|
|
30
|
+
# path-like characters — random prose with "command not found"
|
|
31
|
+
# later in the line won't match.
|
|
32
|
+
# - The "zsh-reverse" pattern requires the literal sequence
|
|
33
|
+
# ``command not found:`` (with trailing colon) so prose
|
|
34
|
+
# sentences like "the command not found" don't match.
|
|
35
|
+
_SHELL_NAME_PREFIX = r"(?:[\w./-]*sh|bash|zsh|dash|ash|ksh|busybox)"
|
|
36
|
+
"""Trailing-``sh`` (including bare ``sh``) or any of the named
|
|
37
|
+
shells. Matches common on-disk paths too (``/bin/sh``,
|
|
38
|
+
``/usr/local/bin/zsh``). Uses ``[\\w./-]*sh`` (zero-or-more) so
|
|
39
|
+
the bare-shell-name prefix is included; ``[\\w./-]+sh`` would
|
|
40
|
+
require at least one character before ``sh`` and miss the most
|
|
41
|
+
common case."""
|
|
42
|
+
|
|
43
|
+
_SH_NOT_FOUND_PATTERNS = (
|
|
44
|
+
# bash / sh form with optional line-number prefix:
|
|
45
|
+
# "bash: line 1: foo: command not found"
|
|
46
|
+
# "sh: foo: command not found"
|
|
47
|
+
re.compile(
|
|
48
|
+
rf"^{_SHELL_NAME_PREFIX}:\s*(?:line\s+\d+:\s*)?(?P<cmd>[^:\s][^:\n]*?):"
|
|
49
|
+
r"\s*command not found\s*$",
|
|
50
|
+
re.MULTILINE,
|
|
51
|
+
),
|
|
52
|
+
# dash / busybox "not found" form (no "command "):
|
|
53
|
+
# "sh: 1: foo: not found"
|
|
54
|
+
# "/bin/sh: foo: not found"
|
|
55
|
+
re.compile(
|
|
56
|
+
rf"^{_SHELL_NAME_PREFIX}:\s*(?:\d+:\s*)?(?P<cmd>[^:\s][^:\n]*?):"
|
|
57
|
+
r"\s*not found\s*$",
|
|
58
|
+
re.MULTILINE,
|
|
59
|
+
),
|
|
60
|
+
# zsh reversed form:
|
|
61
|
+
# "zsh: command not found: foo"
|
|
62
|
+
re.compile(
|
|
63
|
+
rf"^{_SHELL_NAME_PREFIX}:\s*command not found:\s*(?P<cmd>\S+)\s*$",
|
|
64
|
+
re.MULTILINE,
|
|
65
|
+
),
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _extract_missing_binary(stderr: str) -> str | None:
|
|
70
|
+
"""Pull the offending binary name out of a POSIX shell
|
|
71
|
+
"command not found" stderr.
|
|
72
|
+
|
|
73
|
+
Handles the major shell variants: bash, sh (POSIX), zsh, dash,
|
|
74
|
+
ash, busybox. Each shell's diagnostic format is matched by a distinct regex in
|
|
75
|
+
:data:`_SH_NOT_FOUND_PATTERNS`. The line-anchored shell-name
|
|
76
|
+
prefix prevents false positives from prose containing the
|
|
77
|
+
substring "command not found".
|
|
78
|
+
|
|
79
|
+
Returns the binary name from the LAST match by line
|
|
80
|
+
position (chronologically last shell diagnostic). The "last"
|
|
81
|
+
rule preserves the operator's most-recent failure when
|
|
82
|
+
multiple diagnostics are present — earlier-in-stderr
|
|
83
|
+
diagnostics could be from an earlier subcommand that
|
|
84
|
+
succeeded post-diagnostic (rare but possible in conditional
|
|
85
|
+
pipelines). Returns ``None`` when no pattern matches; caller
|
|
86
|
+
falls back to the full ``test_command`` string.
|
|
87
|
+
"""
|
|
88
|
+
# Collect (start_pos, captured_command) tuples from every
|
|
89
|
+
# pattern. Sort by start_pos descending so we return the
|
|
90
|
+
# last-occurring diagnostic.
|
|
91
|
+
matches: list[tuple[int, str]] = []
|
|
92
|
+
for pattern in _SH_NOT_FOUND_PATTERNS:
|
|
93
|
+
for m in pattern.finditer(stderr):
|
|
94
|
+
matches.append((m.start(), m.group("cmd").strip()))
|
|
95
|
+
if not matches:
|
|
96
|
+
return None
|
|
97
|
+
matches.sort(key=lambda t: t[0], reverse=True)
|
|
98
|
+
return matches[0][1]
|
syncade/transcript.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Claude Code session-transcript parsing.
|
|
2
|
+
|
|
3
|
+
Turns a Claude Code session JSONL into a clean, role-labeled **dialogue text** the
|
|
4
|
+
cold drafter (:mod:`syncade.spec_draft`) reads to manufacture a spec. This is the
|
|
5
|
+
**only** Claude-Code-format-aware component in syncade — the firewall's "front
|
|
6
|
+
door" coupling is quarantined here; the agnostic core never imports it and only
|
|
7
|
+
ever sees the plain text this produces.
|
|
8
|
+
|
|
9
|
+
What is kept vs dropped (the firewall starts here — intent in, "what was built"
|
|
10
|
+
out):
|
|
11
|
+
|
|
12
|
+
- **Kept:** the text of ``type: "user"`` and ``type: "assistant"`` turns, in
|
|
13
|
+
order, each labeled ``User:`` / ``Assistant:``.
|
|
14
|
+
- **Dropped:** ``tool_use`` / ``tool_result`` blocks (those are the actions/
|
|
15
|
+
execution — i.e. *what was built*, which is the diff's job, not intent),
|
|
16
|
+
assistant ``thinking`` blocks (private reasoning, not what the user affirmed),
|
|
17
|
+
``isSidechain`` turns (embedded subagent transcripts), non-dialogue entries
|
|
18
|
+
(``queue-operation`` etc.), and harness-injected wrappers (``system-reminder`` /
|
|
19
|
+
``local-command-caveat`` / ``local-command-stdout`` / the slash-command
|
|
20
|
+
``command-name`` / ``command-message`` / ``command-args`` markers) that are
|
|
21
|
+
never build intent.
|
|
22
|
+
|
|
23
|
+
Pure stdlib (``json`` / ``re`` / ``pathlib``); no syncade imports. The drafter's
|
|
24
|
+
prompt does the *semantic* firewall (forward-looking intent vs backward-looking
|
|
25
|
+
justification); this module only does the structural strip.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
import re
|
|
32
|
+
import sys
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
|
|
35
|
+
# Harness-injected wrapper blocks that appear inside user turns but are never the
|
|
36
|
+
# user's build intent: the auto-injected context/reminder blocks, local-command
|
|
37
|
+
# boilerplate/output, AND slash-command invocation markers. Stripped (content
|
|
38
|
+
# included). (QA finding 2026-06-03: real sessions are full of `/model`,
|
|
39
|
+
# `/compact`, etc. whose `<command-name>/<command-message>/<command-args>` wrappers
|
|
40
|
+
# are meta-noise, not intent about what to build — they were leaking into the
|
|
41
|
+
# drafter's dialogue. A turn that is ONLY a slash command drops out as empty.)
|
|
42
|
+
_HARNESS_TAGS = (
|
|
43
|
+
"system-reminder",
|
|
44
|
+
"local-command-caveat",
|
|
45
|
+
"local-command-stdout",
|
|
46
|
+
"command-name",
|
|
47
|
+
"command-message",
|
|
48
|
+
"command-args",
|
|
49
|
+
)
|
|
50
|
+
_HARNESS_BLOCK_RE = re.compile(
|
|
51
|
+
r"<(" + "|".join(_HARNESS_TAGS) + r")>.*?</\1>",
|
|
52
|
+
re.DOTALL,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class TranscriptError(Exception):
|
|
57
|
+
"""A transcript could not be read or yielded no dialogue (missing/unreadable
|
|
58
|
+
file, or only noise). The CLI surfaces it as a stop-before-the-drafter per
|
|
59
|
+
CLAUDE.md's "Exit-code convention for CLI mode handlers"."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _text_from_content(content: object) -> str:
|
|
63
|
+
"""Extract the human-readable text from a turn's ``message.content``: a plain
|
|
64
|
+
string verbatim, or the ``text`` blocks of a content list (``thinking`` /
|
|
65
|
+
``tool_use`` / ``tool_result`` blocks dropped). Anything else → ``""``."""
|
|
66
|
+
if isinstance(content, str):
|
|
67
|
+
return content
|
|
68
|
+
if isinstance(content, list):
|
|
69
|
+
parts: list[str] = []
|
|
70
|
+
for block in content:
|
|
71
|
+
if isinstance(block, dict) and block.get("type") == "text":
|
|
72
|
+
text = block.get("text")
|
|
73
|
+
if isinstance(text, str):
|
|
74
|
+
parts.append(text)
|
|
75
|
+
return "\n".join(parts)
|
|
76
|
+
return ""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _strip_harness(text: str) -> str:
|
|
80
|
+
"""Remove harness-injected wrapper blocks (see :data:`_HARNESS_TAGS`) and
|
|
81
|
+
trim. What remains is the turn's actual dialogue text."""
|
|
82
|
+
return _HARNESS_BLOCK_RE.sub("", text).strip()
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def parse_transcript(jsonl_path: Path) -> str:
|
|
86
|
+
"""Parse a Claude Code session JSONL into a role-labeled dialogue text.
|
|
87
|
+
|
|
88
|
+
Keeps user + assistant turn text in order (``User:`` / ``Assistant:``),
|
|
89
|
+
dropping tool/thinking blocks, sidechains, non-dialogue entries, and
|
|
90
|
+
harness-injected wrappers. Individual malformed (non-parseable) JSONL lines
|
|
91
|
+
are **skipped, not fatal** — a live session transcript can legitimately have a
|
|
92
|
+
trailing mid-write partial line — BUT if any are skipped a **loud stderr
|
|
93
|
+
warning** is emitted (it survives ``--quiet``): silently dropping intent is the
|
|
94
|
+
failure mode this guards against, so the skip is announced, never silent.
|
|
95
|
+
Raises :class:`TranscriptError` only if the file is missing/unreadable or
|
|
96
|
+
yields no dialogue at all.
|
|
97
|
+
"""
|
|
98
|
+
if not jsonl_path.is_file():
|
|
99
|
+
raise TranscriptError(f"transcript not found: {jsonl_path}")
|
|
100
|
+
try:
|
|
101
|
+
raw = jsonl_path.read_text(encoding="utf-8", errors="replace")
|
|
102
|
+
except OSError as exc:
|
|
103
|
+
raise TranscriptError(f"could not read transcript {jsonl_path}: {exc}") from exc
|
|
104
|
+
|
|
105
|
+
turns: list[str] = []
|
|
106
|
+
skipped_malformed = 0
|
|
107
|
+
for line in raw.splitlines():
|
|
108
|
+
line = line.strip()
|
|
109
|
+
if not line:
|
|
110
|
+
continue
|
|
111
|
+
try:
|
|
112
|
+
entry = json.loads(line)
|
|
113
|
+
except (json.JSONDecodeError, ValueError):
|
|
114
|
+
# Skip a malformed line (robust to a few bad/partial lines, e.g. a
|
|
115
|
+
# live session's trailing mid-write line) — but count it; the skip is
|
|
116
|
+
# announced loudly below, never silent.
|
|
117
|
+
skipped_malformed += 1
|
|
118
|
+
continue
|
|
119
|
+
if not isinstance(entry, dict):
|
|
120
|
+
continue
|
|
121
|
+
kind = entry.get("type")
|
|
122
|
+
if kind not in ("user", "assistant"):
|
|
123
|
+
continue
|
|
124
|
+
if entry.get("isSidechain") is True:
|
|
125
|
+
continue
|
|
126
|
+
message = entry.get("message")
|
|
127
|
+
if not isinstance(message, dict):
|
|
128
|
+
continue
|
|
129
|
+
text = _strip_harness(_text_from_content(message.get("content")))
|
|
130
|
+
if not text:
|
|
131
|
+
continue
|
|
132
|
+
label = "User" if kind == "user" else "Assistant"
|
|
133
|
+
turns.append(f"{label}: {text}")
|
|
134
|
+
|
|
135
|
+
if not turns:
|
|
136
|
+
raise TranscriptError(
|
|
137
|
+
f"no user/assistant dialogue found in transcript {jsonl_path} "
|
|
138
|
+
"(only tool calls, sidechains, or harness noise)"
|
|
139
|
+
)
|
|
140
|
+
if skipped_malformed:
|
|
141
|
+
# Loud + unconditional (bypasses any logger so it survives --quiet, like
|
|
142
|
+
# syncade's deprecation / scope-fallback warnings): a manufactured draft
|
|
143
|
+
# that silently dropped intent is exactly the risk we refuse to ship.
|
|
144
|
+
print(
|
|
145
|
+
f"[syncade] transcript: skipped {skipped_malformed} malformed/unparseable "
|
|
146
|
+
f"line(s) in {jsonl_path} — the manufactured draft may be missing some "
|
|
147
|
+
"dialogue; check the source if the result looks incomplete.",
|
|
148
|
+
file=sys.stderr,
|
|
149
|
+
)
|
|
150
|
+
return "\n\n".join(turns) + "\n"
|
syncade/usage.py
ADDED
|
@@ -0,0 +1,407 @@
|
|
|
1
|
+
"""Per-subprocess token usage + cost (PR-v2-04).
|
|
2
|
+
|
|
3
|
+
Both CLIs emit usage in the envelope syncade already captures; the adapters
|
|
4
|
+
discard it. These pure extractors pull it back out. Claude reports cost directly
|
|
5
|
+
(``total_cost_usd``); codex reports tokens only, so its cost is derived from a
|
|
6
|
+
pricing table (see :func:`priced` + :mod:`syncade.pricing_config`). Missing or
|
|
7
|
+
malformed usage degrades to ``None`` / ``"unknown"`` — never a fabricated number.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import math
|
|
13
|
+
from dataclasses import dataclass, replace
|
|
14
|
+
|
|
15
|
+
from syncade.pricing_config import PricingConfig
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _required_int_field(payload: dict, key: str) -> int | None:
|
|
19
|
+
"""Return a required int field, or None when absent/malformed."""
|
|
20
|
+
value = payload.get(key)
|
|
21
|
+
if type(value) is int and value >= 0:
|
|
22
|
+
return value
|
|
23
|
+
return None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _optional_int_field(payload: dict, key: str) -> int | None:
|
|
27
|
+
"""Return an optional int field, 0 when absent, or None when malformed."""
|
|
28
|
+
if key not in payload:
|
|
29
|
+
return 0
|
|
30
|
+
value = payload.get(key)
|
|
31
|
+
if type(value) is int and value >= 0:
|
|
32
|
+
return value
|
|
33
|
+
return None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _cost_or_none(value: object) -> float | None:
|
|
37
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
38
|
+
cost = float(value)
|
|
39
|
+
if math.isfinite(cost) and cost >= 0:
|
|
40
|
+
return cost
|
|
41
|
+
return None
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _nonnegative_int(value: object) -> bool:
|
|
45
|
+
return type(value) is int and value >= 0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _valid_cost(value: object) -> bool:
|
|
49
|
+
if value is None:
|
|
50
|
+
return True
|
|
51
|
+
return _cost_or_none(value) is not None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _has_valid_numbers(usage: Usage) -> bool:
|
|
55
|
+
return (
|
|
56
|
+
_nonnegative_int(usage.input_tokens)
|
|
57
|
+
and _nonnegative_int(usage.output_tokens)
|
|
58
|
+
and _nonnegative_int(usage.cached_input_tokens)
|
|
59
|
+
and _nonnegative_int(usage.reasoning_output_tokens)
|
|
60
|
+
and _valid_cost(usage.cost_usd)
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class Usage:
|
|
66
|
+
"""Token usage + optional cost for one reviewer / synth / producer call."""
|
|
67
|
+
|
|
68
|
+
model: str
|
|
69
|
+
input_tokens: int
|
|
70
|
+
output_tokens: int
|
|
71
|
+
cached_input_tokens: int = 0
|
|
72
|
+
reasoning_output_tokens: int = 0
|
|
73
|
+
cost_usd: float | None = None
|
|
74
|
+
cost_source: str = "unknown" # "provider" | "estimated" | "unknown"
|
|
75
|
+
auth_mode: str = "unknown" # "subscription" | "api" | "none" | "unknown"
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def total_tokens(self) -> int:
|
|
79
|
+
return self.input_tokens + self.output_tokens + self.reasoning_output_tokens
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def cost_incomplete_tokens(self) -> int:
|
|
83
|
+
"""Tokens whose cost we could NOT establish — 0 when fully priced, else all of them.
|
|
84
|
+
|
|
85
|
+
The live-Usage twin of the persisted ``actor_stats.cost_incomplete_tokens`` column,
|
|
86
|
+
and it MUST use the same predicate the writer does
|
|
87
|
+
(:mod:`syncade.metrics.aggregate`: ``cost_usd is None or cost_source == "unknown"``).
|
|
88
|
+
A retry combines a priced attempt with an unpriced one via :func:`_add_usage`,
|
|
89
|
+
which keeps the partial cost but marks ``cost_source="unknown"``; a `cost_usd is
|
|
90
|
+
None`-only test then calls it fully priced and drops the lower-bound hedge.
|
|
91
|
+
|
|
92
|
+
This exists so ``billing.from_usages`` and ``billing.from_rows`` read ONE rule.
|
|
93
|
+
Having the classifier in one module (PR-v2-24) was worthless while its two entry
|
|
94
|
+
points still disagreed on what "priced" means — which the panel caught.
|
|
95
|
+
"""
|
|
96
|
+
if self.cost_usd is not None and self.cost_source != "unknown":
|
|
97
|
+
return 0
|
|
98
|
+
return self.total_tokens
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def billed_usd(self) -> float | None:
|
|
102
|
+
"""Money that actually left the user's account. ``0.0`` on a subscription.
|
|
103
|
+
|
|
104
|
+
**``cost_usd`` is not spend.** It is an API-EQUIVALENT VALUATION, and it is
|
|
105
|
+
fiction whenever the call rode a subscription — which is the common case, and
|
|
106
|
+
which syncade reported as spend for its entire history:
|
|
107
|
+
|
|
108
|
+
- ``cost_source="estimated"`` (codex) prices tokens from a table. Those calls run
|
|
109
|
+
on a ChatGPT plan unless the user logged in with an API key. Marginal cost: $0.
|
|
110
|
+
- ``cost_source="provider"`` is no better, and that surprised me. ``claude`` emits
|
|
111
|
+
``total_cost_usd`` even on an OAuth subscription session — measured $0.1426 for
|
|
112
|
+
a two-token reply that cost the user nothing. Syncade trusted it *most*.
|
|
113
|
+
|
|
114
|
+
So billing reality is orthogonal to where the number came from, and only
|
|
115
|
+
``auth_mode`` knows it. ``None`` when we cannot tell — never a guess, per I5
|
|
116
|
+
(cost is never fabricated).
|
|
117
|
+
"""
|
|
118
|
+
if self.auth_mode == "subscription":
|
|
119
|
+
return 0.0
|
|
120
|
+
if self.auth_mode == "api":
|
|
121
|
+
return self.cost_usd
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def usage_from_claude_envelope(envelope: dict) -> Usage | None:
|
|
126
|
+
"""Extract usage from a ``claude -p`` terminal ``{"type":"result"}`` envelope.
|
|
127
|
+
|
|
128
|
+
Claude reports ``total_cost_usd`` directly, so ``cost_source="provider"``.
|
|
129
|
+
"""
|
|
130
|
+
usage = envelope.get("usage")
|
|
131
|
+
if not isinstance(usage, dict):
|
|
132
|
+
return None
|
|
133
|
+
model = envelope.get("model") if isinstance(envelope.get("model"), str) else ""
|
|
134
|
+
model_usage = envelope.get("modelUsage")
|
|
135
|
+
if not model and isinstance(model_usage, dict):
|
|
136
|
+
model = next(iter(model_usage), "")
|
|
137
|
+
cost_usd = _cost_or_none(envelope.get("total_cost_usd"))
|
|
138
|
+
# claude's usage.input_tokens counts ONLY uncached input; the full input the
|
|
139
|
+
# model processed also includes cache reads + cache-creation writes. Fold them
|
|
140
|
+
# into input_tokens so total_tokens doesn't drop them (dogfood finding #1);
|
|
141
|
+
# cached_input_tokens keeps just the cache-read subset for the cost split.
|
|
142
|
+
input_tokens = _required_int_field(usage, "input_tokens")
|
|
143
|
+
output_tokens = _required_int_field(usage, "output_tokens")
|
|
144
|
+
cache_read = _optional_int_field(usage, "cache_read_input_tokens")
|
|
145
|
+
cache_creation = _optional_int_field(usage, "cache_creation_input_tokens")
|
|
146
|
+
if (
|
|
147
|
+
input_tokens is None
|
|
148
|
+
or output_tokens is None
|
|
149
|
+
or cache_read is None
|
|
150
|
+
or cache_creation is None
|
|
151
|
+
):
|
|
152
|
+
return None
|
|
153
|
+
return Usage(
|
|
154
|
+
model=model or "",
|
|
155
|
+
input_tokens=input_tokens + cache_read + cache_creation,
|
|
156
|
+
output_tokens=output_tokens,
|
|
157
|
+
cached_input_tokens=cache_read,
|
|
158
|
+
cost_usd=cost_usd,
|
|
159
|
+
cost_source="provider" if cost_usd is not None else "unknown",
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def usage_from_codex_events(events: list) -> Usage | None:
|
|
164
|
+
"""Extract usage from parsed ``codex exec --json`` JSONL events (the last
|
|
165
|
+
``turn.completed``). Tokens only — cost is derived later via :func:`priced`.
|
|
166
|
+
"""
|
|
167
|
+
completed = [
|
|
168
|
+
e
|
|
169
|
+
for e in events
|
|
170
|
+
if isinstance(e, dict)
|
|
171
|
+
and e.get("type") == "turn.completed"
|
|
172
|
+
and isinstance(e.get("usage"), dict)
|
|
173
|
+
]
|
|
174
|
+
if not completed:
|
|
175
|
+
return None
|
|
176
|
+
usage = completed[-1]["usage"]
|
|
177
|
+
input_tokens = _required_int_field(usage, "input_tokens")
|
|
178
|
+
output_tokens = _required_int_field(usage, "output_tokens")
|
|
179
|
+
cached_input_tokens = _optional_int_field(usage, "cached_input_tokens")
|
|
180
|
+
reasoning_output_tokens = _optional_int_field(usage, "reasoning_output_tokens")
|
|
181
|
+
if (
|
|
182
|
+
input_tokens is None
|
|
183
|
+
or output_tokens is None
|
|
184
|
+
or cached_input_tokens is None
|
|
185
|
+
or reasoning_output_tokens is None
|
|
186
|
+
):
|
|
187
|
+
return None
|
|
188
|
+
return Usage(
|
|
189
|
+
model="", # the event carries no model; the caller sets it from config
|
|
190
|
+
input_tokens=input_tokens,
|
|
191
|
+
output_tokens=output_tokens,
|
|
192
|
+
cached_input_tokens=cached_input_tokens,
|
|
193
|
+
reasoning_output_tokens=reasoning_output_tokens,
|
|
194
|
+
cost_usd=None,
|
|
195
|
+
cost_source="unknown",
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def priced(usage: Usage, pricing: PricingConfig) -> Usage:
|
|
200
|
+
"""Fill ``cost_usd`` / ``cost_source``.
|
|
201
|
+
|
|
202
|
+
Provider-reported cost (claude) passes through untouched. Otherwise cost is
|
|
203
|
+
estimated from the pricing table (``cost_source="estimated"``); an unknown model
|
|
204
|
+
stays ``"unknown"`` — never fabricated. Reasoning tokens bill as output (that is
|
|
205
|
+
how providers bill them); cached input bills at ``cached_input_per_mtok`` or, if
|
|
206
|
+
unset, the input rate.
|
|
207
|
+
"""
|
|
208
|
+
if usage.cost_source == "provider":
|
|
209
|
+
if not _has_valid_numbers(usage):
|
|
210
|
+
raise ValueError("usage contains invalid token or cost fields")
|
|
211
|
+
return usage
|
|
212
|
+
if not _has_valid_numbers(usage):
|
|
213
|
+
raise ValueError("usage contains invalid token or cost fields")
|
|
214
|
+
price = pricing.price_for(usage.model)
|
|
215
|
+
if price is None:
|
|
216
|
+
return usage
|
|
217
|
+
fresh_input = max(0, usage.input_tokens - usage.cached_input_tokens)
|
|
218
|
+
cached_rate = (
|
|
219
|
+
price.cached_input_per_mtok
|
|
220
|
+
if price.cached_input_per_mtok is not None
|
|
221
|
+
else price.input_per_mtok
|
|
222
|
+
)
|
|
223
|
+
cost = (
|
|
224
|
+
fresh_input * price.input_per_mtok
|
|
225
|
+
+ usage.cached_input_tokens * cached_rate
|
|
226
|
+
+ (usage.output_tokens + usage.reasoning_output_tokens) * price.output_per_mtok
|
|
227
|
+
) / 1_000_000
|
|
228
|
+
if not math.isfinite(cost) or cost < 0:
|
|
229
|
+
raise ValueError("pricing produced invalid cost")
|
|
230
|
+
return replace(usage, cost_usd=cost, cost_source="estimated")
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _add_usage(left: Usage | None, right: Usage | None) -> Usage | None:
|
|
234
|
+
"""Add usage from multiple paid attempts of the same subprocess actor."""
|
|
235
|
+
if left is None:
|
|
236
|
+
return right
|
|
237
|
+
if right is None:
|
|
238
|
+
return left
|
|
239
|
+
|
|
240
|
+
if left.cost_usd is None and right.cost_usd is None:
|
|
241
|
+
cost_usd = None
|
|
242
|
+
cost_source = "unknown"
|
|
243
|
+
else:
|
|
244
|
+
cost_usd = (left.cost_usd or 0.0) + (right.cost_usd or 0.0)
|
|
245
|
+
if left.cost_usd is None or right.cost_usd is None:
|
|
246
|
+
cost_source = "unknown"
|
|
247
|
+
elif left.cost_source == right.cost_source:
|
|
248
|
+
cost_source = left.cost_source
|
|
249
|
+
else:
|
|
250
|
+
cost_source = "estimated"
|
|
251
|
+
|
|
252
|
+
return Usage(
|
|
253
|
+
model=left.model or right.model,
|
|
254
|
+
input_tokens=left.input_tokens + right.input_tokens,
|
|
255
|
+
output_tokens=left.output_tokens + right.output_tokens,
|
|
256
|
+
cached_input_tokens=left.cached_input_tokens + right.cached_input_tokens,
|
|
257
|
+
reasoning_output_tokens=left.reasoning_output_tokens + right.reasoning_output_tokens,
|
|
258
|
+
cost_usd=cost_usd,
|
|
259
|
+
cost_source=cost_source,
|
|
260
|
+
auth_mode=_merge_auth_mode(left.auth_mode, right.auth_mode),
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _merge_auth_mode(left: str, right: str) -> str:
|
|
265
|
+
"""Combine the auth modes of two paid attempts of the SAME actor.
|
|
266
|
+
|
|
267
|
+
Dropping this is how a RETRY silently loses its billing classification: the
|
|
268
|
+
dispatcher and the synth driver both accumulate attempts through
|
|
269
|
+
:func:`_add_usage`, so a single transient 429 was enough to turn a
|
|
270
|
+
fully-classified cost into an unclassified one. Caught by syncade's own panel,
|
|
271
|
+
unanimously — the cost was right, the label was gone.
|
|
272
|
+
|
|
273
|
+
Both attempts are the same actor resolved from the same config and env, so they
|
|
274
|
+
agree by construction. The merge is therefore defensive rather than clever: a known
|
|
275
|
+
mode beats an unrecorded one; two genuinely different modes degrade to ``"mixed"``,
|
|
276
|
+
which reports as neither billed nor free rather than picking a side.
|
|
277
|
+
"""
|
|
278
|
+
if left == right:
|
|
279
|
+
return left
|
|
280
|
+
known = {"subscription", "api"}
|
|
281
|
+
if left in known and right not in known:
|
|
282
|
+
return left
|
|
283
|
+
if right in known and left not in known:
|
|
284
|
+
return right
|
|
285
|
+
return "mixed"
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def usage_for(
|
|
289
|
+
raw: object,
|
|
290
|
+
provider: str,
|
|
291
|
+
model: str,
|
|
292
|
+
pricing: PricingConfig | None,
|
|
293
|
+
auth_mode: str = "unknown",
|
|
294
|
+
) -> Usage | None:
|
|
295
|
+
"""Extract + price usage from a completed subprocess result, by provider.
|
|
296
|
+
|
|
297
|
+
``raw`` is the ``SubprocessResult`` whose ``.stdout`` holds the provider
|
|
298
|
+
envelope (the adapters already parse it, they just discard the usage). The
|
|
299
|
+
authoritative ``model`` (from config; codex events carry none) is stamped on
|
|
300
|
+
the result. Returns ``None`` when ``pricing`` is ``None`` (usage disabled), or
|
|
301
|
+
for no result / no usage / an unknown provider — usage is best-effort
|
|
302
|
+
telemetry, never load-bearing on the verdict.
|
|
303
|
+
|
|
304
|
+
``auth_mode`` is the RESOLVED mode this call actually ran in (not the declared one),
|
|
305
|
+
and it is what turns ``cost_usd`` from a number into a fact: on a subscription the
|
|
306
|
+
money billed is $0 and the priced figure is an API-equivalent valuation. Defaults to
|
|
307
|
+
``"unknown"``, which reports as neither — cost is never fabricated.
|
|
308
|
+
"""
|
|
309
|
+
if pricing is None:
|
|
310
|
+
return None
|
|
311
|
+
stdout = getattr(raw, "stdout", "") or ""
|
|
312
|
+
if not stdout:
|
|
313
|
+
return None
|
|
314
|
+
if provider == "anthropic":
|
|
315
|
+
try:
|
|
316
|
+
from syncade.adapters.anthropic import _extract_claude_results
|
|
317
|
+
|
|
318
|
+
envelope, _ = _extract_claude_results(stdout)
|
|
319
|
+
u = usage_from_claude_envelope(envelope) if envelope else None
|
|
320
|
+
except Exception: # noqa: BLE001 - usage telemetry must never affect verdicts
|
|
321
|
+
return None
|
|
322
|
+
elif provider == "openai":
|
|
323
|
+
try:
|
|
324
|
+
from syncade.adapters.openai_parsing import _parse_jsonl_events
|
|
325
|
+
|
|
326
|
+
u = usage_from_codex_events(_parse_jsonl_events(stdout))
|
|
327
|
+
except Exception: # noqa: BLE001 - usage telemetry must never affect verdicts
|
|
328
|
+
return None
|
|
329
|
+
else:
|
|
330
|
+
return None
|
|
331
|
+
if u is None:
|
|
332
|
+
return None
|
|
333
|
+
if not u.model:
|
|
334
|
+
u = replace(u, model=model)
|
|
335
|
+
u = replace(u, auth_mode=auth_mode)
|
|
336
|
+
try:
|
|
337
|
+
return priced(u, pricing)
|
|
338
|
+
except Exception: # noqa: BLE001 - pricing is best-effort telemetry
|
|
339
|
+
return None
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def usage_fields(usage: Usage | None) -> dict[str, object]:
|
|
343
|
+
"""The persistence/manifest dict fields for a usage record — all ``None`` when
|
|
344
|
+
usage is absent. Keeps the reviewer / synth / producer manifest builders DRY and
|
|
345
|
+
the metrics aggregator's ``rev.get("tokens")`` reads uniform across entries.
|
|
346
|
+
|
|
347
|
+
``auth_mode`` rides along with the cost so an artifact is self-describing: a run
|
|
348
|
+
persisted today can be re-read years later and still say whether its dollars were
|
|
349
|
+
money or an API-equivalent valuation. Without it on disk, ``--metrics`` would have to
|
|
350
|
+
guess retroactively — and guessing is what this PR exists to stop.
|
|
351
|
+
"""
|
|
352
|
+
if usage is None or not _has_valid_numbers(usage):
|
|
353
|
+
return {"tokens": None, "cost_usd": None, "cost_source": None, "auth_mode": None}
|
|
354
|
+
return {
|
|
355
|
+
"tokens": usage.total_tokens,
|
|
356
|
+
"cost_usd": usage.cost_usd,
|
|
357
|
+
"cost_source": usage.cost_source,
|
|
358
|
+
"auth_mode": usage.auth_mode,
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def usage_from_fields(
|
|
363
|
+
tokens: object,
|
|
364
|
+
cost_usd: object,
|
|
365
|
+
cost_source: object,
|
|
366
|
+
model: str = "",
|
|
367
|
+
auth_mode: object = None,
|
|
368
|
+
) -> Usage | None:
|
|
369
|
+
"""Reconstruct a (partial) Usage from persisted manifest fields, for resume
|
|
370
|
+
rehydration (finding #2). Persistence keeps only the total tokens + cost +
|
|
371
|
+
source; the input/output split doesn't survive, so the total is folded into
|
|
372
|
+
``input_tokens`` — enough for the loop-manifest / metrics, which read only
|
|
373
|
+
:func:`usage_fields`. Returns ``None`` when no usage was recorded.
|
|
374
|
+
|
|
375
|
+
``auth_mode`` MUST be rehydrated with the rest. Omitting it silently rewrote every
|
|
376
|
+
completed round of a resumed run as ``"unknown"``: the dollars came back, the billing
|
|
377
|
+
classification did not, and a fully-classified run became unclassified just by being
|
|
378
|
+
resumed. The artifact on disk was correct; the reload destroyed it. Caught by
|
|
379
|
+
syncade's own panel, unanimously.
|
|
380
|
+
"""
|
|
381
|
+
if tokens is None and cost_usd is None:
|
|
382
|
+
return None
|
|
383
|
+
if type(tokens) is not int or tokens < 0:
|
|
384
|
+
return None
|
|
385
|
+
cost = _cost_or_none(cost_usd)
|
|
386
|
+
return Usage(
|
|
387
|
+
model=model,
|
|
388
|
+
input_tokens=tokens,
|
|
389
|
+
output_tokens=0,
|
|
390
|
+
cost_usd=cost,
|
|
391
|
+
cost_source=str(cost_source) if cost is not None and cost_source else "unknown",
|
|
392
|
+
auth_mode=str(auth_mode) if auth_mode else "unknown",
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _auth_mode(actor: object) -> str:
|
|
397
|
+
"""Resolved auth mode for ``actor``, for stamping onto its cost record.
|
|
398
|
+
|
|
399
|
+
Function-local import: `auth_preflight` imports `config`, which would cycle if this
|
|
400
|
+
module (imported by the adapters) pulled it in at module scope. Same rule the GC
|
|
401
|
+
follows for `orchestrator`.
|
|
402
|
+
"""
|
|
403
|
+
import os
|
|
404
|
+
|
|
405
|
+
from syncade.auth_preflight import resolve_auth_mode
|
|
406
|
+
|
|
407
|
+
return resolve_auth_mode(actor, dict(os.environ)) # type: ignore[arg-type]
|