tony-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tony_cli/agent.py ADDED
@@ -0,0 +1,767 @@
1
+ import argparse
2
+ import json
3
+ import os
4
+ import sys
5
+ import webbrowser
6
+
7
+ import anthropic
8
+ from dotenv import load_dotenv
9
+ from tony_cli import hosted, install
10
+ from tony_cli.layout import parseReview
11
+ from tony_cli.page import renderPage
12
+ from tony_cli.payload import buildPayload, dumpPayload
13
+ from tony_cli.source.local import (
14
+ FAILED, confine, getDiff, globFiles, grepFiles, isDirty, readFile,
15
+ resolveBase, resolveRepo, resolveRev,
16
+ )
17
+
18
+ # The key comes from the environment first, then from ~/.tony/.env — a home
19
+ # that is not inside any repo, so it can never be committed by accident.
20
+ # Never from the repo being reviewed: tony should not be reading secrets out
21
+ # of a project it is also summarising to an API.
22
+ if not os.environ.get("ANTHROPIC_API_KEY"):
23
+ load_dotenv(os.path.expanduser(os.path.join("~", ".tony", ".env")))
24
+
25
+ MISSING_KEY = """\
26
+ tony: ANTHROPIC_API_KEY is not set.
27
+
28
+ tony reviews your diff with Claude, which needs an Anthropic API key.
29
+ Get one at https://console.anthropic.com/settings/keys, then either:
30
+
31
+ export ANTHROPIC_API_KEY=sk-ant-... # this shell only
32
+
33
+ mkdir -p ~/.tony && \\
34
+ echo 'ANTHROPIC_API_KEY=sk-ant-...' >> ~/.tony/.env # every shell
35
+
36
+ A typical review costs well under a dollar; a very large diff costs more."""
37
+
38
+ SYSTEM = """You explain code changes to the developer who is about to own them — often someone who did not type this code, because an AI wrote it. Your job is to build their mental map of what now exists, fast.
39
+
40
+ Never pad, and never say the same thing twice. Explaining what a block of code does is not padding — that is the job. What you must not do is transcribe syntax: "sets `retries` to 3" tells the reader nothing the line did not. Say what the code does when it runs: "tries each upload up to three times before giving up on it and moving to the next". Describe the mechanism, not the motivation — why it was worth doing belongs in `impact`, not here.
41
+
42
+ You have file and search tools. Use them before you write anything — read the changed files in full, and grep for consumers of anything whose shape changed. Never describe code you did not read.
43
+
44
+ Your entire output is ONE fenced ```json block. No prose outside the fence — nothing before it, nothing after it.
45
+
46
+ THE JSON BLOCK
47
+
48
+ {
49
+ "intent": "One sentence. What this change accomplishes, in plain language.",
50
+ "annotations": [
51
+ {"path": "billing/invoices.py", "line": 84, "title": "Idempotent invoice creation",
52
+ "kind": "changed",
53
+ "prev": "Every call inserted a new row unconditionally, so a client that retried a timed-out request created a second invoice.",
54
+ "now": "Looks up the caller-supplied idempotency key first and returns the existing invoice when the key has been seen before.",
55
+ "impact": "Retries stop double-billing customers. Anything that counted invoice rows to measure volume now sees fewer of them."},
56
+ {"path": "billing/models.py", "line": 31, "title": "Idempotency key column",
57
+ "kind": "added",
58
+ "now": "The unique column the lookup depends on. Existing rows keep NULL, which the unique index permits, so old data needs no backfill."}
59
+ ],
60
+ "risks": [
61
+ {"path": "billing/invoices.py", "line": 90,
62
+ "text": "The key lookup and the insert are not one transaction, so two simultaneous retries can still race past each other and both insert."}
63
+ ]
64
+ }
65
+
66
+ ANNOTATIONS — these are the entire walkthrough. Everything the reader learns, they learn here.
67
+
68
+ COVERAGE — this is the rule that matters most. Every hunk in the diff must be explained. Walk the diff hunk by hunk and account for all of it. If one hunk contains several distinct changes, write several annotations against it. Leaving a hunk unannotated is a failure, not concision. The only things you may skip are lockfiles, generated code, and binary assets. There should never be blocks of code that aren't annotated. The developer reading through needs everything to be annotated so that they could read end to end and understand. This is an alternative to reading code.
69
+
70
+ MIRRORED FILES — repositories often keep two copies of the same code: a sync and an async version of one module, a generated client beside its source, the same fix applied to several platform-specific copies. When two or more changed files receive substantively the same change, do NOT explain it twice.
71
+
72
+ Pick the copy a reader is most likely to open — the sync one, the hand-written one, the one the tests import — and annotate it in full. For each remaining copy, emit exactly ONE annotation:
73
+
74
+ {"path": "src/ahttpx/_parsers.py", "line": 227, "title": "Same change, async copy",
75
+ "kind": "added",
76
+ "now": "Mirrors src/httpx/_parsers.py line for line, with async/await. Read the annotations there."}
77
+
78
+ Anchor it at the first changed line in that file. Do not restate the explanation, and do not write a `prev` or `impact` for it. Two files count as mirrors when the change is the same idea in both, even if the syntax differs — an `async def` against a `def`, an `await` against a plain call. They are NOT mirrors merely because both were touched by the same PR.
79
+
80
+ Assume the reader did not write this code and cannot necessarily read this syntax fluently. Do not assume they know the language's idioms, the framework's conventions, or what any given API call does. When code does something non-obvious — a hook, a ref, a directive, a lifecycle behaviour, an operator whose meaning is not literal — say what it does mechanically, in the same sentence, without a detour.
81
+
82
+ FIELDS
83
+ - `path` is repo-relative. `line` is a line number in the NEW file that the annotation sits above — the first line of the code it describes. For a pure deletion, use the line where the removed code used to begin.
84
+ - `title`: at most six words. A label, not a sentence.
85
+ - `kind`: one of "added", "changed", "removed".
86
+ - "added" — this code is new; nothing was replaced.
87
+ - "changed" — this code replaces behaviour that already existed.
88
+ - "removed" — this code is gone and nothing took its place.
89
+ - `now`: what this code does when it runs, mechanically, in plain language — what the function does, what the loop iterates over and what it does to each item, what a condition decides. Enough that someone who cannot read this language fluently does not have to. As many sentences as the code needs and no more; a three-line assignment needs one, a twenty-line loop may need four. Required for "added" and "changed". Omit for "removed".
90
+ - `prev`: how it worked before — the actual old mechanism, not "it did not exist". One or two sentences. Required for "changed" and "removed". NEVER include it for "added", and never invent it: if you did not read the old version, go read it before writing the annotation.
91
+ - `impact`: what the difference means in practice — what now behaves differently for a user, a caller, or a build. This is the only field for consequences; keep them out of `now`. One or two sentences. Required for "changed" and "removed". Omit for "added".
92
+
93
+ The reader reads `prev`, `now`, and `impact` as three separate panes, so each must stand alone. Do not write `now` as a continuation of `prev`, and do not repeat the same sentence across two fields.
94
+
95
+ Order annotations by file, then by line.
96
+
97
+ IMPACTS — files this change reaches that are NOT in the diff.
98
+
99
+ A diff shows what was edited. It cannot show what breaks. After you have written the annotations, grep for every consumer of anything whose shape changed — a signature, a prop, an export, a schema, a config key, a route, an environment variable — and record each place that now behaves differently.
100
+
101
+ {"impacts": [
102
+ {"symbol": "createInvoice", "fromPath": "billing/invoices.py",
103
+ "path": "dashboard/src/api/billing.ts", "line": 112, "kind": "behavior-change",
104
+ "why": "Calls the endpoint without an idempotency key, so it now takes the new code path where the server generates one per request."}
105
+ ]}
106
+
107
+ - `symbol` is the changed thing this file depends on. It must match something you described in an annotation, so the two can be linked.
108
+ - `fromPath` is the file the symbol lives in — one of the changed files.
109
+ - `path` and `line` locate the consuming code. `line` is the line in that file that depends on the symbol. Verify it by reading the file; never guess.
110
+ - `kind` is one of:
111
+ - "breaks" — this will fail to compile, or throw, or 404. Something is now wrong.
112
+ - "behavior-change" — it still works, but does something different than before.
113
+ - "compatible" — it consumes the changed thing and is fine. Say so explicitly; this is the common case and the reader needs to know you checked.
114
+ - `why` is one sentence: what this file does with the changed thing, and what is different for it now. Not a restatement of the annotation.
115
+
116
+ RULES
117
+ - Only files NOT present in the diff. A file that was edited is covered by its annotations.
118
+ - One entry per consuming site. If a file uses the changed symbol in three places and all three are affected, that is three entries.
119
+ - Report "compatible" consumers too. A blast radius that lists only problems teaches the reader nothing about coverage, and they cannot tell "no impact" from "did not look".
120
+ - Never list a file you did not read. If a consumer might exist that you could not confirm, say so in `risks` instead of inventing an impact.
121
+ - An empty list is valid when nothing outside the diff depends on what changed.
122
+
123
+ RISKS — genuine ones only, and only where you can name the failing path. These are shown behind a toggle and are not the point of the output.
124
+ - `text` is one sentence. What breaks, concretely.
125
+ - Anchor to `path` and `line` when the risk lives in the diff. Omit both when it does not.
126
+ - An empty list is a valid answer. Never manufacture risks, and never restate an annotation as a risk.
127
+
128
+ WALKTHROUGHS — a steppable trace of what the code DOES at runtime.
129
+
130
+ The reader has no working model of how this program executes. They cannot get one from reading the code, and a static picture of boxes and arrows will not give them one either. What builds it is following a single concrete scenario, one step at a time, watching state change.
131
+
132
+ Write at most two. Zero is correct when the change has no runtime behaviour — a copy edit, a rename, a config bump. Never write one that merely restates the annotations.
133
+
134
+ Put them in the `walkthroughs` array:
135
+
136
+ {"walkthroughs": [
137
+ {
138
+ "title": "Resuming a crashed export",
139
+ "trigger": "You run `export --resume` after the previous run died partway",
140
+ "whatChanged": "Before this diff a resumed export started over from the first record instead of picking up where it stopped.",
141
+ "steps": [
142
+ {"say": "The CLI reads the checkpoint file the previous run left behind, which records the last record it managed to write.",
143
+ "path": "exporter/checkpoint.py", "lines": [22, 30],
144
+ "state": {"lastWritten": "4180", "cursor": "0"},
145
+ "phase": "new"},
146
+ {"say": "The database cursor opens at that position instead of at zero, so nothing already exported is fetched a second time.",
147
+ "path": "exporter/run.py", "lines": [57, 61],
148
+ "state": {"cursor": "0 -> 4180"},
149
+ "phase": "changed"}
150
+ ]
151
+ }
152
+ ]}
153
+
154
+ FIELDS
155
+ - `title`: at most six words, naming the scenario.
156
+ - `trigger`: one sentence describing what the user or system does to start it. Concrete and physical — "you click X", "a link is pasted into Slack", "the page finishes loading" — never "the function is invoked".
157
+ - `whatChanged`: ONE sentence naming what this diff altered about THIS flow specifically, written so it makes sense before the reader has stepped through anything. This is the reason the walkthrough exists — if you cannot write it, the walkthrough does not belong.
158
+ - `steps`: in execution order. Between three and seven. Fewer than three is not a trace; more than seven is a lecture.
159
+ - `say`: ONE sentence, plain language, about what happens at this step and why. No jargon unless you define it in the same clause. Do not narrate the syntax — explain the effect.
160
+ - `path` and `lines`: `[start, end]` in the CURRENT file, the code responsible for this step. The reader is shown these exact lines, read from disk, so verify them. `lines` may be omitted for a step that happens outside the codebase (a browser behaviour, a third-party fetch), in which case omit `path` too.
161
+ - `state`: a small map of what is true at this step. Use `"before -> after"` when the step changes something. Keep to three entries or fewer, and name things as the code names them so the reader can connect the two. Omit when nothing observable changes.
162
+ - `phase`: "same" if this step happened before this diff too, "new" if the change introduced it, "changed" if the step existed but now behaves differently, "removed" if the change deleted it. Include "removed" steps in execution order where they used to run — seeing what no longer happens is how the reader understands the change.
163
+
164
+ RULES
165
+ - One scenario per walkthrough. Do not merge two unrelated flows.
166
+ - Trace what the code actually does. Read the files. Never guess at a line range or invent a step.
167
+ - Prefer the scenario the change most affects. If the diff alters what happens when a link is shared, trace a link being shared.
168
+ - Plain language throughout. The reader does not know what a hook, a ref, a prop, or a build step is unless you tell them in passing.
169
+
170
+ RULES
171
+ - Valid JSON. No comments, no trailing commas. Escape newlines inside strings.
172
+ - Skip lockfiles, generated code, and binary assets. Do not annotate them.
173
+ - If the whole diff is trivial, return the intent, an empty annotations array, and stop.
174
+ """
175
+
176
+ GET_DIFF_TOOL = {
177
+ "name": "getDiff",
178
+ "description": (
179
+ "Gets the diffs for a local repository. Returns the diff between the base "
180
+ "branch and head as text. Accepts any path inside the repo - it resolves to "
181
+ "the repo root automatically. Returns an error string if the path is not a "
182
+ "git repository or the revisions are invalid."
183
+ ),
184
+ "input_schema": {
185
+ "type": "object",
186
+ "properties": {
187
+ "repoPath": {
188
+ "type": "string",
189
+ "description": "Absolute path to the repository, or any directory inside it.",
190
+ },
191
+ "base": {
192
+ "type": "string",
193
+ "description": "Branch to diff against. Omit to use the repo's default branch.",
194
+ },
195
+ "head": {
196
+ "type": "string",
197
+ "description": "Revision to diff. Defaults to HEAD.",
198
+ },
199
+ },
200
+ "required": ["repoPath"],
201
+ },
202
+ }
203
+ READ_TOOL = {
204
+ "name": "readFile",
205
+ "description": (
206
+ "Read a file's full contents. Use this to verify what a diff hunk only hints at. "
207
+ "Returns the text, or an error string if the path is not a readable text file. "
208
+ "Long files are truncated with a marker."
209
+ ),
210
+ "input_schema": {
211
+ "type": "object",
212
+ "properties": {
213
+ "path": {
214
+ "type": "string",
215
+ "description": "Absolute path to the file.",
216
+ },
217
+ },
218
+ "required": ["path"],
219
+ },
220
+ }
221
+
222
+ GLOB_TOOL = {
223
+ "name": "globFiles",
224
+ "description": (
225
+ "Find files by name pattern. Matches against both the path relative to root and "
226
+ "the bare filename. Skips .git, node_modules, and build output. Returns one "
227
+ "relative path per line."
228
+ ),
229
+ "input_schema": {
230
+ "type": "object",
231
+ "properties": {
232
+ "pattern": {
233
+ "type": "string",
234
+ "description": "Glob pattern, e.g. '*.py', '*.go', or 'src/**/*.ts'.",
235
+ },
236
+ "root": {
237
+ "type": "string",
238
+ "description": "Absolute path of the directory to search from.",
239
+ },
240
+ },
241
+ "required": ["pattern", "root"],
242
+ },
243
+ }
244
+
245
+ GREP_TOOL = {
246
+ "name": "grepFiles",
247
+ "description": (
248
+ "Search file contents for a Python regex. This is how you find callers, imports, "
249
+ "and references to a changed symbol. Skips .git, node_modules, and build output. "
250
+ "Returns matching lines as 'relative/path:lineno: text'."
251
+ ),
252
+ "input_schema": {
253
+ "type": "object",
254
+ "properties": {
255
+ "pattern": {
256
+ "type": "string",
257
+ "description": "Python regular expression to search for.",
258
+ },
259
+ "root": {
260
+ "type": "string",
261
+ "description": "Absolute path of the directory to search in.",
262
+ },
263
+ },
264
+ "required": ["pattern", "root"],
265
+ },
266
+ }
267
+
268
+ ALL_TOOLS = [GET_DIFF_TOOL, READ_TOOL, GLOB_TOOL, GREP_TOOL]
269
+
270
+ TOOLS = {
271
+ "getDiff": getDiff,
272
+ "readFile": readFile,
273
+ "globFiles": globFiles,
274
+ "grepFiles": grepFiles,
275
+ }
276
+
277
+ # Which argument of each tool names a path that must stay inside the repo.
278
+ PATH_ARGS = {"getDiff": "repoPath", "readFile": "path", "globFiles": "root", "grepFiles": "root"}
279
+
280
+
281
+ def runTool(name, toolInput, root):
282
+ func = TOOLS.get(name)
283
+ if func is None:
284
+ return f"unknown tool: {name}"
285
+
286
+ # Tool inputs come from the model, and the model reads untrusted repo
287
+ # content. Whatever it asks for, it gets nothing outside the repo under
288
+ # review — there is no legitimate reason to read anywhere else, and the
289
+ # results of these calls are sent off-machine in the next API request.
290
+ arg = PATH_ARGS.get(name)
291
+ if arg and arg in toolInput:
292
+ inside = confine(root, str(toolInput[arg]))
293
+ if inside is None:
294
+ return f"refused: {toolInput[arg]} is outside the repository under review"
295
+ toolInput = {**toolInput, arg: inside}
296
+
297
+ try:
298
+ return func(**toolInput)
299
+ except Exception as e:
300
+ return f"tool error:{e}"
301
+
302
+ DEFAULT_MODEL = "claude-opus-5"
303
+ # Streaming, so this is not bounded by how long one HTTP response may take.
304
+ # A review of a large diff has to annotate every hunk, and the old 16k ceiling
305
+ # truncated exactly the reviews that needed the most room.
306
+ DEFAULT_MAX_TOKENS = 64000
307
+
308
+ # How many times the model may come back for more tools before we stop paying.
309
+ # A review of a large diff settles well inside this; a model that has started
310
+ # looping never does, and every turn costs another request.
311
+ MAX_TURNS = 30
312
+
313
+
314
+ def parseRange(spec):
315
+ """'main...HEAD' -> ('main', 'HEAD'). 'main' -> ('main', 'HEAD'). None -> (None, 'HEAD')."""
316
+ if not spec:
317
+ return None, "HEAD"
318
+ if "..." in spec:
319
+ base, _, head = spec.partition("...")
320
+ return base or None, head or "HEAD"
321
+ return spec, "HEAD"
322
+
323
+
324
+ def buildPrompt(repoPath, base, head):
325
+ prompt = f"Review the diff for the repository at {repoPath}"
326
+ if base:
327
+ prompt += f", diffing {base}...{head}"
328
+ return prompt
329
+
330
+
331
+ def review(repoPath, base=None, head="HEAD", model=DEFAULT_MODEL,
332
+ maxTokens=DEFAULT_MAX_TOKENS, verbose=False):
333
+ """Run the review loop.
334
+
335
+ Returns (code, text, diff): 0 on success, non-zero on failure. `text` is the
336
+ model's final message; `diff` is the diff the model actually reviewed, kept
337
+ from its own getDiff call so the page renders the same change it read.
338
+ """
339
+ client = anthropic.Anthropic()
340
+ messages = [{"role": "user", "content": buildPrompt(repoPath, base, head)}]
341
+ diff = ""
342
+ text = ""
343
+
344
+ for _ in range(MAX_TURNS):
345
+ try:
346
+ # Streaming because `max_tokens` is large: a non-streaming request
347
+ # has to deliver the whole answer within one HTTP timeout, and a
348
+ # long review does not. It costs nothing and changes no output.
349
+ #
350
+ # `cache_control` caches the longest stable prefix — the system
351
+ # prompt, the tool list, and the diff, which is usually most of the
352
+ # request. Every turn after the first re-reads it at a tenth of the
353
+ # price instead of paying full freight to send it again.
354
+ with client.messages.stream(
355
+ model=model,
356
+ max_tokens=maxTokens,
357
+ system=SYSTEM,
358
+ messages=messages,
359
+ tools=ALL_TOOLS,
360
+ cache_control={"type": "ephemeral"},
361
+ ) as stream:
362
+ response = stream.get_final_message()
363
+ except anthropic.APIStatusError as e:
364
+ print(f"tony: API error {e.status_code}: {e.message}", file=sys.stderr)
365
+ return 1, text, diff
366
+ except anthropic.APIConnectionError as e:
367
+ print(f"tony: could not reach the API: {e}", file=sys.stderr)
368
+ return 1, text, diff
369
+
370
+ messages.append({"role": "assistant", "content": response.content})
371
+
372
+ if response.stop_reason == "max_tokens":
373
+ print(
374
+ f"tony: response hit the {maxTokens}-token limit and was cut off. "
375
+ "Re-run with a larger --max-tokens.",
376
+ file=sys.stderr,
377
+ )
378
+ text = "\n".join(b.text for b in response.content if b.type == "text")
379
+ return 1, text, diff
380
+
381
+ if response.stop_reason != "tool_use":
382
+ text = "\n".join(b.text for b in response.content if b.type == "text")
383
+ return 0, text, diff
384
+
385
+ results = []
386
+ if verbose:
387
+ u = response.usage
388
+ print(
389
+ f"[usage] in={u.input_tokens} out={u.output_tokens} "
390
+ f"cache_write={getattr(u, 'cache_creation_input_tokens', 0)} "
391
+ f"cache_read={getattr(u, 'cache_read_input_tokens', 0)}",
392
+ file=sys.stderr,
393
+ )
394
+
395
+ for block in response.content:
396
+ if block.type == "tool_use":
397
+ if verbose:
398
+ print(f"[{block.name}] {block.input}", file=sys.stderr)
399
+ result = runTool(block.name, block.input, repoPath)
400
+ if block.name == "getDiff" and result.lstrip().startswith("diff --git "):
401
+ diff = result
402
+ results.append({
403
+ "type": "tool_result",
404
+ "tool_use_id": block.id,
405
+ "content": result,
406
+ })
407
+
408
+ messages.append({"role": "user", "content": results})
409
+
410
+ print(
411
+ f"tony: the review kept asking for more files after {MAX_TURNS} rounds and "
412
+ "was stopped. Try a narrower range.",
413
+ file=sys.stderr,
414
+ )
415
+ return 1, text, diff
416
+
417
+
418
+ def checkWorkingTree(root, head):
419
+ """Every line of code tony shows is read from the working tree.
420
+
421
+ Annotations, walkthrough code windows, and whole blast-radius files are all
422
+ numbered against the NEW side of the diff. If the working tree is not that
423
+ revision, those line numbers point at different code and the page is wrong
424
+ in a way the reader cannot see. Returns a problem string, or None if clean.
425
+ """
426
+ wanted = resolveRev(root, head)
427
+ if wanted is None:
428
+ return f"cannot resolve revision {head!r}"
429
+
430
+ if wanted != resolveRev(root, "HEAD"):
431
+ return (
432
+ f"{head} is not checked out, so the code on disk is not the code being "
433
+ f"reviewed and every line tony showed you could be wrong.\n"
434
+ f" Run `git checkout {head}` first, or pass --stale to accept "
435
+ "mismatched line numbers."
436
+ )
437
+ if isDirty(root):
438
+ return (
439
+ "the working tree has uncommitted changes. tony reviews committed work, "
440
+ "and shows code straight from disk —\n uncommitted edits would make "
441
+ "those lines lie. Commit or stash first (`git stash` restores with "
442
+ "`git stash pop`),\n or pass --stale to accept possibly-wrong lines."
443
+ )
444
+ return None
445
+
446
+
447
+ def viewerFixture():
448
+ """`web/src/fixtures/local.json` in tony's own checkout, if there is one.
449
+
450
+ Only exists when tony is installed from source (`pip install -e .`), which is
451
+ the case while the viewer is being built. Returns None for a plain install.
452
+
453
+ Deliberately not `review.json`, which is tracked and ships as the homepage
454
+ demo: this file holds a review of whatever repo `--viewer` was run in, so
455
+ it is gitignored and must stay that way.
456
+ """
457
+ here = os.path.dirname(os.path.abspath(__file__))
458
+ path = os.path.normpath(
459
+ os.path.join(here, "..", "..", "web", "src", "fixtures", "local.json")
460
+ )
461
+ return path if os.path.isdir(os.path.dirname(path)) else None
462
+
463
+
464
+ def reportPath(repoPath, base, head, ext="html"):
465
+ """Where this review lands: `.tony/<base>...<head>.html` inside the repo."""
466
+ stamp = f"{base or 'default'}...{head}".replace("/", "-").replace(" ", "-")
467
+ return os.path.join(repoPath, ".tony", f"{stamp}.{ext}")
468
+
469
+
470
+ def saveRaw(path, text, diff):
471
+ """Keep the model's answer and the diff it read, so the page can be rebuilt for free."""
472
+ with open(path, "w", encoding="utf-8") as fh:
473
+ json.dump({"review": text, "diff": diff}, fh)
474
+
475
+
476
+ def loadRaw(path):
477
+ """The saved review and diff, or (None, None) if the file is not one."""
478
+ try:
479
+ with open(path, encoding="utf-8") as fh:
480
+ saved = json.load(fh)
481
+ return saved["review"], saved["diff"]
482
+ except (OSError, ValueError, KeyError, TypeError):
483
+ return None, None
484
+
485
+
486
+ def writeReport(path, page):
487
+ outDir = os.path.dirname(path)
488
+ os.makedirs(outDir, exist_ok=True)
489
+ # Reviews are disposable and belong to the reader, not the repo's history.
490
+ ignore = os.path.join(outDir, ".gitignore")
491
+ if not os.path.exists(ignore):
492
+ with open(ignore, "w", encoding="utf-8") as fh:
493
+ fh.write("*\n")
494
+ with open(path, "w", encoding="utf-8") as fh:
495
+ fh.write(page)
496
+
497
+
498
+ def main(argv=None):
499
+ argv = sys.argv[1:] if argv is None else argv
500
+
501
+ # Account commands, dispatched before argparse so `tony login` does not
502
+ # read as a repository path.
503
+ if argv[:1] == ["login"]:
504
+ return hosted.login()
505
+ if argv[:1] == ["logout"]:
506
+ return hosted.logout()
507
+ if argv[:1] == ["whoami"]:
508
+ return hosted.whoami()
509
+ if argv[:1] == ["unpublish"]:
510
+ if len(argv) != 2:
511
+ print("usage: tony unpublish <id>", file=sys.stderr)
512
+ return 2
513
+ return hosted.unpublish(argv[1])
514
+ if argv[:1] == ["update"]:
515
+ return install.update(argv[1:])
516
+ if argv[:1] == ["uninstall"]:
517
+ from tony_cli.uninstall import uninstall
518
+ return uninstall(argv[1:])
519
+
520
+ parser = argparse.ArgumentParser(
521
+ prog="tony",
522
+ description="Review a git diff for impact, not line-by-line.",
523
+ epilog=(
524
+ "commands:\n"
525
+ " tony <path> [BASE...HEAD] review a diff — the default\n"
526
+ " tony login sign in to the tony site\n"
527
+ " tony logout revoke this machine's token\n"
528
+ " tony whoami who this machine is signed in as\n"
529
+ " tony unpublish <id> take one published review down\n"
530
+ " tony update update to the latest release\n"
531
+ " tony uninstall delete tony, its key, and every saved review\n"
532
+ " tony --version what is installed\n"
533
+ "\n"
534
+ " https://tony-cli.com\n"
535
+ ),
536
+ formatter_class=argparse.RawDescriptionHelpFormatter,
537
+ )
538
+ parser.add_argument(
539
+ "path", nargs="?", default=os.getcwd(),
540
+ help="Repository path, or any directory inside it (default: current directory).",
541
+ )
542
+ parser.add_argument(
543
+ "range", nargs="?", default=None, metavar="BASE...HEAD",
544
+ help=(
545
+ "What to diff, in git's range syntax: 'main...HEAD'. A bare branch name "
546
+ "means 'that branch...HEAD'. Omit to use the repo's default branch."
547
+ ),
548
+ )
549
+ parser.add_argument(
550
+ "--model", "-m", default=DEFAULT_MODEL,
551
+ help=f"Model to review with (default: {DEFAULT_MODEL}).",
552
+ )
553
+ parser.add_argument(
554
+ "--max-tokens", type=int, default=DEFAULT_MAX_TOKENS, dest="maxTokens",
555
+ help=f"Response token ceiling (default: {DEFAULT_MAX_TOKENS}).",
556
+ )
557
+ parser.add_argument(
558
+ "--verbose", "-v", action="store_true",
559
+ help="Log every tool call to stderr, to see what the review actually checked.",
560
+ )
561
+ parser.add_argument(
562
+ "--json", action="store_true", dest="asJson",
563
+ help="Print the raw JSON review to stdout instead of writing a page.",
564
+ )
565
+ parser.add_argument(
566
+ "--no-open", action="store_false", dest="openPage",
567
+ help="Write the page but do not open it in a browser.",
568
+ )
569
+ parser.add_argument(
570
+ "--local", action="store_true",
571
+ help="Keep the review on this machine. Skips publishing, needs no account, "
572
+ "and opens the self-contained page in .tony/ instead of a link.",
573
+ )
574
+ parser.add_argument(
575
+ "--payload", nargs="?", const=True, default=False, metavar="PATH",
576
+ help="Write the upload payload (windowed source, no absolute paths). "
577
+ "Defaults to sitting next to the page; give a PATH to put it elsewhere.",
578
+ )
579
+ parser.add_argument(
580
+ "--viewer", action="store_true",
581
+ help="Write the payload straight into the local Astro viewer and reload it there.",
582
+ )
583
+ parser.add_argument(
584
+ "--stale", action="store_true",
585
+ help="Review a revision that is not checked out. The code shown comes from the "
586
+ "working tree, so line numbers may not match the diff.",
587
+ )
588
+ parser.add_argument(
589
+ "--replay", action="store_true",
590
+ help="Rebuild the page from the last saved review for this range, without calling the API.",
591
+ )
592
+ parser.add_argument(
593
+ "--version", action="version",
594
+ version=f"tony {install.installedVersion() or 'unknown'}",
595
+ )
596
+ args = parser.parse_args(argv)
597
+
598
+ if args.maxTokens < 1:
599
+ print(f"tony: --max-tokens must be at least 1, not {args.maxTokens}.",
600
+ file=sys.stderr)
601
+ return 2
602
+
603
+ # Checked before the review rather than after: --viewer is asking for the
604
+ # payload to land somewhere that does not exist on a plain install, and
605
+ # finding that out afterwards means having paid for a review to be told so.
606
+ if args.viewer and not viewerFixture():
607
+ print("tony: --viewer needs tony installed from source — there is no "
608
+ "viewer to load into.", file=sys.stderr)
609
+ return 2
610
+
611
+ base, head = parseRange(args.range)
612
+ repoPath = os.path.abspath(args.path)
613
+
614
+ try:
615
+ root = resolveRepo(repoPath)
616
+ except ValueError as e:
617
+ print(f"tony: {e}", file=sys.stderr)
618
+ return 2
619
+
620
+ # Resolve the base here rather than leaving it to the model, so the run is
621
+ # reproducible and the reader is told what it was compared against.
622
+ if base is None:
623
+ try:
624
+ base = resolveBase(root)
625
+ except ValueError as e:
626
+ print(f"tony: {e}", file=sys.stderr)
627
+ return 2
628
+
629
+ rawPath = reportPath(root, base, head, ext="json")
630
+
631
+ problem = checkWorkingTree(root, head)
632
+ if problem:
633
+ if args.stale:
634
+ print(f"tony: warning — {problem.splitlines()[0]}", file=sys.stderr)
635
+ else:
636
+ print(f"tony: {problem}", file=sys.stderr)
637
+ return 2
638
+
639
+ # Reviews publish by default, so the login is checked before the review
640
+ # runs — discovering it afterwards would waste an API call the user paid for.
641
+ # `--json` prints and exits without ever uploading, so it is as offline as
642
+ # `--local` and must not be gated on an account it will not use.
643
+ offline = args.local or args.asJson
644
+ if not offline and not hosted.savedToken():
645
+ print(
646
+ "tony: you need to sign in before publishing a review.\n\n"
647
+ " tony login # once, with GitHub\n\n"
648
+ " Or keep this one on your machine and skip the account:\n\n"
649
+ f" tony {args.path} {args.range or ''} --local".rstrip() + "\n",
650
+ file=sys.stderr,
651
+ )
652
+ return 2
653
+
654
+ if args.replay:
655
+ if not os.path.exists(rawPath):
656
+ print(f"tony: nothing saved for this range at {rawPath}", file=sys.stderr)
657
+ return 2
658
+ text, diff = loadRaw(rawPath)
659
+ if text is None:
660
+ print(f"tony: the saved review at {rawPath} could not be read. "
661
+ "Re-run without --replay.", file=sys.stderr)
662
+ return 2
663
+ else:
664
+ # An empty range is the most common harmless mistake — a branch already
665
+ # merged, or a typo'd base. Catch it here rather than after paying for a
666
+ # review of nothing.
667
+ probe = getDiff(root, base, head)
668
+ if probe.startswith(FAILED):
669
+ print(f"tony: {probe}", file=sys.stderr)
670
+ return 2
671
+ if not probe.strip():
672
+ print(
673
+ f"tony: {base}...{head} has no changes to review.\n"
674
+ " Check the range — a branch that is already merged shows "
675
+ "nothing against its base.",
676
+ file=sys.stderr,
677
+ )
678
+ return 2
679
+
680
+ if not os.environ.get("ANTHROPIC_API_KEY"):
681
+ print(MISSING_KEY, file=sys.stderr)
682
+ return 2
683
+
684
+ print(f"tony: reviewing {base or 'default branch'}...{head} in {root}", file=sys.stderr)
685
+ code, text, diff = review(root, base, head, args.model, args.maxTokens, args.verbose)
686
+ if code != 0:
687
+ return code
688
+
689
+ if args.asJson:
690
+ print(text)
691
+ return 0
692
+
693
+ if not diff:
694
+ # The model answered without ever pulling the diff, so there is nothing to
695
+ # lay a page out against. The review itself is still worth handing back.
696
+ print("tony: no diff was fetched during the review — printing raw output.",
697
+ file=sys.stderr)
698
+ print(text)
699
+ return 1
700
+
701
+ if not parseReview(text):
702
+ # The model's answer had no usable JSON block — usually a response that
703
+ # hit the token ceiling. A page built from it would show a bare diff and
704
+ # look like a successful run. Fail loudly and keep the raw text instead.
705
+ keep = reportPath(root, base, head, ext="raw.txt")
706
+ os.makedirs(os.path.dirname(keep), exist_ok=True)
707
+ with open(keep, "w", encoding="utf-8") as fh:
708
+ fh.write(text)
709
+ print(
710
+ "tony: the review came back unparseable, so no page was written.\n"
711
+ f" The raw output is at {keep}. If it looks cut off, re-run "
712
+ "with a larger --max-tokens.",
713
+ file=sys.stderr,
714
+ )
715
+ return 1
716
+
717
+ rangeLabel = f"{base or 'default'}...{head}"
718
+ payloadJson = dumpPayload(buildPayload(text, diff, root, rangeLabel))
719
+
720
+ path = reportPath(root, base, head)
721
+ page = renderPage(payloadJson, title=f"tony — {os.path.basename(root)}")
722
+ writeReport(path, page)
723
+ if not args.replay:
724
+ saveRaw(rawPath, text, diff)
725
+
726
+ if args.payload or args.viewer:
727
+ targets = []
728
+ if args.payload is True:
729
+ targets.append(reportPath(root, base, head, ext="payload.json"))
730
+ elif args.payload:
731
+ targets.append(os.path.abspath(args.payload))
732
+ if args.viewer:
733
+ targets.append(viewerFixture())
734
+
735
+ for target in targets:
736
+ os.makedirs(os.path.dirname(target), exist_ok=True)
737
+ with open(target, "w", encoding="utf-8") as fh:
738
+ fh.write(payloadJson)
739
+ print(f"tony: {target}")
740
+
741
+ if args.viewer:
742
+ print("tony: loaded into the viewer — npm run dev in web/, then refresh.")
743
+
744
+ # The local page is written either way: it costs nothing, works offline, and
745
+ # is the fallback when publishing fails so a paid-for review is never lost.
746
+ target = f"file://{path}"
747
+
748
+ if not args.local:
749
+ url, problem = hosted.publish(
750
+ payloadJson, repo=os.path.basename(root), rangeLabel=rangeLabel,
751
+ )
752
+ if problem:
753
+ print(f"tony: could not publish — {problem}", file=sys.stderr)
754
+ print(f"tony: the review is still here: {path}", file=sys.stderr)
755
+ else:
756
+ print(url)
757
+ target = url
758
+
759
+ if args.local:
760
+ print(f"tony: {path}")
761
+ if args.openPage:
762
+ webbrowser.open(target)
763
+ return 0
764
+
765
+
766
+ if __name__ == "__main__":
767
+ sys.exit(main())