pushback 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. pushback-0.1.2/PKG-INFO +281 -0
  2. pushback-0.1.2/README.md +254 -0
  3. {pushback-0.1.0 → pushback-0.1.2}/pyproject.toml +39 -36
  4. pushback-0.1.2/src/pushback/__init__.py +1 -0
  5. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/cli.py +261 -231
  6. pushback-0.1.2/src/pushback/label.py +232 -0
  7. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/prompt.py +93 -71
  8. pushback-0.1.2/src/pushback/report.py +124 -0
  9. pushback-0.1.2/src/pushback.egg-info/PKG-INFO +281 -0
  10. {pushback-0.1.0 → pushback-0.1.2}/tests/test_pushback.py +456 -343
  11. pushback-0.1.0/PKG-INFO +0 -168
  12. pushback-0.1.0/README.md +0 -144
  13. pushback-0.1.0/src/pushback/__init__.py +0 -1
  14. pushback-0.1.0/src/pushback/label.py +0 -120
  15. pushback-0.1.0/src/pushback/report.py +0 -81
  16. pushback-0.1.0/src/pushback.egg-info/PKG-INFO +0 -168
  17. {pushback-0.1.0 → pushback-0.1.2}/LICENSE +0 -0
  18. {pushback-0.1.0 → pushback-0.1.2}/setup.cfg +0 -0
  19. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/audit.py +0 -0
  20. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/export.py +0 -0
  21. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/extract.py +0 -0
  22. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/rules.py +0 -0
  23. {pushback-0.1.0 → pushback-0.1.2}/src/pushback/stats.py +0 -0
  24. {pushback-0.1.0 → pushback-0.1.2}/src/pushback.egg-info/SOURCES.txt +0 -0
  25. {pushback-0.1.0 → pushback-0.1.2}/src/pushback.egg-info/dependency_links.txt +0 -0
  26. {pushback-0.1.0 → pushback-0.1.2}/src/pushback.egg-info/entry_points.txt +0 -0
  27. {pushback-0.1.0 → pushback-0.1.2}/src/pushback.egg-info/requires.txt +0 -0
  28. {pushback-0.1.0 → pushback-0.1.2}/src/pushback.egg-info/top_level.txt +0 -0
@@ -0,0 +1,281 @@
1
+ Metadata-Version: 2.4
2
+ Name: pushback
3
+ Version: 0.1.2
4
+ Summary: How often do you correct your coding agent, and at what kind of work? Measured from your own Claude Code transcripts.
5
+ Author: Intikhab Azam
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/intikhab49/pushback
8
+ Project-URL: Repository, https://github.com/intikhab49/pushback
9
+ Project-URL: Issues, https://github.com/intikhab49/pushback/issues
10
+ Keywords: claude-code,coding-agents,ai-agents,llm,claude,claude-md,dpo,preference-data,llm-evaluation
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Environment :: Console
17
+ Classifier: Intended Audience :: Developers
18
+ Classifier: Topic :: Software Development
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Requires-Python: >=3.10
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: anthropic>=1.0
24
+ Provides-Extra: test
25
+ Requires-Dist: pytest>=8; extra == "test"
26
+ Dynamic: license-file
27
+
28
+ <p align="center">
29
+ <img src="assets/hero.svg" width="100%" alt="pushback: measure how often you correct your AI coding agent, find the CLAUDE.md rules it keeps breaking, and export your corrections as preference pairs">
30
+ </p>
31
+
32
+ <h1 align="center">pushback: how often do you correct your AI coding agent?</h1>
33
+
34
+ <p align="center">
35
+ <a href="https://pypi.org/project/pushback/"><img alt="PyPI" src="https://img.shields.io/pypi/v/pushback?style=for-the-badge&color=FF5C39&labelColor=12121F"></a>
36
+ <a href="https://github.com/intikhab49/pushback/actions/workflows/ci.yml"><img alt="tests" src="https://img.shields.io/github/actions/workflow/status/intikhab49/pushback/ci.yml?branch=master&style=for-the-badge&label=tests&color=A3F547&labelColor=12121F"></a>
37
+ <img alt="Python 3.10+" src="https://img.shields.io/badge/python-3.10%2B-5B8CFF?style=for-the-badge&labelColor=12121F">
38
+ <img alt="Works with Claude Code" src="https://img.shields.io/badge/works%20with-Claude%20Code-FFB224?style=for-the-badge&labelColor=12121F">
39
+ <a href="LICENSE"><img alt="MIT license" src="https://img.shields.io/badge/license-MIT-B794FF?style=for-the-badge&labelColor=12121F"></a>
40
+ </p>
41
+
42
+ <p align="center">
43
+ <a href="#results">Results</a> ·
44
+ <a href="#quick-start">Quick start</a> ·
45
+ <a href="#how-it-works">How it works</a> ·
46
+ <a href="#turn-recurring-corrections-into-claudemd-rules">Rules</a> ·
47
+ <a href="#export-your-corrections-as-preference-pairs">Preference pairs</a> ·
48
+ <a href="#privacy">Privacy</a> ·
49
+ <a href="#how-the-numbers-are-computed">Method</a>
50
+ </p>
51
+
52
+ `pushback` is a command-line tool that reads your **Claude Code transcripts** and
53
+ shows what you use your AI coding agent for, how often you correct it on each
54
+ kind of work (code, writing, media, research, ops, meta), and how often a
55
+ correction needs correcting again. It drafts
56
+ **CLAUDE.md rules** from the corrections you keep repeating, shows which of
57
+ your existing rules keep getting broken, and exports your corrections as
58
+ **preference pairs** (prompt / chosen / rejected). A hand-audit step tells you
59
+ how far to trust every number.
60
+
61
+ ```
62
+ pip install pushback
63
+ ```
64
+
65
+ ## Results
66
+
67
+ The author's own 1,633 messages to Claude Code:
68
+
69
+ <p align="center">
70
+ <img src="assets/rates.svg" width="100%" alt="Correction rate by task: media 43.0%, writing 24.7%, research 6.9%, code 5.9%, meta 4.7%, ops 4.5%. One in three corrections needed a second correction. Three existing rules kept getting broken.">
71
+ </p>
72
+
73
+ | task | messages | share of use | corrections | rate | 95% CI | corrected again |
74
+ |---|---:|---:|---:|---:|---|---:|
75
+ | writing | 429 | 26% | 106 | 24.7% | 20.9%–29.0% | 41% (41/99) |
76
+ | code | 340 | 21% | 20 | 5.9% | 3.8%–8.9% | 11% (2/18) |
77
+ | meta | 297 | 18% | 14 | 4.7% | 2.8%–7.8% | 36% (5/14) |
78
+ | ops | 265 | 16% | 12 | 4.5% | 2.6%–7.7% | 8% (1/12) |
79
+ | research | 188 | 12% | 13 | 6.9% | 4.1%–11.5% | 8% (1/13) |
80
+ | media | 114 | 7% | 49 | 43.0% | 34.3%–52.2% | 44% (20/45) |
81
+
82
+ Most corrected topics (at least 5 messages each):
83
+
84
+ | topic | messages | corrections | rate | corrected again |
85
+ |---|---:|---:|---:|---:|
86
+ | writing / social-post | 135 | 47 | 35% | 23/45 |
87
+ | writing / client-message | 144 | 26 | 18% | 7/23 |
88
+ | media / image | 47 | 24 | 51% | 10/21 |
89
+ | media / video | 36 | 12 | 33% | 6/12 |
90
+ | writing / proposal-report | 26 | 11 | 42% | 5/11 |
91
+ | writing / video-script | 36 | 10 | 28% | 3/9 |
92
+ | code / frontend | 31 | 6 | 19% | 1/6 |
93
+ | code / debugging | 27 | 5 | 19% | 1/4 |
94
+
95
+ - Against 91 hand-checked messages the correction labels had **precision 0.97 and recall 0.80**. Topic labels were spot-checked, not audited.
96
+ - **1 in 3 corrections needed a second one.** 70 of 201 fixes got corrected again (35%), 41% on writing and 44% on images.
97
+ - **3 rules the author had already written kept getting broken**, with 6 to 8 corrections each.
98
+
99
+ > [!NOTE]
100
+ > These rates describe a workflow, not a model. The author's code work runs
101
+ > through skills, reference files, memory and a plan before the agent writes
102
+ > anything, and CI catches mistakes before a human has to. Writing got none of
103
+ > that. A low rate means the process around the agent is doing its job.
104
+
105
+ That's one person's logs. Run it on yours.
106
+
107
+ ## Quick start
108
+
109
+ ```bash
110
+ pip install pushback
111
+
112
+ pushback extract # reads ~/.claude/projects, writes ./pushback-data/
113
+ pushback label # labels each message with Claude (resumable)
114
+ pushback audit # hand-check 40 messages
115
+ pushback report --markdown # the table, with the audit folded in
116
+ pushback rules # draft CLAUDE.md rules from your recurring corrections
117
+ pushback export # your corrections as prompt/chosen/rejected pairs
118
+ ```
119
+
120
+ `label` uses the Anthropic SDK, so it picks up `ANTHROPIC_API_KEY` or an
121
+ `ant auth login` profile. It defaults to `claude-opus-5` at low effort; change
122
+ it with `--model`. To send through a gateway, set `ANTHROPIC_BASE_URL`.
123
+
124
+ > [!TIP]
125
+ > **No API key?** Every step works without one:
126
+ > ```bash
127
+ > pushback label --prompt-file # writes the labelling job as batch files
128
+ > # in Claude Code: "read <path it prints>/INSTRUCTIONS.md and follow it"
129
+ > pushback label --from-responses # reads the answers back, with the same checks as the API path
130
+ > ```
131
+ > Your messages are still read by Claude, through Claude Code on your
132
+ > subscription. Unanswered or broken batches are listed and stay unlabelled.
133
+
134
+ | command | what it does | leaves your machine? |
135
+ |---|---|---|
136
+ | `extract` | pulls the messages you typed, plus full agent turns, out of transcripts | no |
137
+ | `label` | tags each message with a task, a topic and whether it's a correction | yes: to the API after asking, or through Claude Code with `--prompt-file` |
138
+ | `audit` | shows you a stratified sample to judge by hand | no |
139
+ | `report` | what you use the agent for, how often you correct it per task and per topic (frontend, cli-tool, social-post, ...), how often a fix gets corrected again, with 95% intervals | no |
140
+ | `rules` | groups recurring corrections into CLAUDE.md rules | yes, or no with `--prompt-file` |
141
+ | `export` | writes preference pairs as JSONL, with secrets scrubbed | no |
142
+
143
+ ## How it works
144
+
145
+ <p align="center">
146
+ <img src="assets/pipeline.svg" width="100%" alt="How pushback works: extract messages from Claude Code transcripts, label them, audit a sample, then report correction rates, draft CLAUDE.md rules, or export preference pairs">
147
+ </p>
148
+
149
+ Every message gets a **task** (code, writing, media, research, ops, meta) and a
150
+ **topic** from a fixed list for that task, so results stay comparable between
151
+ people:
152
+
153
+ | task | topics |
154
+ |---|---|
155
+ | code | frontend, backend-api, cli-tool, database, tests, auth-security, integrations, data-ml, scripts-automation, agent-prompts, debugging, pr-workflow |
156
+ | writing | social-post, client-message, email, docs-readme, video-script, proposal-report, spec-plan |
157
+ | media | image, video, diagram, ui-design |
158
+ | research | benchmark-experiment, data-analysis, market-leads, paper-reading, trading-analysis |
159
+ | ops | deploy-hosting, git-github, accounts-credentials, config-settings, browser-automation, ci |
160
+ | meta | planning, memory-context, advice, status-check |
161
+
162
+ Anything that fits none of them is `other`.
163
+
164
+ A **correction** is a message where you reject, fix or redirect something the
165
+ agent just did, said, wrote or proposed. New tasks, answers to its questions,
166
+ picking between its options, approvals and pasted logs don't count. The full
167
+ rubric is in [`src/pushback/prompt.py`](src/pushback/prompt.py). If you change
168
+ it, run `audit` again.
169
+
170
+ Rejected tool calls and interrupts carry no text, so they're counted separately
171
+ and grouped by the action you stopped (a code edit, a shell command, and so on).
172
+
173
+ ## Turn recurring corrections into CLAUDE.md rules
174
+
175
+ ```bash
176
+ pushback rules # drafts rules into pushback-data/rules.md
177
+ pushback rules --existing CLAUDE.md notes/*.md # also check against the rules you already have
178
+ pushback rules --prompt-file # no API key? writes the request to a file instead
179
+ pushback rules --from-response reply.json # ...and reads Claude's answer back
180
+ ```
181
+
182
+ The model groups corrections that share a cause and drafts one CLAUDE.md
183
+ instruction per group. Support is counted from the correction ids it cites,
184
+ not from its own numbers, and a rule needs at least 3 real corrections.
185
+
186
+ When you pass your existing rule files, the output splits in two:
187
+
188
+ - **Rules you already have that keep getting broken.** These are the useful
189
+ ones. A rule that exists and still gets corrected isn't working: it's too
190
+ vague, or the agent doesn't read it at the right moment.
191
+ - **New rules to consider.** Recurring corrections with no rule behind them.
192
+
193
+ > [!TIP]
194
+ > No API key? `--prompt-file` writes the whole request to `rules-prompt.md`.
195
+ > Ask Claude Code to answer it, save the JSON reply, then run `--from-response`.
196
+ > `label` has the same option, so the whole pipeline runs without a key.
197
+
198
+ ## Export your corrections as preference pairs
199
+
200
+ ```bash
201
+ pushback export # all pairs -> pushback-data/dpo.jsonl
202
+ pushback export --ctype writing_content tone_style # the cleanest pairs
203
+ pushback export --minimal # only prompt/chosen/rejected (TRL DPO columns)
204
+ ```
205
+
206
+ Each correction you made becomes one row:
207
+
208
+ | field | what it holds |
209
+ |---|---|
210
+ | `prompt` | your message the agent was answering |
211
+ | `rejected` | the agent turn you corrected |
212
+ | `chosen` | the agent's reply to your correction, kept only if your next message wasn't another correction |
213
+ | `feedback` | the correction itself, for critique-and-revise formats |
214
+
215
+ On the author's logs, 214 corrections gave 124 pairs. The biggest loss:
216
+ 70 corrections (a third) had their fix corrected too, so there was no accepted
217
+ answer to pair with.
218
+
219
+ > [!WARNING]
220
+ > Read this before you train on it.
221
+ > - "You didn't correct it again" is weak evidence that the fix was good.
222
+ > - The fix was written after seeing your feedback. For corrections of a wrong
223
+ > assumption ("I already sent that"), `chosen` answers the correction rather
224
+ > than the prompt. Those make poor DPO pairs, so filter with `--ctype`.
225
+ > - Only agent text is exported. File edits and commands are tool calls, so a
226
+ > code correction's pair may hold the explanation without the diff. Use
227
+ > `--max-tools` to drop tool-heavy turns.
228
+ > - Secrets are scrubbed on a best-effort basis (API key shapes, tokens,
229
+ > `NAME_KEY=value`). Read the file before you use it.
230
+ > - If the agent is Claude, Anthropic's terms restrict using its outputs to
231
+ > build competing models. Treat this as a personal dataset or eval set.
232
+
233
+ ## Privacy
234
+
235
+ - `extract`, `audit`, `report` and `export` never leave your machine. `report` prints counts only.
236
+ - `label` and `rules` send each message, plus the end of the agent reply before
237
+ it, to the model provider. If your logs contain client work, check that
238
+ provider's data policy first. Both ask before they send anything. With
239
+ `--prompt-file` nothing is sent by pushback; Claude Code reads the files instead.
240
+ - `pushback-data/` holds your raw messages. It's in `.gitignore`. Keep it there.
241
+ - Your transcripts probably contain other people's information. Keep exports
242
+ local unless every conversation in them is yours to share.
243
+
244
+ > [!IMPORTANT]
245
+ > **Your history is shorter than you think.** Claude Code deletes transcripts
246
+ > older than 30 days by default. To keep more, set `cleanupPeriodDays` in
247
+ > `~/.claude/settings.json`, for example `"cleanupPeriodDays": 365`. It only
248
+ > saves transcripts that still exist.
249
+
250
+ ## How the numbers are computed
251
+
252
+ - Rates per task come with Wilson 95% intervals.
253
+ - `audit` samples half from messages the model flagged and half from the rest.
254
+ The report estimates the true count as
255
+ `flagged × precision + unflagged × miss rate`, with a range built from both
256
+ strata's intervals.
257
+ - A batch that fails stays unlabelled and is retried on the next run. It's
258
+ never counted as "not a correction".
259
+ - Message ids are `session:index`, so re-running `extract` never shifts your labels.
260
+
261
+ ## Limits
262
+
263
+ - Claude Code transcripts only, for now. Codex and Cursor logs are not read yet.
264
+ - Task type is judged from the last agent reply and your message, not the whole session.
265
+ - The rubric was tuned on one person's logs.
266
+ - Labels aren't perfectly stable between runs. Two independent runs on the same
267
+ 60 messages agreed on **correction 93%** of the time, on **task 70%**, and on
268
+ **topic 76%** when the task matched. Disagreements sit at fuzzy edges
269
+ (writing vs meta, code vs ops), mostly short "ok do that" messages. Treat
270
+ small differences between topics as noise.
271
+
272
+ ## Contributing
273
+
274
+ Issues and pull requests are welcome, especially readers for other agents'
275
+ transcript formats. See [CONTRIBUTING.md](CONTRIBUTING.md).
276
+
277
+ ## License
278
+
279
+ [MIT](LICENSE) © Intikhab Azam
280
+
281
+ <sub>Keywords: Claude Code analytics · AI coding agent evaluation · correction rate · CLAUDE.md rules generator · agent transcripts · human feedback · preference pairs · DPO dataset · LLM evaluation · developer productivity</sub>
@@ -0,0 +1,254 @@
1
+ <p align="center">
2
+ <img src="assets/hero.svg" width="100%" alt="pushback: measure how often you correct your AI coding agent, find the CLAUDE.md rules it keeps breaking, and export your corrections as preference pairs">
3
+ </p>
4
+
5
+ <h1 align="center">pushback: how often do you correct your AI coding agent?</h1>
6
+
7
+ <p align="center">
8
+ <a href="https://pypi.org/project/pushback/"><img alt="PyPI" src="https://img.shields.io/pypi/v/pushback?style=for-the-badge&color=FF5C39&labelColor=12121F"></a>
9
+ <a href="https://github.com/intikhab49/pushback/actions/workflows/ci.yml"><img alt="tests" src="https://img.shields.io/github/actions/workflow/status/intikhab49/pushback/ci.yml?branch=master&style=for-the-badge&label=tests&color=A3F547&labelColor=12121F"></a>
10
+ <img alt="Python 3.10+" src="https://img.shields.io/badge/python-3.10%2B-5B8CFF?style=for-the-badge&labelColor=12121F">
11
+ <img alt="Works with Claude Code" src="https://img.shields.io/badge/works%20with-Claude%20Code-FFB224?style=for-the-badge&labelColor=12121F">
12
+ <a href="LICENSE"><img alt="MIT license" src="https://img.shields.io/badge/license-MIT-B794FF?style=for-the-badge&labelColor=12121F"></a>
13
+ </p>
14
+
15
+ <p align="center">
16
+ <a href="#results">Results</a> ·
17
+ <a href="#quick-start">Quick start</a> ·
18
+ <a href="#how-it-works">How it works</a> ·
19
+ <a href="#turn-recurring-corrections-into-claudemd-rules">Rules</a> ·
20
+ <a href="#export-your-corrections-as-preference-pairs">Preference pairs</a> ·
21
+ <a href="#privacy">Privacy</a> ·
22
+ <a href="#how-the-numbers-are-computed">Method</a>
23
+ </p>
24
+
25
+ `pushback` is a command-line tool that reads your **Claude Code transcripts** and
26
+ shows what you use your AI coding agent for, how often you correct it on each
27
+ kind of work (code, writing, media, research, ops, meta), and how often a
28
+ correction needs correcting again. It drafts
29
+ **CLAUDE.md rules** from the corrections you keep repeating, shows which of
30
+ your existing rules keep getting broken, and exports your corrections as
31
+ **preference pairs** (prompt / chosen / rejected). A hand-audit step tells you
32
+ how far to trust every number.
33
+
34
+ ```
35
+ pip install pushback
36
+ ```
37
+
38
+ ## Results
39
+
40
+ The author's own 1,633 messages to Claude Code:
41
+
42
+ <p align="center">
43
+ <img src="assets/rates.svg" width="100%" alt="Correction rate by task: media 43.0%, writing 24.7%, research 6.9%, code 5.9%, meta 4.7%, ops 4.5%. One in three corrections needed a second correction. Three existing rules kept getting broken.">
44
+ </p>
45
+
46
+ | task | messages | share of use | corrections | rate | 95% CI | corrected again |
47
+ |---|---:|---:|---:|---:|---|---:|
48
+ | writing | 429 | 26% | 106 | 24.7% | 20.9%–29.0% | 41% (41/99) |
49
+ | code | 340 | 21% | 20 | 5.9% | 3.8%–8.9% | 11% (2/18) |
50
+ | meta | 297 | 18% | 14 | 4.7% | 2.8%–7.8% | 36% (5/14) |
51
+ | ops | 265 | 16% | 12 | 4.5% | 2.6%–7.7% | 8% (1/12) |
52
+ | research | 188 | 12% | 13 | 6.9% | 4.1%–11.5% | 8% (1/13) |
53
+ | media | 114 | 7% | 49 | 43.0% | 34.3%–52.2% | 44% (20/45) |
54
+
55
+ Most corrected topics (at least 5 messages each):
56
+
57
+ | topic | messages | corrections | rate | corrected again |
58
+ |---|---:|---:|---:|---:|
59
+ | writing / social-post | 135 | 47 | 35% | 23/45 |
60
+ | writing / client-message | 144 | 26 | 18% | 7/23 |
61
+ | media / image | 47 | 24 | 51% | 10/21 |
62
+ | media / video | 36 | 12 | 33% | 6/12 |
63
+ | writing / proposal-report | 26 | 11 | 42% | 5/11 |
64
+ | writing / video-script | 36 | 10 | 28% | 3/9 |
65
+ | code / frontend | 31 | 6 | 19% | 1/6 |
66
+ | code / debugging | 27 | 5 | 19% | 1/4 |
67
+
68
+ - Against 91 hand-checked messages the correction labels had **precision 0.97 and recall 0.80**. Topic labels were spot-checked, not audited.
69
+ - **1 in 3 corrections needed a second one.** 70 of 201 fixes got corrected again (35%), 41% on writing and 44% on images.
70
+ - **3 rules the author had already written kept getting broken**, with 6 to 8 corrections each.
71
+
72
+ > [!NOTE]
73
+ > These rates describe a workflow, not a model. The author's code work runs
74
+ > through skills, reference files, memory and a plan before the agent writes
75
+ > anything, and CI catches mistakes before a human has to. Writing got none of
76
+ > that. A low rate means the process around the agent is doing its job.
77
+
78
+ That's one person's logs. Run it on yours.
79
+
80
+ ## Quick start
81
+
82
+ ```bash
83
+ pip install pushback
84
+
85
+ pushback extract # reads ~/.claude/projects, writes ./pushback-data/
86
+ pushback label # labels each message with Claude (resumable)
87
+ pushback audit # hand-check 40 messages
88
+ pushback report --markdown # the table, with the audit folded in
89
+ pushback rules # draft CLAUDE.md rules from your recurring corrections
90
+ pushback export # your corrections as prompt/chosen/rejected pairs
91
+ ```
92
+
93
+ `label` uses the Anthropic SDK, so it picks up `ANTHROPIC_API_KEY` or an
94
+ `ant auth login` profile. It defaults to `claude-opus-5` at low effort; change
95
+ it with `--model`. To send through a gateway, set `ANTHROPIC_BASE_URL`.
96
+
97
+ > [!TIP]
98
+ > **No API key?** Every step works without one:
99
+ > ```bash
100
+ > pushback label --prompt-file # writes the labelling job as batch files
101
+ > # in Claude Code: "read <path it prints>/INSTRUCTIONS.md and follow it"
102
+ > pushback label --from-responses # reads the answers back, with the same checks as the API path
103
+ > ```
104
+ > Your messages are still read by Claude, through Claude Code on your
105
+ > subscription. Unanswered or broken batches are listed and stay unlabelled.
106
+
107
+ | command | what it does | leaves your machine? |
108
+ |---|---|---|
109
+ | `extract` | pulls the messages you typed, plus full agent turns, out of transcripts | no |
110
+ | `label` | tags each message with a task, a topic and whether it's a correction | yes: to the API after asking, or through Claude Code with `--prompt-file` |
111
+ | `audit` | shows you a stratified sample to judge by hand | no |
112
+ | `report` | what you use the agent for, how often you correct it per task and per topic (frontend, cli-tool, social-post, ...), how often a fix gets corrected again, with 95% intervals | no |
113
+ | `rules` | groups recurring corrections into CLAUDE.md rules | yes, or no with `--prompt-file` |
114
+ | `export` | writes preference pairs as JSONL, with secrets scrubbed | no |
115
+
116
+ ## How it works
117
+
118
+ <p align="center">
119
+ <img src="assets/pipeline.svg" width="100%" alt="How pushback works: extract messages from Claude Code transcripts, label them, audit a sample, then report correction rates, draft CLAUDE.md rules, or export preference pairs">
120
+ </p>
121
+
122
+ Every message gets a **task** (code, writing, media, research, ops, meta) and a
123
+ **topic** from a fixed list for that task, so results stay comparable between
124
+ people:
125
+
126
+ | task | topics |
127
+ |---|---|
128
+ | code | frontend, backend-api, cli-tool, database, tests, auth-security, integrations, data-ml, scripts-automation, agent-prompts, debugging, pr-workflow |
129
+ | writing | social-post, client-message, email, docs-readme, video-script, proposal-report, spec-plan |
130
+ | media | image, video, diagram, ui-design |
131
+ | research | benchmark-experiment, data-analysis, market-leads, paper-reading, trading-analysis |
132
+ | ops | deploy-hosting, git-github, accounts-credentials, config-settings, browser-automation, ci |
133
+ | meta | planning, memory-context, advice, status-check |
134
+
135
+ Anything that fits none of them is `other`.
136
+
137
+ A **correction** is a message where you reject, fix or redirect something the
138
+ agent just did, said, wrote or proposed. New tasks, answers to its questions,
139
+ picking between its options, approvals and pasted logs don't count. The full
140
+ rubric is in [`src/pushback/prompt.py`](src/pushback/prompt.py). If you change
141
+ it, run `audit` again.
142
+
143
+ Rejected tool calls and interrupts carry no text, so they're counted separately
144
+ and grouped by the action you stopped (a code edit, a shell command, and so on).
145
+
146
+ ## Turn recurring corrections into CLAUDE.md rules
147
+
148
+ ```bash
149
+ pushback rules # drafts rules into pushback-data/rules.md
150
+ pushback rules --existing CLAUDE.md notes/*.md # also check against the rules you already have
151
+ pushback rules --prompt-file # no API key? writes the request to a file instead
152
+ pushback rules --from-response reply.json # ...and reads Claude's answer back
153
+ ```
154
+
155
+ The model groups corrections that share a cause and drafts one CLAUDE.md
156
+ instruction per group. Support is counted from the correction ids it cites,
157
+ not from its own numbers, and a rule needs at least 3 real corrections.
158
+
159
+ When you pass your existing rule files, the output splits in two:
160
+
161
+ - **Rules you already have that keep getting broken.** These are the useful
162
+ ones. A rule that exists and still gets corrected isn't working: it's too
163
+ vague, or the agent doesn't read it at the right moment.
164
+ - **New rules to consider.** Recurring corrections with no rule behind them.
165
+
166
+ > [!TIP]
167
+ > No API key? `--prompt-file` writes the whole request to `rules-prompt.md`.
168
+ > Ask Claude Code to answer it, save the JSON reply, then run `--from-response`.
169
+ > `label` has the same option, so the whole pipeline runs without a key.
170
+
171
+ ## Export your corrections as preference pairs
172
+
173
+ ```bash
174
+ pushback export # all pairs -> pushback-data/dpo.jsonl
175
+ pushback export --ctype writing_content tone_style # the cleanest pairs
176
+ pushback export --minimal # only prompt/chosen/rejected (TRL DPO columns)
177
+ ```
178
+
179
+ Each correction you made becomes one row:
180
+
181
+ | field | what it holds |
182
+ |---|---|
183
+ | `prompt` | your message the agent was answering |
184
+ | `rejected` | the agent turn you corrected |
185
+ | `chosen` | the agent's reply to your correction, kept only if your next message wasn't another correction |
186
+ | `feedback` | the correction itself, for critique-and-revise formats |
187
+
188
+ On the author's logs, 214 corrections gave 124 pairs. The biggest loss:
189
+ 70 corrections (a third) had their fix corrected too, so there was no accepted
190
+ answer to pair with.
191
+
192
+ > [!WARNING]
193
+ > Read this before you train on it.
194
+ > - "You didn't correct it again" is weak evidence that the fix was good.
195
+ > - The fix was written after seeing your feedback. For corrections of a wrong
196
+ > assumption ("I already sent that"), `chosen` answers the correction rather
197
+ > than the prompt. Those make poor DPO pairs, so filter with `--ctype`.
198
+ > - Only agent text is exported. File edits and commands are tool calls, so a
199
+ > code correction's pair may hold the explanation without the diff. Use
200
+ > `--max-tools` to drop tool-heavy turns.
201
+ > - Secrets are scrubbed on a best-effort basis (API key shapes, tokens,
202
+ > `NAME_KEY=value`). Read the file before you use it.
203
+ > - If the agent is Claude, Anthropic's terms restrict using its outputs to
204
+ > build competing models. Treat this as a personal dataset or eval set.
205
+
206
+ ## Privacy
207
+
208
+ - `extract`, `audit`, `report` and `export` never leave your machine. `report` prints counts only.
209
+ - `label` and `rules` send each message, plus the end of the agent reply before
210
+ it, to the model provider. If your logs contain client work, check that
211
+ provider's data policy first. Both ask before they send anything. With
212
+ `--prompt-file` nothing is sent by pushback; Claude Code reads the files instead.
213
+ - `pushback-data/` holds your raw messages. It's in `.gitignore`. Keep it there.
214
+ - Your transcripts probably contain other people's information. Keep exports
215
+ local unless every conversation in them is yours to share.
216
+
217
+ > [!IMPORTANT]
218
+ > **Your history is shorter than you think.** Claude Code deletes transcripts
219
+ > older than 30 days by default. To keep more, set `cleanupPeriodDays` in
220
+ > `~/.claude/settings.json`, for example `"cleanupPeriodDays": 365`. It only
221
+ > saves transcripts that still exist.
222
+
223
+ ## How the numbers are computed
224
+
225
+ - Rates per task come with Wilson 95% intervals.
226
+ - `audit` samples half from messages the model flagged and half from the rest.
227
+ The report estimates the true count as
228
+ `flagged × precision + unflagged × miss rate`, with a range built from both
229
+ strata's intervals.
230
+ - A batch that fails stays unlabelled and is retried on the next run. It's
231
+ never counted as "not a correction".
232
+ - Message ids are `session:index`, so re-running `extract` never shifts your labels.
233
+
234
+ ## Limits
235
+
236
+ - Claude Code transcripts only, for now. Codex and Cursor logs are not read yet.
237
+ - Task type is judged from the last agent reply and your message, not the whole session.
238
+ - The rubric was tuned on one person's logs.
239
+ - Labels aren't perfectly stable between runs. Two independent runs on the same
240
+ 60 messages agreed on **correction 93%** of the time, on **task 70%**, and on
241
+ **topic 76%** when the task matched. Disagreements sit at fuzzy edges
242
+ (writing vs meta, code vs ops), mostly short "ok do that" messages. Treat
243
+ small differences between topics as noise.
244
+
245
+ ## Contributing
246
+
247
+ Issues and pull requests are welcome, especially readers for other agents'
248
+ transcript formats. See [CONTRIBUTING.md](CONTRIBUTING.md).
249
+
250
+ ## License
251
+
252
+ [MIT](LICENSE) © Intikhab Azam
253
+
254
+ <sub>Keywords: Claude Code analytics · AI coding agent evaluation · correction rate · CLAUDE.md rules generator · agent transcripts · human feedback · preference pairs · DPO dataset · LLM evaluation · developer productivity</sub>
@@ -1,36 +1,39 @@
1
- [build-system]
2
- requires = ["setuptools>=68"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [project]
6
- name = "pushback"
7
- version = "0.1.0"
8
- description = "How often do you correct your coding agent, and at what kind of work? Measured from your own Claude Code transcripts."
9
- readme = "README.md"
10
- license = {text = "MIT"}
11
- requires-python = ">=3.10"
12
- dependencies = ["anthropic>=1.0"]
13
- authors = [{name = "Intikhab Azam"}]
14
- keywords = ["claude-code", "coding-agents", "ai-agents", "llm", "claude", "claude-md", "dpo", "preference-data", "llm-evaluation"]
15
- classifiers = [
16
- "Programming Language :: Python :: 3",
17
- "License :: OSI Approved :: MIT License",
18
- "Environment :: Console",
19
- "Intended Audience :: Developers",
20
- "Topic :: Software Development",
21
- "Topic :: Scientific/Engineering :: Artificial Intelligence",
22
- ]
23
-
24
- [project.urls]
25
- Homepage = "https://github.com/intikhab49/pushback"
26
- Repository = "https://github.com/intikhab49/pushback"
27
- Issues = "https://github.com/intikhab49/pushback/issues"
28
-
29
- [project.optional-dependencies]
30
- test = ["pytest>=8"]
31
-
32
- [project.scripts]
33
- pushback = "pushback.cli:main"
34
-
35
- [tool.setuptools.packages.find]
36
- where = ["src"]
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pushback"
7
+ version = "0.1.2"
8
+ description = "How often do you correct your coding agent, and at what kind of work? Measured from your own Claude Code transcripts."
9
+ readme = "README.md"
10
+ license = {text = "MIT"}
11
+ requires-python = ">=3.10"
12
+ dependencies = ["anthropic>=1.0"]
13
+ authors = [{name = "Intikhab Azam"}]
14
+ keywords = ["claude-code", "coding-agents", "ai-agents", "llm", "claude", "claude-md", "dpo", "preference-data", "llm-evaluation"]
15
+ classifiers = [
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3.10",
18
+ "Programming Language :: Python :: 3.11",
19
+ "Programming Language :: Python :: 3.12",
20
+ "License :: OSI Approved :: MIT License",
21
+ "Environment :: Console",
22
+ "Intended Audience :: Developers",
23
+ "Topic :: Software Development",
24
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
25
+ ]
26
+
27
+ [project.urls]
28
+ Homepage = "https://github.com/intikhab49/pushback"
29
+ Repository = "https://github.com/intikhab49/pushback"
30
+ Issues = "https://github.com/intikhab49/pushback/issues"
31
+
32
+ [project.optional-dependencies]
33
+ test = ["pytest>=8"]
34
+
35
+ [project.scripts]
36
+ pushback = "pushback.cli:main"
37
+
38
+ [tool.setuptools.packages.find]
39
+ where = ["src"]
@@ -0,0 +1 @@
1
+ __version__ = "0.1.2"