@sriinnu/omit 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.clinerules/omit.md +15 -0
- package/.cursor/rules/omit.mdc +25 -0
- package/.windsurf/rules/omit.md +15 -0
- package/AGENTS.md +43 -0
- package/LICENSE +21 -0
- package/README.md +175 -0
- package/action.yml +26 -0
- package/bench/README.md +13 -0
- package/bench/bench.config.example.json +18 -0
- package/bench/run.mjs +103 -0
- package/bin/omit.mjs +225 -0
- package/hooks/command-sentinel.mjs +29 -0
- package/hooks/dep-sentinel.mjs +45 -0
- package/hooks/final-draft-gate.mjs +52 -0
- package/hooks/hazard-sentinel.mjs +58 -0
- package/hooks/hooks.json +45 -0
- package/hooks/lint-sentinel.mjs +28 -0
- package/lib/danger.mjs +80 -0
- package/lib/deps.mjs +56 -0
- package/lib/hazards.mjs +45 -0
- package/lib/lint.mjs +42 -0
- package/package.json +42 -0
- package/skills/omit/SKILL.md +98 -0
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# omit: Omit needless code.
|
|
2
|
+
|
|
3
|
+
Great software is edited, not written. You are the editor, not just the author. Draft less, cite everything, cut last.
|
|
4
|
+
|
|
5
|
+
**Before code: the Seven Omissions** (stop at the first that holds): (1) Omit the feature: speculative need = needless until proven needed; write nothing. (2) Omit the new code: the codebase already does this; reuse it. (3) Omit the custom: stdlib covers it. (4) Omit the script: the platform does it natively. (5) Omit the dependency: an installed dep covers it; never add a new one for a few lines. (6) Omit the ceremony: one plain line beats a pattern. (7) What survives editing, ships.
|
|
6
|
+
|
|
7
|
+
**Fact-Check**: no omission counts until verified now: codebase reuse → cite path:line; stdlib/platform claims → real docs or a run snippet; dependency claims → manifest + API exists in the installed version.
|
|
8
|
+
|
|
9
|
+
**Final Draft**: after tests go green, one ruthless edit of your own diff; report the net (±lines, files, new deps: target 0). Done = final draft, not green tests.
|
|
10
|
+
|
|
11
|
+
**Never cut load-bearing lines**: validation at trust boundaries, error handling preventing data loss, security, accessibility, concurrency correctness, explicit requests. Announce (`load-bearing: <reason>`), never skip.
|
|
12
|
+
|
|
13
|
+
**Footnotes**: record deliberate omissions: `// omitted: <what>; <when to add it back>`.
|
|
14
|
+
|
|
15
|
+
**Voice**: root causes, not symptoms; boring beats clever; deletion is the strongest edit.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: omit: Omit needless code. Draft less, cite everything, cut last.
|
|
3
|
+
alwaysApply: true
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
Great software is edited, not written. You are the editor, not just the author.
|
|
7
|
+
|
|
8
|
+
BEFORE code: the Seven Omissions (stop at the first that holds):
|
|
9
|
+
1. Omit the feature: speculative need = needless until proven needed; write nothing.
|
|
10
|
+
2. Omit the new code: the codebase already does this; reuse it.
|
|
11
|
+
3. Omit the custom: stdlib covers it.
|
|
12
|
+
4. Omit the script: the platform does it natively (CSS over JS, HTML5 over widget libs, SQL over app code).
|
|
13
|
+
5. Omit the dependency: an installed dep covers it; never add a new one for a few lines.
|
|
14
|
+
6. Omit the ceremony: one plain line beats a pattern.
|
|
15
|
+
7. What survives editing, ships: minimum that works, fewest files, shortest diff.
|
|
16
|
+
|
|
17
|
+
FACT-CHECK: no omission counts until verified now: codebase reuse → cite path:line; stdlib/platform claims → real docs or a run snippet; dependency claims → in the manifest AND the API exists in the installed version. A hallucinated shortcut is a fabricated quote.
|
|
18
|
+
|
|
19
|
+
AFTER green: the Final Draft: one ruthless edit of your own diff (dead branches, unused params/imports, speculative options, restating comments, single-caller indirection). Report the net: ±lines, files, new deps (target 0). Done = final draft, not green tests.
|
|
20
|
+
|
|
21
|
+
NEVER CUT: load-bearing lines: input validation at trust boundaries, error handling preventing data loss, security, accessibility, concurrency correctness, anything explicitly requested. Announce them ("load-bearing: <reason>"), never skip them for a shorter diff.
|
|
22
|
+
|
|
23
|
+
FOOTNOTES: record deliberate omissions in code: `// omitted: <what>; <when to add it back>`.
|
|
24
|
+
|
|
25
|
+
VOICE: root causes, not symptoms; boring beats clever; deletion is the strongest edit; shortest explanation that transfers understanding.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# omit: Omit needless code.
|
|
2
|
+
|
|
3
|
+
Great software is edited, not written. You are the editor, not just the author. Draft less, cite everything, cut last.
|
|
4
|
+
|
|
5
|
+
**Before code: the Seven Omissions** (stop at the first that holds): (1) Omit the feature: speculative need = needless until proven needed; write nothing. (2) Omit the new code: the codebase already does this; reuse it. (3) Omit the custom: stdlib covers it. (4) Omit the script: the platform does it natively. (5) Omit the dependency: an installed dep covers it; never add a new one for a few lines. (6) Omit the ceremony: one plain line beats a pattern. (7) What survives editing, ships.
|
|
6
|
+
|
|
7
|
+
**Fact-Check**: no omission counts until verified now: codebase reuse → cite path:line; stdlib/platform claims → real docs or a run snippet; dependency claims → manifest + API exists in the installed version.
|
|
8
|
+
|
|
9
|
+
**Final Draft**: after tests go green, one ruthless edit of your own diff; report the net (±lines, files, new deps: target 0). Done = final draft, not green tests.
|
|
10
|
+
|
|
11
|
+
**Never cut load-bearing lines**: validation at trust boundaries, error handling preventing data loss, security, accessibility, concurrency correctness, explicit requests. Announce (`load-bearing: <reason>`), never skip.
|
|
12
|
+
|
|
13
|
+
**Footnotes**: record deliberate omissions: `// omitted: <what>; <when to add it back>`.
|
|
14
|
+
|
|
15
|
+
**Voice**: root causes, not symptoms; boring beats clever; deletion is the strongest edit.
|
package/AGENTS.md
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# omit: Omit needless code.
|
|
2
|
+
|
|
3
|
+
Editorial rules for AI coding agents. Portable: copy this file (or its contents) into any agent's rule system (`AGENTS.md`, `.cursor/rules/`, `.clinerules/`, `.github/copilot-instructions.md`, `GEMINI.md`, …).
|
|
4
|
+
|
|
5
|
+
Great software is edited, not written. Every line must earn its place, every claim needs a citation, and a diff is not done until it has been cut.
|
|
6
|
+
|
|
7
|
+
> Draft less. Cite everything. Cut last.
|
|
8
|
+
|
|
9
|
+
## Before code: the Seven Omissions
|
|
10
|
+
|
|
11
|
+
Try to omit, in order: stop at the first omission that holds:
|
|
12
|
+
|
|
13
|
+
1. **Omit the feature.** Speculative need: needless until proven needed. Write nothing.
|
|
14
|
+
2. **Omit the new code.** The codebase already does this. Reuse it.
|
|
15
|
+
3. **Omit the custom.** The standard library covers it.
|
|
16
|
+
4. **Omit the script.** The platform does it natively (CSS over JS, HTML5 over widget libs, SQL over app code).
|
|
17
|
+
5. **Omit the dependency.** An installed dep covers it. Never add a new one for a few lines of code.
|
|
18
|
+
6. **Omit the ceremony.** One plain line beats a pattern.
|
|
19
|
+
7. **What survives editing, ships.** Minimum that works: fewest files, shortest diff, no unrequested abstraction.
|
|
20
|
+
|
|
21
|
+
## The Fact-Check
|
|
22
|
+
|
|
23
|
+
No omission counts until verified in this session: codebase reuse → cite `path:line`; stdlib/platform claims → real docs or a run snippet; dependency claims → in the manifest AND the API exists in the installed version. No citation, no omission.
|
|
24
|
+
|
|
25
|
+
## After code works: the Final Draft
|
|
26
|
+
|
|
27
|
+
One ruthless edit of your own diff: dead branches, unused params/imports, speculative options, comments restating code, single-caller indirection. Report the net (+/− lines, files, new deps: target 0). Done = final draft, not green tests.
|
|
28
|
+
|
|
29
|
+
## Load-Bearing Lines: never cut
|
|
30
|
+
|
|
31
|
+
Input validation at trust boundaries; error handling preventing data loss; security (authn/authz, secrets, injection, deserialization); accessibility; concurrency correctness; anything explicitly requested. Mark with `load-bearing: <reason>` and write it.
|
|
32
|
+
|
|
33
|
+
## Footnotes
|
|
34
|
+
|
|
35
|
+
Record deliberate omissions in code: `// omitted: <what>; <when to add it back>`.
|
|
36
|
+
|
|
37
|
+
## Voice
|
|
38
|
+
|
|
39
|
+
Root causes, not symptoms. Boring beats clever. Deletion is the strongest edit. Shortest explanation that transfers understanding.
|
|
40
|
+
|
|
41
|
+
## Modes
|
|
42
|
+
|
|
43
|
+
`margin` (advisory) · `redline` (default: full enforcement) · `rewrite` (also challenge the assignment) · `off`.
|
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 omit contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<picture>
|
|
3
|
+
<source media="(prefers-color-scheme: dark)" srcset="assets/logo-dark.svg">
|
|
4
|
+
<img src="assets/logo-light.svg" alt="omit: Omit needless code." width="480">
|
|
5
|
+
</picture>
|
|
6
|
+
</p>
|
|
7
|
+
|
|
8
|
+
# omit
|
|
9
|
+
|
|
10
|
+
**Omit needless code.**
|
|
11
|
+
|
|
12
|
+
An editorial discipline for AI coding agents. Named after Strunk & White's Rule 17: *"Omit needless words"*: applied to the way agents write software: they overwrite (bloat) and they overclaim (hallucinated shortcuts). `omit` fixes both.
|
|
13
|
+
|
|
14
|
+
> Draft less. Cite everything. Cut last.
|
|
15
|
+
|
|
16
|
+
## The problem
|
|
17
|
+
|
|
18
|
+
AI agents are prolific authors and terrible editors. Left alone they add abstractions nobody asked for, pull in dependencies for three lines of logic, and: when told to "keep it simple": confidently reach for stdlib APIs that don't exist. Minimalism-only rulesets fix the bloat and make the overclaiming *worse*: the pressure to write less rewards inventing shortcuts.
|
|
19
|
+
|
|
20
|
+
## The system
|
|
21
|
+
|
|
22
|
+
`omit` turns the agent from author into editor. Four parts:
|
|
23
|
+
|
|
24
|
+
| Part | What it does |
|
|
25
|
+
|---|---|
|
|
26
|
+
| **The Seven Omissions** | Before writing anything, try seven ways to *not* write it: omit the feature, the new code, the custom, the script, the dependency, the ceremony: stopping at the first omission that holds. What survives editing, ships. |
|
|
27
|
+
| **The Fact-Check** | No omission counts without a citation verified this session: `path:line` for "the codebase has this", real docs or a run snippet for "stdlib covers it", manifest + installed API for "the dep handles it". A hallucinated shortcut is a fabricated quote. |
|
|
28
|
+
| **The Final Draft** | Working code is a first draft. After tests go green, one ruthless edit of the agent's own diff: then a net report: files, ±lines, new deps (target: 0). Done means final draft, not green tests. |
|
|
29
|
+
| **Load-Bearing Lines** | Editing cuts fat, not walls. Validation, error handling, security, accessibility, concurrency correctness, and explicit requests are never cut: and adding them is announced, never smuggled or skipped. |
|
|
30
|
+
|
|
31
|
+
Deliberate omissions go on the record as footnotes in the code:
|
|
32
|
+
|
|
33
|
+
```js
|
|
34
|
+
// omitted: retries: single caller tolerates failure; add backoff if this goes multi-tenant
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Enforcement: asked for vs. made to
|
|
38
|
+
|
|
39
|
+
Every other skill in this genre is words the agent can ignore under context pressure. omit ships mechanisms that run *outside* the model:
|
|
40
|
+
|
|
41
|
+
| Mechanism | What it does |
|
|
42
|
+
|---|---|
|
|
43
|
+
| **Command sentinel** (hook) | Inspects every shell command BEFORE it runs and blocks the classic agent disasters: `rm -rf ~`, recursive deletes of system/drive roots, deletes through unset variables (`rm -rf $OUT/*` with `$OUT` empty), `dd` to block devices, `mkfs`, fork bombs. The user's machine is load-bearing. |
|
|
44
|
+
| **Dep sentinel** (hook) | A new dependency hits a manifest with no receipt in `.omit/receipts.jsonl` → the edit is objected to on the spot. Cite why omissions 2-5 failed, or revert. |
|
|
45
|
+
| **Hazard sentinel** (hook) | Hardcoded API keys/secrets and injection-prone patterns (string-built SQL, `eval`, shell concatenation, `innerHTML`, unsafe deserialization) are blocked the moment they land in a file. Secrets have no override; injection lines need a reviewed `omit-allow: <reason>`. |
|
|
46
|
+
| **Lint sentinel** (hook) | omit ships no lint rules. It detects the linter the repo already configured (eslint, biome, ruff, flake8) and runs it on every edited file, so the agent hears objections immediately instead of at CI time. |
|
|
47
|
+
| **Final Draft gate** (hook) | The session cannot end with an edited tree and no `.omit/final-draft.md` net report. The deletion pass is a gate, not a suggestion. |
|
|
48
|
+
| **Receipts ledger** | Every Fact-Check citation is appended to `.omit/receipts.jsonl`: an auditable trail of the agent's claims your reviewers can actually read. |
|
|
49
|
+
|
|
50
|
+
Hooks install automatically with the Claude Code plugin. Escape hatch for humans: `OMIT_OFF=1`.
|
|
51
|
+
|
|
52
|
+
## Any provider, same gates
|
|
53
|
+
|
|
54
|
+
The enforcement logic lives in a zero-dependency CLI, not in any one vendor's hook system: Claude Code's hooks are just thin adapters over it. For Cursor, Codex, Copilot, or anything else, enforce at the two chokepoints every agent passes through:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
npx @sriinnu/omit hook install # git pre-commit: audits the staged diff,
|
|
58
|
+
# fails on secrets, injections, uncited deps
|
|
59
|
+
npx @sriinnu/omit audit # net diff, new deps, hazards, omit score
|
|
60
|
+
npx @sriinnu/omit check <files> # hazard-scan specific files (wire into any hook system)
|
|
61
|
+
npx @sriinnu/omit lint [files] # run the repo's OWN linter on changed files
|
|
62
|
+
npx @sriinnu/omit guard "<cmd>" # is this shell command a disaster? (wire into any hook system)
|
|
63
|
+
npx @sriinnu/omit gate # the pre-commit check, callable from anywhere
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
And server-side, the GitHub Action comments the verdict on every PR regardless of what wrote the code:
|
|
67
|
+
|
|
68
|
+
```yaml
|
|
69
|
+
# .github/workflows/omit.yml
|
|
70
|
+
on: pull_request
|
|
71
|
+
permissions: { pull-requests: write }
|
|
72
|
+
jobs:
|
|
73
|
+
omit:
|
|
74
|
+
runs-on: ubuntu-latest
|
|
75
|
+
steps:
|
|
76
|
+
- uses: actions/checkout@v4
|
|
77
|
+
with: { fetch-depth: 0 }
|
|
78
|
+
- uses: sriinnu/omit@main
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
```
|
|
82
|
+
### omit verdict
|
|
83
|
+
- net: +61 −204 lines across 4 files
|
|
84
|
+
- new deps: 0 ✅
|
|
85
|
+
- hazards: 0 ✅
|
|
86
|
+
- footnotes: 3 recorded · load-bearing: 1 marked
|
|
87
|
+
- omit score: 91/100 🟢
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## The referee (experimental)
|
|
91
|
+
|
|
92
|
+
`bench/` is METHODOLOGY.md made runnable: paired agentic runs of the same tasks under baseline, omit, or **any competing skill**, metrics computed from the actual git diffs, all transcripts kept. The category argues about self-reported numbers; omit ships the measuring instrument. See `bench/README.md`.
|
|
93
|
+
|
|
94
|
+
## Modes
|
|
95
|
+
|
|
96
|
+
| Mode | Behavior |
|
|
97
|
+
|---|---|
|
|
98
|
+
| `margin` | Build as asked; note in the margin what could have been omitted |
|
|
99
|
+
| `redline` | **Default.** Full enforcement: Seven Omissions, Fact-Check, Final Draft |
|
|
100
|
+
| `rewrite` | Also question the assignment itself before building |
|
|
101
|
+
| `off` | Disabled until re-invoked |
|
|
102
|
+
|
|
103
|
+
Say `omit redline` (or any mode) in chat, or use `/omit <mode>` where slash commands are supported.
|
|
104
|
+
|
|
105
|
+
## Install
|
|
106
|
+
|
|
107
|
+
New here? **[GETTING-STARTED.md](GETTING-STARTED.md)** has a copy-paste setup for every agent.
|
|
108
|
+
|
|
109
|
+
**Claude Code (plugin marketplace)**: one command pair, gets you the skill plus `/omit` and `/omit-edit`:
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
/plugin marketplace add sriinnu/omit
|
|
113
|
+
/plugin install omit@omit
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
**Global command**: install once from GitHub, use everywhere:
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
npm install -g github:sriinnu/omit
|
|
120
|
+
omit init cursor # or: omit audit / omit gate / omit hook install
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
**npm / npx**: drops the right rule file into the current repo (never overwrites existing files):
|
|
124
|
+
|
|
125
|
+
```
|
|
126
|
+
npx @sriinnu/omit init # AGENTS.md (default)
|
|
127
|
+
npx @sriinnu/omit init claude # .claude/skills/omit/SKILL.md
|
|
128
|
+
npx @sriinnu/omit init cursor # .cursor/rules/omit.mdc
|
|
129
|
+
npx @sriinnu/omit init cline # .clinerules/omit.md
|
|
130
|
+
npx @sriinnu/omit init windsurf # .windsurf/rules/omit.md
|
|
131
|
+
npx @sriinnu/omit init all # everything above
|
|
132
|
+
npx @sriinnu/omit hook install # git pre-commit gate (works with ANY agent)
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
**Claude Code (manual)**: copy the skill into your project or user skills directory:
|
|
136
|
+
|
|
137
|
+
```
|
|
138
|
+
skills/omit/SKILL.md → .claude/skills/omit/SKILL.md (project)
|
|
139
|
+
~/.claude/skills/omit/SKILL.md (all projects)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
**Codex / Takumi / any AGENTS.md-aware agent**: copy `AGENTS.md` into your repo root (or append to an existing one), or `npx @sriinnu/omit init codex`.
|
|
143
|
+
|
|
144
|
+
**GitHub Copilot**: copy `.github/copilot-instructions.md` into your repo, or `npx @sriinnu/omit init copilot`.
|
|
145
|
+
|
|
146
|
+
**Cursor**: copy `.cursor/rules/omit.mdc` into your repo.
|
|
147
|
+
|
|
148
|
+
**Cline**: copy `.clinerules/omit.md` into your repo.
|
|
149
|
+
|
|
150
|
+
**Windsurf**: copy `.windsurf/rules/omit.md` into your repo.
|
|
151
|
+
|
|
152
|
+
**Anything else**: paste the contents of `AGENTS.md` into the agent's custom-instructions/rules mechanism. It's plain markdown; there is nothing to build.
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
// omitted: an MCP server: MCP exposes tools and data; omit is a behavioral
|
|
156
|
+
// discipline, and rule files + skills already deliver it. Add one only if
|
|
157
|
+
// omit ever grows verifiable tooling (e.g., a standalone diff auditor).
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
## Commands (Claude Code)
|
|
161
|
+
|
|
162
|
+
- `/omit [margin|redline|rewrite|off]`: switch or show the current mode
|
|
163
|
+
- `/omit-edit`: run an editor's pass over the current diff: flag bloat, uncited claims, missing footnotes, and cut opportunities
|
|
164
|
+
|
|
165
|
+
## Benchmarks
|
|
166
|
+
|
|
167
|
+
None yet: and we won't publish numbers we can't hand you the harness for. `benchmarks/METHODOLOGY.md` defines the measurement we consider honest (paired tasks, agentic baseline, net LOC / new deps / defect rate / load-bearing violations, full transcripts). Reproducible runs are the most welcome PR this repo can receive.
|
|
168
|
+
|
|
169
|
+
## Prior art
|
|
170
|
+
|
|
171
|
+
The minimalism-pressure idea was popularized by [ponytail](https://github.com/DietrichGebert/ponytail), which deserves its stars. `omit` differs where it matters: shortcuts require citations, the diff is edited *after* it works, safety lines are enumerated and never cut, and what's left out is footnoted instead of silent.
|
|
172
|
+
|
|
173
|
+
## License
|
|
174
|
+
|
|
175
|
+
MIT
|
package/action.yml
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
name: omit audit
|
|
2
|
+
description: Comment the omit verdict (net diff, new deps, hazards, score) on pull requests: works whatever agent or editor wrote the code.
|
|
3
|
+
branding:
|
|
4
|
+
icon: scissors
|
|
5
|
+
color: gray-dark
|
|
6
|
+
inputs:
|
|
7
|
+
base:
|
|
8
|
+
description: Base ref to diff against
|
|
9
|
+
required: false
|
|
10
|
+
default: ''
|
|
11
|
+
runs:
|
|
12
|
+
using: composite
|
|
13
|
+
steps:
|
|
14
|
+
- name: Run omit audit
|
|
15
|
+
shell: bash
|
|
16
|
+
run: |
|
|
17
|
+
BASE="${{ inputs.base }}"
|
|
18
|
+
if [ -z "$BASE" ]; then BASE="origin/${{ github.event.pull_request.base.ref }}"; fi
|
|
19
|
+
git fetch --depth=1 origin "${{ github.event.pull_request.base.ref }}" || true
|
|
20
|
+
node "${{ github.action_path }}/bin/omit.mjs" audit --base "$BASE" --markdown > omit-verdict.md
|
|
21
|
+
cat omit-verdict.md
|
|
22
|
+
- name: Comment on PR
|
|
23
|
+
shell: bash
|
|
24
|
+
env:
|
|
25
|
+
GH_TOKEN: ${{ github.token }}
|
|
26
|
+
run: gh pr comment ${{ github.event.pull_request.number }} --body-file omit-verdict.md --repo ${{ github.repository }}
|
package/bench/README.md
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# omit bench (experimental)
|
|
2
|
+
|
|
3
|
+
The runnable referee for `benchmarks/METHODOLOGY.md`: paired agentic runs of the same tasks under different rule files: baseline, omit, or any competing skill: with every metric computed from the actual git diff and every transcript kept.
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
node bench/run.mjs bench/bench.config.json
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Copy `bench.config.example.json`, point `tasks[].repo` at real fixture repos with real test suites, and pick your arms. Output lands in `bench/runs/<timestamp>/`: per-run logs, `results.json`, and a median summary table.
|
|
10
|
+
|
|
11
|
+
Status: **experimental**: the harness runs, but no official numbers exist yet because none have been produced by a run anyone can reproduce. That is the standard; it doesn't bend for the project that set it.
|
|
12
|
+
|
|
13
|
+
`// omitted: cost/token accounting: agent CLIs report spend differently; add per-agent adapters when the first real run needs them`
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"agentCmd": "claude",
|
|
3
|
+
"agentArgs": ["-p", "{prompt}", "--dangerously-skip-permissions"],
|
|
4
|
+
"timeoutMinutes": 20,
|
|
5
|
+
"arms": [
|
|
6
|
+
{ "name": "baseline" },
|
|
7
|
+
{ "name": "omit", "rules": "./AGENTS.md" },
|
|
8
|
+
{ "name": "other-skill", "rules": "../some-other-skill/AGENTS.md" }
|
|
9
|
+
],
|
|
10
|
+
"tasks": [
|
|
11
|
+
{
|
|
12
|
+
"id": "add-pagination",
|
|
13
|
+
"repo": "../fixtures/fastapi-app",
|
|
14
|
+
"prompt": "Add cursor-based pagination to the /items endpoint. Existing tests must keep passing.",
|
|
15
|
+
"testCmd": "python -m pytest -q"
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
package/bench/run.mjs
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// omit bench: EXPERIMENTAL referee harness. Runs paired agentic tasks across
|
|
3
|
+
// arms (baseline / omit / any other rule file) and reports the metrics from
|
|
4
|
+
// benchmarks/METHODOLOGY.md with full per-run logs. No numbers without receipts.
|
|
5
|
+
//
|
|
6
|
+
// node bench/run.mjs bench/bench.config.json
|
|
7
|
+
import { cpSync, mkdirSync, mkdtempSync, readFileSync, writeFileSync, existsSync } from 'node:fs'
|
|
8
|
+
import { execSync, spawnSync } from 'node:child_process'
|
|
9
|
+
import { tmpdir } from 'node:os'
|
|
10
|
+
import { basename, join, resolve } from 'node:path'
|
|
11
|
+
import { fileURLToPath } from 'node:url'
|
|
12
|
+
import { isManifest, addedDeps } from '../lib/deps.mjs'
|
|
13
|
+
import { findHazards } from '../lib/hazards.mjs'
|
|
14
|
+
|
|
15
|
+
const configPath = process.argv[2]
|
|
16
|
+
if (!configPath) {
|
|
17
|
+
console.error('usage: node bench/run.mjs <config.json> (see bench.config.example.json)')
|
|
18
|
+
process.exit(1)
|
|
19
|
+
}
|
|
20
|
+
const cfg = JSON.parse(readFileSync(configPath, 'utf8'))
|
|
21
|
+
const benchRoot = join(dirnameOf(import.meta.url), 'runs', String(Date.now()))
|
|
22
|
+
mkdirSync(benchRoot, { recursive: true })
|
|
23
|
+
|
|
24
|
+
function dirnameOf(url) {
|
|
25
|
+
return join(fileURLToPath(url), '..')
|
|
26
|
+
}
|
|
27
|
+
const sh = (cmd, cwd) => execSync(cmd, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] })
|
|
28
|
+
|
|
29
|
+
const results = []
|
|
30
|
+
for (const task of cfg.tasks) {
|
|
31
|
+
for (const arm of cfg.arms) {
|
|
32
|
+
const dir = mkdtempSync(join(tmpdir(), `omit-bench-`))
|
|
33
|
+
cpSync(resolve(task.repo), dir, { recursive: true })
|
|
34
|
+
if (arm.rules) cpSync(resolve(arm.rules), join(dir, arm.rulesDest ?? 'AGENTS.md'))
|
|
35
|
+
if (!existsSync(join(dir, '.git'))) {
|
|
36
|
+
sh('git init -q && git add -A && git -c user.email=bench@omit -c user.name=bench commit -qm base', dir)
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
console.log(`▶ ${task.id} × ${arm.name}`)
|
|
40
|
+
const t0 = Date.now()
|
|
41
|
+
// {prompt} in agentArgs is replaced inline; if absent, the prompt goes to stdin
|
|
42
|
+
// (safer quoting, and `claude -p` reads stdin in print mode).
|
|
43
|
+
const inline = cfg.agentArgs.some((a) => a.includes('{prompt}'))
|
|
44
|
+
const run = spawnSync(cfg.agentCmd, cfg.agentArgs.map((a) => a.replaceAll('{prompt}', task.prompt)), {
|
|
45
|
+
cwd: dir,
|
|
46
|
+
encoding: 'utf8',
|
|
47
|
+
input: inline ? undefined : task.prompt,
|
|
48
|
+
timeout: (cfg.timeoutMinutes ?? 20) * 60_000,
|
|
49
|
+
shell: process.platform === 'win32',
|
|
50
|
+
})
|
|
51
|
+
const durationMs = Date.now() - t0
|
|
52
|
+
|
|
53
|
+
// metrics vs the base commit
|
|
54
|
+
let added = 0, deleted = 0, newDeps = [], hazards = []
|
|
55
|
+
try {
|
|
56
|
+
sh('git add -A', dir)
|
|
57
|
+
for (const row of sh('git diff --cached --numstat', dir).split('\n').filter(Boolean)) {
|
|
58
|
+
const [a, d, path] = row.split('\t')
|
|
59
|
+
if (a === '-') continue
|
|
60
|
+
added += +a
|
|
61
|
+
deleted += +d
|
|
62
|
+
if (isManifest(path)) newDeps.push(...addedDeps(basename(path), sh(`git diff --cached -- "${path}"`, dir)))
|
|
63
|
+
}
|
|
64
|
+
const addedLines = sh('git diff --cached', dir).split('\n').filter((l) => l.startsWith('+') && !l.startsWith('+++')).map((l) => l.slice(1))
|
|
65
|
+
hazards = findHazards(addedLines)
|
|
66
|
+
} catch {}
|
|
67
|
+
|
|
68
|
+
let testPass = null
|
|
69
|
+
if (task.testCmd) {
|
|
70
|
+
const t = spawnSync(task.testCmd, { cwd: dir, shell: true, encoding: 'utf8', timeout: 10 * 60_000 })
|
|
71
|
+
testPass = t.status === 0
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const rec = { task: task.id, arm: arm.name, added, deleted, net: added - deleted, newDeps, hazards: hazards.length, testPass, durationMs, dir }
|
|
75
|
+
results.push(rec)
|
|
76
|
+
writeFileSync(join(benchRoot, `${task.id}--${arm.name}.log`), `${run.stdout ?? ''}\n--- stderr ---\n${run.stderr ?? ''}`)
|
|
77
|
+
console.log(` net ${rec.net >= 0 ? '+' : ''}${rec.net} · deps +${newDeps.length} · hazards ${rec.hazards} · tests ${testPass === null ? 'n/a' : testPass ? 'pass' : 'FAIL'} · ${(durationMs / 1000).toFixed(0)}s`)
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
writeFileSync(join(benchRoot, 'results.json'), JSON.stringify(results, null, 2))
|
|
82
|
+
|
|
83
|
+
// summary table: metric medians per arm
|
|
84
|
+
const arms = [...new Set(results.map((r) => r.arm))]
|
|
85
|
+
const med = (xs) => (xs.length ? xs.slice().sort((a, b) => a - b)[Math.floor(xs.length / 2)] : null)
|
|
86
|
+
let table = `| metric | ${arms.join(' | ')} |\n|---|${arms.map(() => '---').join('|')}|\n`
|
|
87
|
+
for (const [label, pick] of [
|
|
88
|
+
['median net LOC', (r) => r.net],
|
|
89
|
+
['new deps (total)', null],
|
|
90
|
+
['test failures', null],
|
|
91
|
+
['hazards (total)', null],
|
|
92
|
+
]) {
|
|
93
|
+
const cells = arms.map((a) => {
|
|
94
|
+
const rs = results.filter((r) => r.arm === a)
|
|
95
|
+
if (label === 'median net LOC') return med(rs.map(pick))
|
|
96
|
+
if (label === 'new deps (total)') return rs.reduce((n, r) => n + r.newDeps.length, 0)
|
|
97
|
+
if (label === 'test failures') return rs.filter((r) => r.testPass === false).length
|
|
98
|
+
return rs.reduce((n, r) => n + r.hazards, 0)
|
|
99
|
+
})
|
|
100
|
+
table += `| ${label} | ${cells.join(' | ')} |\n`
|
|
101
|
+
}
|
|
102
|
+
writeFileSync(join(benchRoot, 'summary.md'), table)
|
|
103
|
+
console.log(`\n${table}\nfull transcripts and results: ${benchRoot}`)
|