codepraxis 0.3.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codepraxis-0.4.0/ARCHITECTURE.md +319 -0
- codepraxis-0.4.0/PKG-INFO +278 -0
- codepraxis-0.4.0/README.md +253 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/pyproject.toml +1 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/__init__.py +1 -1
- codepraxis-0.4.0/src/codepraxis/cli.py +524 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/catalog.py +33 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/example.py +1 -1
- codepraxis-0.4.0/src/codepraxis/commands/guide.py +79 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/login.py +1 -1
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/publish.py +25 -16
- codepraxis-0.4.0/src/codepraxis/commands/status.py +170 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/domain/contract.py +14 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/remote/client.py +4 -1
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/remote/config.py +3 -3
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/remote/executor.py +1 -1
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/templates/commands/new.md +3 -3
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/templates/plugin.json +1 -1
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/test_publish.py +90 -0
- codepraxis-0.3.2/PKG-INFO +0 -249
- codepraxis-0.3.2/README.md +0 -224
- codepraxis-0.3.2/src/codepraxis/cli.py +0 -345
- {codepraxis-0.3.2 → codepraxis-0.4.0}/.gitignore +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/CONTRIBUTING.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/RELEASING.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/__main__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/lint.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/commands/validate.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/domain/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/domain/pack.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/domain/results.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/errors.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/executor.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/local/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/local/backends.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/local/executor.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/local/worker.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/local/workspace.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/execution/remote/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/packio/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/packio/archive.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/packio/discovery.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/packio/loader.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/packio/toc.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/installer.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/templates/commands/validate.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/templates/marketplace.json +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/plugin/templates/skills/pack-authoring/SKILL.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/reporting/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/reporting/human.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/reporting/json_reporter.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/reporting/reporter.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/generator.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/README.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/backend.conf +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/course_toc.json +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/feature.md +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/main.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/metadata.json +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/solution.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/scaffold/templates/test_1.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/registry.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/rule.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/rules/__init__.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/rules/hygiene.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/rules/instructions.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/src/codepraxis/validation/rules/testcases.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/conformance/test_corpus.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/conftest.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/test_catalog.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/test_classification.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/test_toc.py +0 -0
- {codepraxis-0.3.2 → codepraxis-0.4.0}/tests/test_validation_rules.py +0 -0
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
# How this works
|
|
2
|
+
|
|
3
|
+
This explains the shape of the CLI and why it is shaped that way. It is written
|
|
4
|
+
for someone new to the codebase. If you only want to *use* the tool, read the
|
|
5
|
+
README or run `codepraxis guide`.
|
|
6
|
+
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
## What we are actually building
|
|
10
|
+
|
|
11
|
+
CodePraxis gives candidates real engineering work instead of algorithm puzzles.
|
|
12
|
+
A company sends someone a question; they open a browser and land in a real
|
|
13
|
+
editor, in a real container, with real code in front of them. They fix or build
|
|
14
|
+
something, and hidden tests decide whether it worked.
|
|
15
|
+
|
|
16
|
+
This CLI is how a company **makes** those questions.
|
|
17
|
+
|
|
18
|
+
The hard part is not the file format. The hard part is that a good question is
|
|
19
|
+
hard to write, and a bad one is expensive: it either passes everyone, fails
|
|
20
|
+
everyone, or measures something other than the job. So most of the design here
|
|
21
|
+
is about making the good version the easy one.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
## The core idea: a question is a folder
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
challenges/webhook-debug/
|
|
29
|
+
spec.md the plan — what this tests and why
|
|
30
|
+
publish.json catalog identity — id, title, difficulty, time limit
|
|
31
|
+
metadata.json the workspace directory name
|
|
32
|
+
backend.conf question type and language
|
|
33
|
+
setup.sh installs what the question needs
|
|
34
|
+
source/ what the candidate starts from
|
|
35
|
+
._tests/test_1.py the tests, which the candidate never sees
|
|
36
|
+
._course_data/
|
|
37
|
+
course_toc.json picks the active test module
|
|
38
|
+
feature.md the instructions tab
|
|
39
|
+
solution/ the reference solution — a SIBLING, never inside
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Two files are ours, not the runner's. `spec.md` is the plan, kept in version
|
|
43
|
+
control beside the code it describes so that an edit six months later can
|
|
44
|
+
recover the original intent. `publish.json` is the question's catalog identity;
|
|
45
|
+
before it was adopted, those values were passed as one-off command-line flags,
|
|
46
|
+
which meant a question's identity lived in someone's shell history.
|
|
47
|
+
|
|
48
|
+
`solution/` sits outside the pack on purpose. That is what keeps it out of the
|
|
49
|
+
archive that goes to candidates.
|
|
50
|
+
|
|
51
|
+
### The rule everything rests on
|
|
52
|
+
|
|
53
|
+
Every question is run **twice**:
|
|
54
|
+
|
|
55
|
+
- with `source/` **plus** `solution/` overlaid — this must pass every test
|
|
56
|
+
- with `source/` **alone** — this must fail
|
|
57
|
+
|
|
58
|
+
The second one is what authors forget. If the starter passes, the tests do not
|
|
59
|
+
discriminate and every candidate scores full marks. A question that cannot fail
|
|
60
|
+
is not measuring anything.
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## The lifecycle
|
|
65
|
+
|
|
66
|
+
```
|
|
67
|
+
plan → build → review → try → ship → edit
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
| Step | What happens | Where it runs |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| **plan** | Talk it through, agree what to test, write `spec.md`. No code. | Claude, via the plugin |
|
|
73
|
+
| **build** | Write the pack, tests, setup script and reference solution. | Claude, via the plugin |
|
|
74
|
+
| **review** | Audit the result and repair what fails. Runs inside build. | Claude + CLI checks |
|
|
75
|
+
| **try** | Open it exactly as a candidate would. Nothing published. | CLI → platform |
|
|
76
|
+
| **ship** | Publish as a draft, then promote to live. | CLI → platform |
|
|
77
|
+
| **edit** | Pull a question back down, change it, publish a new version. | CLI → platform |
|
|
78
|
+
|
|
79
|
+
Two of those steps are Claude workflows and four are CLI operations. That split
|
|
80
|
+
matters: **judgement lives in the plugin, mechanics live in the CLI.** Deciding
|
|
81
|
+
whether a question is any good is not something a Python function can do.
|
|
82
|
+
Deciding whether the tests pass is not something a language model should be
|
|
83
|
+
trusted with.
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## Why plan mode carries most of the weight
|
|
88
|
+
|
|
89
|
+
Everything after planning is mechanical. Planning is the only step where a
|
|
90
|
+
wrong answer produces a bad question, so it gets two phases and two gates.
|
|
91
|
+
|
|
92
|
+
**Phase one is conversation.** Claude establishes three things early, because
|
|
93
|
+
everything else derives from them: how long the assessment is, what stack it
|
|
94
|
+
uses, and where it is coming from — a fresh idea, an existing question in some
|
|
95
|
+
other format, or a repository to mine. Then it keeps asking until it
|
|
96
|
+
understands, which is three more questions for some subjects and ten for
|
|
97
|
+
others.
|
|
98
|
+
|
|
99
|
+
It stops asking when it can write down three things: the signal in one
|
|
100
|
+
falsifiable sentence, the exact invocation and output contract, and at least
|
|
101
|
+
three cases that fail for different reasons. Without a stopping rule an
|
|
102
|
+
open-ended interview becomes an interrogation.
|
|
103
|
+
|
|
104
|
+
**Phase two is the document.** Only written once a short sketch has been agreed,
|
|
105
|
+
because producing a polished plan for the wrong question wastes everyone's time.
|
|
106
|
+
|
|
107
|
+
### Gate A — can a model just solve it?
|
|
108
|
+
|
|
109
|
+
Candidates have an AI agent and an LLM endpoint inside the container. So a
|
|
110
|
+
question a model answers from the brief alone is not an assessment.
|
|
111
|
+
|
|
112
|
+
The draft brief is handed to a model, cold, and the attempt is judged against
|
|
113
|
+
the case table. If it solves it, **the plan stops** and Claude says which
|
|
114
|
+
property made it trivial. Partial success is the target. A total miss usually
|
|
115
|
+
means the brief is unclear rather than the question being hard.
|
|
116
|
+
|
|
117
|
+
What survives the probe, strongest first: context the model has never seen
|
|
118
|
+
(your repo, your schema quirk, your log format), debugging rather than
|
|
119
|
+
authoring, decisions with no single right answer, integration volume, and
|
|
120
|
+
hidden cases that punish the obvious approach.
|
|
121
|
+
|
|
122
|
+
There is a real tension here. The clearer a brief is, the more one-shot-able it
|
|
123
|
+
becomes. The resolution is to be **precise about the contract and silent about
|
|
124
|
+
the approach**.
|
|
125
|
+
|
|
126
|
+
### Gate B — can we actually provision it?
|
|
127
|
+
|
|
128
|
+
Any stack is allowed. `setup.sh` runs on every container load, so Java, Go,
|
|
129
|
+
Rust or a database server are all just installs. The gate does not grant
|
|
130
|
+
permission; it checks that setup really provisions what the question needs and
|
|
131
|
+
prices the cost. Only genuine impossibility stops a plan — Docker-in-Docker
|
|
132
|
+
without privilege, or anything needing more than one container.
|
|
133
|
+
|
|
134
|
+
---
|
|
135
|
+
|
|
136
|
+
## Things about the container that change question design
|
|
137
|
+
|
|
138
|
+
These are real constraints, and several of them silently break otherwise
|
|
139
|
+
sensible questions.
|
|
140
|
+
|
|
141
|
+
- **`curl` is not there.** It is installed during the image build and purged on
|
|
142
|
+
the way out. `wget`, pip and npm survive.
|
|
143
|
+
- **Git cannot reach a network.** `git-upload-pack` and friends are removed.
|
|
144
|
+
Candidates can commit locally — and some questions grade that history — but
|
|
145
|
+
nothing clones or pushes.
|
|
146
|
+
- **Setup runs on every load, by every candidate.** Not once at build time. A
|
|
147
|
+
three-minute install is three minutes of a sixty-minute assessment. It also
|
|
148
|
+
means unpinned versions rot: `pip install openai` resolves to whatever is
|
|
149
|
+
current that day, and a breaking release later fails during someone's
|
|
150
|
+
assessment rather than during ours. Pin everything.
|
|
151
|
+
- **Setup runs in parallel with the first test load.** The runner imports the
|
|
152
|
+
test module right after spawning setup, which is why test modules must import
|
|
153
|
+
setup-installed packages inside methods rather than at the top.
|
|
154
|
+
- **Test cases run one at a time.** A sequential loop, each case waiting for its
|
|
155
|
+
own timeout window plus the default. Total submit time is the sum of every
|
|
156
|
+
case, so a question with ten slow cases makes the candidate wait minutes on
|
|
157
|
+
every submit.
|
|
158
|
+
|
|
159
|
+
Free in the image: Python 3, Node 20, gcc, g++, make, clangd, .NET 8, the ARM
|
|
160
|
+
toolchain, QEMU, Renode, git, tmux, and an editor with Python and C++ tooling.
|
|
161
|
+
|
|
162
|
+
---
|
|
163
|
+
|
|
164
|
+
## How the code is laid out
|
|
165
|
+
|
|
166
|
+
```
|
|
167
|
+
src/codepraxis/
|
|
168
|
+
cli.py argument parsing and wiring, nothing else
|
|
169
|
+
commands/ one module per user-facing action
|
|
170
|
+
domain/ what a pack is, what a result is, the runner contract
|
|
171
|
+
packio/ finding, loading, archiving packs
|
|
172
|
+
validation/ the lint rules
|
|
173
|
+
execution/
|
|
174
|
+
local/ the pure-Python harness
|
|
175
|
+
remote/ the platform client
|
|
176
|
+
reporting/ human and JSON output
|
|
177
|
+
scaffold/ templates for `codepraxis new`
|
|
178
|
+
plugin/ the Claude Code plugin, shipped inside the wheel
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
A few decisions worth knowing:
|
|
182
|
+
|
|
183
|
+
**No runtime dependencies.** The local harness has to run on an author's
|
|
184
|
+
machine with nothing but a Python interpreter. Even the network code uses
|
|
185
|
+
`urllib` rather than `requests`. Keep it that way.
|
|
186
|
+
|
|
187
|
+
**`cli.py` only wires things up.** Executors and reporters are chosen there and
|
|
188
|
+
injected, so commands do not know whether they are running locally or remotely,
|
|
189
|
+
or whether output is going to a terminal or to JSON.
|
|
190
|
+
|
|
191
|
+
**`domain/contract.py` is a mirror.** Every constant in it has a counterpart in
|
|
192
|
+
the production runner, with the source file and line recorded next to it.
|
|
193
|
+
Mirrors drift, so `tests/conformance/` replays the harness against a corpus of
|
|
194
|
+
real questions and asserts the verdicts. That corpus is private and lives
|
|
195
|
+
outside this repository — point `PRAXIS_CONFORMANCE_PACKS` at it.
|
|
196
|
+
|
|
197
|
+
---
|
|
198
|
+
|
|
199
|
+
## Local and remote are a trust boundary
|
|
200
|
+
|
|
201
|
+
| | Runs | Speed | Counts? |
|
|
202
|
+
|---|---|---|---|
|
|
203
|
+
| `validate --local` | Pure Python, on your machine | seconds | No — advisory |
|
|
204
|
+
| `validate --remote` | The real runner image | ~1 min | **Yes — gates publish** |
|
|
205
|
+
|
|
206
|
+
Local reproduces how the runner loads, orders and scores a question, so it
|
|
207
|
+
catches most mistakes in the inner loop. It is **not** the container: it does
|
|
208
|
+
not run `setup.sh`, does not have the image's packages, and has no LLM proxy.
|
|
209
|
+
Anything it cannot check is reported as a note rather than passing silently.
|
|
210
|
+
|
|
211
|
+
Publishing always requires a remote run. Local results never qualify, because a
|
|
212
|
+
published question can be sent to candidates immediately.
|
|
213
|
+
|
|
214
|
+
---
|
|
215
|
+
|
|
216
|
+
## Publishing rules
|
|
217
|
+
|
|
218
|
+
- **Remote validation runs first**, and the run is consumed, so one validation
|
|
219
|
+
cannot justify publishing repeatedly.
|
|
220
|
+
- **A reference solution is required.** It is what proves the question is
|
|
221
|
+
solvable.
|
|
222
|
+
- **Draft by default.** `--live` is deliberate.
|
|
223
|
+
- **The company comes from your API key.** The CLI never sends a company id.
|
|
224
|
+
Ownership is derived server-side, so a compromised or mistyped client cannot
|
|
225
|
+
publish into someone else's catalog.
|
|
226
|
+
- **`--challenge-id` adds a version** to an existing question, keeping its id,
|
|
227
|
+
assignments and history. Without it, publishing always creates a new
|
|
228
|
+
question — right the first time, a duplicate every time after.
|
|
229
|
+
|
|
230
|
+
---
|
|
231
|
+
|
|
232
|
+
## The command surface
|
|
233
|
+
|
|
234
|
+
Everything is a subcommand. The old flag forms (`--publish`, `--list`,
|
|
235
|
+
`--edit`, `--delete`, `--login`, `--install`, `--example`) still work, are
|
|
236
|
+
hidden from help, and print a pointer to their replacement. They shipped in a
|
|
237
|
+
released version, so CI jobs depend on them, and breaking those silently is
|
|
238
|
+
worse than carrying the aliases.
|
|
239
|
+
|
|
240
|
+
```
|
|
241
|
+
codepraxis where am I, what is next
|
|
242
|
+
codepraxis guide the whole thing explained
|
|
243
|
+
codepraxis login
|
|
244
|
+
codepraxis install claude-plugin
|
|
245
|
+
codepraxis new <name>
|
|
246
|
+
codepraxis lint [question]
|
|
247
|
+
codepraxis validate [question] --local | --remote
|
|
248
|
+
codepraxis ship [question] [--live] [--challenge-id N]
|
|
249
|
+
codepraxis list
|
|
250
|
+
codepraxis edit <id> [--title ...] [--open]
|
|
251
|
+
codepraxis delete <id>
|
|
252
|
+
codepraxis example
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
A bare `codepraxis` prints status, not help. It used to dump the argparse
|
|
256
|
+
listing — a wall of flags that tells a new author nothing about what to *do* —
|
|
257
|
+
and it is the first thing everyone types, which makes it the best onboarding
|
|
258
|
+
surface we have. It reads the directory and answers one question, in the spirit
|
|
259
|
+
of `git status`:
|
|
260
|
+
|
|
261
|
+
```
|
|
262
|
+
codepraxis — authoring as Acme Corp
|
|
263
|
+
|
|
264
|
+
challenges/webhook-debug
|
|
265
|
+
plan ready
|
|
266
|
+
pack 8 cases
|
|
267
|
+
solution present
|
|
268
|
+
published #214 · draft
|
|
269
|
+
|
|
270
|
+
Next:
|
|
271
|
+
codepraxis ship --live challenges/webhook-debug
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
It only reports what it can determine from local files. Whether a question has
|
|
275
|
+
*passed* validation is deliberately not guessed at — the CLI does not record run
|
|
276
|
+
results, and inventing a state we cannot observe is worse than omitting it.
|
|
277
|
+
|
|
278
|
+
Two more onboarding rules: **every command ends by printing the next one**, and
|
|
279
|
+
the guide leaves out anything Claude handles for the author. The `testCases`
|
|
280
|
+
contract and the pack layout are not in it, because an author using the plugin
|
|
281
|
+
never writes either by hand and showing them makes the job look harder than it
|
|
282
|
+
is.
|
|
283
|
+
|
|
284
|
+
---
|
|
285
|
+
|
|
286
|
+
## The Claude Code plugin
|
|
287
|
+
|
|
288
|
+
`codepraxis install claude-plugin` writes a self-contained local marketplace
|
|
289
|
+
into `.codepraxis/claude-plugin/`. Nothing is fetched from the network — the
|
|
290
|
+
templates ship inside the wheel and are copied out with `importlib.resources`,
|
|
291
|
+
so it works from a wheel, a zip or an editable install.
|
|
292
|
+
|
|
293
|
+
```
|
|
294
|
+
.codepraxis/claude-plugin/
|
|
295
|
+
.claude-plugin/marketplace.json
|
|
296
|
+
codepraxis/
|
|
297
|
+
.claude-plugin/plugin.json
|
|
298
|
+
commands/ the slash commands
|
|
299
|
+
skills/ loaded automatically when Claude touches a question
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
Installing refuses to overwrite an existing install without `--force`, because
|
|
303
|
+
an author may have edited the commands and silently reverting that is worse
|
|
304
|
+
than failing.
|
|
305
|
+
|
|
306
|
+
The skills are split by rate of change: **pack-contract** is mechanical and
|
|
307
|
+
changes when the runner changes; **question-design** is editorial and changes
|
|
308
|
+
when our opinion of a good question changes. Mixing them in one file meant
|
|
309
|
+
every editorial tweak touched the same file as the runner mirror.
|
|
310
|
+
|
|
311
|
+
---
|
|
312
|
+
|
|
313
|
+
## Where to start reading
|
|
314
|
+
|
|
315
|
+
- `cli.py` — the whole surface in one file
|
|
316
|
+
- `domain/contract.py` — everything the runner requires, with provenance
|
|
317
|
+
- `packio/loader.py` — how a directory becomes a `Pack`
|
|
318
|
+
- `execution/local/executor.py` — the harness that mirrors the runner
|
|
319
|
+
- `plugin/templates/` — the prompts that drive plan and build
|
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: codepraxis
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Author and validate CodePraxis challenge packs.
|
|
5
|
+
Project-URL: Documentation, https://docs.codepraxis.com/authoring
|
|
6
|
+
Author: CodePraxis
|
|
7
|
+
License: Proprietary
|
|
8
|
+
Keywords: assessment,authoring,challenge,codepraxis
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: build>=1; extra == 'dev'
|
|
21
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
22
|
+
Requires-Dist: ruff==0.16.2; extra == 'dev'
|
|
23
|
+
Requires-Dist: twine>=5; extra == 'dev'
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# codepraxis
|
|
27
|
+
|
|
28
|
+
Build real-world coding assessments from your own repository.
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install codepraxis
|
|
32
|
+
|
|
33
|
+
codepraxis example # see a real question, no account needed
|
|
34
|
+
codepraxis login
|
|
35
|
+
codepraxis install claude-plugin # design and build questions with Claude
|
|
36
|
+
codepraxis # where am I, what is next
|
|
37
|
+
codepraxis guide # the whole thing explained
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## What this is
|
|
41
|
+
|
|
42
|
+
CodePraxis gives candidates real engineering work instead of algorithm puzzles.
|
|
43
|
+
They open a browser, land in a real editor in a real container, and fix or build
|
|
44
|
+
something. Hidden tests decide whether it worked.
|
|
45
|
+
|
|
46
|
+
This CLI is how you make those questions — keeping them in your own git
|
|
47
|
+
repository, editing them with your own tools, and checking them before they
|
|
48
|
+
reach anyone.
|
|
49
|
+
|
|
50
|
+
A question is a directory: starter code, a test module, instructions.
|
|
51
|
+
|
|
52
|
+
## The steps
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
plan → build → try → ship
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
| | |
|
|
59
|
+
|---|---|
|
|
60
|
+
| **plan** | Talk it through, agree what to test. No code yet. |
|
|
61
|
+
| **build** | Claude writes the question, tests it, and fixes what fails. |
|
|
62
|
+
| **try** | Open it exactly as a candidate would. Nothing published. |
|
|
63
|
+
| **ship** | Publish as a draft, then go live. |
|
|
64
|
+
| **edit** | Pull one back down to change it later. |
|
|
65
|
+
|
|
66
|
+
Run `codepraxis` at any point to see where you are and what to type next.
|
|
67
|
+
|
|
68
|
+
## Working with Claude Code
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
codepraxis install claude-plugin
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Writes a local plugin into `.codepraxis/claude-plugin/`, then prints the two
|
|
75
|
+
commands that enable it. You get:
|
|
76
|
+
|
|
77
|
+
- **`/codepraxis:plan`** — design a question: what it tests, what the candidate
|
|
78
|
+
starts from, the cases, how long it should take
|
|
79
|
+
- **`/codepraxis:build`** — write it, run it, repair what fails
|
|
80
|
+
- **`/codepraxis:validate`** — check an existing question and fix it
|
|
81
|
+
- a **pack-authoring skill** that loads automatically when Claude touches a
|
|
82
|
+
question, so it already knows the contract
|
|
83
|
+
|
|
84
|
+
Re-run with `--force` to overwrite an existing install.
|
|
85
|
+
|
|
86
|
+
Planning refuses to proceed on a question a model can solve from the brief
|
|
87
|
+
alone. Candidates have an AI agent in the container, so a question that is one
|
|
88
|
+
prompt away from done measures nothing.
|
|
89
|
+
|
|
90
|
+
## Doing it by hand
|
|
91
|
+
|
|
92
|
+
The CLI works without Claude:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
codepraxis new my-question # scaffold one that already passes
|
|
96
|
+
codepraxis lint my-question # static checks, no execution
|
|
97
|
+
codepraxis validate my-question # run it, fast and advisory
|
|
98
|
+
codepraxis ship my-question # validates in the runner, then publishes
|
|
99
|
+
codepraxis list # your company's questions
|
|
100
|
+
codepraxis edit 214 --open # update details, open a preview
|
|
101
|
+
codepraxis ship my-question --challenge-id 214 # publish an edit, not a duplicate
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Two kinds of checking
|
|
105
|
+
|
|
106
|
+
| Command | Runs | Speed | Counts? |
|
|
107
|
+
|---|---|---|---|
|
|
108
|
+
| `codepraxis lint` | Static rules over the files | milliseconds | Advisory |
|
|
109
|
+
| `codepraxis validate --local` | Pure-Python harness on your machine | seconds | No — advisory |
|
|
110
|
+
| `codepraxis validate --remote` | The real runner image, on CodePraxis | ~1 min | **Yes — gates publish** |
|
|
111
|
+
|
|
112
|
+
`lint` never imports your code, so it is safe on a question you did not write
|
|
113
|
+
and fast enough for every save.
|
|
114
|
+
|
|
115
|
+
`--local` reproduces how the runner loads, orders and scores a question, so it
|
|
116
|
+
catches most mistakes in the inner loop. It is **not** the container: it does
|
|
117
|
+
not run `setup.sh` and does not have the image's package set. Anything it
|
|
118
|
+
cannot check is reported as a `note` rather than silently passing. Publishing
|
|
119
|
+
always requires a remote run.
|
|
120
|
+
|
|
121
|
+
## The rule that matters most
|
|
122
|
+
|
|
123
|
+
Every question is validated twice:
|
|
124
|
+
|
|
125
|
+
- **solution** — `source/` overlaid with `solution/`. Must pass everything.
|
|
126
|
+
- **starter** — `source/` alone. Must *fail*.
|
|
127
|
+
|
|
128
|
+
The starter run is the one authors forget. A question whose starter already
|
|
129
|
+
passes has tests that do not discriminate, and every candidate scores full
|
|
130
|
+
marks.
|
|
131
|
+
|
|
132
|
+
```
|
|
133
|
+
webhook-debug (local)
|
|
134
|
+
solution 18/18 passed
|
|
135
|
+
starter 0/18 passed
|
|
136
|
+
PASSED 8.9s
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Three verdicts: **PASSED**, **FAILED**, and **INCONCLUSIVE** (this tier lacked
|
|
140
|
+
the infrastructure to judge it — not a failure, and it does not fail the
|
|
141
|
+
command). `--json` always emits a single document with a `packs` array.
|
|
142
|
+
|
|
143
|
+
## Questions that call a model
|
|
144
|
+
|
|
145
|
+
By default there is no model endpoint locally, so cases that need one are
|
|
146
|
+
reported **unverifiable** rather than failed — a question is not broken just
|
|
147
|
+
because your laptop has no LLM proxy. Point it at a real endpoint and they run
|
|
148
|
+
for real:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
export OPENAI_API_KEY=... # or --llm-api-key
|
|
152
|
+
export OPENAI_BASE_URL=... # or --llm-base-url
|
|
153
|
+
codepraxis validate my-question
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Once a key is configured the leniency stops: a model failure is then a real
|
|
157
|
+
failure, because it can be judged.
|
|
158
|
+
|
|
159
|
+
## Layout
|
|
160
|
+
|
|
161
|
+
```
|
|
162
|
+
challenges/my-question/
|
|
163
|
+
├── spec.md the plan — what this tests and why
|
|
164
|
+
├── publish.json title, difficulty, time limit, tech stack
|
|
165
|
+
├── metadata.json {"name": "..."} — becomes the workspace directory
|
|
166
|
+
├── backend.conf {"BACKEND": "AI", "LANGUAGE": "PYTHON"}
|
|
167
|
+
├── setup.sh optional; installs dependencies (remote only)
|
|
168
|
+
├── source/ what the candidate starts from
|
|
169
|
+
├── ._tests/test_1.py a `testCases` class
|
|
170
|
+
└── ._course_data/
|
|
171
|
+
├── course_toc.json selects the active test module
|
|
172
|
+
└── feature.md the Instructions tab
|
|
173
|
+
solution/ sibling, never uploaded — the reference solution
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
`setup.sh` runs on **every** container load, not once at build time. Pin your
|
|
177
|
+
versions: an unpinned install resolves to whatever is current that day, and a
|
|
178
|
+
breaking release later fails during a candidate's assessment.
|
|
179
|
+
|
|
180
|
+
## Authentication
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
codepraxis login
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
Prompts for an API key (hidden input), verifies it, and stores it at
|
|
187
|
+
`~/.config/codepraxis/config.json` with `0600` permissions. It prints which
|
|
188
|
+
company the key publishes as — worth reading, because that is what every
|
|
189
|
+
publish is scoped to.
|
|
190
|
+
|
|
191
|
+
In CI, skip the prompt:
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
export CODEPRAXIS_TOKEN=...
|
|
195
|
+
export CODEPRAXIS_API_URL=... # optional; defaults to the production API
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## Publishing
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
codepraxis ship my-question # draft, with confirmation
|
|
202
|
+
codepraxis ship my-question --live # straight to candidates
|
|
203
|
+
codepraxis ship my-question --yes # non-interactive, for CI
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Publishing is deliberately strict, because a published question can be assigned
|
|
207
|
+
to candidates immediately:
|
|
208
|
+
|
|
209
|
+
- **Remote validation runs first.** Local results never qualify. Reuse an
|
|
210
|
+
earlier passing run with `--validation-run-id`.
|
|
211
|
+
- **A reference solution is required.** It is what proves the question is
|
|
212
|
+
solvable.
|
|
213
|
+
- **It publishes as a draft** unless you pass `--live`.
|
|
214
|
+
- **The company comes from your API key.** The CLI never sends a company id —
|
|
215
|
+
ownership is derived server-side, so a compromised or mistyped client cannot
|
|
216
|
+
publish into someone else's catalog.
|
|
217
|
+
|
|
218
|
+
### Publishing an edit
|
|
219
|
+
|
|
220
|
+
Changing the *content* of a question means re-publishing it. Pass the id, or
|
|
221
|
+
you get a second copy:
|
|
222
|
+
|
|
223
|
+
```bash
|
|
224
|
+
codepraxis ship my-question --challenge-id 214
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
With `--challenge-id` the platform adds a new version and keeps the question's
|
|
228
|
+
id, assignments and history.
|
|
229
|
+
|
|
230
|
+
### Deleting
|
|
231
|
+
|
|
232
|
+
```bash
|
|
233
|
+
codepraxis delete 214
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
Asks first, and the platform refuses once the question has been assigned to
|
|
237
|
+
anyone — deleting it then would orphan attempts and reports. To take an
|
|
238
|
+
assigned question out of circulation, set it back to draft with
|
|
239
|
+
`codepraxis edit 214 --status draft`.
|
|
240
|
+
|
|
241
|
+
## Older command forms
|
|
242
|
+
|
|
243
|
+
`--publish`, `--list`, `--edit`, `--delete`, `--login`, `--install` and
|
|
244
|
+
`--example` still work. They are hidden from help and print a pointer to the
|
|
245
|
+
subcommand that replaced them.
|
|
246
|
+
|
|
247
|
+
## Development
|
|
248
|
+
|
|
249
|
+
```bash
|
|
250
|
+
python3.11 -m pip install -e '.[dev]'
|
|
251
|
+
pytest
|
|
252
|
+
ruff check src tests scripts
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
The package has **no runtime dependencies**. The harness must run on an
|
|
256
|
+
author's machine with nothing but a Python interpreter, so keep it that way —
|
|
257
|
+
the remote tier uses `urllib` from the standard library for the same reason.
|
|
258
|
+
|
|
259
|
+
See [ARCHITECTURE.md](ARCHITECTURE.md) for how the pieces fit together.
|
|
260
|
+
|
|
261
|
+
### Conformance tests
|
|
262
|
+
|
|
263
|
+
The harness mirrors the production runner's behaviour (`setupCodeBase.py`,
|
|
264
|
+
`koro/test_loader.py`, `koro/test_runner.py`). Mirrors drift, so
|
|
265
|
+
`tests/conformance/` replays the harness across a corpus of real questions and
|
|
266
|
+
asserts the expected verdicts.
|
|
267
|
+
|
|
268
|
+
That corpus is **private and lives outside this repository**:
|
|
269
|
+
|
|
270
|
+
```bash
|
|
271
|
+
PRAXIS_CONFORMANCE_PACKS=/path/to/question-bank pytest tests/conformance
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
Without the variable the conformance tests skip.
|
|
275
|
+
|
|
276
|
+
> **Do not vendor questions into this repository.** This package is published
|
|
277
|
+
> publicly. No challenge content, no reference solutions, no fixtures derived
|
|
278
|
+
> from real questions. Scaffold templates must be written from scratch.
|