org-knowledge-layer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okl/__init__.py +12 -0
- okl/__main__.py +8 -0
- okl/bootstrap.py +83 -0
- okl/cli.py +484 -0
- okl/client.py +160 -0
- okl/core.py +223 -0
- okl/drift.py +119 -0
- okl/mcp_server.py +75 -0
- okl/scaffold/MANIFEST.md +59 -0
- okl/scaffold/ci/method-gates.yml +32 -0
- okl/scaffold/ci/okl-verify.yml +59 -0
- okl/scaffold/claude/agents/architecture-reviewer.md +41 -0
- okl/scaffold/claude/commands/check-rules.md +24 -0
- okl/scaffold/claude/commands/feature-spec.md +37 -0
- okl/scaffold/claude/rules/example-area.md +22 -0
- okl/scaffold/claude/skills/RECOMMENDED-COMPANIONS.md +40 -0
- okl/scaffold/claude/skills/encoding-loop/SKILL.md +48 -0
- okl/scaffold/claude/skills/verify-before-claiming/SKILL.md +56 -0
- okl/scaffold/evals/README.md +32 -0
- okl/scaffold/evals/cases.jsonl +1 -0
- okl/scaffold/evals/run_evals.py +109 -0
- okl/scaffold/gates/check-canon-size.sh +11 -0
- okl/scaffold/gates/check-doc-orphans.sh +19 -0
- okl/scaffold/gates/check-retractions.sh +22 -0
- okl/scaffold/gates/check-tombstones.sh +22 -0
- okl/scaffold/gates/run-gates.sh +31 -0
- okl/scaffold/hooks/hooks.json +16 -0
- okl/scaffold/hooks/stop-okl-encode.sh +78 -0
- okl/scaffold/hooks/userpromptsubmit-okl-check.sh +68 -0
- okl/scaffold/plugin/plugin.json +10 -0
- okl/scaffold/profiles/dotnet/README.md +12 -0
- okl/scaffold/profiles/dotnet/rules/architecture.md +55 -0
- okl/scaffold/profiles/dotnet/rules/messaging.md +31 -0
- okl/scaffold/profiles/dotnet/rules/performance-and-data.md +36 -0
- okl/scaffold/profiles/dotnet/rules/security.md +42 -0
- okl/scaffold/profiles/geospatial/README.md +6 -0
- okl/scaffold/profiles/geospatial/rules/geospatial-ml.md +38 -0
- okl/scaffold/profiles/python-rag/README.md +13 -0
- okl/scaffold/profiles/python-rag/rules/fastapi-backend.md +37 -0
- okl/scaffold/profiles/python-rag/rules/project-structure.md +28 -0
- okl/scaffold/profiles/python-rag/rules/rag-pipeline.md +73 -0
- okl/scaffold/profiles/react/README.md +18 -0
- okl/scaffold/profiles/react/rules/frontend.md +57 -0
- okl/scaffold/registries/RETRACTIONS.md +19 -0
- okl/scaffold/registries/tombstones.txt +7 -0
- okl/scaffold/root/CLAUDE.md +55 -0
- okl/scaffold/root/METHOD.md +64 -0
- okl/scaffold_cmd.py +110 -0
- okl/seed/dotnet-canon.json +489 -0
- okl/seed/dotnet-decisions.json +328 -0
- okl/seed/dotnet-defects.json +133 -0
- okl/seed/dotnet-review-surfaces.json +147 -0
- okl/seed/frontend-canon.json +116 -0
- okl/seed/geospatial-deeptime-defects.json +59 -0
- okl/seed/geospatial-defects.json +154 -0
- okl/seed/geospatial-enforcement-defects.json +121 -0
- okl/seed/geospatial-eval-defects.json +25 -0
- okl/seed/rag-defects.json +120 -0
- okl/seed/react-defects.json +45 -0
- okl/seed.py +55 -0
- okl/service.py +137 -0
- okl/store.py +432 -0
- org_knowledge_layer-0.1.0.dist-info/METADATA +475 -0
- org_knowledge_layer-0.1.0.dist-info/RECORD +67 -0
- org_knowledge_layer-0.1.0.dist-info/WHEEL +4 -0
- org_knowledge_layer-0.1.0.dist-info/entry_points.txt +2 -0
- org_knowledge_layer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# Installed by `okl init` into .github/workflows/okl-verify.yml (also stamped by `okl scaffold`).
|
|
2
|
+
#
|
|
3
|
+
# The knowledge layer's CI verifier — real checks, not placeholders:
|
|
4
|
+
# 1. `okl drift --gate` — fails the build when a rule's governed files changed after the
|
|
5
|
+
# rule was last verified (a stale rule is a rule nobody re-checked).
|
|
6
|
+
# 2. `gates/run-gates.sh` — the repo's mechanical method gates, when the scaffold is present.
|
|
7
|
+
#
|
|
8
|
+
# Shared-layer connection is optional: without OKL_SERVICE_URL the drift gate runs against
|
|
9
|
+
# the repo's local .okl store. Configure the secrets to verify against the org layer and to
|
|
10
|
+
# let gate scripts emit receipts (`okl link <gate_id> VERIFIED_ON <defect_id>`) — emit those
|
|
11
|
+
# from inside the gate that proved itself against real drift, where the ids are known.
|
|
12
|
+
name: okl-verify
|
|
13
|
+
|
|
14
|
+
on:
|
|
15
|
+
pull_request:
|
|
16
|
+
push:
|
|
17
|
+
branches: [main]
|
|
18
|
+
|
|
19
|
+
jobs:
|
|
20
|
+
okl-verify:
|
|
21
|
+
runs-on: ubuntu-latest
|
|
22
|
+
steps:
|
|
23
|
+
- uses: actions/checkout@v4
|
|
24
|
+
with:
|
|
25
|
+
fetch-depth: 0 # drift needs history: it compares file mtimes/commits vs verified_at
|
|
26
|
+
- uses: actions/setup-python@v5
|
|
27
|
+
with:
|
|
28
|
+
python-version: "3.13"
|
|
29
|
+
- name: Install okl
|
|
30
|
+
# Consumer repos install the released package; in okl's own repo this
|
|
31
|
+
# workflow dogfoods the working tree (pip install okl would 404 — not on PyPI yet).
|
|
32
|
+
run: |
|
|
33
|
+
if grep -q '^name = "okl"' pyproject.toml 2>/dev/null; then
|
|
34
|
+
pip install -e .
|
|
35
|
+
else
|
|
36
|
+
pip install okl
|
|
37
|
+
fi
|
|
38
|
+
|
|
39
|
+
- name: Connect to the shared layer (optional — skipped when secrets are unset)
|
|
40
|
+
env:
|
|
41
|
+
OKL_SERVICE_URL: ${{ secrets.OKL_SERVICE_URL }}
|
|
42
|
+
OKL_TOKEN: ${{ secrets.OKL_TOKEN }}
|
|
43
|
+
run: |
|
|
44
|
+
if [ -n "${OKL_SERVICE_URL:-}" ]; then
|
|
45
|
+
okl connect "$OKL_SERVICE_URL" --token "$OKL_TOKEN"
|
|
46
|
+
else
|
|
47
|
+
echo "no OKL_SERVICE_URL secret — verifying against the repo-local store"
|
|
48
|
+
fi
|
|
49
|
+
|
|
50
|
+
- name: Drift gate — rules whose governed code changed after last verification
|
|
51
|
+
run: okl drift --gate
|
|
52
|
+
|
|
53
|
+
- name: Method gates (when the scaffold kit is present)
|
|
54
|
+
run: |
|
|
55
|
+
if [ -x gates/run-gates.sh ]; then
|
|
56
|
+
bash gates/run-gates.sh
|
|
57
|
+
else
|
|
58
|
+
echo "no gates/run-gates.sh — method-kit gates not installed (okl scaffold adds them)"
|
|
59
|
+
fi
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: architecture-reviewer
|
|
3
|
+
description: Reviews a diff or a proposed change against this repo's encoded rules and the org knowledge layer. Invoke on non-pattern-conforming changes (new bounded context, novel dependency, security-model change, multi-step refactor) and before merging anything touching published results. Returns a pattern-checklist verdict, not a rewrite.
|
|
4
|
+
tools: Read, Grep, Glob, Bash
|
|
5
|
+
model: inherit
|
|
6
|
+
memory: project # persists to .claude/agent-memory/architecture-reviewer/ (committed, team-shared)
|
|
7
|
+
skills:
|
|
8
|
+
- encoding-loop
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Architecture reviewer
|
|
12
|
+
|
|
13
|
+
You are a reviewer, not an implementer. You read the change and judge it against the encoded body —
|
|
14
|
+
you do not rewrite it. Your output is a checklist verdict with specific file:line findings.
|
|
15
|
+
|
|
16
|
+
## Before you start
|
|
17
|
+
1. Run `okl check --task "<one-line summary of the change under review>"` and treat the returned
|
|
18
|
+
armed gates, retractions, tombstones, and THREAT prior-art as binding review criteria.
|
|
19
|
+
2. Read `.claude/rules/` entries whose `paths:` match the changed files.
|
|
20
|
+
|
|
21
|
+
## Pattern checklist (extend as the repo encodes new rules)
|
|
22
|
+
- **Smallest-surface / speculative coupling** — is there an abstraction (interface, factory, layer)
|
|
23
|
+
with exactly one implementation and no test substitution and no concrete second impl on the
|
|
24
|
+
roadmap? If so, flag it as speculative — the concrete class should be used directly.
|
|
25
|
+
- **Assert-from-memory** — does any spec/config/scaffold claim something is "correct"/"valid"
|
|
26
|
+
without a mechanical `validate` step? Flag it; this is the #14-class defect.
|
|
27
|
+
- **Retracted claim restated** — does any prose restate a claim in `registries/RETRACTIONS.md`
|
|
28
|
+
without also retracting it? Fail.
|
|
29
|
+
- **Resurrected identifier** — does the diff reintroduce anything in `registries/tombstones.txt`? Fail.
|
|
30
|
+
- **Missing gate receipt** — does a fix reference a defect class that has a gate, without the gate
|
|
31
|
+
running in CI on this change?
|
|
32
|
+
- **Report-the-result** — if this change is a "fix" that confirms a hypothesis, was the control run?
|
|
33
|
+
|
|
34
|
+
<!-- <<FILL: STACK-SPECIFIC REVIEW CHECKS>>
|
|
35
|
+
Add checks specific to this stack (e.g. IDOR predicate in the SQL Where clause; N+1 queries;
|
|
36
|
+
async-over-sync; mass-assignment of server-controlled fields). Keep each as one scan rule. -->
|
|
37
|
+
|
|
38
|
+
## Memory
|
|
39
|
+
Use your persistent memory dir to accumulate this repo's recurring findings and architectural
|
|
40
|
+
decisions across reviews. When you see the same finding class a third time, propose promoting it to
|
|
41
|
+
a mechanical gate (surface 5) via the encoding-loop skill.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Run all mechanical drift-gates locally (the same set CI enforces) and print the worklist of any drift found.
|
|
3
|
+
disable-model-invocation: true
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# /check-rules — run the drift-gates locally
|
|
7
|
+
|
|
8
|
+
Run the full gate suite and report, exactly as CI will:
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
bash gates/run-gates.sh --untracked
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
This runs:
|
|
15
|
+
- **retractions** — fails any doc that states a retracted claim without retracting it
|
|
16
|
+
- **tombstones** — fails any doc/comment/config resurrecting a retired identifier
|
|
17
|
+
- **doc-orphans** — fails any spec/ADR/audit not reachable from the docs hub
|
|
18
|
+
- **stale-open-items** — fails any registry item marked "open" that has been resolved
|
|
19
|
+
- **canon-size** — warns/fails if `CLAUDE.md` exceeds the size budget
|
|
20
|
+
|
|
21
|
+
<!-- <<FILL: STACK-SPECIFIC GATES>> Add your build/analyzer/test gates here and in gates/run-gates.sh. -->
|
|
22
|
+
|
|
23
|
+
If anything fails, encode the fix at the smallest surface (see the `encoding-loop` skill) and
|
|
24
|
+
re-run. Do not merge with a red gate; a broken gate that reports "clean" is worse than no gate.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Draft a structured feature spec — value gate, significance check, prior-art check, and a validate plan. The ritual at the start of any non-trivial work.
|
|
3
|
+
argument-hint: "<one-line feature description>"
|
|
4
|
+
disable-model-invocation: true
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# /feature-spec — the front of the loop
|
|
8
|
+
|
|
9
|
+
Produce a short spec for: **$ARGUMENTS**
|
|
10
|
+
|
|
11
|
+
Do NOT write implementation code. Produce the spec, then stop for review.
|
|
12
|
+
|
|
13
|
+
## 1. Value gate
|
|
14
|
+
- What problem does this solve, and for whom? What breaks if we don't build it?
|
|
15
|
+
- What is the smallest version that delivers the value? (Build that; note the rest as deferred.)
|
|
16
|
+
- Set a stop condition up front (a token/time budget for experiments). When it runs out, the
|
|
17
|
+
default is "we learned what we needed; we don't continue."
|
|
18
|
+
|
|
19
|
+
## 2. Significance check
|
|
20
|
+
- Is this **pattern-conforming** (matches an existing shape → gate review suffices) or
|
|
21
|
+
**non-pattern-conforming** (new bounded context, novel dependency, security-model change,
|
|
22
|
+
multi-step refactor → stay present during implementation, route through architecture-reviewer)?
|
|
23
|
+
|
|
24
|
+
## 3. Prior-art / knowledge check (mandatory)
|
|
25
|
+
- Run `okl check --task "$ARGUMENTS"`. Record any armed gates to adopt, retractions to respect,
|
|
26
|
+
and THREAT prior-art. If this is a research/novelty claim, treat prior art as falsifying until shown otherwise.
|
|
27
|
+
|
|
28
|
+
## 4. Contract + validate plan
|
|
29
|
+
- Inputs, outputs, invariants, error cases.
|
|
30
|
+
- **Every assertion in this spec that could be wrong from memory gets a mechanical `validate` step.**
|
|
31
|
+
(Do the class paths exist? Does the config parse? Are the class counts right?) List them as
|
|
32
|
+
gate/test items, not prose claims. This is the one SDD rung we keep.
|
|
33
|
+
|
|
34
|
+
## 5. Handoff
|
|
35
|
+
- List the tasks. Flag which are pattern-conforming vs not. Name the gates that must be green to merge.
|
|
36
|
+
|
|
37
|
+
<!-- <<FILL: STACK-SPECIFIC SPEC SECTIONS>> e.g. migration plan, API version bump, deploy steps. -->
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Conventions for <AREA> — loaded only when the agent touches matching files.
|
|
3
|
+
paths:
|
|
4
|
+
- "<<FILL: glob, e.g. **/*.sql or src/payments/**>>"
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# <AREA> rules (path-scoped — surface 2, lazy-loaded)
|
|
8
|
+
|
|
9
|
+
<!-- This is a TEMPLATE for a path-scoped rule file. Copy it per area, set `paths:`, keep it short.
|
|
10
|
+
The point of path-scoped rules is that CLAUDE.md stays lean: rules load into context ONLY when
|
|
11
|
+
the agent is working on files that match the glob. Target ~30–60 lines each.
|
|
12
|
+
|
|
13
|
+
Example (delete and replace):
|
|
14
|
+
|
|
15
|
+
# Database rules (paths: **/*.sql, **/migrations/**)
|
|
16
|
+
- Every migration is reversible; ship the down-migration in the same PR.
|
|
17
|
+
- No `SELECT *` in application queries; name columns so a schema change breaks loudly.
|
|
18
|
+
- Money is DECIMAL, never float; never round in the DB layer.
|
|
19
|
+
-->
|
|
20
|
+
|
|
21
|
+
- <<FILL: rule 1>>
|
|
22
|
+
- <<FILL: rule 2>>
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# Recommended companion skills (install separately)
|
|
2
|
+
|
|
3
|
+
This kit ships two first-party skills — **`encoding-loop`** (turn a finding into a
|
|
4
|
+
promoted, recorded rule) and **`verify-before-claiming`** (evidence before assertion).
|
|
5
|
+
They are the two disciplines specific to this method.
|
|
6
|
+
|
|
7
|
+
The broader engineering-discipline skills that pair well with it — systematic debugging,
|
|
8
|
+
test-driven development, plan writing/execution, git-worktree isolation — are **not
|
|
9
|
+
bundled here on purpose.** The best-maintained open versions live in third-party skill
|
|
10
|
+
collections, and vendoring them would mean redistributing someone else's work (with its
|
|
11
|
+
own license and attribution) and shipping cross-references to sibling skills this kit
|
|
12
|
+
doesn't contain. Point at them instead of copying:
|
|
13
|
+
|
|
14
|
+
## Superpowers (obra)
|
|
15
|
+
|
|
16
|
+
A collection of Claude Code skills for disciplined agentic development. The skills below
|
|
17
|
+
reference each other via the `superpowers:` namespace, so install the collection as a
|
|
18
|
+
unit rather than cherry-picking files:
|
|
19
|
+
|
|
20
|
+
- `systematic-debugging` — root-cause-before-symptom, pressure-resistant debugging
|
|
21
|
+
- `test-driven-development` — red/green/refactor discipline
|
|
22
|
+
- `writing-plans` / `executing-plans` — spec → reviewed multi-step execution
|
|
23
|
+
- `using-git-worktrees` — isolated workspace per feature
|
|
24
|
+
- `verification-before-completion` — the upstream cousin of this kit's `verify-before-claiming`
|
|
25
|
+
|
|
26
|
+
Install: follow the Superpowers project's own instructions (it manages its own namespace
|
|
27
|
+
and inter-skill references). Verify its license permits your use before redistributing it
|
|
28
|
+
inside your own repo.
|
|
29
|
+
|
|
30
|
+
## How they fit with this method
|
|
31
|
+
|
|
32
|
+
- Use **`encoding-loop`** whenever any of the companion skills surfaces a durable lesson —
|
|
33
|
+
a debugging root cause, a test that should always exist, a plan step that must not be
|
|
34
|
+
skipped — and record it to the knowledge layer so the next task inherits it.
|
|
35
|
+
- Use **`verify-before-claiming`** as the last step of any companion skill that ends in a
|
|
36
|
+
"done"/"passing"/"fixed" claim.
|
|
37
|
+
|
|
38
|
+
Keeping the general skills external and the method skills first-party is deliberate: the
|
|
39
|
+
kit stays cleanly installable and license-clean, and you always run the companions'
|
|
40
|
+
latest upstream versions rather than a frozen copy that drifts.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: encoding-loop
|
|
3
|
+
description: Use when you discover a rule, antipattern, or lesson worth keeping ("we should never write this again" / "we should always do this when") and need to encode it at the right surface and record it to the org knowledge layer. Fires on review findings, debugging discoveries, audits, and prior-art hits.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Encoding loop — turn a finding into a durable, promoted rule
|
|
7
|
+
|
|
8
|
+
A trigger surfaced a candidate rule. The response is always the same two moves.
|
|
9
|
+
|
|
10
|
+
## 1. Pick the smallest sufficient surface (softest→strongest)
|
|
11
|
+
|
|
12
|
+
| # | Surface | Use when |
|
|
13
|
+
|---|---|---|
|
|
14
|
+
| 1 | `docs/` + diagram | the *why* needs more than a line, or reviewers need a picture |
|
|
15
|
+
| 2 | `CLAUDE.md` | every session needs it, always-on (keep it lean — prefer a rule file) |
|
|
16
|
+
| 2 | `.claude/rules/<area>.md` with `paths:` glob | scoped to a file category (loads only in context) |
|
|
17
|
+
| 3 | `.claude/skills/` or `.claude/commands/` | a multi-step procedure with real logic |
|
|
18
|
+
| 4 | `.coderabbit.yaml` path_instructions / `architecture-reviewer` checklist | catch at PR-review time |
|
|
19
|
+
| 5 | a gate in `gates/` + CI | mechanical, build-breaking — for rules that keep being broken |
|
|
20
|
+
|
|
21
|
+
Default to the *softest* surface that could hold the rule. **Promote down (toward 5) only as it
|
|
22
|
+
earns it** — a rule that keeps being violated moves to a sterner tier, not a sterner paragraph.
|
|
23
|
+
Most rules never leave tier 1, and that is fine.
|
|
24
|
+
|
|
25
|
+
## 2. Record it to the org layer so other repos inherit it
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
okl record --type <Defect|Gate|Rule|Claim|Retraction|Tombstone|Decision|PriorArt> \
|
|
29
|
+
--title "..." --body "..." --scope <org|repo> [--verified]
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
- **`scope=org`** — a fact about the world: prior art, an API contract, a data-source gotcha, a
|
|
33
|
+
portable gate. It will surface in every connected repo's `okl check`.
|
|
34
|
+
- **`scope=repo`** — a quirk true only of this codebase. Stays local.
|
|
35
|
+
|
|
36
|
+
If the finding retracts a prior claim, also add it to `registries/RETRACTIONS.md`; if it retires an
|
|
37
|
+
identifier, add it to `registries/tombstones.txt`. The CI gates then fail any doc that contradicts.
|
|
38
|
+
|
|
39
|
+
## 3. Verify the gate (if you made one)
|
|
40
|
+
|
|
41
|
+
A gate you have only seen pass is a gate you have not tested. Run it against the real drifted file
|
|
42
|
+
from git history and confirm it FAILS, then confirm it passes on the fix. Only then is it a gate.
|
|
43
|
+
|
|
44
|
+
## Gotchas
|
|
45
|
+
- A merged fix without the encoded rule is a half-finished job — the next instance slips through.
|
|
46
|
+
- A surface nobody runs is documentation, not enforcement. If it is not read/executed, it will drift.
|
|
47
|
+
- The drift you notice is the drift that embarrasses you; the flattering kind (stale "open" items)
|
|
48
|
+
needs a mechanical `verified_at`/TTL check, not vigilance.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: verify-before-claiming
|
|
3
|
+
description: Use before stating that work is done, a test passes, a build is green, a bug is fixed, a file exists, or any factual claim about the state of the code or the world. Requires running the check and reading its real output before asserting the result — evidence precedes assertion, always.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Verify before claiming
|
|
7
|
+
|
|
8
|
+
A claim about state — "the tests pass," "the file is created," "the endpoint returns
|
|
9
|
+
404," "19 tests green" — is only worth as much as the check behind it. Stating a result
|
|
10
|
+
you *expect* as if it were a result you *observed* is the single most common way
|
|
11
|
+
AI-assisted work goes fast, fluent, plausible, and wrong.
|
|
12
|
+
|
|
13
|
+
This skill has one rule: **run the check, read the output, then make the claim — in that
|
|
14
|
+
order.** If you have not run the check this turn, you do not have the result; you have a
|
|
15
|
+
guess. Say "I expect" for a guess and "I confirmed" only for an observation.
|
|
16
|
+
|
|
17
|
+
## When this fires
|
|
18
|
+
|
|
19
|
+
- About to write "done", "fixed", "passing", "green", "works", "created", "verified".
|
|
20
|
+
- About to put a count, a status, or a value into prose, a commit message, a README, a
|
|
21
|
+
PR description, or a report.
|
|
22
|
+
- About to tell a human that a thing is true about the code or the system.
|
|
23
|
+
|
|
24
|
+
## The procedure
|
|
25
|
+
|
|
26
|
+
1. **Name the check.** What single command or observation would make this claim true or
|
|
27
|
+
false? (`pytest -q`, `ls path`, `curl -s -o /dev/null -w '%{http_code}'`, `git status`.)
|
|
28
|
+
2. **Run it this turn.** Not "I ran it earlier" — state drifts; a file moved, a kernel
|
|
29
|
+
reset, an install got lost. Re-run it now.
|
|
30
|
+
3. **Read the real output.** The exit code and the actual lines — not the first line, not
|
|
31
|
+
what you assume follows.
|
|
32
|
+
4. **Claim only what the output shows.** "18 passed, 1 skipped" — not "all green." If the
|
|
33
|
+
output surprises you, the output wins; investigate before you claim.
|
|
34
|
+
5. **If you cannot run the check, say so.** "I could not run the suite in this
|
|
35
|
+
environment, so I have not confirmed the count" is a true statement. "Tests pass" when
|
|
36
|
+
you didn't run them is not.
|
|
37
|
+
|
|
38
|
+
## Anti-patterns this exists to stop
|
|
39
|
+
|
|
40
|
+
- **Reciting a remembered number.** "19 passing tests" carried from three turns ago, when
|
|
41
|
+
the suite was never run this session (or was, and now one fails). Re-run; report today's
|
|
42
|
+
result.
|
|
43
|
+
- **Reporting the expected instead of the observed.** Writing the result the code *should*
|
|
44
|
+
produce as the result it *did* produce. These diverge exactly when it matters.
|
|
45
|
+
- **"Ran it" ≠ "passed it".** A command that executed is not a command that succeeded.
|
|
46
|
+
Check the exit code, not just that it ran.
|
|
47
|
+
- **Success by absence.** "No errors shown" is not "it worked" — a check that did not run
|
|
48
|
+
produces no errors either. Absence of a failure signal and presence of a success signal
|
|
49
|
+
are different things (see also: the knowledge layer's fail-closed rule).
|
|
50
|
+
|
|
51
|
+
## The link to the rest of the method
|
|
52
|
+
|
|
53
|
+
This is the personal-discipline version of what the mechanical gates enforce at CI time
|
|
54
|
+
and what `okl check` enforces at task time: **a claim with no check behind it is
|
|
55
|
+
documentation, not verification.** When a verification you skipped later turns into a
|
|
56
|
+
defect, that defect is worth encoding (`okl record`) so the next task is warned.
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# Evals — the measurement tier (surface 5, for behavior)
|
|
2
|
+
|
|
3
|
+
The gates in `gates/` catch drift in *artifacts* (docs, identifiers, canon size). This tier catches
|
|
4
|
+
drift in *behavior* — did the change make the system measurably worse? It is here because the RAG service's
|
|
5
|
+
single most dangerous defect was a metric that could not report its own unreliability
|
|
6
|
+
("LLM Judge 5.0/5.0 while 19 of 20 cases crashed").
|
|
7
|
+
|
|
8
|
+
## The three rules this harness enforces (earned, not invented)
|
|
9
|
+
|
|
10
|
+
1. **A metric that cannot report its own failure rate is not a metric.** `run_evals.py` leads with
|
|
11
|
+
its failure count and prints `❌ RESULTS NOT USABLE` above a configurable failure-rate threshold —
|
|
12
|
+
*before* any score. A green average over a crashed suite is the failure mode to prevent.
|
|
13
|
+
2. **The judge must not grade its own homework.** If you use an LLM judge, it must be a different
|
|
14
|
+
model from the generator. The harness refuses to run if `JUDGE_MODEL == GENERATOR_MODEL`.
|
|
15
|
+
3. **Bake in the error-analysis cross-tab.** Every eval run emits the `retrieval_hit × judge_score`
|
|
16
|
+
(or your task's equivalent `precondition × outcome`) cross-tab, because that 2×2 is the only thing
|
|
17
|
+
that told the RAG service which component was actually failing. It runs on every eval, not as a one-off.
|
|
18
|
+
|
|
19
|
+
## Golden set from real failures — not invented fixtures
|
|
20
|
+
|
|
21
|
+
Rule 4 of the method: *fixtures you invented cannot falsify assumptions you hold.* Seed `cases.jsonl`
|
|
22
|
+
from real inputs that actually broke (the RAG service used its 25 real EDGAR filenames verbatim), not from
|
|
23
|
+
hand-written happy-path examples.
|
|
24
|
+
|
|
25
|
+
## Files
|
|
26
|
+
- `run_evals.py` — the harness (framework-agnostic; adapt `evaluate_one()` to your system)
|
|
27
|
+
- `cases.jsonl` — the golden set (`<<FILL>>` with real cases)
|
|
28
|
+
- results write to `results/` with the failure count and cross-tab at the top
|
|
29
|
+
|
|
30
|
+
## Wire into CI
|
|
31
|
+
Add to `ci/` after the gates: a regression fails the build if the usable pass-rate drops below the
|
|
32
|
+
committed baseline. Never quote a number from a run whose failure rate tripped `RESULTS NOT USABLE`.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"id": "example-1", "input": "<<FILL: a real input that actually broke>>", "expected": "<<FILL: expected behavior>>"}
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Eval harness — measures behavior and, crucially, reports its own unreliability first.
|
|
3
|
+
|
|
4
|
+
Framework-agnostic: plug your system into `evaluate_one()`. The invariants it enforces are the
|
|
5
|
+
earned lessons, not any particular eval library (works alongside deepeval/ragas/promptfoo or none).
|
|
6
|
+
|
|
7
|
+
Usage: python run_evals.py [--cases cases.jsonl] [--fail-rate 0.20]
|
|
8
|
+
Env: GENERATOR_MODEL, JUDGE_MODEL (must differ — rule: no grading your own homework)
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
from collections import Counter
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def evaluate_one(case: dict) -> dict:
|
|
22
|
+
"""<<FILL>>: run YOUR system on `case` and return a result dict.
|
|
23
|
+
|
|
24
|
+
Required keys in the returned dict:
|
|
25
|
+
- "ok": bool — did the case COMPLETE without crashing (not "did it pass")
|
|
26
|
+
- "score": float|None — the judge/quality score, or None if it didn't complete
|
|
27
|
+
- "precondition": bool — the x-axis of the cross-tab (e.g. retrieval_hit); task-specific
|
|
28
|
+
- "outcome_bad": bool — the y-axis (e.g. answer judged bad)
|
|
29
|
+
- "error": str|None
|
|
30
|
+
Replace the stub below with a real call into your app.
|
|
31
|
+
"""
|
|
32
|
+
# STUB — deterministic placeholder so the harness runs before you wire your system.
|
|
33
|
+
q = case.get("input", "")
|
|
34
|
+
return {"ok": True, "score": 5.0 if q else 0.0,
|
|
35
|
+
"precondition": bool(q), "outcome_bad": not bool(q), "error": None}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def main(argv=None) -> int:
|
|
39
|
+
ap = argparse.ArgumentParser()
|
|
40
|
+
ap.add_argument("--cases", default=str(Path(__file__).parent / "cases.jsonl"))
|
|
41
|
+
ap.add_argument("--fail-rate", type=float, default=0.20,
|
|
42
|
+
help="usable-results threshold; above this, results are NOT USABLE")
|
|
43
|
+
args = ap.parse_args(argv)
|
|
44
|
+
|
|
45
|
+
# Rule 2: the judge must not be the generator.
|
|
46
|
+
gen, judge = os.environ.get("GENERATOR_MODEL"), os.environ.get("JUDGE_MODEL")
|
|
47
|
+
if gen and judge and gen == judge:
|
|
48
|
+
print(f"❌ REFUSING TO RUN: judge model == generator model ({gen}). "
|
|
49
|
+
"A model grading its own output is a mirror, not a signal.", file=sys.stderr)
|
|
50
|
+
return 3
|
|
51
|
+
|
|
52
|
+
path = Path(args.cases)
|
|
53
|
+
if not path.exists():
|
|
54
|
+
print(f"no cases file at {path} — add a golden set (see README). Nothing to measure.")
|
|
55
|
+
return 0
|
|
56
|
+
cases = [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
|
|
57
|
+
if not cases:
|
|
58
|
+
print("cases.jsonl is empty — add real failing cases before trusting any number.")
|
|
59
|
+
return 0
|
|
60
|
+
|
|
61
|
+
results = []
|
|
62
|
+
for c in cases:
|
|
63
|
+
try:
|
|
64
|
+
r = evaluate_one(c)
|
|
65
|
+
except Exception as e: # a crash is a completed-with-failure, counted as such
|
|
66
|
+
r = {"ok": False, "score": None, "precondition": False, "outcome_bad": True, "error": repr(e)}
|
|
67
|
+
results.append(r)
|
|
68
|
+
|
|
69
|
+
n = len(results)
|
|
70
|
+
crashed = [r for r in results if not r.get("ok")]
|
|
71
|
+
fail_rate = len(crashed) / n
|
|
72
|
+
|
|
73
|
+
# ---- Rule 1: LEAD with the failure count, before any score. ----
|
|
74
|
+
print("=" * 60)
|
|
75
|
+
print(f"EVAL RUN {datetime.now(timezone.utc).isoformat()}")
|
|
76
|
+
print(f"cases: {n} | completed: {n - len(crashed)} | crashed: {len(crashed)} "
|
|
77
|
+
f"| failure-rate: {fail_rate:.0%}")
|
|
78
|
+
usable = fail_rate <= args.fail_rate
|
|
79
|
+
if not usable:
|
|
80
|
+
print(f"❌ RESULTS NOT USABLE — failure rate {fail_rate:.0%} exceeds {args.fail_rate:.0%}. "
|
|
81
|
+
"Do not quote any score below.")
|
|
82
|
+
completed = [r for r in results if r.get("ok") and r.get("score") is not None]
|
|
83
|
+
if completed:
|
|
84
|
+
avg = sum(r["score"] for r in completed) / len(completed)
|
|
85
|
+
label = "avg score (COMPLETED ONLY — not the whole suite)" if not usable else "avg score"
|
|
86
|
+
print(f"{label}: {avg:.2f} over {len(completed)} completed case(s)")
|
|
87
|
+
|
|
88
|
+
# ---- Rule 3: the error-analysis cross-tab, every run. ----
|
|
89
|
+
print("\nerror-analysis cross-tab (precondition × outcome):")
|
|
90
|
+
ct = Counter((r.get("precondition", False), not r.get("outcome_bad", True)) for r in results)
|
|
91
|
+
print(" outcome GOOD outcome BAD")
|
|
92
|
+
print(f" precond OK : {ct[(True, True)]:>6} {ct[(True, False)]:>6}")
|
|
93
|
+
print(f" precond MISS : {ct[(False, True)]:>6} {ct[(False, False)]:>6}")
|
|
94
|
+
print(" (if BAD concentrates in precond-OK, the downstream stage is the problem, not the precondition.)")
|
|
95
|
+
|
|
96
|
+
# write results
|
|
97
|
+
outdir = Path(__file__).parent / "results"
|
|
98
|
+
outdir.mkdir(exist_ok=True)
|
|
99
|
+
stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
|
100
|
+
(outdir / f"eval-{stamp}.json").write_text(json.dumps(
|
|
101
|
+
{"n": n, "failure_rate": fail_rate, "usable": usable, "results": results}, indent=2))
|
|
102
|
+
print(f"\nwrote results/eval-{stamp}.json")
|
|
103
|
+
|
|
104
|
+
# CI semantics: non-zero if results are not usable (blocks quoting a bogus number)
|
|
105
|
+
return 0 if usable else 1
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
if __name__ == "__main__":
|
|
109
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Warn/fail on CLAUDE.md size (keep the canon lean; detail belongs in rules/skills/docs).
|
|
3
|
+
set -uo pipefail
|
|
4
|
+
cd "$(dirname "$0")/.."
|
|
5
|
+
[ -f CLAUDE.md ] || { echo "no CLAUDE.md — skipping"; exit 0; }
|
|
6
|
+
WARN="${CANON_WARN:-200}"; FAIL="${CANON_FAIL:-300}"
|
|
7
|
+
n=$(wc -l < CLAUDE.md)
|
|
8
|
+
if [ "$n" -gt "$FAIL" ]; then echo " ✗ CLAUDE.md is $n lines (> $FAIL). Split into .claude/rules/*.md."; exit 1; fi
|
|
9
|
+
if [ "$n" -gt "$WARN" ]; then echo " ⚠ CLAUDE.md is $n lines (> $WARN soft budget). Consider splitting."; fi
|
|
10
|
+
echo " CLAUDE.md: $n lines"
|
|
11
|
+
exit 0
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Fail if a doc under docs/ is not reachable (linked) from a hub file or another doc.
|
|
3
|
+
# A doc nobody links to drifts unseen — this is the doc-orphan reachability gate.
|
|
4
|
+
set -uo pipefail
|
|
5
|
+
cd "$(dirname "$0")/.."
|
|
6
|
+
[ -d docs ] || { echo "no docs/ — skipping"; exit 0; }
|
|
7
|
+
HUBS="$(ls docs/README.md docs/index.md CLAUDE.md METHOD.md 2>/dev/null || true)"
|
|
8
|
+
[ -z "$HUBS" ] && { echo "no hub file — skipping"; exit 0; }
|
|
9
|
+
rc=0
|
|
10
|
+
for doc in docs/*.md; do
|
|
11
|
+
[ -e "$doc" ] || continue
|
|
12
|
+
base="$(basename "$doc")"
|
|
13
|
+
# skip hub files and kit-shipped reference docs (not project docs to link)
|
|
14
|
+
case "$base" in README.md|index.md|method-kit-manifest.md) continue;; esac
|
|
15
|
+
if ! grep -RqlF "$base" $HUBS docs/ --include='*.md' 2>/dev/null; then
|
|
16
|
+
echo " ✗ orphan doc (unlinked from any hub or sibling): $doc"; rc=1
|
|
17
|
+
fi
|
|
18
|
+
done
|
|
19
|
+
exit $rc
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Fail if a tracked doc restates a retracted claim without retracting it.
|
|
3
|
+
# Each retraction entry declares a quoted claim; if that exact quote appears in a doc OTHER than the
|
|
4
|
+
# retraction registry, flag it. (A restated claim needs its own retraction note or removal.)
|
|
5
|
+
set -uo pipefail
|
|
6
|
+
cd "$(dirname "$0")/.."
|
|
7
|
+
REG="registries/RETRACTIONS.md"
|
|
8
|
+
[ -f "$REG" ] || { echo "no RETRACTIONS.md — skipping"; exit 0; }
|
|
9
|
+
rc=0
|
|
10
|
+
tmp="$(mktemp)"
|
|
11
|
+
grep -oE '"[^"]{12,}"' "$REG" | sed 's/^"//;s/"$//' > "$tmp" || true
|
|
12
|
+
while IFS= read -r claim; do
|
|
13
|
+
[ -z "$claim" ] && continue
|
|
14
|
+
hits="$(grep -RnF "$claim" . --include='*.md' 2>/dev/null | grep -v "$REG" || true)"
|
|
15
|
+
if [ -n "$hits" ]; then
|
|
16
|
+
echo " ✗ retracted claim restated: \"$claim\""
|
|
17
|
+
echo "$hits" | sed 's/^/ /'
|
|
18
|
+
rc=1
|
|
19
|
+
fi
|
|
20
|
+
done < "$tmp"
|
|
21
|
+
rm -f "$tmp"
|
|
22
|
+
exit $rc
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Fail if a retired identifier (tombstones.txt) reappears in tracked source/docs/config.
|
|
3
|
+
set -uo pipefail
|
|
4
|
+
cd "$(dirname "$0")/.."
|
|
5
|
+
TS="registries/tombstones.txt"
|
|
6
|
+
[ -f "$TS" ] || { echo "no tombstones.txt — skipping"; exit 0; }
|
|
7
|
+
rc=0
|
|
8
|
+
while IFS= read -r line; do
|
|
9
|
+
case "$line" in ''|'#'*) continue;; esac
|
|
10
|
+
id="$(printf '%s' "$line" | cut -f1)"
|
|
11
|
+
[ -z "$id" ] && continue
|
|
12
|
+
hits="$(grep -RnF "$id" . \
|
|
13
|
+
--include='*.md' --include='*.py' --include='*.ts' --include='*.tsx' \
|
|
14
|
+
--include='*.cs' --include='*.yaml' --include='*.yml' --include='*.json' --include='*.sh' \
|
|
15
|
+
2>/dev/null | grep -v 'registries/' || true)"
|
|
16
|
+
if [ -n "$hits" ]; then
|
|
17
|
+
echo " ✗ tombstoned identifier '$id' resurrected:"
|
|
18
|
+
echo "$hits" | sed 's/^/ /'
|
|
19
|
+
rc=1
|
|
20
|
+
fi
|
|
21
|
+
done < "$TS"
|
|
22
|
+
exit $rc
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# run-gates.sh — the mechanical drift-gates (surface 5). Portable; stack-agnostic.
|
|
3
|
+
# Run locally via `/check-rules` or `bash gates/run-gates.sh`; wired into CI (see ci/).
|
|
4
|
+
#
|
|
5
|
+
# Each gate is a separate script returning non-zero on drift. This runner aggregates them so one
|
|
6
|
+
# red gate fails the whole check (fail-closed). Add stack-specific gates in the <<FILL>> block.
|
|
7
|
+
set -uo pipefail
|
|
8
|
+
cd "$(dirname "$0")/.." # repo root
|
|
9
|
+
GATES_DIR="gates"
|
|
10
|
+
fail=0
|
|
11
|
+
|
|
12
|
+
run() {
|
|
13
|
+
local name="$1"; shift
|
|
14
|
+
echo "── gate: $name ──"
|
|
15
|
+
if "$@"; then echo " ✓ $name"; else echo " ✗ $name FAILED"; fail=1; fi
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
run "retractions" bash "$GATES_DIR/check-retractions.sh"
|
|
19
|
+
run "tombstones" bash "$GATES_DIR/check-tombstones.sh"
|
|
20
|
+
run "doc-orphans" bash "$GATES_DIR/check-doc-orphans.sh"
|
|
21
|
+
run "canon-size" bash "$GATES_DIR/check-canon-size.sh"
|
|
22
|
+
|
|
23
|
+
# <<FILL: STACK-SPECIFIC GATES — e.g. class-path import check, analyzer, type-check, tests>>
|
|
24
|
+
# run "class-paths" bash "$GATES_DIR/check-scaffold-classpaths.sh"
|
|
25
|
+
# run "tests" your-test-command
|
|
26
|
+
|
|
27
|
+
if [ "$fail" -ne 0 ]; then
|
|
28
|
+
echo; echo "GATES RED — do not merge. Encode the fix at the smallest surface, then re-run."
|
|
29
|
+
exit 1
|
|
30
|
+
fi
|
|
31
|
+
echo; echo "All gates green."
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
{
|
|
2
|
+
"UserPromptSubmit": [
|
|
3
|
+
{
|
|
4
|
+
"hooks": [
|
|
5
|
+
{ "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/hooks/userpromptsubmit-okl-check.sh" }
|
|
6
|
+
]
|
|
7
|
+
}
|
|
8
|
+
],
|
|
9
|
+
"Stop": [
|
|
10
|
+
{
|
|
11
|
+
"hooks": [
|
|
12
|
+
{ "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/hooks/stop-okl-encode.sh" }
|
|
13
|
+
]
|
|
14
|
+
}
|
|
15
|
+
]
|
|
16
|
+
}
|