org-knowledge-layer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. okl/__init__.py +12 -0
  2. okl/__main__.py +8 -0
  3. okl/bootstrap.py +83 -0
  4. okl/cli.py +484 -0
  5. okl/client.py +160 -0
  6. okl/core.py +223 -0
  7. okl/drift.py +119 -0
  8. okl/mcp_server.py +75 -0
  9. okl/scaffold/MANIFEST.md +59 -0
  10. okl/scaffold/ci/method-gates.yml +32 -0
  11. okl/scaffold/ci/okl-verify.yml +59 -0
  12. okl/scaffold/claude/agents/architecture-reviewer.md +41 -0
  13. okl/scaffold/claude/commands/check-rules.md +24 -0
  14. okl/scaffold/claude/commands/feature-spec.md +37 -0
  15. okl/scaffold/claude/rules/example-area.md +22 -0
  16. okl/scaffold/claude/skills/RECOMMENDED-COMPANIONS.md +40 -0
  17. okl/scaffold/claude/skills/encoding-loop/SKILL.md +48 -0
  18. okl/scaffold/claude/skills/verify-before-claiming/SKILL.md +56 -0
  19. okl/scaffold/evals/README.md +32 -0
  20. okl/scaffold/evals/cases.jsonl +1 -0
  21. okl/scaffold/evals/run_evals.py +109 -0
  22. okl/scaffold/gates/check-canon-size.sh +11 -0
  23. okl/scaffold/gates/check-doc-orphans.sh +19 -0
  24. okl/scaffold/gates/check-retractions.sh +22 -0
  25. okl/scaffold/gates/check-tombstones.sh +22 -0
  26. okl/scaffold/gates/run-gates.sh +31 -0
  27. okl/scaffold/hooks/hooks.json +16 -0
  28. okl/scaffold/hooks/stop-okl-encode.sh +78 -0
  29. okl/scaffold/hooks/userpromptsubmit-okl-check.sh +68 -0
  30. okl/scaffold/plugin/plugin.json +10 -0
  31. okl/scaffold/profiles/dotnet/README.md +12 -0
  32. okl/scaffold/profiles/dotnet/rules/architecture.md +55 -0
  33. okl/scaffold/profiles/dotnet/rules/messaging.md +31 -0
  34. okl/scaffold/profiles/dotnet/rules/performance-and-data.md +36 -0
  35. okl/scaffold/profiles/dotnet/rules/security.md +42 -0
  36. okl/scaffold/profiles/geospatial/README.md +6 -0
  37. okl/scaffold/profiles/geospatial/rules/geospatial-ml.md +38 -0
  38. okl/scaffold/profiles/python-rag/README.md +13 -0
  39. okl/scaffold/profiles/python-rag/rules/fastapi-backend.md +37 -0
  40. okl/scaffold/profiles/python-rag/rules/project-structure.md +28 -0
  41. okl/scaffold/profiles/python-rag/rules/rag-pipeline.md +73 -0
  42. okl/scaffold/profiles/react/README.md +18 -0
  43. okl/scaffold/profiles/react/rules/frontend.md +57 -0
  44. okl/scaffold/registries/RETRACTIONS.md +19 -0
  45. okl/scaffold/registries/tombstones.txt +7 -0
  46. okl/scaffold/root/CLAUDE.md +55 -0
  47. okl/scaffold/root/METHOD.md +64 -0
  48. okl/scaffold_cmd.py +110 -0
  49. okl/seed/dotnet-canon.json +489 -0
  50. okl/seed/dotnet-decisions.json +328 -0
  51. okl/seed/dotnet-defects.json +133 -0
  52. okl/seed/dotnet-review-surfaces.json +147 -0
  53. okl/seed/frontend-canon.json +116 -0
  54. okl/seed/geospatial-deeptime-defects.json +59 -0
  55. okl/seed/geospatial-defects.json +154 -0
  56. okl/seed/geospatial-enforcement-defects.json +121 -0
  57. okl/seed/geospatial-eval-defects.json +25 -0
  58. okl/seed/rag-defects.json +120 -0
  59. okl/seed/react-defects.json +45 -0
  60. okl/seed.py +55 -0
  61. okl/service.py +137 -0
  62. okl/store.py +432 -0
  63. org_knowledge_layer-0.1.0.dist-info/METADATA +475 -0
  64. org_knowledge_layer-0.1.0.dist-info/RECORD +67 -0
  65. org_knowledge_layer-0.1.0.dist-info/WHEEL +4 -0
  66. org_knowledge_layer-0.1.0.dist-info/entry_points.txt +2 -0
  67. org_knowledge_layer-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,59 @@
1
+ # Installed by `okl init` into .github/workflows/okl-verify.yml (also stamped by `okl scaffold`).
2
+ #
3
+ # The knowledge layer's CI verifier — real checks, not placeholders:
4
+ # 1. `okl drift --gate` — fails the build when a rule's governed files changed after the
5
+ # rule was last verified (a stale rule is a rule nobody re-checked).
6
+ # 2. `gates/run-gates.sh` — the repo's mechanical method gates, when the scaffold is present.
7
+ #
8
+ # Shared-layer connection is optional: without OKL_SERVICE_URL the drift gate runs against
9
+ # the repo's local .okl store. Configure the secrets to verify against the org layer and to
10
+ # let gate scripts emit receipts (`okl link <gate_id> VERIFIED_ON <defect_id>`) — emit those
11
+ # from inside the gate that proved itself against real drift, where the ids are known.
12
+ name: okl-verify
13
+
14
+ on:
15
+ pull_request:
16
+ push:
17
+ branches: [main]
18
+
19
+ jobs:
20
+ okl-verify:
21
+ runs-on: ubuntu-latest
22
+ steps:
23
+ - uses: actions/checkout@v4
24
+ with:
25
+ fetch-depth: 0 # drift needs history: it compares file mtimes/commits vs verified_at
26
+ - uses: actions/setup-python@v5
27
+ with:
28
+ python-version: "3.13"
29
+ - name: Install okl
30
+ # Consumer repos install the released package; in okl's own repo this
31
+ # workflow dogfoods the working tree (pip install okl would 404 — not on PyPI yet).
32
+ run: |
33
+ if grep -q '^name = "okl"' pyproject.toml 2>/dev/null; then
34
+ pip install -e .
35
+ else
36
+ pip install okl
37
+ fi
38
+
39
+ - name: Connect to the shared layer (optional — skipped when secrets are unset)
40
+ env:
41
+ OKL_SERVICE_URL: ${{ secrets.OKL_SERVICE_URL }}
42
+ OKL_TOKEN: ${{ secrets.OKL_TOKEN }}
43
+ run: |
44
+ if [ -n "${OKL_SERVICE_URL:-}" ]; then
45
+ okl connect "$OKL_SERVICE_URL" --token "$OKL_TOKEN"
46
+ else
47
+ echo "no OKL_SERVICE_URL secret — verifying against the repo-local store"
48
+ fi
49
+
50
+ - name: Drift gate — rules whose governed code changed after last verification
51
+ run: okl drift --gate
52
+
53
+ - name: Method gates (when the scaffold kit is present)
54
+ run: |
55
+ if [ -x gates/run-gates.sh ]; then
56
+ bash gates/run-gates.sh
57
+ else
58
+ echo "no gates/run-gates.sh — method-kit gates not installed (okl scaffold adds them)"
59
+ fi
@@ -0,0 +1,41 @@
1
+ ---
2
+ name: architecture-reviewer
3
+ description: Reviews a diff or a proposed change against this repo's encoded rules and the org knowledge layer. Invoke on non-pattern-conforming changes (new bounded context, novel dependency, security-model change, multi-step refactor) and before merging anything touching published results. Returns a pattern-checklist verdict, not a rewrite.
4
+ tools: Read, Grep, Glob, Bash
5
+ model: inherit
6
+ memory: project # persists to .claude/agent-memory/architecture-reviewer/ (committed, team-shared)
7
+ skills:
8
+ - encoding-loop
9
+ ---
10
+
11
+ # Architecture reviewer
12
+
13
+ You are a reviewer, not an implementer. You read the change and judge it against the encoded body —
14
+ you do not rewrite it. Your output is a checklist verdict with specific file:line findings.
15
+
16
+ ## Before you start
17
+ 1. Run `okl check --task "<one-line summary of the change under review>"` and treat the returned
18
+ armed gates, retractions, tombstones, and THREAT prior-art as binding review criteria.
19
+ 2. Read `.claude/rules/` entries whose `paths:` match the changed files.
20
+
21
+ ## Pattern checklist (extend as the repo encodes new rules)
22
+ - **Smallest-surface / speculative coupling** — is there an abstraction (interface, factory, layer)
23
+ with exactly one implementation and no test substitution and no concrete second impl on the
24
+ roadmap? If so, flag it as speculative — the concrete class should be used directly.
25
+ - **Assert-from-memory** — does any spec/config/scaffold claim something is "correct"/"valid"
26
+ without a mechanical `validate` step? Flag it; this is the #14-class defect.
27
+ - **Retracted claim restated** — does any prose restate a claim in `registries/RETRACTIONS.md`
28
+ without also retracting it? Fail.
29
+ - **Resurrected identifier** — does the diff reintroduce anything in `registries/tombstones.txt`? Fail.
30
+ - **Missing gate receipt** — does a fix reference a defect class that has a gate, without the gate
31
+ running in CI on this change?
32
+ - **Report-the-result** — if this change is a "fix" that confirms a hypothesis, was the control run?
33
+
34
+ <!-- <<FILL: STACK-SPECIFIC REVIEW CHECKS>>
35
+ Add checks specific to this stack (e.g. IDOR predicate in the SQL Where clause; N+1 queries;
36
+ async-over-sync; mass-assignment of server-controlled fields). Keep each as one scan rule. -->
37
+
38
+ ## Memory
39
+ Use your persistent memory dir to accumulate this repo's recurring findings and architectural
40
+ decisions across reviews. When you see the same finding class a third time, propose promoting it to
41
+ a mechanical gate (surface 5) via the encoding-loop skill.
@@ -0,0 +1,24 @@
1
+ ---
2
+ description: Run all mechanical drift-gates locally (the same set CI enforces) and print the worklist of any drift found.
3
+ disable-model-invocation: true
4
+ ---
5
+
6
+ # /check-rules — run the drift-gates locally
7
+
8
+ Run the full gate suite and report, exactly as CI will:
9
+
10
+ ```bash
11
+ bash gates/run-gates.sh --untracked
12
+ ```
13
+
14
+ This runs:
15
+ - **retractions** — fails any doc that states a retracted claim without retracting it
16
+ - **tombstones** — fails any doc/comment/config resurrecting a retired identifier
17
+ - **doc-orphans** — fails any spec/ADR/audit not reachable from the docs hub
18
+ - **stale-open-items** — fails any registry item marked "open" that has been resolved
19
+ - **canon-size** — warns/fails if `CLAUDE.md` exceeds the size budget
20
+
21
+ <!-- <<FILL: STACK-SPECIFIC GATES>> Add your build/analyzer/test gates here and in gates/run-gates.sh. -->
22
+
23
+ If anything fails, encode the fix at the smallest surface (see the `encoding-loop` skill) and
24
+ re-run. Do not merge with a red gate; a broken gate that reports "clean" is worse than no gate.
@@ -0,0 +1,37 @@
1
+ ---
2
+ description: Draft a structured feature spec — value gate, significance check, prior-art check, and a validate plan. The ritual at the start of any non-trivial work.
3
+ argument-hint: "<one-line feature description>"
4
+ disable-model-invocation: true
5
+ ---
6
+
7
+ # /feature-spec — the front of the loop
8
+
9
+ Produce a short spec for: **$ARGUMENTS**
10
+
11
+ Do NOT write implementation code. Produce the spec, then stop for review.
12
+
13
+ ## 1. Value gate
14
+ - What problem does this solve, and for whom? What breaks if we don't build it?
15
+ - What is the smallest version that delivers the value? (Build that; note the rest as deferred.)
16
+ - Set a stop condition up front (a token/time budget for experiments). When it runs out, the
17
+ default is "we learned what we needed; we don't continue."
18
+
19
+ ## 2. Significance check
20
+ - Is this **pattern-conforming** (matches an existing shape → gate review suffices) or
21
+ **non-pattern-conforming** (new bounded context, novel dependency, security-model change,
22
+ multi-step refactor → stay present during implementation, route through architecture-reviewer)?
23
+
24
+ ## 3. Prior-art / knowledge check (mandatory)
25
+ - Run `okl check --task "$ARGUMENTS"`. Record any armed gates to adopt, retractions to respect,
26
+ and THREAT prior-art. If this is a research/novelty claim, treat prior art as falsifying until shown otherwise.
27
+
28
+ ## 4. Contract + validate plan
29
+ - Inputs, outputs, invariants, error cases.
30
+ - **Every assertion in this spec that could be wrong from memory gets a mechanical `validate` step.**
31
+ (Do the class paths exist? Does the config parse? Are the class counts right?) List them as
32
+ gate/test items, not prose claims. This is the one SDD rung we keep.
33
+
34
+ ## 5. Handoff
35
+ - List the tasks. Flag which are pattern-conforming vs not. Name the gates that must be green to merge.
36
+
37
+ <!-- <<FILL: STACK-SPECIFIC SPEC SECTIONS>> e.g. migration plan, API version bump, deploy steps. -->
@@ -0,0 +1,22 @@
1
+ ---
2
+ description: Conventions for <AREA> — loaded only when the agent touches matching files.
3
+ paths:
4
+ - "<<FILL: glob, e.g. **/*.sql or src/payments/**>>"
5
+ ---
6
+
7
+ # <AREA> rules (path-scoped — surface 2, lazy-loaded)
8
+
9
+ <!-- This is a TEMPLATE for a path-scoped rule file. Copy it per area, set `paths:`, keep it short.
10
+ The point of path-scoped rules is that CLAUDE.md stays lean: rules load into context ONLY when
11
+ the agent is working on files that match the glob. Target ~30–60 lines each.
12
+
13
+ Example (delete and replace):
14
+
15
+ # Database rules (paths: **/*.sql, **/migrations/**)
16
+ - Every migration is reversible; ship the down-migration in the same PR.
17
+ - No `SELECT *` in application queries; name columns so a schema change breaks loudly.
18
+ - Money is DECIMAL, never float; never round in the DB layer.
19
+ -->
20
+
21
+ - <<FILL: rule 1>>
22
+ - <<FILL: rule 2>>
@@ -0,0 +1,40 @@
1
+ # Recommended companion skills (install separately)
2
+
3
+ This kit ships two first-party skills — **`encoding-loop`** (turn a finding into a
4
+ promoted, recorded rule) and **`verify-before-claiming`** (evidence before assertion).
5
+ They are the two disciplines specific to this method.
6
+
7
+ The broader engineering-discipline skills that pair well with it — systematic debugging,
8
+ test-driven development, plan writing/execution, git-worktree isolation — are **not
9
+ bundled here on purpose.** The best-maintained open versions live in third-party skill
10
+ collections, and vendoring them would mean redistributing someone else's work (with its
11
+ own license and attribution) and shipping cross-references to sibling skills this kit
12
+ doesn't contain. Point at them instead of copying:
13
+
14
+ ## Superpowers (obra)
15
+
16
+ A collection of Claude Code skills for disciplined agentic development. The skills below
17
+ reference each other via the `superpowers:` namespace, so install the collection as a
18
+ unit rather than cherry-picking files:
19
+
20
+ - `systematic-debugging` — root-cause-before-symptom, pressure-resistant debugging
21
+ - `test-driven-development` — red/green/refactor discipline
22
+ - `writing-plans` / `executing-plans` — spec → reviewed multi-step execution
23
+ - `using-git-worktrees` — isolated workspace per feature
24
+ - `verification-before-completion` — the upstream cousin of this kit's `verify-before-claiming`
25
+
26
+ Install: follow the Superpowers project's own instructions (it manages its own namespace
27
+ and inter-skill references). Verify its license permits your use before redistributing it
28
+ inside your own repo.
29
+
30
+ ## How they fit with this method
31
+
32
+ - Use **`encoding-loop`** whenever any of the companion skills surfaces a durable lesson —
33
+ a debugging root cause, a test that should always exist, a plan step that must not be
34
+ skipped — and record it to the knowledge layer so the next task inherits it.
35
+ - Use **`verify-before-claiming`** as the last step of any companion skill that ends in a
36
+ "done"/"passing"/"fixed" claim.
37
+
38
+ Keeping the general skills external and the method skills first-party is deliberate: the
39
+ kit stays cleanly installable and license-clean, and you always run the companions'
40
+ latest upstream versions rather than a frozen copy that drifts.
@@ -0,0 +1,48 @@
1
+ ---
2
+ name: encoding-loop
3
+ description: Use when you discover a rule, antipattern, or lesson worth keeping ("we should never write this again" / "we should always do this when") and need to encode it at the right surface and record it to the org knowledge layer. Fires on review findings, debugging discoveries, audits, and prior-art hits.
4
+ ---
5
+
6
+ # Encoding loop — turn a finding into a durable, promoted rule
7
+
8
+ A trigger surfaced a candidate rule. The response is always the same two moves.
9
+
10
+ ## 1. Pick the smallest sufficient surface (softest→strongest)
11
+
12
+ | # | Surface | Use when |
13
+ |---|---|---|
14
+ | 1 | `docs/` + diagram | the *why* needs more than a line, or reviewers need a picture |
15
+ | 2 | `CLAUDE.md` | every session needs it, always-on (keep it lean — prefer a rule file) |
16
+ | 2 | `.claude/rules/<area>.md` with `paths:` glob | scoped to a file category (loads only in context) |
17
+ | 3 | `.claude/skills/` or `.claude/commands/` | a multi-step procedure with real logic |
18
+ | 4 | `.coderabbit.yaml` path_instructions / `architecture-reviewer` checklist | catch at PR-review time |
19
+ | 5 | a gate in `gates/` + CI | mechanical, build-breaking — for rules that keep being broken |
20
+
21
+ Default to the *softest* surface that could hold the rule. **Promote down (toward 5) only as it
22
+ earns it** — a rule that keeps being violated moves to a sterner tier, not a sterner paragraph.
23
+ Most rules never leave tier 1, and that is fine.
24
+
25
+ ## 2. Record it to the org layer so other repos inherit it
26
+
27
+ ```
28
+ okl record --type <Defect|Gate|Rule|Claim|Retraction|Tombstone|Decision|PriorArt> \
29
+ --title "..." --body "..." --scope <org|repo> [--verified]
30
+ ```
31
+
32
+ - **`scope=org`** — a fact about the world: prior art, an API contract, a data-source gotcha, a
33
+ portable gate. It will surface in every connected repo's `okl check`.
34
+ - **`scope=repo`** — a quirk true only of this codebase. Stays local.
35
+
36
+ If the finding retracts a prior claim, also add it to `registries/RETRACTIONS.md`; if it retires an
37
+ identifier, add it to `registries/tombstones.txt`. The CI gates then fail any doc that contradicts.
38
+
39
+ ## 3. Verify the gate (if you made one)
40
+
41
+ A gate you have only seen pass is a gate you have not tested. Run it against the real drifted file
42
+ from git history and confirm it FAILS, then confirm it passes on the fix. Only then is it a gate.
43
+
44
+ ## Gotchas
45
+ - A merged fix without the encoded rule is a half-finished job — the next instance slips through.
46
+ - A surface nobody runs is documentation, not enforcement. If it is not read/executed, it will drift.
47
+ - The drift you notice is the drift that embarrasses you; the flattering kind (stale "open" items)
48
+ needs a mechanical `verified_at`/TTL check, not vigilance.
@@ -0,0 +1,56 @@
1
+ ---
2
+ name: verify-before-claiming
3
+ description: Use before stating that work is done, a test passes, a build is green, a bug is fixed, a file exists, or any factual claim about the state of the code or the world. Requires running the check and reading its real output before asserting the result — evidence precedes assertion, always.
4
+ ---
5
+
6
+ # Verify before claiming
7
+
8
+ A claim about state — "the tests pass," "the file is created," "the endpoint returns
9
+ 404," "19 tests green" — is only worth as much as the check behind it. Stating a result
10
+ you *expect* as if it were a result you *observed* is the single most common way
11
+ AI-assisted work goes fast, fluent, plausible, and wrong.
12
+
13
+ This skill has one rule: **run the check, read the output, then make the claim — in that
14
+ order.** If you have not run the check this turn, you do not have the result; you have a
15
+ guess. Say "I expect" for a guess and "I confirmed" only for an observation.
16
+
17
+ ## When this fires
18
+
19
+ - About to write "done", "fixed", "passing", "green", "works", "created", "verified".
20
+ - About to put a count, a status, or a value into prose, a commit message, a README, a
21
+ PR description, or a report.
22
+ - About to tell a human that a thing is true about the code or the system.
23
+
24
+ ## The procedure
25
+
26
+ 1. **Name the check.** What single command or observation would make this claim true or
27
+ false? (`pytest -q`, `ls path`, `curl -s -o /dev/null -w '%{http_code}'`, `git status`.)
28
+ 2. **Run it this turn.** Not "I ran it earlier" — state drifts; a file moved, a kernel
29
+ reset, an install got lost. Re-run it now.
30
+ 3. **Read the real output.** The exit code and the actual lines — not the first line, not
31
+ what you assume follows.
32
+ 4. **Claim only what the output shows.** "18 passed, 1 skipped" — not "all green." If the
33
+ output surprises you, the output wins; investigate before you claim.
34
+ 5. **If you cannot run the check, say so.** "I could not run the suite in this
35
+ environment, so I have not confirmed the count" is a true statement. "Tests pass" when
36
+ you didn't run them is not.
37
+
38
+ ## Anti-patterns this exists to stop
39
+
40
+ - **Reciting a remembered number.** "19 passing tests" carried from three turns ago, when
41
+ the suite was never run this session (or was, and now one fails). Re-run; report today's
42
+ result.
43
+ - **Reporting the expected instead of the observed.** Writing the result the code *should*
44
+ produce as the result it *did* produce. These diverge exactly when it matters.
45
+ - **"Ran it" ≠ "passed it".** A command that executed is not a command that succeeded.
46
+ Check the exit code, not just that it ran.
47
+ - **Success by absence.** "No errors shown" is not "it worked" — a check that did not run
48
+ produces no errors either. Absence of a failure signal and presence of a success signal
49
+ are different things (see also: the knowledge layer's fail-closed rule).
50
+
51
+ ## The link to the rest of the method
52
+
53
+ This is the personal-discipline version of what the mechanical gates enforce at CI time
54
+ and what `okl check` enforces at task time: **a claim with no check behind it is
55
+ documentation, not verification.** When a verification you skipped later turns into a
56
+ defect, that defect is worth encoding (`okl record`) so the next task is warned.
@@ -0,0 +1,32 @@
1
+ # Evals — the measurement tier (surface 5, for behavior)
2
+
3
+ The gates in `gates/` catch drift in *artifacts* (docs, identifiers, canon size). This tier catches
4
+ drift in *behavior* — did the change make the system measurably worse? It is here because the RAG service's
5
+ single most dangerous defect was a metric that could not report its own unreliability
6
+ ("LLM Judge 5.0/5.0 while 19 of 20 cases crashed").
7
+
8
+ ## The three rules this harness enforces (earned, not invented)
9
+
10
+ 1. **A metric that cannot report its own failure rate is not a metric.** `run_evals.py` leads with
11
+ its failure count and prints `❌ RESULTS NOT USABLE` above a configurable failure-rate threshold —
12
+ *before* any score. A green average over a crashed suite is the failure mode to prevent.
13
+ 2. **The judge must not grade its own homework.** If you use an LLM judge, it must be a different
14
+ model from the generator. The harness refuses to run if `JUDGE_MODEL == GENERATOR_MODEL`.
15
+ 3. **Bake in the error-analysis cross-tab.** Every eval run emits the `retrieval_hit × judge_score`
16
+ (or your task's equivalent `precondition × outcome`) cross-tab, because that 2×2 is the only thing
17
+ that told the RAG service which component was actually failing. It runs on every eval, not as a one-off.
18
+
19
+ ## Golden set from real failures — not invented fixtures
20
+
21
+ Rule 4 of the method: *fixtures you invented cannot falsify assumptions you hold.* Seed `cases.jsonl`
22
+ from real inputs that actually broke (the RAG service used its 25 real EDGAR filenames verbatim), not from
23
+ hand-written happy-path examples.
24
+
25
+ ## Files
26
+ - `run_evals.py` — the harness (framework-agnostic; adapt `evaluate_one()` to your system)
27
+ - `cases.jsonl` — the golden set (`<<FILL>>` with real cases)
28
+ - results write to `results/` with the failure count and cross-tab at the top
29
+
30
+ ## Wire into CI
31
+ Add to `ci/` after the gates: a regression fails the build if the usable pass-rate drops below the
32
+ committed baseline. Never quote a number from a run whose failure rate tripped `RESULTS NOT USABLE`.
@@ -0,0 +1 @@
1
+ {"id": "example-1", "input": "<<FILL: a real input that actually broke>>", "expected": "<<FILL: expected behavior>>"}
@@ -0,0 +1,109 @@
1
+ #!/usr/bin/env python3
2
+ """Eval harness — measures behavior and, crucially, reports its own unreliability first.
3
+
4
+ Framework-agnostic: plug your system into `evaluate_one()`. The invariants it enforces are the
5
+ earned lessons, not any particular eval library (works alongside deepeval/ragas/promptfoo or none).
6
+
7
+ Usage: python run_evals.py [--cases cases.jsonl] [--fail-rate 0.20]
8
+ Env: GENERATOR_MODEL, JUDGE_MODEL (must differ — rule: no grading your own homework)
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import json
14
+ import os
15
+ import sys
16
+ from collections import Counter
17
+ from datetime import datetime, timezone
18
+ from pathlib import Path
19
+
20
+
21
+ def evaluate_one(case: dict) -> dict:
22
+ """<<FILL>>: run YOUR system on `case` and return a result dict.
23
+
24
+ Required keys in the returned dict:
25
+ - "ok": bool — did the case COMPLETE without crashing (not "did it pass")
26
+ - "score": float|None — the judge/quality score, or None if it didn't complete
27
+ - "precondition": bool — the x-axis of the cross-tab (e.g. retrieval_hit); task-specific
28
+ - "outcome_bad": bool — the y-axis (e.g. answer judged bad)
29
+ - "error": str|None
30
+ Replace the stub below with a real call into your app.
31
+ """
32
+ # STUB — deterministic placeholder so the harness runs before you wire your system.
33
+ q = case.get("input", "")
34
+ return {"ok": True, "score": 5.0 if q else 0.0,
35
+ "precondition": bool(q), "outcome_bad": not bool(q), "error": None}
36
+
37
+
38
+ def main(argv=None) -> int:
39
+ ap = argparse.ArgumentParser()
40
+ ap.add_argument("--cases", default=str(Path(__file__).parent / "cases.jsonl"))
41
+ ap.add_argument("--fail-rate", type=float, default=0.20,
42
+ help="usable-results threshold; above this, results are NOT USABLE")
43
+ args = ap.parse_args(argv)
44
+
45
+ # Rule 2: the judge must not be the generator.
46
+ gen, judge = os.environ.get("GENERATOR_MODEL"), os.environ.get("JUDGE_MODEL")
47
+ if gen and judge and gen == judge:
48
+ print(f"❌ REFUSING TO RUN: judge model == generator model ({gen}). "
49
+ "A model grading its own output is a mirror, not a signal.", file=sys.stderr)
50
+ return 3
51
+
52
+ path = Path(args.cases)
53
+ if not path.exists():
54
+ print(f"no cases file at {path} — add a golden set (see README). Nothing to measure.")
55
+ return 0
56
+ cases = [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
57
+ if not cases:
58
+ print("cases.jsonl is empty — add real failing cases before trusting any number.")
59
+ return 0
60
+
61
+ results = []
62
+ for c in cases:
63
+ try:
64
+ r = evaluate_one(c)
65
+ except Exception as e: # a crash is a completed-with-failure, counted as such
66
+ r = {"ok": False, "score": None, "precondition": False, "outcome_bad": True, "error": repr(e)}
67
+ results.append(r)
68
+
69
+ n = len(results)
70
+ crashed = [r for r in results if not r.get("ok")]
71
+ fail_rate = len(crashed) / n
72
+
73
+ # ---- Rule 1: LEAD with the failure count, before any score. ----
74
+ print("=" * 60)
75
+ print(f"EVAL RUN {datetime.now(timezone.utc).isoformat()}")
76
+ print(f"cases: {n} | completed: {n - len(crashed)} | crashed: {len(crashed)} "
77
+ f"| failure-rate: {fail_rate:.0%}")
78
+ usable = fail_rate <= args.fail_rate
79
+ if not usable:
80
+ print(f"❌ RESULTS NOT USABLE — failure rate {fail_rate:.0%} exceeds {args.fail_rate:.0%}. "
81
+ "Do not quote any score below.")
82
+ completed = [r for r in results if r.get("ok") and r.get("score") is not None]
83
+ if completed:
84
+ avg = sum(r["score"] for r in completed) / len(completed)
85
+ label = "avg score (COMPLETED ONLY — not the whole suite)" if not usable else "avg score"
86
+ print(f"{label}: {avg:.2f} over {len(completed)} completed case(s)")
87
+
88
+ # ---- Rule 3: the error-analysis cross-tab, every run. ----
89
+ print("\nerror-analysis cross-tab (precondition × outcome):")
90
+ ct = Counter((r.get("precondition", False), not r.get("outcome_bad", True)) for r in results)
91
+ print(" outcome GOOD outcome BAD")
92
+ print(f" precond OK : {ct[(True, True)]:>6} {ct[(True, False)]:>6}")
93
+ print(f" precond MISS : {ct[(False, True)]:>6} {ct[(False, False)]:>6}")
94
+ print(" (if BAD concentrates in precond-OK, the downstream stage is the problem, not the precondition.)")
95
+
96
+ # write results
97
+ outdir = Path(__file__).parent / "results"
98
+ outdir.mkdir(exist_ok=True)
99
+ stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
100
+ (outdir / f"eval-{stamp}.json").write_text(json.dumps(
101
+ {"n": n, "failure_rate": fail_rate, "usable": usable, "results": results}, indent=2))
102
+ print(f"\nwrote results/eval-{stamp}.json")
103
+
104
+ # CI semantics: non-zero if results are not usable (blocks quoting a bogus number)
105
+ return 0 if usable else 1
106
+
107
+
108
+ if __name__ == "__main__":
109
+ raise SystemExit(main())
@@ -0,0 +1,11 @@
1
+ #!/usr/bin/env bash
2
+ # Warn/fail on CLAUDE.md size (keep the canon lean; detail belongs in rules/skills/docs).
3
+ set -uo pipefail
4
+ cd "$(dirname "$0")/.."
5
+ [ -f CLAUDE.md ] || { echo "no CLAUDE.md — skipping"; exit 0; }
6
+ WARN="${CANON_WARN:-200}"; FAIL="${CANON_FAIL:-300}"
7
+ n=$(wc -l < CLAUDE.md)
8
+ if [ "$n" -gt "$FAIL" ]; then echo " ✗ CLAUDE.md is $n lines (> $FAIL). Split into .claude/rules/*.md."; exit 1; fi
9
+ if [ "$n" -gt "$WARN" ]; then echo " ⚠ CLAUDE.md is $n lines (> $WARN soft budget). Consider splitting."; fi
10
+ echo " CLAUDE.md: $n lines"
11
+ exit 0
@@ -0,0 +1,19 @@
1
+ #!/usr/bin/env bash
2
+ # Fail if a doc under docs/ is not reachable (linked) from a hub file or another doc.
3
+ # A doc nobody links to drifts unseen — this is the doc-orphan reachability gate.
4
+ set -uo pipefail
5
+ cd "$(dirname "$0")/.."
6
+ [ -d docs ] || { echo "no docs/ — skipping"; exit 0; }
7
+ HUBS="$(ls docs/README.md docs/index.md CLAUDE.md METHOD.md 2>/dev/null || true)"
8
+ [ -z "$HUBS" ] && { echo "no hub file — skipping"; exit 0; }
9
+ rc=0
10
+ for doc in docs/*.md; do
11
+ [ -e "$doc" ] || continue
12
+ base="$(basename "$doc")"
13
+ # skip hub files and kit-shipped reference docs (not project docs to link)
14
+ case "$base" in README.md|index.md|method-kit-manifest.md) continue;; esac
15
+ if ! grep -RqlF "$base" $HUBS docs/ --include='*.md' 2>/dev/null; then
16
+ echo " ✗ orphan doc (unlinked from any hub or sibling): $doc"; rc=1
17
+ fi
18
+ done
19
+ exit $rc
@@ -0,0 +1,22 @@
1
+ #!/usr/bin/env bash
2
+ # Fail if a tracked doc restates a retracted claim without retracting it.
3
+ # Each retraction entry declares a quoted claim; if that exact quote appears in a doc OTHER than the
4
+ # retraction registry, flag it. (A restated claim needs its own retraction note or removal.)
5
+ set -uo pipefail
6
+ cd "$(dirname "$0")/.."
7
+ REG="registries/RETRACTIONS.md"
8
+ [ -f "$REG" ] || { echo "no RETRACTIONS.md — skipping"; exit 0; }
9
+ rc=0
10
+ tmp="$(mktemp)"
11
+ grep -oE '"[^"]{12,}"' "$REG" | sed 's/^"//;s/"$//' > "$tmp" || true
12
+ while IFS= read -r claim; do
13
+ [ -z "$claim" ] && continue
14
+ hits="$(grep -RnF "$claim" . --include='*.md' 2>/dev/null | grep -v "$REG" || true)"
15
+ if [ -n "$hits" ]; then
16
+ echo " ✗ retracted claim restated: \"$claim\""
17
+ echo "$hits" | sed 's/^/ /'
18
+ rc=1
19
+ fi
20
+ done < "$tmp"
21
+ rm -f "$tmp"
22
+ exit $rc
@@ -0,0 +1,22 @@
1
+ #!/usr/bin/env bash
2
+ # Fail if a retired identifier (tombstones.txt) reappears in tracked source/docs/config.
3
+ set -uo pipefail
4
+ cd "$(dirname "$0")/.."
5
+ TS="registries/tombstones.txt"
6
+ [ -f "$TS" ] || { echo "no tombstones.txt — skipping"; exit 0; }
7
+ rc=0
8
+ while IFS= read -r line; do
9
+ case "$line" in ''|'#'*) continue;; esac
10
+ id="$(printf '%s' "$line" | cut -f1)"
11
+ [ -z "$id" ] && continue
12
+ hits="$(grep -RnF "$id" . \
13
+ --include='*.md' --include='*.py' --include='*.ts' --include='*.tsx' \
14
+ --include='*.cs' --include='*.yaml' --include='*.yml' --include='*.json' --include='*.sh' \
15
+ 2>/dev/null | grep -v 'registries/' || true)"
16
+ if [ -n "$hits" ]; then
17
+ echo " ✗ tombstoned identifier '$id' resurrected:"
18
+ echo "$hits" | sed 's/^/ /'
19
+ rc=1
20
+ fi
21
+ done < "$TS"
22
+ exit $rc
@@ -0,0 +1,31 @@
1
+ #!/usr/bin/env bash
2
+ # run-gates.sh — the mechanical drift-gates (surface 5). Portable; stack-agnostic.
3
+ # Run locally via `/check-rules` or `bash gates/run-gates.sh`; wired into CI (see ci/).
4
+ #
5
+ # Each gate is a separate script returning non-zero on drift. This runner aggregates them so one
6
+ # red gate fails the whole check (fail-closed). Add stack-specific gates in the <<FILL>> block.
7
+ set -uo pipefail
8
+ cd "$(dirname "$0")/.." # repo root
9
+ GATES_DIR="gates"
10
+ fail=0
11
+
12
+ run() {
13
+ local name="$1"; shift
14
+ echo "── gate: $name ──"
15
+ if "$@"; then echo " ✓ $name"; else echo " ✗ $name FAILED"; fail=1; fi
16
+ }
17
+
18
+ run "retractions" bash "$GATES_DIR/check-retractions.sh"
19
+ run "tombstones" bash "$GATES_DIR/check-tombstones.sh"
20
+ run "doc-orphans" bash "$GATES_DIR/check-doc-orphans.sh"
21
+ run "canon-size" bash "$GATES_DIR/check-canon-size.sh"
22
+
23
+ # <<FILL: STACK-SPECIFIC GATES — e.g. class-path import check, analyzer, type-check, tests>>
24
+ # run "class-paths" bash "$GATES_DIR/check-scaffold-classpaths.sh"
25
+ # run "tests" your-test-command
26
+
27
+ if [ "$fail" -ne 0 ]; then
28
+ echo; echo "GATES RED — do not merge. Encode the fix at the smallest surface, then re-run."
29
+ exit 1
30
+ fi
31
+ echo; echo "All gates green."
@@ -0,0 +1,16 @@
1
+ {
2
+ "UserPromptSubmit": [
3
+ {
4
+ "hooks": [
5
+ { "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/hooks/userpromptsubmit-okl-check.sh" }
6
+ ]
7
+ }
8
+ ],
9
+ "Stop": [
10
+ {
11
+ "hooks": [
12
+ { "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/hooks/stop-okl-encode.sh" }
13
+ ]
14
+ }
15
+ ]
16
+ }