org-knowledge-layer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okl/__init__.py +12 -0
- okl/__main__.py +8 -0
- okl/bootstrap.py +83 -0
- okl/cli.py +484 -0
- okl/client.py +160 -0
- okl/core.py +223 -0
- okl/drift.py +119 -0
- okl/mcp_server.py +75 -0
- okl/scaffold/MANIFEST.md +59 -0
- okl/scaffold/ci/method-gates.yml +32 -0
- okl/scaffold/ci/okl-verify.yml +59 -0
- okl/scaffold/claude/agents/architecture-reviewer.md +41 -0
- okl/scaffold/claude/commands/check-rules.md +24 -0
- okl/scaffold/claude/commands/feature-spec.md +37 -0
- okl/scaffold/claude/rules/example-area.md +22 -0
- okl/scaffold/claude/skills/RECOMMENDED-COMPANIONS.md +40 -0
- okl/scaffold/claude/skills/encoding-loop/SKILL.md +48 -0
- okl/scaffold/claude/skills/verify-before-claiming/SKILL.md +56 -0
- okl/scaffold/evals/README.md +32 -0
- okl/scaffold/evals/cases.jsonl +1 -0
- okl/scaffold/evals/run_evals.py +109 -0
- okl/scaffold/gates/check-canon-size.sh +11 -0
- okl/scaffold/gates/check-doc-orphans.sh +19 -0
- okl/scaffold/gates/check-retractions.sh +22 -0
- okl/scaffold/gates/check-tombstones.sh +22 -0
- okl/scaffold/gates/run-gates.sh +31 -0
- okl/scaffold/hooks/hooks.json +16 -0
- okl/scaffold/hooks/stop-okl-encode.sh +78 -0
- okl/scaffold/hooks/userpromptsubmit-okl-check.sh +68 -0
- okl/scaffold/plugin/plugin.json +10 -0
- okl/scaffold/profiles/dotnet/README.md +12 -0
- okl/scaffold/profiles/dotnet/rules/architecture.md +55 -0
- okl/scaffold/profiles/dotnet/rules/messaging.md +31 -0
- okl/scaffold/profiles/dotnet/rules/performance-and-data.md +36 -0
- okl/scaffold/profiles/dotnet/rules/security.md +42 -0
- okl/scaffold/profiles/geospatial/README.md +6 -0
- okl/scaffold/profiles/geospatial/rules/geospatial-ml.md +38 -0
- okl/scaffold/profiles/python-rag/README.md +13 -0
- okl/scaffold/profiles/python-rag/rules/fastapi-backend.md +37 -0
- okl/scaffold/profiles/python-rag/rules/project-structure.md +28 -0
- okl/scaffold/profiles/python-rag/rules/rag-pipeline.md +73 -0
- okl/scaffold/profiles/react/README.md +18 -0
- okl/scaffold/profiles/react/rules/frontend.md +57 -0
- okl/scaffold/registries/RETRACTIONS.md +19 -0
- okl/scaffold/registries/tombstones.txt +7 -0
- okl/scaffold/root/CLAUDE.md +55 -0
- okl/scaffold/root/METHOD.md +64 -0
- okl/scaffold_cmd.py +110 -0
- okl/seed/dotnet-canon.json +489 -0
- okl/seed/dotnet-decisions.json +328 -0
- okl/seed/dotnet-defects.json +133 -0
- okl/seed/dotnet-review-surfaces.json +147 -0
- okl/seed/frontend-canon.json +116 -0
- okl/seed/geospatial-deeptime-defects.json +59 -0
- okl/seed/geospatial-defects.json +154 -0
- okl/seed/geospatial-enforcement-defects.json +121 -0
- okl/seed/geospatial-eval-defects.json +25 -0
- okl/seed/rag-defects.json +120 -0
- okl/seed/react-defects.json +45 -0
- okl/seed.py +55 -0
- okl/service.py +137 -0
- okl/store.py +432 -0
- org_knowledge_layer-0.1.0.dist-info/METADATA +475 -0
- org_knowledge_layer-0.1.0.dist-info/RECORD +67 -0
- org_knowledge_layer-0.1.0.dist-info/WHEEL +4 -0
- org_knowledge_layer-0.1.0.dist-info/entry_points.txt +2 -0
- org_knowledge_layer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Enforcement/encoding-loop lessons earned in the geospatial pipeline (2026-08-02) while hardening PR #75's CI. All WORLD-FACTS — portable to any repo — so org-scoped: a repo's first `okl check` returns them before the same mistake recurs. Tagged 'method' (encoding-loop/CI discipline), plus 'geospatial'/'data-quality' where the class is subject-specific. No 'ci' tag exists in the controlled vocabulary yet; add one to store.py KNOWN_TAGS if this class grows. Review, then `okl seed seed/geospatial-enforcement-defects.json`.",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{
|
|
5
|
+
"key": "d_invoked_by_nothing",
|
|
6
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
7
|
+
"found_by": "auditing the encoding-loop surfaces — the arch-reviewer, once actually run, caught a must-fix an automatic surface missed",
|
|
8
|
+
"title": "An 'enforcement surface' that only runs on-demand runs never",
|
|
9
|
+
"body": "The architecture-reviewer agent was listed in CLAUDE.md as one of the encoding-loop's enforcement surfaces, but was invoked by nothing — no hook, no CI job, only a human typing /check-rules. For the repo's whole life it never ran. A review agent, lint rule, or checklist with no mechanical trigger is documentation, not enforcement: it provides zero protection while appearing on the ledger as covered.",
|
|
10
|
+
"symptom": "a review agent / rule / checklist is described as enforcement but nothing automatically triggers it",
|
|
11
|
+
"fix": "wire it to a mechanical trigger (hook or CI job), or relabel it honestly as on-demand-only",
|
|
12
|
+
"tags": "method"
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"key": "g_review_agent_in_ci",
|
|
16
|
+
"type": "Gate", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
17
|
+
"found_by": "architecture-review.yml — runs the reviewer agent headless on every PR diff",
|
|
18
|
+
"title": "Run the architecture/review agent headless in CI on every PR diff",
|
|
19
|
+
"body": "A CI job runs the review agent against `git diff origin/BASE...HEAD` on every PR and fails on any must-fix finding. Self-owned — does not depend on a third-party review SaaS. Soft-passes until its API-key secret is set so it never wedges merges before setup, then becomes a hard gate via branch protection. Portable to any repo that has a review agent or a written architecture checklist.",
|
|
20
|
+
"symptom": "architecture rules live only as an agent/checklist that a human must remember to run",
|
|
21
|
+
"fix": "add a CI job that runs the agent on the diff and fails on must-fix; soft-pass until the key is configured",
|
|
22
|
+
"tags": "method,agent-safety"
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"key": "r_surface_needs_trigger",
|
|
26
|
+
"type": "Rule", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
27
|
+
"found_by": "CLAUDE.md 'What is AUTOMATIC vs what you must RUN' ledger",
|
|
28
|
+
"title": "Every enforcement surface needs a mechanical trigger and an honest AUTOMATIC-vs-RUN ledger",
|
|
29
|
+
"body": "Keep an explicit table of what runs automatically (hooks, CI) vs what a human must remember to run; a surface with no trigger belongs in the second column or nowhere. Update the ledger in the SAME change that adds or closes a gate — a ledger claiming coverage it doesn't have is worse than none, because it stops people looking.",
|
|
30
|
+
"symptom": "a rules doc lists surfaces without saying which actually run automatically",
|
|
31
|
+
"fix": "maintain an AUTOMATIC-vs-RUN ledger; update it whenever a gate is added or closed",
|
|
32
|
+
"tags": "method"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"key": "d_unpinned_linter",
|
|
36
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
37
|
+
"found_by": "ci-python.yml lint gate failing a branch that changed no linted code",
|
|
38
|
+
"title": "Unpinned linter/formatter in CI retro-fails a branch that changed nothing",
|
|
39
|
+
"body": "CI ran `pip install ruff` (unpinned). A new release (0.15→0.16) promoted rules to default (RUF059, I001, RUF012, SIM117) and failed a previously-green branch on pre-existing code the PR never touched. Any CI step installing a linter/formatter/type-checker without a pinned version is a time-bomb that fires on the vendor's release schedule, not yours.",
|
|
40
|
+
"symptom": "a CI lint/format/type step fails on code the PR didn't change, right after a tool release",
|
|
41
|
+
"fix": "pin the exact tool version; treat bumps as deliberate PRs that address the new findings",
|
|
42
|
+
"tags": "method"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"key": "r_pin_gating_tools",
|
|
46
|
+
"type": "Rule", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
47
|
+
"found_by": "the 0.15→0.16 ruff incident above",
|
|
48
|
+
"title": "Pin every tool that gates CI; an unpinned linter is a time-bomb",
|
|
49
|
+
"body": "Any tool whose output can fail a build — ruff, black, mypy, eslint, prettier, a Sonar scanner image — must be version-pinned in CI. Unpinned, the gate's behavior changes on the vendor's release cadence and fails work that changed nothing. Pin it; bump it in its own PR where the new findings are visible and addressed.",
|
|
50
|
+
"symptom": "a CI gate step runs `pip install <tool>` / `npm i -g <tool>` with no version",
|
|
51
|
+
"fix": "pin the exact version; bumps are deliberate PRs",
|
|
52
|
+
"tags": "method"
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"key": "d_ungated_zone",
|
|
56
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
57
|
+
"found_by": "the arch-reviewer's 'ungated zone' checklist section — code outside the source roots",
|
|
58
|
+
"title": "Code outside the linter/type/scan source-roots is gated by nothing",
|
|
59
|
+
"body": "Python under olmoearth_run_data/ and docs/ was covered by no linter, no type-checker, no Sonar rule, and no AI-review path rule: CI ran ruff only on the package, sonar.sources listed three dirs, CodeRabbit path_instructions matched the package only, and the CI trigger paths didn't even fire on those dirs. Untyped params, bare excepts, and hardcoded-constant drift shipped straight through review because no gate watched that directory. Every repo has these zones; they must be enumerated, not assumed covered.",
|
|
60
|
+
"symptom": "a .py/.ts/.cs dir ships code but appears in no linter/type/scan source-root and no CI trigger path",
|
|
61
|
+
"fix": "add the dir to CI trigger paths + the lint/scan invocation; list remaining ungated zones in the ledger",
|
|
62
|
+
"tags": "method"
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"key": "g_gate_every_dir",
|
|
66
|
+
"type": "Gate", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
67
|
+
"found_by": "ci-python.yml — added olmoearth_run_data to triggers + a dedicated repo-root lint step",
|
|
68
|
+
"title": "Gate every executable directory; lint each import-tree from its own root",
|
|
69
|
+
"body": "Add each shipping dir to the CI trigger paths and the lint/type/scan invocation. Subtlety: ruff's isort first-party detection is CWD-anchored, so `ruff check ../otherdir` from a subpackage classifies the other tree's sibling imports differently than the repo root would, and flags I001 in CI though it's clean locally. Give each import tree its own lint step with working-directory at that tree's root.",
|
|
70
|
+
"symptom": "lint clean locally but I001/import-order fails in CI, or a dir has no gate at all",
|
|
71
|
+
"fix": "one lint step per import-tree run from that tree's root; add every shipping dir to CI trigger paths",
|
|
72
|
+
"tags": "method"
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"key": "d_cwd_isort",
|
|
76
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
77
|
+
"found_by": "reproducing the CI I001 locally by matching the working directory",
|
|
78
|
+
"title": "ruff isort is CWD-anchored: sibling imports sort differently from a subdir vs the repo root",
|
|
79
|
+
"body": "`ruff check ../olmoearth_run_data` invoked from the python-etl working dir classified the zone's sibling-module imports as third-party and demanded a different grouping than `ruff check olmoearth_run_data` from the repo root — so the same files passed locally (root) and failed in CI (subdir) with I001. The invocation's working directory determines first-party detection; reproduce a CI lint failure with the SAME cwd CI uses.",
|
|
80
|
+
"symptom": "ruff I001 in CI on files that pass ruff locally",
|
|
81
|
+
"fix": "lint each import tree from its own root (separate step, working-directory at that root)",
|
|
82
|
+
"tags": "method"
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"key": "d_masked_gate_order",
|
|
86
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
87
|
+
"found_by": "a test failure that only appeared once the lint step ahead of it went green",
|
|
88
|
+
"title": "A failing early gate step masks a later gate's failure in the same job",
|
|
89
|
+
"body": "The CI python job ran ruff before pytest in one job; while ruff failed (the unpinned-bump debt) the job short-circuited before pytest, hiding a real test failure (a test fake's signature missing a newly-added kwarg). Fixing ruff 'revealed' a second red that had been there all along. A single job chaining gates reports only the first failure; green-after-fix is not proof the rest pass.",
|
|
90
|
+
"symptom": "fixing one CI failure immediately surfaces a different one that was always broken",
|
|
91
|
+
"fix": "after greening a gate, re-run the full downstream; prefer independent jobs so one red can't hide another",
|
|
92
|
+
"tags": "method"
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
"key": "r_single_source_sets",
|
|
96
|
+
"type": "Rule", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
97
|
+
"found_by": "architecture-reviewer must-fix: hardcoded 2020 in 5 places + a re-defined label frozenset",
|
|
98
|
+
"title": "Single-source a derived fact — including SETS and class schemes, not just scalars",
|
|
99
|
+
"body": "A canonical constant (IMAGERY_YEAR=2020) or set (INVASIVE_LABELS=frozenset({TAMARISK, RUSSIAN_OLIVE})) must be imported wherever used, never re-written as a literal. Re-hardcoding 2020 in five places, or re-defining the frozenset locally, is the same drift class: two definitions of one fact are two facts free to diverge. Applies equally to class-index schemes, label sets, peak-season months, and vintages.",
|
|
100
|
+
"symptom": "a module re-writes a constant, frozenset, or class-list that already exists as a canonical import",
|
|
101
|
+
"fix": "import the canonical constant/set; never restate it as a literal",
|
|
102
|
+
"tags": "method,data-quality"
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"key": "d_month_range_28",
|
|
106
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
107
|
+
"found_by": "CodeRabbit review of a monthly-composite date range",
|
|
108
|
+
"title": "Monthly STAC query ending at day-28 drops days 29-31 of imagery",
|
|
109
|
+
"body": "A per-month scene search built as `{y}-{m}-01`/`{y}-{m}-28` silently omits days 29, 30, 31 — fewer candidate scenes and a possibly-worse least-cloudy pick. End the range at the first day of the FOLLOWING month (with December→January rollover) so every day is covered uniformly (28/29/30/31). It was inherited by copy into a second script before being caught.",
|
|
110
|
+
"symptom": "a monthly composite/search date range hard-codes -28 (or -30) as the month end",
|
|
111
|
+
"fix": "end at first-of-next-month with Dec→Jan rollover; covers all month lengths",
|
|
112
|
+
"tags": "geospatial"
|
|
113
|
+
}
|
|
114
|
+
],
|
|
115
|
+
"edges": [
|
|
116
|
+
{ "src": "g_review_agent_in_ci", "rel": "CATCHES", "dst": "d_invoked_by_nothing" },
|
|
117
|
+
{ "src": "g_review_agent_in_ci", "rel": "ENCODES", "dst": "r_surface_needs_trigger" },
|
|
118
|
+
{ "src": "g_gate_every_dir", "rel": "CATCHES", "dst": "d_ungated_zone" },
|
|
119
|
+
{ "src": "g_gate_every_dir", "rel": "CATCHES", "dst": "d_cwd_isort" }
|
|
120
|
+
]
|
|
121
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Experiment-methodology lessons from the geospatial repo Stage-2 (2026-08-04). The beetle-inversion RF arm returned a clean NEGATIVE: a pre-registered control vetoed an underpowered cross-era result, and leave-one-trip-out CV crashed on single-cluster field data. Both are WORLD-FACTS → org scope, portable to any repo running transfer/cross-era evals. Review, then `okl seed seed/geospatial-eval-defects.json`.",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{
|
|
5
|
+
"key": "r_preregistered_control_vetoes",
|
|
6
|
+
"type": "Rule", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
7
|
+
"found_by": "phase3c beetle-inversion — the Russian-olive control dropped 0.285 where it should be flat",
|
|
8
|
+
"title": "A pre-registered negative control can VETO a transfer/cross-era result the sample is too small or confounded to support",
|
|
9
|
+
"body": "In a cross-era / cross-sensor / cross-domain comparison, add a control that SHOULD NOT move and declare it in advance. If the control moves, its movement is the noise floor — the claimed effect must clear it or the result is unproven. Pre-registering the control lets it VETO the result instead of being quietly dropped when inconvenient. In phase3c the tamarisk 'signal' moved ~0.05 while the Russian-olive control (the beetle doesn't touch it) moved 0.285 pre-beetle, so the beetle claim was unprovable on that data — and saying so honestly was only possible because the control was declared first.",
|
|
10
|
+
"symptom": "a small or confounded transfer experiment shows a modest effect in the hoped-for direction",
|
|
11
|
+
"fix": "pre-register a negative control that should be flat; require the effect to exceed the control's movement; if it doesn't, report a null, not a hint",
|
|
12
|
+
"tags": "eval-integrity,method,data-quality"
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"key": "d_loo_group_single_class",
|
|
16
|
+
"type": "Defect", "scope": "org", "repo": "geospatial-ml-pipeline", "verified": true,
|
|
17
|
+
"found_by": "phase3c IndexError: predict_proba(...)[:, 1] out of bounds — a leave-one-trip-out fold was single-class",
|
|
18
|
+
"title": "Leave-one-group-out CV crashes or degenerates when a group can be single-class (one-cluster field data)",
|
|
19
|
+
"body": "Field campaigns cluster geographically, so grouping CV by field trip / site can hand a fold a training set with only ONE class present. The classifier then learns one class, predict_proba returns a single column, and a hard `[:, 1]` throws (or the fold is meaningless). Two fixes: (1) group by SPATIAL BLOCK — bin coordinates into tiles — so blocks within the one cluster mix classes; (2) map the positive-class column via `estimator.classes_` and handle the single-class fold gracefully, never hard-index `[:, 1]`.",
|
|
20
|
+
"symptom": "predict_proba(...)[:, 1] IndexError, or a grouped-CV fold with only one class present",
|
|
21
|
+
"fix": "spatial-block folds (bin lat/lon) instead of by-trip; index the positive column via classes_ and guard the degenerate fold",
|
|
22
|
+
"tags": "eval-integrity,method,data-quality"
|
|
23
|
+
}
|
|
24
|
+
]
|
|
25
|
+
}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Seed for the OKL — real, dated, receipted lessons from the RAG service (Python/FastAPI RAG: Qdrant hybrid, Redis, Postgres, Ollama, deepeval). Eval-integrity and retrieval-vs-identity lessons are org-scoped (portable to any RAG/agent/ML-eval work); pipeline-plumbing quirks are scoped repo:python-rag-service.",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{
|
|
5
|
+
"key": "qz_judge_crash",
|
|
6
|
+
"type": "Defect",
|
|
7
|
+
"scope": "org",
|
|
8
|
+
"repo": "python-rag-service",
|
|
9
|
+
"found_by": "commit 60bbed7",
|
|
10
|
+
"verified": true,
|
|
11
|
+
"title": "'LLM Judge 5.0/5.0' while 19 of 20 cases crashed — a metric that can't report its own failure rate",
|
|
12
|
+
"body": "The eval harness averaged only the cases that completed; one survived, scored 5, printed a perfect score. The most dangerous bug of the effort — it made everything downstream unfalsifiable. Portable fix: the summary LEADS with its own failure count and prints RESULTS NOT USABLE above a failure-rate threshold, before any score.",
|
|
13
|
+
"symptom": "a summary score prints while some cases threw or were skipped",
|
|
14
|
+
"fix": "lead the report with the crash/failure count; mark results unusable above threshold",
|
|
15
|
+
"tags": "eval-integrity"
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"key": "qz_judge_self",
|
|
19
|
+
"type": "Defect",
|
|
20
|
+
"scope": "org",
|
|
21
|
+
"repo": "python-rag-service",
|
|
22
|
+
"found_by": "findings-log Part 2 #5",
|
|
23
|
+
"verified": true,
|
|
24
|
+
"title": "The LLM judge was grading its own homework (same model as the generator)",
|
|
25
|
+
"body": "The judge defaulted to the same model as the generator — a mirror, not a signal (its default was even an embedding model that can't generate). Portable rule: the judge must be a different model from the generator; refuse to run if they're equal.",
|
|
26
|
+
"symptom": "the eval's judge model equals (or defaults to) the generator model",
|
|
27
|
+
"fix": "require judge≠generator; refuse to run if they're equal",
|
|
28
|
+
"tags": "eval-integrity"
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
"key": "qz_crosstab",
|
|
32
|
+
"type": "Rule",
|
|
33
|
+
"scope": "org",
|
|
34
|
+
"repo": "python-rag-service",
|
|
35
|
+
"found_by": "findings-log Part 2 #6",
|
|
36
|
+
"verified": true,
|
|
37
|
+
"title": "Run the retrieval_hit × judge_score cross-tab — it names which component is actually failing",
|
|
38
|
+
"body": "A whole session optimized retrieval; the cross-tab revealed generation was the dominant failure mode. The conclusion 'agentic retrieval is architecturally weaker' was exactly backwards, and that eval-results doc was retracted. Portable: bake the precondition×outcome cross-tab into every eval run, not as a one-off.",
|
|
39
|
+
"tags": "eval-integrity"
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
"key": "qz_fixtures",
|
|
43
|
+
"type": "Defect",
|
|
44
|
+
"scope": "org",
|
|
45
|
+
"repo": "python-rag-service",
|
|
46
|
+
"found_by": "findings-log Part 3",
|
|
47
|
+
"verified": true,
|
|
48
|
+
"title": "Invented fixtures can't falsify assumptions the author holds",
|
|
49
|
+
"body": "Entity tests used invented fixtures (Apple_10K.html) and passed at 100% while the live index had exhibit number '10.1' indexed as a company. Fixtures written by the person who wrote the parser encode the same assumptions as the parser. Portable rule: test against real inputs sampled from the actual corpus, verbatim (the RAG service switched to its 25 real filenames).",
|
|
50
|
+
"tags": "eval-integrity,data-quality"
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"key": "qz_identity",
|
|
54
|
+
"type": "Rule",
|
|
55
|
+
"scope": "org",
|
|
56
|
+
"repo": "python-rag-service",
|
|
57
|
+
"found_by": "findings-log Part 3/4 (the through-line)",
|
|
58
|
+
"verified": true,
|
|
59
|
+
"title": "Don't infer by resemblance what you can look up — document identity is indexed & pre-filtered, not ranked",
|
|
60
|
+
"body": "Semantic search ranks by content and cannot distinguish one document from another (every contract has a termination clause). A corpus-wide question ('do any two docs share a company?') is a GROUP BY, not a similarity question — a ranked retriever cannot answer it at any top-k. Make the model choose the constraint and make code satisfy it. Portable across RAG/agent retrieval design.",
|
|
61
|
+
"tags": "retrieval-design"
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"key": "qz_wrong_corpus",
|
|
65
|
+
"type": "Defect",
|
|
66
|
+
"scope": "repo",
|
|
67
|
+
"repo": "python-rag-service",
|
|
68
|
+
"found_by": "commits 36db337, 8cab711",
|
|
69
|
+
"verified": true,
|
|
70
|
+
"title": "Different eval runs were measuring different Qdrant collections",
|
|
71
|
+
"body": "Four collections existed (python-rag-service_hybrid / _ids / _full (zero entities) / _entities); the container was hand-pointed at one while config.py named another. Runs were compared as if the same system. Fix: one canonical collection, strays deleted+tombstoned. Stack-specific (Qdrant), scoped repo.",
|
|
72
|
+
"tags": "python-rag,eval-integrity"
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"key": "qz_broken_tool",
|
|
76
|
+
"type": "Defect",
|
|
77
|
+
"scope": "org",
|
|
78
|
+
"repo": "python-rag-service",
|
|
79
|
+
"found_by": "commits 8df8bbc, c1b80f5",
|
|
80
|
+
"verified": true,
|
|
81
|
+
"title": "An agent whose primary tool throws on every call measures a broken tool, not an architecture",
|
|
82
|
+
"body": "search_filtered defaulted to a paid Voyage reranker with no key configured and threw on every call; the agent burned its whole step budget retrying because the error told it nothing actionable. 'Agentic loses on the real corpus' measured that, not architecture. Portable: empty/failed results should return actionable guidance (e.g. the entity vocabulary) so the agent can self-correct.",
|
|
83
|
+
"tags": "agent-safety"
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"key": "qz_unequal_budget",
|
|
87
|
+
"type": "Defect",
|
|
88
|
+
"scope": "org",
|
|
89
|
+
"repo": "python-rag-service",
|
|
90
|
+
"found_by": "findings-log Part 2 #4",
|
|
91
|
+
"verified": true,
|
|
92
|
+
"title": "Comparing arms with unequal budgets attributes the gap to the wrong cause",
|
|
93
|
+
"body": "search_corpus inherited top_k=3 (the FACTUAL default) while classic used 8 on the same COMPARISON question — comparing an agent given 3 chunks against a pipeline given 8 and blaming the architecture. Portable: give each arm the same retrieval/token budget before concluding anything about the arms.",
|
|
94
|
+
"tags": "eval-integrity"
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"key": "qz_agent_contract",
|
|
98
|
+
"type": "Rule",
|
|
99
|
+
"scope": "org",
|
|
100
|
+
"repo": "python-rag-service",
|
|
101
|
+
"found_by": "docs/agent-contract.md",
|
|
102
|
+
"verified": true,
|
|
103
|
+
"title": "The agent's capability surface is an allowlist; side-effect machinery is built with the first mutating tool",
|
|
104
|
+
"body": "A tool exists only if registered; unknown names are never executed ('the loop must never invent capabilities'). Budgets (max steps, token budget, tool timeout) are explicit and env-overridable; finish is the explicit stop. A tool with side effects requires arg-schema validation, a pre-execution control mode (audit/approval/block), and severity-gated alerting BEFORE it ships — built deliberately with the first mutating tool, not speculatively. Portable agent-safety rule.",
|
|
105
|
+
"tags": "agent-safety"
|
|
106
|
+
}
|
|
107
|
+
],
|
|
108
|
+
"edges": [
|
|
109
|
+
{
|
|
110
|
+
"src": "qz_crosstab",
|
|
111
|
+
"rel": "REFUTES",
|
|
112
|
+
"dst": "qz_judge_crash"
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
"src": "qz_fixtures",
|
|
116
|
+
"rel": "RECURS_IN",
|
|
117
|
+
"dst": "python-rag-service"
|
|
118
|
+
}
|
|
119
|
+
]
|
|
120
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Seed for the OKL — React/frontend-SPA lessons ported from the .NET platform's frontend canon (frontend/CLAUDE.md + the vendored vercel-react-best-practices skill, frozen 2026-06-12). React is backend-agnostic, so these are org-scoped: they propagate to any repo with a React UI. This closes the coverage gap the A/B experiment measured — react_fetch reproduced the useEffect-fetching defect 3/3 in BOTH arms because no React node existed in the store to inject.",
|
|
3
|
+
"nodes": [
|
|
4
|
+
{
|
|
5
|
+
"key": "rx_useeffect_fetch",
|
|
6
|
+
"type": "Defect",
|
|
7
|
+
"scope": "org",
|
|
8
|
+
"repo": "dotnet-microservices",
|
|
9
|
+
"found_by": "frontend/CLAUDE.md + vercel-react-best-practices (You Might Not Need an Effect)",
|
|
10
|
+
"verified": true,
|
|
11
|
+
"title": "Hand-rolled data fetching in useEffect+useState reimplements caching/dedup/races badly",
|
|
12
|
+
"body": "cause: fetching server data inside a useEffect and storing it in useState re-derives — poorly — the caching, request dedup, race-cancellation, and refetch logic a query library already solves. .NET platform canon: ALL server data goes through TanStack Query v5; fetching-in-useEffect is banned. Query keys are [feature, entity, params].",
|
|
13
|
+
"symptom": "a component fetches from an API inside useEffect and stores the result in useState",
|
|
14
|
+
"fix": "use TanStack Query (useQuery) with a [feature,entity,params] key; reserve useEffect for non-data side effects only",
|
|
15
|
+
"tags": "react"
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"key": "rx_tokens_localstorage",
|
|
19
|
+
"type": "Defect",
|
|
20
|
+
"scope": "org",
|
|
21
|
+
"repo": "dotnet-microservices",
|
|
22
|
+
"found_by": "frontend/CLAUDE.md (auth)",
|
|
23
|
+
"verified": true,
|
|
24
|
+
"title": "Access tokens in localStorage are XSS-exfiltratable",
|
|
25
|
+
"body": "cause: tokens in localStorage are readable by any script on the page, so one XSS steals the session. .NET platform canon: PKCE auth-code flow (oidc-client-ts → Keycloak), tokens held in memory; the localStorage-vs-BFF trade-off is documented, not defaulted.",
|
|
26
|
+
"symptom": "auth tokens written to or read from localStorage/sessionStorage",
|
|
27
|
+
"fix": "hold tokens in memory via the auth library; use PKCE; if persistence is required, document the BFF/cookie trade-off explicitly",
|
|
28
|
+
"tags": "react,security"
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
"key": "rx_fight_compiler",
|
|
32
|
+
"type": "Rule",
|
|
33
|
+
"scope": "org",
|
|
34
|
+
"repo": "dotnet-microservices",
|
|
35
|
+
"found_by": "vercel-react-best-practices",
|
|
36
|
+
"verified": true,
|
|
37
|
+
"title": "Don't fight the React Compiler — no manual memo when it's on",
|
|
38
|
+
"body": "With the React Compiler enabled, hand-written useMemo/useCallback/React.memo add noise and can pessimize. Let the compiler memoize; keep components pure. Bundle budget is CI-checked (≤200 KB gz); virtualize lists over ~50 items.",
|
|
39
|
+
"symptom": "manual useMemo/useCallback/React.memo added for performance while the React Compiler is enabled",
|
|
40
|
+
"fix": "remove the manual memoization; keep the component pure and let the compiler handle it",
|
|
41
|
+
"tags": "react"
|
|
42
|
+
}
|
|
43
|
+
],
|
|
44
|
+
"edges": []
|
|
45
|
+
}
|
okl/seed.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Seed the layer from a JSON file of nodes and edges.
|
|
2
|
+
|
|
3
|
+
The canonical first seed is the geospatial pipeline's defect table + the gates
|
|
4
|
+
that catch each defect — so a brand-new repo's very first `okl check` can return
|
|
5
|
+
a real, earned lesson (e.g. error #14's class-path gate) rather than an empty set.
|
|
6
|
+
|
|
7
|
+
File format:
|
|
8
|
+
{
|
|
9
|
+
"nodes": [ {"key": "d14", "type": "Defect", "title": "...", "scope": "org", ...}, ... ],
|
|
10
|
+
"edges": [ {"src": "g_classpath", "rel": "CATCHES", "dst": "d14"}, ... ]
|
|
11
|
+
}
|
|
12
|
+
`key` is a local alias used only to wire edges within the file; the store
|
|
13
|
+
assigns the real id. Edges reference keys, which are resolved to ids on load.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from .client import Client
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def seed_from_file(client: Client, path: str) -> int:
|
|
25
|
+
"""Ingest a *-defects.json file. Idempotent: re-seeding the same file REPLACES
|
|
26
|
+
its own nodes instead of duplicating them.
|
|
27
|
+
|
|
28
|
+
Each node's stable id is derived from the seed file's stem + the node's local
|
|
29
|
+
`key` (e.g. `seed:geospatial-defects:d14`). `key` remains the in-file alias used
|
|
30
|
+
to wire edges; here it does double duty as the persistent identity so a second
|
|
31
|
+
`okl seed` run upserts the same rows. A node without a `key` falls back to a
|
|
32
|
+
fresh random id (non-idempotent) — every bundled seed node has a key.
|
|
33
|
+
"""
|
|
34
|
+
p = Path(path)
|
|
35
|
+
data = json.loads(p.read_text())
|
|
36
|
+
ns = f"seed:{p.stem}"
|
|
37
|
+
keymap: dict[str, str] = {}
|
|
38
|
+
for node in data.get("nodes", []):
|
|
39
|
+
key = node.pop("key", None)
|
|
40
|
+
# Seed nodes carry provenance from ANOTHER repo. Client.record defaults a
|
|
41
|
+
# missing `repo` to the CURRENT repo, which would mislabel it — pass None
|
|
42
|
+
# explicitly ("unknown") and warn, rather than inherit that default.
|
|
43
|
+
if "repo" not in node:
|
|
44
|
+
node["repo"] = None
|
|
45
|
+
print(f" ! {key or node.get('title', '?')}: seed node has no 'repo' — "
|
|
46
|
+
"recording provenance as unknown, not this repo", file=sys.stderr)
|
|
47
|
+
stable_id = f"{ns}:{key}" if key else None
|
|
48
|
+
node_id = client.record(id=stable_id, **node) if stable_id else client.record(**node)
|
|
49
|
+
if key:
|
|
50
|
+
keymap[key] = node_id
|
|
51
|
+
for edge in data.get("edges", []):
|
|
52
|
+
src = keymap.get(edge["src"], edge["src"])
|
|
53
|
+
dst = keymap.get(edge["dst"], edge["dst"])
|
|
54
|
+
client.link(src, edge["rel"], dst)
|
|
55
|
+
return len(data.get("nodes", []))
|
okl/service.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""The shared OKL service — a small FastAPI app exposing check/record/search/link.
|
|
2
|
+
|
|
3
|
+
This is the "option 3" piece: one always-on service that owns the database, so
|
|
4
|
+
any repo on any machine reaches the SAME curated knowledge via a URL. Storage is
|
|
5
|
+
selected by OKL_DATABASE_URL (sqlite:///okl.db default, or postgres://... when
|
|
6
|
+
you deploy) — the code here never changes when you promote the backend.
|
|
7
|
+
|
|
8
|
+
Run: `okl serve` (or `uvicorn okl.service:app`).
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from . import core
|
|
16
|
+
from .store import Store
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
from fastapi import FastAPI, Header, HTTPException
|
|
20
|
+
from pydantic import BaseModel
|
|
21
|
+
except ImportError as e: # pragma: no cover
|
|
22
|
+
raise RuntimeError("The service needs FastAPI — install okl[service]") from e
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class CheckReq(BaseModel):
|
|
26
|
+
repo: str
|
|
27
|
+
task: str
|
|
28
|
+
limit: int = 12
|
|
29
|
+
interests: list[str] | None = None # the calling repo's declared subject tags
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class RecordReq(BaseModel):
|
|
33
|
+
type: str
|
|
34
|
+
title: str
|
|
35
|
+
scope: str
|
|
36
|
+
repo: str | None = None
|
|
37
|
+
body: str | None = None
|
|
38
|
+
status: str | None = None
|
|
39
|
+
found_by: str | None = None
|
|
40
|
+
ttl_days: int | None = None
|
|
41
|
+
owner: str | None = None
|
|
42
|
+
files: str | None = None
|
|
43
|
+
symptom: str | None = None
|
|
44
|
+
fix: str | None = None
|
|
45
|
+
tags: str | None = None
|
|
46
|
+
id: str | None = None
|
|
47
|
+
verified: bool = False
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class SearchReq(BaseModel):
|
|
51
|
+
query: str
|
|
52
|
+
scope: str | None = None
|
|
53
|
+
node_types: list[str] | None = None
|
|
54
|
+
limit: int = 25
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class LinkReq(BaseModel):
|
|
58
|
+
src: str
|
|
59
|
+
rel: str
|
|
60
|
+
dst: str
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class VerifyReq(BaseModel):
|
|
64
|
+
id: str
|
|
65
|
+
evidence: str # the observed check that passed (command + timestamp)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def create_app(store: Store | None = None) -> FastAPI:
|
|
69
|
+
app = FastAPI(title="OKL — the sixth surface", version="0.1.0")
|
|
70
|
+
_store = store or Store(os.environ.get("OKL_DATABASE_URL"))
|
|
71
|
+
# Optional shared-secret gate. If OKL_TOKEN is set, writes require it.
|
|
72
|
+
write_token = os.environ.get("OKL_TOKEN")
|
|
73
|
+
|
|
74
|
+
def _auth(authorization: str | None) -> None:
|
|
75
|
+
if write_token and authorization != f"Bearer {write_token}":
|
|
76
|
+
raise HTTPException(status_code=401, detail="missing or bad bearer token")
|
|
77
|
+
|
|
78
|
+
@app.get("/health")
|
|
79
|
+
def health() -> dict[str, Any]:
|
|
80
|
+
return {"ok": True, "nodes": len(_store.all_nodes()), "backend": _store.url.split(":")[0]}
|
|
81
|
+
|
|
82
|
+
@app.post("/check")
|
|
83
|
+
def check(req: CheckReq) -> dict[str, Any]:
|
|
84
|
+
return core.check(_store, req.repo, req.task, limit=req.limit,
|
|
85
|
+
interests=req.interests)
|
|
86
|
+
|
|
87
|
+
@app.post("/record")
|
|
88
|
+
def record(req: RecordReq, authorization: str | None = Header(default=None)) -> dict[str, str]:
|
|
89
|
+
_auth(authorization)
|
|
90
|
+
try:
|
|
91
|
+
node_id = core.record(_store, **req.model_dump())
|
|
92
|
+
except ValueError as e:
|
|
93
|
+
# Validation (unknown tag, bad scope) is the CALLER's error and must carry the
|
|
94
|
+
# message (e.g. the tag vocabulary) — a 500 hides it and reads as an outage.
|
|
95
|
+
raise HTTPException(status_code=400, detail=str(e)) from e
|
|
96
|
+
return {"id": node_id}
|
|
97
|
+
|
|
98
|
+
@app.post("/search")
|
|
99
|
+
def search(req: SearchReq) -> dict[str, Any]:
|
|
100
|
+
return {"results": core.search(_store, req.query, req.scope, req.node_types, req.limit)}
|
|
101
|
+
|
|
102
|
+
@app.post("/link")
|
|
103
|
+
def link(req: LinkReq, authorization: str | None = Header(default=None)) -> dict[str, bool]:
|
|
104
|
+
_auth(authorization)
|
|
105
|
+
core.link(_store, req.src, req.rel, req.dst)
|
|
106
|
+
return {"ok": True}
|
|
107
|
+
|
|
108
|
+
@app.post("/verify")
|
|
109
|
+
def verify(req: VerifyReq, authorization: str | None = Header(default=None)) -> dict[str, Any]:
|
|
110
|
+
_auth(authorization)
|
|
111
|
+
try:
|
|
112
|
+
return core.verify(_store, req.id, req.evidence)
|
|
113
|
+
except ValueError as e:
|
|
114
|
+
raise HTTPException(status_code=400, detail=str(e)) from e
|
|
115
|
+
|
|
116
|
+
@app.get("/metric/recurrence")
|
|
117
|
+
def recurrence() -> dict[str, Any]:
|
|
118
|
+
rows = _store.recurrence_after_arming()
|
|
119
|
+
return {"recurrence_after_arming": rows, "count": len(rows)}
|
|
120
|
+
|
|
121
|
+
@app.get("/nodes")
|
|
122
|
+
def nodes() -> dict[str, Any]:
|
|
123
|
+
from dataclasses import asdict
|
|
124
|
+
rows = [asdict(n) for n in _store.all_nodes()]
|
|
125
|
+
return {"nodes": rows, "count": len(rows)}
|
|
126
|
+
|
|
127
|
+
return app
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
app = None
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def run(host: str = "0.0.0.0", port: int = 8080) -> None:
|
|
134
|
+
import uvicorn
|
|
135
|
+
global app
|
|
136
|
+
app = create_app()
|
|
137
|
+
uvicorn.run(app, host=host, port=port)
|