org-knowledge-layer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okl/__init__.py +12 -0
- okl/__main__.py +8 -0
- okl/bootstrap.py +83 -0
- okl/cli.py +484 -0
- okl/client.py +160 -0
- okl/core.py +223 -0
- okl/drift.py +119 -0
- okl/mcp_server.py +75 -0
- okl/scaffold/MANIFEST.md +59 -0
- okl/scaffold/ci/method-gates.yml +32 -0
- okl/scaffold/ci/okl-verify.yml +59 -0
- okl/scaffold/claude/agents/architecture-reviewer.md +41 -0
- okl/scaffold/claude/commands/check-rules.md +24 -0
- okl/scaffold/claude/commands/feature-spec.md +37 -0
- okl/scaffold/claude/rules/example-area.md +22 -0
- okl/scaffold/claude/skills/RECOMMENDED-COMPANIONS.md +40 -0
- okl/scaffold/claude/skills/encoding-loop/SKILL.md +48 -0
- okl/scaffold/claude/skills/verify-before-claiming/SKILL.md +56 -0
- okl/scaffold/evals/README.md +32 -0
- okl/scaffold/evals/cases.jsonl +1 -0
- okl/scaffold/evals/run_evals.py +109 -0
- okl/scaffold/gates/check-canon-size.sh +11 -0
- okl/scaffold/gates/check-doc-orphans.sh +19 -0
- okl/scaffold/gates/check-retractions.sh +22 -0
- okl/scaffold/gates/check-tombstones.sh +22 -0
- okl/scaffold/gates/run-gates.sh +31 -0
- okl/scaffold/hooks/hooks.json +16 -0
- okl/scaffold/hooks/stop-okl-encode.sh +78 -0
- okl/scaffold/hooks/userpromptsubmit-okl-check.sh +68 -0
- okl/scaffold/plugin/plugin.json +10 -0
- okl/scaffold/profiles/dotnet/README.md +12 -0
- okl/scaffold/profiles/dotnet/rules/architecture.md +55 -0
- okl/scaffold/profiles/dotnet/rules/messaging.md +31 -0
- okl/scaffold/profiles/dotnet/rules/performance-and-data.md +36 -0
- okl/scaffold/profiles/dotnet/rules/security.md +42 -0
- okl/scaffold/profiles/geospatial/README.md +6 -0
- okl/scaffold/profiles/geospatial/rules/geospatial-ml.md +38 -0
- okl/scaffold/profiles/python-rag/README.md +13 -0
- okl/scaffold/profiles/python-rag/rules/fastapi-backend.md +37 -0
- okl/scaffold/profiles/python-rag/rules/project-structure.md +28 -0
- okl/scaffold/profiles/python-rag/rules/rag-pipeline.md +73 -0
- okl/scaffold/profiles/react/README.md +18 -0
- okl/scaffold/profiles/react/rules/frontend.md +57 -0
- okl/scaffold/registries/RETRACTIONS.md +19 -0
- okl/scaffold/registries/tombstones.txt +7 -0
- okl/scaffold/root/CLAUDE.md +55 -0
- okl/scaffold/root/METHOD.md +64 -0
- okl/scaffold_cmd.py +110 -0
- okl/seed/dotnet-canon.json +489 -0
- okl/seed/dotnet-decisions.json +328 -0
- okl/seed/dotnet-defects.json +133 -0
- okl/seed/dotnet-review-surfaces.json +147 -0
- okl/seed/frontend-canon.json +116 -0
- okl/seed/geospatial-deeptime-defects.json +59 -0
- okl/seed/geospatial-defects.json +154 -0
- okl/seed/geospatial-enforcement-defects.json +121 -0
- okl/seed/geospatial-eval-defects.json +25 -0
- okl/seed/rag-defects.json +120 -0
- okl/seed/react-defects.json +45 -0
- okl/seed.py +55 -0
- okl/service.py +137 -0
- okl/store.py +432 -0
- org_knowledge_layer-0.1.0.dist-info/METADATA +475 -0
- org_knowledge_layer-0.1.0.dist-info/RECORD +67 -0
- org_knowledge_layer-0.1.0.dist-info/WHEEL +4 -0
- org_knowledge_layer-0.1.0.dist-info/entry_points.txt +2 -0
- org_knowledge_layer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Stop hook — the write-side mechanical catch for the encoding loop.
|
|
3
|
+
#
|
|
4
|
+
# The read side (okl check) is enforced by the PreToolUse hook; nothing enforced the WRITE
|
|
5
|
+
# side, so a session could end without recording what it learned ("a merged fix without the
|
|
6
|
+
# rule is a half-finished job"). This hook asks the question at the ship moment, once:
|
|
7
|
+
# if the session changed the working tree, block the first stop (exit 2) with a prompt to
|
|
8
|
+
# either `okl record` the lesson or state that there is none. It never fires twice in one
|
|
9
|
+
# session (marker file) and never loops (stop_hook_active guard).
|
|
10
|
+
set -uo pipefail
|
|
11
|
+
|
|
12
|
+
# Same resolver as pretooluse-okl-check.sh (env → pinned config → PATH → python3 -m okl);
|
|
13
|
+
# the reminder is best-effort, so an unresolvable okl silently disables it rather than blocking.
|
|
14
|
+
resolve_okl() {
|
|
15
|
+
if [ -n "${OKL_BIN:-}" ]; then printf '%s' "$OKL_BIN"; return 0; fi
|
|
16
|
+
local d="$PWD"
|
|
17
|
+
while [ "$d" != "/" ]; do
|
|
18
|
+
if [ -f "$d/.okl/config.json" ]; then
|
|
19
|
+
local bin
|
|
20
|
+
bin=$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1])).get("okl_bin") or "")' \
|
|
21
|
+
"$d/.okl/config.json" 2>/dev/null || true)
|
|
22
|
+
if [ -n "$bin" ]; then printf '%s' "$bin"; return 0; fi
|
|
23
|
+
break
|
|
24
|
+
fi
|
|
25
|
+
d=$(dirname "$d")
|
|
26
|
+
done
|
|
27
|
+
if command -v okl >/dev/null 2>&1; then printf '%s' "okl"; return 0; fi
|
|
28
|
+
if python3 -c "import okl" >/dev/null 2>&1; then printf '%s' "python3 -m okl"; return 0; fi
|
|
29
|
+
return 1
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
OKL=$(resolve_okl) || exit 0
|
|
33
|
+
|
|
34
|
+
payload=$(cat 2>/dev/null || true)
|
|
35
|
+
parsed=$(printf '%s' "$payload" | python3 -c '
|
|
36
|
+
import json, sys
|
|
37
|
+
try:
|
|
38
|
+
d = json.load(sys.stdin)
|
|
39
|
+
except Exception:
|
|
40
|
+
d = {}
|
|
41
|
+
print(d.get("session_id", ""))
|
|
42
|
+
print("true" if d.get("stop_hook_active") else "false")
|
|
43
|
+
' 2>/dev/null) || parsed=""
|
|
44
|
+
session_id=$(printf '%s\n' "$parsed" | sed -n 1p)
|
|
45
|
+
stop_hook_active=$(printf '%s\n' "$parsed" | sed -n 2p)
|
|
46
|
+
[ -n "$stop_hook_active" ] || stop_hook_active="false"
|
|
47
|
+
|
|
48
|
+
# Never loop: if we already blocked once and Claude is stopping again, let it stop.
|
|
49
|
+
[ "${stop_hook_active}" = "true" ] && exit 0
|
|
50
|
+
|
|
51
|
+
# Only fire when the session plausibly did work: uncommitted changes, or a commit in the
|
|
52
|
+
# last hour (covers commit-then-stop sessions).
|
|
53
|
+
changed=0
|
|
54
|
+
if [ -n "$(git status --porcelain 2>/dev/null)" ]; then
|
|
55
|
+
changed=1
|
|
56
|
+
elif last=$(git log -1 --format=%ct 2>/dev/null); then
|
|
57
|
+
now=$(date +%s)
|
|
58
|
+
[ $((now - last)) -lt 3600 ] && changed=1
|
|
59
|
+
fi
|
|
60
|
+
[ "$changed" = "1" ] || exit 0
|
|
61
|
+
|
|
62
|
+
# Once per session (fall back to a repo-scoped marker when no session id is provided).
|
|
63
|
+
marker="${TMPDIR:-/tmp}/okl-encode-reminder-${session_id:-$(pwd | cksum | cut -d' ' -f1)}"
|
|
64
|
+
[ -e "$marker" ] && exit 0
|
|
65
|
+
touch "$marker" 2>/dev/null || true
|
|
66
|
+
|
|
67
|
+
cat >&2 <<'MSG'
|
|
68
|
+
ENCODING LOOP — before this session ends: did it surface a lesson worth keeping?
|
|
69
|
+
A non-obvious failure mode, a rule discovered the hard way, a decision that shouldn't be
|
|
70
|
+
silently reversed? If yes, record it now (choose the scope deliberately — 'org' spreads
|
|
71
|
+
to every repo, 'repo' stays local — and tag the subject):
|
|
72
|
+
|
|
73
|
+
okl record --type Defect|Rule|Decision --scope org|repo --tags "<subjects>" \
|
|
74
|
+
--title "..." --symptom "..." --body "cause: ..." --fix "..."
|
|
75
|
+
|
|
76
|
+
If the session genuinely learned nothing durable, state that explicitly and finish.
|
|
77
|
+
MSG
|
|
78
|
+
exit 2
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# UserPromptSubmit hook — inject the org's relevant lessons into the model's context
|
|
3
|
+
# BEFORE it starts the task. This event is the only correct one for delivery: its stdout
|
|
4
|
+
# (exit 0) is added to Claude's context, and its stdin carries the actual prompt text, so
|
|
5
|
+
# the briefing is retrieved for the task the user really asked for.
|
|
6
|
+
#
|
|
7
|
+
# (The earlier PreToolUse version fired on every edit and printed the briefing to a channel
|
|
8
|
+
# the model never sees — PreToolUse exit-0 stdout goes to the transcript only. Discovered by
|
|
9
|
+
# an end-to-end test: hook fired, briefing correct, defect reproduced anyway.)
|
|
10
|
+
#
|
|
11
|
+
# FAILS CLOSED via exit 2: if the knowledge layer is unreachable, the prompt is blocked
|
|
12
|
+
# rather than letting the agent proceed blind. A check that reports "clean" while broken is
|
|
13
|
+
# worse than no check.
|
|
14
|
+
set -uo pipefail
|
|
15
|
+
|
|
16
|
+
# Resolve how to invoke okl (env → pinned config → PATH → python3 -m okl); hooks run in
|
|
17
|
+
# whatever environment the harness spawns, which often lacks the venv/pipx bin dir.
|
|
18
|
+
resolve_okl() {
|
|
19
|
+
if [ -n "${OKL_BIN:-}" ]; then printf '%s' "$OKL_BIN"; return 0; fi
|
|
20
|
+
local d="$PWD"
|
|
21
|
+
while [ "$d" != "/" ]; do
|
|
22
|
+
if [ -f "$d/.okl/config.json" ]; then
|
|
23
|
+
local bin
|
|
24
|
+
bin=$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1])).get("okl_bin") or "")' \
|
|
25
|
+
"$d/.okl/config.json" 2>/dev/null || true)
|
|
26
|
+
if [ -n "$bin" ]; then printf '%s' "$bin"; return 0; fi
|
|
27
|
+
break
|
|
28
|
+
fi
|
|
29
|
+
d=$(dirname "$d")
|
|
30
|
+
done
|
|
31
|
+
if command -v okl >/dev/null 2>&1; then printf '%s' "okl"; return 0; fi
|
|
32
|
+
if python3 -c "import okl" >/dev/null 2>&1; then printf '%s' "python3 -m okl"; return 0; fi
|
|
33
|
+
return 1
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
if ! OKL=$(resolve_okl); then
|
|
37
|
+
[ "${OKL_OFFLINE:-0}" = "1" ] && exit 0
|
|
38
|
+
echo "okl NOT FOUND — blocking (a check that can't run must not pass as clean)." >&2
|
|
39
|
+
echo "Install it (pip install okl), set OKL_BIN, or re-run 'okl init' from a shell where it" >&2
|
|
40
|
+
echo "works (pins okl_bin into .okl/config.json). OKL_OFFLINE=1 proceeds without the layer." >&2
|
|
41
|
+
exit 2
|
|
42
|
+
fi
|
|
43
|
+
|
|
44
|
+
# The task is the prompt itself (stdin JSON: {"prompt": "..."}); OKL_TASK overrides;
|
|
45
|
+
# last-commit-message is only the fallback of last resort.
|
|
46
|
+
payload=$(cat 2>/dev/null || true)
|
|
47
|
+
prompt=$(printf '%s' "$payload" | python3 -c '
|
|
48
|
+
import json, sys
|
|
49
|
+
try:
|
|
50
|
+
print((json.load(sys.stdin).get("prompt") or "").strip()[:2000])
|
|
51
|
+
except Exception:
|
|
52
|
+
print("")
|
|
53
|
+
' 2>/dev/null || true)
|
|
54
|
+
TASK="${OKL_TASK:-${prompt:-$(git log -1 --pretty=%s 2>/dev/null || echo 'general work')}}"
|
|
55
|
+
|
|
56
|
+
# $OKL unquoted on purpose: it may be a command + args ("python3 -m okl").
|
|
57
|
+
if out=$($OKL check --task "$TASK" --format agent 2>/dev/null); then
|
|
58
|
+
printf '%s\n' "$out" # stdout → the model's context
|
|
59
|
+
exit 0
|
|
60
|
+
fi
|
|
61
|
+
|
|
62
|
+
if [ "${OKL_OFFLINE:-0}" = "1" ]; then
|
|
63
|
+
echo "OKL offline (OKL_OFFLINE=1 acknowledged) — proceeding without the layer." >&2
|
|
64
|
+
exit 0
|
|
65
|
+
fi
|
|
66
|
+
echo "OKL UNREACHABLE — blocking this prompt. A check that reports 'clean' while broken is worse than no check." >&2
|
|
67
|
+
echo "Fix connectivity, or set OKL_OFFLINE=1 to explicitly proceed without the org knowledge layer." >&2
|
|
68
|
+
exit 2
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "encoding-loop-method",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "The encoding-loop engineering method as an installable Claude Code plugin: lean canon, the encoding-loop skill, an architecture-reviewer subagent, feature-spec and check-rules commands, a fail-closed pre-task hook wired to the okl org knowledge layer, and portable drift-gates + an eval tier.",
|
|
5
|
+
"author": "Joshua Dell",
|
|
6
|
+
"skills": ["./skills/encoding-loop"],
|
|
7
|
+
"agents": ["./agents/architecture-reviewer.md"],
|
|
8
|
+
"commands": ["./commands/feature-spec.md", "./commands/check-rules.md"],
|
|
9
|
+
"hooks": "./hooks/hooks.json"
|
|
10
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# .NET microservices profile
|
|
2
|
+
|
|
3
|
+
Canon extracted verbatim from the .NET platform (`.NET 10 / Aspire / RabbitMQ-Wolverine / gRPC / EF Core`).
|
|
4
|
+
Installed to `.claude/rules/` as path-scoped rule files (load only when matching files are in context):
|
|
5
|
+
|
|
6
|
+
- `architecture.md` — VSA + DDD + CQRS; consumer-substitution interface rule; VSA→Clean promotion signal
|
|
7
|
+
- `security.md` — IDOR→404, JWT ClockSkew, server-controlled fields, rate-limiter scale-out
|
|
8
|
+
- `performance-and-data.md` — DbContext-direct (no repository wrappers), EF projection, async, Guid v7
|
|
9
|
+
- `messaging.md` — Wolverine/RabbitMQ topology, outbox atomicity, handler-discovery≠DI, gRPC/REST versioning
|
|
10
|
+
|
|
11
|
+
Has a React frontend? The React canon is a **separate, backend-agnostic profile** — stack it on:
|
|
12
|
+
`okl scaffold --profile dotnet --profile react`.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: SOLID / DDD / VSA architecture rules for .NET microservices
|
|
3
|
+
paths: ["**/*.cs"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Architecture (.NET microservices — VSA + DDD + CQRS)
|
|
7
|
+
|
|
8
|
+
> Ported verbatim from the .NET platform CLAUDE.md. Stack: .NET 10, Aspire, RabbitMQ (Wolverine),
|
|
9
|
+
> gRPC, EF Core, React. DDD + CQRS + event-driven.
|
|
10
|
+
|
|
11
|
+
## Vertical Slice Architecture is the default for every service
|
|
12
|
+
|
|
13
|
+
- Organize by **feature**, not by kind. `ServiceName/Features/` holds one file per use case:
|
|
14
|
+
command/query record + validator + handler co-located. Saga event-handlers live here too.
|
|
15
|
+
- `Domain/` holds only what is *genuinely shared* across features — aggregates, value objects, enums,
|
|
16
|
+
and consumer-substitution ports. `Infrastructure/` holds EF Core, caching, gateways, DI. `Program.cs`
|
|
17
|
+
is the composition root.
|
|
18
|
+
- **Feature-file soft cap ~300 lines.** Past it, extract the validator or line-item record into a
|
|
19
|
+
sibling file in `Features/` — the cap is on size, not file-count-per-slice.
|
|
20
|
+
- **Don't apply both VSA and Clean across one service.** Pick one shape per service and commit. The
|
|
21
|
+
cross-service pattern diff is intentional — it's the project's lesson, not an inconsistency.
|
|
22
|
+
|
|
23
|
+
## Promotion signal — when to consider Clean Architecture
|
|
24
|
+
|
|
25
|
+
- VSA stays the default. Consider a multi-project split ONLY at 5+ aggregates per service with
|
|
26
|
+
cross-cutting domain rules several features coordinate on, AND `Domain/` growing faster than `Features/`.
|
|
27
|
+
- The **dependency rule** (Domain → nothing; IO at the edges) is already in force in VSA at every scale —
|
|
28
|
+
it is not complexity-gated. Only the multi-project *structure* is gated.
|
|
29
|
+
- Escalate enforcement as the cost of a violated boundary rises: **convention → architecture tests
|
|
30
|
+
(NetArchTest/analyzer) → project split.** The middle rung enforces the same boundary the 4-project
|
|
31
|
+
split does, deterministically, without the project ceremony. Reach for the split only when you want
|
|
32
|
+
the *compiler* (not a test) to hold the line, or need separate deploy/versioning units.
|
|
33
|
+
|
|
34
|
+
## SOLID (the load-bearing parts)
|
|
35
|
+
|
|
36
|
+
- Domain → nothing. Application → Domain. Infrastructure → Domain + Application. Api → all (composition root).
|
|
37
|
+
- A service with no domain entities doesn't need a Domain project — ports (`I*Sender`, `I*Resolver`)
|
|
38
|
+
live in `Application/Interfaces/`.
|
|
39
|
+
- **Interfaces earn their keep through consumer substitution, not "future swap."** A port/adapter
|
|
40
|
+
interface is justified only if **(a)** it's substituted by tests today (`grep "Substitute.For<IFoo"`),
|
|
41
|
+
**(b)** 2+ concrete impls are registered today, or **(c)** a second impl is on a *concrete* near-term
|
|
42
|
+
roadmap. If none hold, it's speculative coupling — delete it, take the concrete class.
|
|
43
|
+
- **Factory / `[FromKeyedServices]` is the shape ONCE condition (b) holds — not before.** A factory
|
|
44
|
+
that returns the same single impl for every input is the same speculative coupling as a deleted
|
|
45
|
+
`I*Repository`. Introduce it the day the second impl actually ships.
|
|
46
|
+
|
|
47
|
+
## DDD
|
|
48
|
+
|
|
49
|
+
- **Rich domain entities only when someone observes the invariant.** Persisted entity with non-trivial
|
|
50
|
+
observable invariants → state changes go through methods, never public setters, with validating
|
|
51
|
+
`static Create()` factories. In-memory, single-use, discarded-after-handler → skip the aggregate shape,
|
|
52
|
+
inline the validation or use a FluentValidation rule.
|
|
53
|
+
- Value objects for Money (amount+currency), Quantity (non-negative). Aggregates control their children —
|
|
54
|
+
no mutable collection exposure (`IReadOnlyList<T>` over `private readonly List<T>`; add via `AddLine()`).
|
|
55
|
+
- Domain events for state changes that affect other bounded contexts.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Wolverine / RabbitMQ messaging, outbox, gRPC, versioning rules
|
|
3
|
+
paths: ["**/Program.cs", "**/Features/**/*.cs", "**/*RecoveryJob*.cs"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Communication patterns (.NET — Wolverine / RabbitMQ / gRPC)
|
|
7
|
+
|
|
8
|
+
> Ported verbatim from the .NET platform CLAUDE.md. Each bullet is a dated distributed-systems trap.
|
|
9
|
+
|
|
10
|
+
## Transport & topology
|
|
11
|
+
|
|
12
|
+
- **Messaging transport is RabbitMQ** in every environment (dev matches prod). Wolverine maps saga pub/sub onto fanout exchanges (one per event family) with a queue per consumer. **Azure Service Bus was evaluated and removed** — its local emulator can't run the saga; re-add only if Azure becomes a real target.
|
|
13
|
+
- **Fanout exchanges silently DISCARD unroutable messages; AutoProvision declares topology lazily per service.** An event published before a consumer's first boot is dropped while the outbox marks it delivered. Rule: **each publisher declares its own exchange AND its consumers' queues+bindings** (`BindExchange(...).ToQueue(...)`), names from `MessagingExchanges`/`MessagingQueues` constants — never inline literals (a typo'd name is auto-provisioned as an empty object and the consumer starves silently).
|
|
14
|
+
- **Wolverine durability is per-direction.** `UseDurableOutboxOnAllSendingEndpoints()` covers ONLY the send side; default listeners are buffered (acked before handlers run — a crash loses the buffer). Store-backed services call `UseDurableInboxOnAllListeners()`; store-less services use `.ProcessInline()`. Every new `ListenToRabbitQueue` gets one of the two.
|
|
15
|
+
- **Durability ≠ replay.** Durable pub/sub (RabbitMQ durable queues + transactional outbox + idempotent handlers) does NOT lose messages. Reach for a stream (Kafka/Event Hubs/Redis Streams) only when you need replay-from-offset, multi-day retention, an ordered append-only log, or N independent re-reading consumers — not merely "don't lose messages."
|
|
16
|
+
|
|
17
|
+
## Outbox atomicity
|
|
18
|
+
|
|
19
|
+
- **Transactional publishing must use the enlisted context, NOT constructor-injected `IEventPublisher`.** Only the `IMessageContext` Wolverine injects as a `HandleAsync` parameter (or an `IDbContextOutbox` in non-handler code) is enlisted in the outbox transaction. A constructor-injected `IMessageBus`/`IEventPublisher` publishes inline under Wolverine 6 — before commit — breaking outbox atomicity.
|
|
20
|
+
- **Outbox outside a handler:** `BeginTransactionAsync` → entity write + `PublishAsync` → **`SaveChangesAsync`** → `CommitAsync`. Skipping the `SaveChangesAsync` between publish and commit silently drops the staged envelope.
|
|
21
|
+
|
|
22
|
+
## Handler discovery, gRPC, REST versioning
|
|
23
|
+
|
|
24
|
+
- **Wolverine handler discovery is NOT DI registration — two separate containers.** `opts.Discovery.IncludeAssembly(...)` builds Wolverine's internal message→handler map; Wolverine constructs handlers itself. `serviceProvider.GetRequiredService<MyHandler>()` throws unless you also `AddScoped<MyHandler>()`. The path that hits this is integration tests resolving handlers directly — every such handler needs an explicit `AddScoped<T>()`.
|
|
25
|
+
- **No `IRequestHandler`/`IFooHandler` interface per handler.** Handlers are plain classes; Wolverine's bus is the abstraction. Handler interfaces fail the consumer-substitution test the same way `IFooRepository` did.
|
|
26
|
+
- **gRPC** (sync) for real-time inter-service queries, versioned via `.proto` `package`. **REST** (HTTP) for frontend only, URL-segment versioned — always `app.MapV1ApiGroup("Tag", "resource")`, never hand-rolled `NewVersionedApi(...).MapGroup(...).HasApiVersion(...)` chains.
|
|
27
|
+
|
|
28
|
+
## Package management (Aspire)
|
|
29
|
+
|
|
30
|
+
- Central Package Management via `Directory.Packages.props`; `.csproj` references packages without versions.
|
|
31
|
+
- **Aspire SDK and runtime packages must match** (minor included) — bump together. **Aspire 13+ Azure resources need explicit local-dev fallbacks** (gate on `IsPublishMode`). **`WithReference(x)` ≠ wait-for-healthy** — every `WithReference` on a non-trivial dependency gets a matching `.WaitFor(x)`.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: EF Core, async, concurrency, pagination performance rules
|
|
3
|
+
paths: ["**/Features/**/*.cs", "**/Infrastructure/**/*.cs", "**/Domain/**/*.cs"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Performance & data correctness (.NET / EF Core)
|
|
7
|
+
|
|
8
|
+
> Ported verbatim from the .NET platform CLAUDE.md "Performance Rules" + "Data access". Always-on headlines;
|
|
9
|
+
> full rationale lives in the source repo's docs/performance-and-data-correctness.md.
|
|
10
|
+
|
|
11
|
+
## Data access: DbContext directly, no repository wrappers
|
|
12
|
+
|
|
13
|
+
- **Handlers take `DbContext` (or `IDbContextFactory<T>`) directly. No `IFooRepository` interfaces.**
|
|
14
|
+
`DbContext` is already the Unit of Work; `DbSet<T>` is already the Repository. The only reason to wrap
|
|
15
|
+
was unit-test mocking — replaced by integration tests against Testcontainers.
|
|
16
|
+
- **Reads project to DTOs inside the IQueryable:** `context.Orders.AsNoTracking().Where(...).Select(o => new OrderSummaryDto { ... }).ToListAsync(ct)` — in the handler, no method wrapping, no in-memory mapper. The projection IS the read contract. EF auto-splits projected collection navigations (no cartesian rows).
|
|
17
|
+
- **Writes load the aggregate tracked and call `SaveChangesAsync`.** Optimistic concurrency tokens fire on `SaveChanges`.
|
|
18
|
+
- **Exception: outbox-atomic non-handler code** (`BackgroundService` sweepers) needs an explicit transactional wrap (`BeginTransactionAsync` → work → `SaveChangesAsync` → `CommitAsync`) so Wolverine's staged outbox envelopes persist atomically.
|
|
19
|
+
|
|
20
|
+
## The always-on rules
|
|
21
|
+
|
|
22
|
+
- **Reads: `AsNoTracking()` + `.Select(...)` into a DTO inside the IQueryable.** Plain `AsNoTracking()` returning an entity is a half-fix.
|
|
23
|
+
- **No N+1** — `Include` or projection; never query inside a `foreach` over another query's results.
|
|
24
|
+
- **Non-sargable predicates defeat indexes — fix at write time.** `u.Email.ToLower() == x` can't use a B-tree index; normalize on insert (`EmailNormalized`) or use a case-insensitive collation. Leading-wildcard `LIKE '%text%'` needs `tsvector`/Elasticsearch when load justifies it.
|
|
25
|
+
- **Async on request paths:** `await` everywhere. Never `.Result`/`.Wait()`/`.GetAwaiter().GetResult()` (banned at build time). Every async method propagates `CancellationToken`.
|
|
26
|
+
- **Parallelize independent awaits with `Task.WhenAll`** — but not dependent ops, not a shared `DbContext` (use `IDbContextFactory<T>`), and when N calls hit the SAME service prefer a batch endpoint (one round-trip, server-atomic).
|
|
27
|
+
- **Long-running work (>~1s) belongs on the message bus** — reshape as 202 Accepted: validate + persist tracking row + publish Wolverine message + return. Same for handlers themselves.
|
|
28
|
+
- **Fan-out belongs on the message bus, not a synchronous handler loop** — one message per recipient/batch, throttled with `MaxDegreeOfParallelism`.
|
|
29
|
+
- **Pagination:** every list endpoint paginates with a server-side cap (≤100); keyset for large offsets.
|
|
30
|
+
- **Bulk ops:** `ExecuteUpdateAsync`/`ExecuteDeleteAsync`, never load thousands of rows to mutate.
|
|
31
|
+
- **Optimistic concurrency:** every updatable aggregate has a token (Postgres `xmin` or row-version). Last-write-wins is not acceptable.
|
|
32
|
+
- **Entity IDs use `Guid.CreateVersion7()`, not `Guid.NewGuid()`** — time-ordered, so PK inserts append-extend the B-tree index instead of fragmenting it. Apply in aggregate factories. (Not for IDs where mint time is sensitive.)
|
|
33
|
+
- **`DbContext` is not thread-safe** — parallel queries require `IDbContextFactory<T>`, one per task.
|
|
34
|
+
- **Migrations are immutable once applied** — destructive changes need a multi-step plan.
|
|
35
|
+
- **Measure before optimizing** — BenchmarkDotNet / `dotnet-counters` / `ToQueryString()`. No caching/compiled-queries/`AsSplitQuery()` on intuition.
|
|
36
|
+
- **Dapper is the sanctioned escape hatch from EF**, not a peer — only for provider-specific SQL, proven EF bottleneck, or LINQ-obscured aggregation; share the EF connection via `ctx.Database.GetDbConnection()`.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Security rules — IDOR, JWT, server-controlled fields, rate limiting
|
|
3
|
+
paths: ["**/Endpoints/**/*.cs", "**/Features/**/*.cs", "**/ServiceDefaults/**/*.cs", "**/Program.cs"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Security (.NET microservices)
|
|
7
|
+
|
|
8
|
+
> Ported verbatim from the .NET platform CLAUDE.md "Security Requirements". Each rule earned its place from a
|
|
9
|
+
> dated defect.
|
|
10
|
+
|
|
11
|
+
## Authorization — the IDOR pattern (CWE-639)
|
|
12
|
+
|
|
13
|
+
A missing scope check is an IDOR, and IDORs slip through tests-by-omission. Canonical buyer-scoped shape:
|
|
14
|
+
|
|
15
|
+
- Endpoint reads `ClaimTypes.NameIdentifier` from the JWT → passes as `RequestingBuyerId` into the query/command.
|
|
16
|
+
- **Read handlers push the ownership predicate INTO the EF `Where` clause** (`Where(o => o.Id == OrderId && o.BuyerId == RequestingBuyerId)`). Non-owner rows never leave the DB — tighter than a post-materialization C# check a buggy refactor could weaken.
|
|
17
|
+
- **Write handlers** load the aggregate tracked, check ownership on the loaded entity, return `false`/`null` on mismatch (NOT throw, NOT 403).
|
|
18
|
+
- Endpoint translates `null`/`false` → **404**. Returning **403 leaks existence** ("this exists, just not yours"). 404 is indistinguishable from "not found."
|
|
19
|
+
- **An integration test asserting buyer X cannot read buyer Y's entity is REQUIRED** for every new scoped-entity endpoint. Its absence is how the original `GET /orders/{id}` IDOR survived the codebase's lifetime.
|
|
20
|
+
|
|
21
|
+
## JWT validation (explicit, not implicit)
|
|
22
|
+
|
|
23
|
+
- `ValidateIssuerSigningKey = true` (explicit is auditable).
|
|
24
|
+
- `ClockSkew = TimeSpan.FromSeconds(30)` — the default is **5 minutes**, which on 5-minute access tokens doubles every token's effective lifetime.
|
|
25
|
+
- `ValidateAudience` / `ValidateIssuer` / `ValidateLifetime` all `true`.
|
|
26
|
+
- **`RequireHttpsMetadata` is fail-closed outside Development** — never derived silently from the authority scheme. An http authority in Production must fail loudly at startup (plaintext OIDC/JWKS = MITM can inject signing keys). Legitimate internal-http opts out explicitly via `Authentication:RequireHttpsMetadata=false` (logs a warning).
|
|
27
|
+
- Keycloak token policy pinned in `auth-realm.json`, never realm defaults: 5-min access tokens, single-use rotated refresh tokens, session idle 30m / max 10h.
|
|
28
|
+
|
|
29
|
+
## Server-controlled fields — computed server-side, never trusted from the client
|
|
30
|
+
|
|
31
|
+
Money (price, currency, tax), authorization identifiers (`BuyerId`/`SellerId` — must match JWT `sub`),
|
|
32
|
+
state-machine columns (`Status`), and security flags (`IsAdmin`, `IsDeleted`) are server-controlled.
|
|
33
|
+
A `[FromBody]` DTO with a `Price` field is a price-tampering vuln (client submits `Price = 0.01` for a
|
|
34
|
+
$999 product). The handler fetches the authoritative value from its source (CatalogService gRPC for
|
|
35
|
+
`Price`+`Currency`, JWT `sub` for buyer identity, DB for `Status`) and uses *that* — the request DTO is
|
|
36
|
+
untrusted input.
|
|
37
|
+
|
|
38
|
+
## Error handling & transport security
|
|
39
|
+
|
|
40
|
+
- Never expose internal state, stack traces, or entity IDs. Response `traceId` uses `Activity.TraceId.ToString()` (32 hex) only, NOT `Activity.Id` (the full W3C traceparent leaks span structure).
|
|
41
|
+
- HTTPS redirection enforced in prod; explicit CORS allowing only known frontend origins.
|
|
42
|
+
- **Rate limiting on search + payment endpoints minimum.** In-memory limiters silently weaken to N× the limit at N instances — swap to a Redis-backed limiter once a service runs 2+ instances, with the `INCR`+`EXPIRE` pair wrapped in a Lua `EVAL` for atomicity. Single-instance today = in-memory is correct *for now* + a comment naming the swap trigger.
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# Geospatial ML profile
|
|
2
|
+
|
|
3
|
+
Canon extracted from the geospatial pipeline (`rslearn / OlmoEarth / Sentinel-1&2 / STAC / GDAL`).
|
|
4
|
+
Installed to `.claude/rules/`:
|
|
5
|
+
|
|
6
|
+
- `geospatial-ml.md` — label quality & time-matching, spatial CV, class_path/materialize/num_classes/temporal-decoder verification, tile-store + CPL_TMPDIR storage traps, adversarial prior-art audit
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Geospatial ML rules — rslearn/OlmoEarth, label quality, spatial CV, storage
|
|
3
|
+
paths: ["**/*.py", "**/*.yaml", "**/*.yml"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Geospatial ML (rslearn / OlmoEarth / Sentinel / remote sensing)
|
|
7
|
+
|
|
8
|
+
> Ported from the geospatial pipeline method.md. Stack: rslearn, OlmoEarth, Sentinel-1/2, STAC,
|
|
9
|
+
> Planetary Computer, GDAL. Each rule is a dated defect (error # from method.md).
|
|
10
|
+
|
|
11
|
+
## Label quality — the failures that compile
|
|
12
|
+
|
|
13
|
+
- **Don't rasterize every polygon as the positive class** (error #1: ~45% of positive labels were wrong — urban/agriculture/upland/water taught as geospatial). Inspect what a label layer actually contains before fitting to it.
|
|
14
|
+
- **Labels and imagery must be time-matched** (error #3: NAIP 2020 labels fit against Sentinel-2 2024 = 4-year gap of self-inflicted label noise). Check the acquisition date in the source metadata of *both* sides.
|
|
15
|
+
- **A model scored against wrong labels produces a meaningless metric** — verify label provenance before trusting any F1/AUC/κ.
|
|
16
|
+
|
|
17
|
+
## Spatial cross-validation
|
|
18
|
+
|
|
19
|
+
- **Unshuffled KFold on a spatial grid leaks nothing and looks broken** (error #11: AUC 0.23 looked like a broken encoder; it was unshuffled KFold on a spatial grid — shuffled gave 0.85). Use spatial-block or shuffled CV; refuse a convenient bad number until you've ruled out the split.
|
|
20
|
+
- **Count/plot points on a map before trusting a public dataset's coordinate order** (error #12: `Virgin_River` rows had x/y transposed — 119 points in the wrong hemisphere).
|
|
21
|
+
|
|
22
|
+
## rslearn / OlmoEarth scaffolding — verify, don't assert from memory
|
|
23
|
+
|
|
24
|
+
- **Import every `class_path` before writing it into a config** (error #14: 5 of 23 class_paths didn't exist — every class name right, every module path wrong, written from memory, never imported; they fail at runner startup on a rented GPU). A mechanical `check-scaffold-classpaths.sh` that imports all of them is the gate.
|
|
25
|
+
- **Verify materialize wrote files — never trust the exit code** (error #15: `rslearn dataset materialize` exited 0 having written zero files; `NotImplementedError` on all 238 windows swallowed into a worker pool). `verify_materialized()` checks the rasters are on disk.
|
|
26
|
+
- **`num_classes` = max crosswalk id + 1 when class 0 is reserved** (error #18: `num_classes: 4` crashed `Target 4 is out of bounds` because the crosswalk emits 4 real classes and `zero_is_invalid` reserves class 0 → needs 5). A `test_class_scheme_contract.py` asserts this mechanically.
|
|
27
|
+
- **Decoders on a temporal cube must read the true last-two axes** (error #19: `SegmentationPoolingDecoder` read `image.shape[1:3]` = `(timesteps, H)` on a 4-D `[bands, timesteps, H, W]` input, predicting 12×2 against a 2×2 target). Use a temporal-aware adapter; catch it with a laptop dry-run, not a GPU.
|
|
28
|
+
|
|
29
|
+
## Storage & temp dirs (cloud-optimized geotiff / GDAL)
|
|
30
|
+
|
|
31
|
+
- **Budget the tile store, not the output** (error #16: "~1.2 GB" estimate was right for materialized chips, ~10× low for `ingest` which pulls whole 110 km granules → 11 GB tile store).
|
|
32
|
+
- **Redirect EVERY temp mechanism, not the first one** (errors #16/#17: `TMPDIR` on the boot disk filled `/` to zero; moving `TMPDIR` to the data drive still leaked 2.8 GB to `/` because **GDAL keeps its own `CPL_TMPDIR`**). Set both `TMPDIR` and `CPL_TMPDIR` to the data drive.
|
|
33
|
+
|
|
34
|
+
## Novelty & prior art (the claim is falsifiable)
|
|
35
|
+
|
|
36
|
+
- **Audit the literature adversarially before building** (error #5: a "nobody has mapped X" novelty claim was later found to be falsified by prior art — the geospatial repo's method.md attributes this to Evangelista et al. 2018 2-epoch change maps and a 2018 CO-RIP basin-wide RF study at κ 0.80, both found *after* building; these citations are transcribed from method.md and **not independently re-verified here — confirm against the literature before relying on them**). Run `/paper-audit` looking for reasons a paper *refutes* your novelty claim.
|
|
37
|
+
- **Run the control before the experiment** — a bad number on the interesting question is uninterpretable (broken pipeline / too few labels / real effect all predict the same failure).
|
|
38
|
+
- **Report the result that came out** (error #4: the mean-pooling defect was real, fixing it moved F1 0.021→0.065 vs RF 0.701 — the hypothesis was wrong, and was published wrong). A real defect is not proof it was the cause.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Python RAG / agentic profile
|
|
2
|
+
|
|
3
|
+
Canon extracted from the RAG service (`FastAPI / Qdrant hybrid / Redis / Postgres / Ollama / deepeval`).
|
|
4
|
+
Installed to `.claude/rules/`:
|
|
5
|
+
|
|
6
|
+
- `rag-pipeline.md` — retrieval-vs-identity (index & pre-filter, GROUP BY ≠ similarity), eval integrity (failure-count-first, judge≠generator, cross-tab, real fixtures), the read-only agent contract, embedder-not-hot-swappable
|
|
7
|
+
- `fastapi-backend.md` — async event-loop discipline (no sync/CPU work on the loop), middleware order & auth, streaming scrub-before-emit, one-orchestrator-per-pipeline, config through the settings layer
|
|
8
|
+
- `project-structure.md` — one-way app→library dependency, install-as-package (no sys.path hacks), extras for heavy deps, CLI/composition-root hygiene
|
|
9
|
+
|
|
10
|
+
Note: the RAG service ships a React UI, but its own canon (docs/) contains no stated frontend rule set —
|
|
11
|
+
so this profile ports backend + library rules only. React rules live in the separate, backend-agnostic
|
|
12
|
+
`react` profile — stack it on if your repo has a React frontend:
|
|
13
|
+
`okl scaffold --profile python-rag --profile react`.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: FastAPI request-path rules — async event loop, middleware order, streaming, config
|
|
3
|
+
paths: ["**/main.py", "**/app/**/*.py", "**/backend/**/*.py"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# FastAPI backend (async request path)
|
|
7
|
+
|
|
8
|
+
> Ported from the RAG service docs/code-review.md — each is a dated, file:line finding.
|
|
9
|
+
|
|
10
|
+
## The async event loop is shared — never block it
|
|
11
|
+
|
|
12
|
+
- **No CPU/IO-heavy synchronous work in an async handler** (main.py:1889: synchronous PDF work in an async handler stalled the event loop for all concurrent requests, including in-flight SSE streams). Offload to a thread/executor or a background task.
|
|
13
|
+
- **No synchronous network calls inside an async path** (rag_pipeline.py:526: a synchronous `requests.post` to Qdrant inside async `retrieve()` blocked the whole event loop; semantic_cache.py:319: `clear()` uses the blocking `KEYS` command and `clear()`/`get_stats()` run sync Redis calls on the async loop). Use the async client.
|
|
14
|
+
- **Don't drive a loop-bound async provider via `asyncio.run()` on executor threads** (document_grader.py:363: broke in the default config and silently disabled CRAG grading).
|
|
15
|
+
|
|
16
|
+
## Middleware order and auth are security-critical
|
|
17
|
+
|
|
18
|
+
- **Auth vs CORS ordering** (main.py:567: `APIKeyMiddleware` added after `CORSMiddleware` made it outermost, so CORS-preflight OPTIONS got 401 and blocked all browser clients). Order middleware deliberately; test a browser preflight.
|
|
19
|
+
- **A rate-limit tier must not be granted on an unvalidated header** (main.py:527: the higher per-key tier was granted to any request merely carrying an `X-API-Key` header — the key was never validated, so each forged key got its own quota bucket).
|
|
20
|
+
- **`slowapi` finds the request parameter by NAME** (main.py:978: `/query/stream` 500'd on every request when rate limiting was on because the parameter named `request` was the Pydantic body). Name the `Request` parameter `request`.
|
|
21
|
+
|
|
22
|
+
## Streaming must not bypass the validation the non-streaming path runs
|
|
23
|
+
|
|
24
|
+
- **Scrub/validate before the first token reaches the client** (main.py:1218: `AnswerScrubber` PII redaction was bypassed on `/query/stream` — raw LLM tokens streamed before validation ran). Streaming scrub-before-emit is the rule.
|
|
25
|
+
- **Don't reimplement a pipeline per endpoint** (main.py:976: `/query/stream` reimplemented ~300 lines of the `/query` pipeline and drifted — no audit log, no metrics, no OTel span — while config claimed audit covered both). One orchestrator behind both `/query` and `/query/stream`.
|
|
26
|
+
|
|
27
|
+
## Errors, persistence, admin state
|
|
28
|
+
|
|
29
|
+
- **Never leak raw exception text to clients** (main.py:1288: raw exception text leaked on both endpoints, bypassing the DEBUG gate the general exception handler implements). Return a generic error + correlation id.
|
|
30
|
+
- **Non-essential persistence stays off the critical path** (main.py:928: trace + conversation writes sat unguarded on the request path, so a Postgres/Redis outage turned an already-generated answer into a 500). Make observability writes best-effort.
|
|
31
|
+
- **Don't mutate shared pipeline state mid-flight** (main.py:2120: `POST /admin/models` mutated shared pipeline state with no lock, reached into private methods, and left `app.state.llm_provider` stale so metrics/audit reported the wrong model). Guard with a lock; go through the public seam.
|
|
32
|
+
- **An advertised parameter that's a silent no-op is a defect** (main.py:1539: `GET /api/search` advertised a `mode` param that was silently ignored).
|
|
33
|
+
|
|
34
|
+
## Config layering — one settings source
|
|
35
|
+
|
|
36
|
+
- **Read config through the settings layer, not `os.environ` at import time** (opik_tracer.py:66 and MODEL_PROFILES: `os.getenv` at import time while `config.py` defined the same settings via pydantic-settings, so values set only in `.env` silently left features disabled; MODEL_PROFILES was even keyed on a dev VM's absolute path).
|
|
37
|
+
- **One embedding-dimension source of truth** (rag_pipeline.py:288: resolving `EMBEDDING_DIM` via `os.environ.get(..., 768)` bypassed the backend's `settings.EMBEDDING_DIM=1024` and silently built a cache index with the wrong vector dimension). Two parallel pydantic Settings systems with conflicting defaults is the root cause — collapse them.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Python project structure — one-way dependency, packaging, CLI hygiene
|
|
3
|
+
paths: ["**/pyproject.toml", "**/Dockerfile*", "**/main.py", "**/src/**/*.py"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Project structure & packaging (Python)
|
|
7
|
+
|
|
8
|
+
> Ported from the RAG service docs/project-structure.md + code-review.md cross-cutting findings.
|
|
9
|
+
|
|
10
|
+
## One dependency direction, installed as a package
|
|
11
|
+
|
|
12
|
+
- **The app imports the library; the library never imports the app.** Two Python roots, one direction:
|
|
13
|
+
the reusable library in `src/<pkg>/`, the app in `app/` consuming it as an *installed* package.
|
|
14
|
+
- **Install the library as a package — no `sys.path`/`PYTHONPATH` hacks** (Dockerfile:74: the library was never installed, just path-hacked via `PYTHONPATH=/app/src` with dependencies hand-mirrored in `requirements.txt`, so the two manifests drifted; rag_pipeline.py:36: a `sys.path` hack counted two `..` hops too many and pointed at a directory that never existed — dead code documenting a false import mechanism). `pip install .` against a curated manifest, no path edits.
|
|
15
|
+
- **Heavy optional capabilities live behind extras** (pyproject.toml:20: the library declared streamlit, deepeval, docling, sentence-transformers, pytesseract, langchain as *mandatory* core deps though they serve only CLI/eval/dashboard paths — making the package uninstallable in a lean backend and forcing the `requirements.txt` workaround). Keep `pip install .` lean; gate the rest behind `[ingestion]`, `[eval]`, `[dashboard]` extras.
|
|
16
|
+
|
|
17
|
+
## Organize by seam, not concern-per-folder
|
|
18
|
+
|
|
19
|
+
- Feature/service code stays adjacent by seam so a change to a provider Protocol and its consumers is one directory away, not six. Every standard concern (app shell, agent core, memory, routing, security, eval, observability, deploy) has exactly one home.
|
|
20
|
+
- Deliberate deviations are allowed and documented (no agent-framework folder when the loop is ~200 explicit lines; no vendor security wrapper when the harness is in-repo and readable).
|
|
21
|
+
|
|
22
|
+
## CLI / composition-root hygiene
|
|
23
|
+
|
|
24
|
+
- **Consistent embedder defaults across commands** (main.py:551: `ingest` defaulted to `--embedder openai` (1536-dim) but `query` defaulted to `--embedder voyage` (1024-dim), so the documented happy path broke).
|
|
25
|
+
- **Don't ship dead course/tutorial artifacts as commands** (main.py:1318: 8 of 16 CLI commands were course-week artifacts whose default paths didn't exist; main.py:1 tutorial narration was ~1/3 of the file, burying the signal).
|
|
26
|
+
- **Fail the same way on the same bad input** (main.py:567: `query` silently swallowed an invalid `--embedder` (`except ValueError: pass`) and proceeded with a None dimension while `ingest` hard-failed on the same input).
|
|
27
|
+
- **Never rewrite a data file in place with a plain `open('w')`** (main.py:472: `dedup --apply` rewrote `documents.jsonl` in place, so a crash mid-write destroyed the processed corpus). Write to a temp file and atomically rename.
|
|
28
|
+
- **Business logic belongs in a service, not the composition root** (main.py:1133: `eval-pairwise`/`build-golden-dataset` embedded prompt templates, dual LLM clients, and regex JSON parsing inside the CLI entrypoint, contradicting the file's own stated design; Qdrant payload-index creation was inlined in the ingest handler instead of the infrastructure layer).
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: RAG pipeline rules — retrieval vs identity, eval integrity, agent contract
|
|
3
|
+
paths: ["**/*.py"]
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# RAG / agentic pipeline (Python / FastAPI / Qdrant / eval)
|
|
7
|
+
|
|
8
|
+
> Ported from the RAG service docs/findings-log.md + agent-contract.md. Stack: FastAPI, Qdrant (hybrid
|
|
9
|
+
> dense+BM25+RRF), Redis, Postgres, Ollama/OpenAI-compatible LLM, deepeval. Each rule is a numbered
|
|
10
|
+
> finding.
|
|
11
|
+
|
|
12
|
+
## The through-line (the one bug wearing different clothes)
|
|
13
|
+
|
|
14
|
+
> **Something inferred an answer by resemblance when an exact answer was available.**
|
|
15
|
+
> The fix is always the same move: **make the model choose the constraint, and make code satisfy it.**
|
|
16
|
+
|
|
17
|
+
- Semantic search *ranks* by content — it cannot distinguish one document from another (every contract
|
|
18
|
+
has a termination clause). **Document identity must be indexed and pre-filtered, not ranked** (Part 3).
|
|
19
|
+
- A corpus-wide question (`multi-12`/`multi-13`: "list every company", "do any two docs share one?") is
|
|
20
|
+
a **`GROUP BY`, not a similarity question** — a ranked retriever cannot answer it at any top-k, because
|
|
21
|
+
a question about the whole corpus has no best-matching chunk. Read the index and aggregate in code (Part 4).
|
|
22
|
+
- Don't match filename tokens loosely when the filename has a grammar (EDGAR `COMPANY_DATE-EX-N.N-TYPE`):
|
|
23
|
+
`"sla" in "Tesla"`, exhibit numbers `10.1` indexed as companies. **Parse the structure; match on token
|
|
24
|
+
boundaries; require a party to contain two consecutive letters** (Part 3).
|
|
25
|
+
|
|
26
|
+
## Eval integrity (the most dangerous defects)
|
|
27
|
+
|
|
28
|
+
- **A metric that cannot report its own failure rate is not a metric** (#1: "LLM Judge 5.0/5.0" while 19
|
|
29
|
+
of 20 cases crashed — the harness averaged only the completed ones). The summary **leads with its own
|
|
30
|
+
failure count** and prints `❌ RESULTS NOT USABLE` above a 20% failure rate.
|
|
31
|
+
- **The judge must not grade its own homework** (#5: the LLM judge defaulted to the same model as the
|
|
32
|
+
generator — a mirror, not a signal; its default was even an embedding model that can't generate). Judge
|
|
33
|
+
≠ generator, always.
|
|
34
|
+
- **Run the error-analysis cross-tab** `retrieval_hit × judge_score` (#6). It's what revealed *generation*,
|
|
35
|
+
not retrieval, was the dominant failure mode — after a whole session optimizing retrieval. The conclusion
|
|
36
|
+
"agentic retrieval is architecturally weaker" was exactly backwards and got the eval-results doc retracted.
|
|
37
|
+
- **Fixtures you invented cannot falsify assumptions you hold** (Part 3): entity tests used invented
|
|
38
|
+
fixtures (`Apple_10K.html`) and passed at 100% while the live index had `10.1` as a company. Tests are
|
|
39
|
+
now the **25 real filenames, verbatim**.
|
|
40
|
+
- **Use the instrumentation you already have** (#6 meta-lesson): OpenTelemetry, Prometheus, per-stage trace
|
|
41
|
+
timings existed the whole time while debugging happened by `curl` and `grep`.
|
|
42
|
+
|
|
43
|
+
## Measure the same system
|
|
44
|
+
|
|
45
|
+
- **Runs must target the same corpus/config** (#2: four Qdrant collections existed — `python-rag-service_hybrid`,
|
|
46
|
+
`python-rag-service_ids`, `python-rag-service_full` (zero entities), `python-rag-service_entities` — and the container was
|
|
47
|
+
hand-pointed at one while `config.py` named another). One canonical collection; strays deleted + tombstoned.
|
|
48
|
+
- **Give each arm the same budget** (#4: the agent's `search_corpus` inherited `top_k=3` while classic used
|
|
49
|
+
8 on the same question — comparing 3 chunks against 8 and blaming the architecture).
|
|
50
|
+
- **A tool that throws on every call makes the agent measure a broken tool, not an architecture** (#3:
|
|
51
|
+
`search_filtered` defaulted to a paid Voyage reranker with no key and threw every call; the agent burned
|
|
52
|
+
its whole step budget retrying). Empty results return the entity vocabulary so the agent can self-correct.
|
|
53
|
+
|
|
54
|
+
## The agent contract (read-only tools, budgets, allowlist)
|
|
55
|
+
|
|
56
|
+
- **The agent's capability surface is the tool registry — allowlist, not denylist.** A tool exists only if
|
|
57
|
+
registered; unknown names are never executed (`test_unknown_tool_is_rejected` — "the loop must never
|
|
58
|
+
invent capabilities").
|
|
59
|
+
- **Budgets are explicit and env-overridable:** `AGENT_MAX_STEPS`, `AGENT_TOKEN_BUDGET`,
|
|
60
|
+
`AGENT_TOOL_TIMEOUT_S`. Exhaustion finishes with what was gathered and flags the trace. `finish` is the
|
|
61
|
+
explicit stop signal; there is no unbounded path through the loop.
|
|
62
|
+
- **Build side-effect machinery with the first mutating tool, not speculatively.** Adding a read-only tool
|
|
63
|
+
= registry entry + tests + a doc row. A tool with side effects requires argument-schema validation, a
|
|
64
|
+
pre-execution control mode (audit/approval/block), and severity-gated alerting *before* it ships — none
|
|
65
|
+
of which exists today, deliberately.
|
|
66
|
+
|
|
67
|
+
## Config / model swapping
|
|
68
|
+
|
|
69
|
+
- **The LLM is hot-swappable (any OpenAI-compatible endpoint, no re-index); the embedder is NOT** — its
|
|
70
|
+
output dimension is baked into the Qdrant collection at ingest. Changing `EMBEDDER_TYPE` means setting
|
|
71
|
+
`EMBEDDING_DIM` to match AND re-indexing.
|
|
72
|
+
- **Library dependency direction is one-way:** the app imports the library, never the reverse; the library
|
|
73
|
+
is pip-installed (no `PYTHONPATH`/`sys.path` edits). Heavy optional capabilities live behind extras.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# React frontend profile
|
|
2
|
+
|
|
3
|
+
Portable React SPA canon, extracted verbatim from the .NET platform's `frontend/CLAUDE.md` (the .NET platform
|
|
4
|
+
storefront). **Backend-agnostic** — stack it onto any backend profile whose repo has a React frontend:
|
|
5
|
+
|
|
6
|
+
okl scaffold --profile dotnet --profile react
|
|
7
|
+
okl scaffold --profile python-rag --profile react
|
|
8
|
+
|
|
9
|
+
Installed to `.claude/rules/`:
|
|
10
|
+
|
|
11
|
+
- `frontend.md` — path-scoped to `frontend/**` and `**/frontend/**`: TanStack Query server state
|
|
12
|
+
(useEffect-fetching banned), effects discipline ("You Might Not Need an Effect"), React-Compiler
|
|
13
|
+
render/bundle perf (don't reflexively memoize; ≤200 KB gz budget), PKCE SPA auth + in-memory tokens
|
|
14
|
+
(BFF trade-off documented), MSW + Playwright testing.
|
|
15
|
+
|
|
16
|
+
Reference stack: Vite + React 19 + TS strict, CSR SPA, TanStack Query/Router, Zustand, Tailwind v4 +
|
|
17
|
+
shadcn/ui, oidc-client-ts → Keycloak. Deep reference: the vendored `vercel-react-best-practices`
|
|
18
|
+
skill (70 rules), where the canon wins on any disagreement.
|