loki-mode 9.8.1 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
package/mcp/__init__.py
CHANGED
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "loki-mode",
|
|
3
3
|
"mcpName": "io.github.asklokesh/loki-mode",
|
|
4
|
-
"version": "9.
|
|
4
|
+
"version": "9.12.0",
|
|
5
5
|
"description": "Loki Mode by Autonomi. Autonomous spec-to-product system: takes a PRD, GitHub issue, OpenAPI/JSON/YAML, or one-line brief to a deployed app via the RARV-C closure loop with 8 quality gates. Provider-agnostic (Claude Code, OpenAI Codex, Cline, Aider).",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"agent",
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
|
|
3
3
|
"name": "loki-mode",
|
|
4
4
|
"displayName": "Loki Mode",
|
|
5
|
-
"version": "9.
|
|
5
|
+
"version": "9.12.0",
|
|
6
6
|
"description": "Autonomous spec-to-product build system with a built-in trust layer (RARV-C closure loop, 8 quality gates, completion council). Ships Loki's spec-hardening, drift-detection, and deterministic PR verification commands plus the Loki MCP server.",
|
|
7
7
|
"author": {
|
|
8
8
|
"name": "Autonomi",
|
|
@@ -1,6 +1,23 @@
|
|
|
1
1
|
# Confidence-Based Routing Reference
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
> **STATUS: DESIGN NOTE, NOT IMPLEMENTED.** Nothing in this document is wired
|
|
4
|
+
> into the runtime. Do not configure against it.
|
|
5
|
+
>
|
|
6
|
+
> Verified against source at v9.8.1:
|
|
7
|
+
> - `LOKI_CONFIDENCE_ROUTING` is read once (`autonomy/run.sh:1319`) and only
|
|
8
|
+
> printed to a log line (`autonomy/run.sh:21182`). No code branches on it.
|
|
9
|
+
> - `LOKI_CONFIDENCE_AUTO_APPROVE`, `LOKI_CONFIDENCE_DIRECT`,
|
|
10
|
+
> `LOKI_CONFIDENCE_SUPERVISOR` and `LOKI_CONFIDENCE_CALIBRATION_DAYS` are
|
|
11
|
+
> read nowhere in the repo.
|
|
12
|
+
> - `calculate_task_confidence()`, `save_confidence_calibration()`, the
|
|
13
|
+
> `force_routing` / `require_human_review` task metadata, and the Brier-score
|
|
14
|
+
> calibration dashboard do not exist in source.
|
|
15
|
+
>
|
|
16
|
+
> The Python and bash blocks below are illustrative pseudocode for a proposed
|
|
17
|
+
> design. They are kept as a design record. For the routing that actually runs,
|
|
18
|
+
> see `skills/model-selection.md` and `get_rarv_tier()` in `autonomy/run.sh`.
|
|
19
|
+
|
|
20
|
+
Design pattern drawn from HN discussions and the Claude Agent SDK guide.
|
|
4
21
|
|
|
5
22
|
---
|
|
6
23
|
|
|
@@ -94,16 +94,21 @@ as an alternative to relying on the exit code.
|
|
|
94
94
|
|
|
95
95
|
## Wiring as a gate
|
|
96
96
|
|
|
97
|
-
The detector
|
|
98
|
-
`
|
|
99
|
-
(`autonomy/run.sh:
|
|
100
|
-
(`autonomy/run.sh:14676`). The full wrapper is documented in the header of
|
|
101
|
-
`tests/detect-invariant-violations.sh`. It:
|
|
97
|
+
The detector IS wired into `autonomy/run.sh`. The wrapper is
|
|
98
|
+
`_invariant_gate_and_surface()` (`autonomy/run.sh:12715`), invoked from the
|
|
99
|
+
completion path (`autonomy/run.sh:23889`). It:
|
|
102
100
|
|
|
103
101
|
- honors `LOKI_SCAN_DIR=TARGET_DIR` (the detector scans the target, not loki-mode)
|
|
104
102
|
- treats detector-not-found and timeout (exit 124) as inconclusive (does not block)
|
|
105
103
|
- persists findings to `${TARGET_DIR}/.loki/quality/invariant-findings.txt`
|
|
106
|
-
- opts out with `LOKI_GATE_INVARIANT=false`
|
|
107
104
|
|
|
108
|
-
|
|
109
|
-
|
|
105
|
+
Two flags control it, both PLURAL:
|
|
106
|
+
|
|
107
|
+
| Flag | Default | Effect |
|
|
108
|
+
|------|---------|--------|
|
|
109
|
+
| `LOKI_GATE_INVARIANTS` | `true` | Run and surface findings as advisory (`autonomy/run.sh:23333`) |
|
|
110
|
+
| `LOKI_GATE_INVARIANTS_BLOCK` | `false` | Opt in to letting CRITICAL/HIGH block completion (`autonomy/run.sh:23889`) |
|
|
111
|
+
|
|
112
|
+
There is no `LOKI_GATE_INVARIANT` (singular). That name appears only in the
|
|
113
|
+
superseded proposal comment in the header of
|
|
114
|
+
`tests/detect-invariant-violations.sh` and is read by nothing.
|
|
@@ -8,7 +8,6 @@ normal RARV-C phase execution.
|
|
|
8
8
|
This reference describes which phase does what.
|
|
9
9
|
|
|
10
10
|
## BOOTSTRAP (before iteration 1)
|
|
11
|
-
- `analyze_git_intelligence()` runs (from v6.75.0).
|
|
12
11
|
- Magic-specific bootstrap:
|
|
13
12
|
- `magic.core.design_tokens.DesignTokens.extract_from_codebase(save=True)`
|
|
14
13
|
scans existing React and Web Components to learn the project's design
|
|
@@ -1,19 +1,31 @@
|
|
|
1
1
|
# Multi-Provider Architecture Reference
|
|
2
2
|
|
|
3
|
-
> **
|
|
3
|
+
> **Status:** Production | **Provider list and model config verified against source at v9.8.1**
|
|
4
|
+
>
|
|
5
|
+
> Values below are illustrative. `providers/*.sh` is the authority; prefer
|
|
6
|
+
> reading it over trusting a literal copied into this document.
|
|
4
7
|
|
|
5
|
-
Loki Mode supports
|
|
8
|
+
Loki Mode supports five AI CLI providers with a unified abstraction layer. This document provides detailed technical reference for the multi-provider system.
|
|
6
9
|
|
|
7
10
|
---
|
|
8
11
|
|
|
9
12
|
## Provider Overview
|
|
10
13
|
|
|
14
|
+
Source of truth: `SUPPORTED_PROVIDERS` in `providers/loader.sh:8`.
|
|
15
|
+
|
|
11
16
|
| Provider | CLI | Status | Features |
|
|
12
17
|
|----------|-----|--------|----------|
|
|
13
18
|
| **Claude Code** | `claude` | Full | Subagents, Parallel, Task Tool, MCP |
|
|
14
19
|
| **OpenAI Codex** | `codex` | Degraded | Sequential, Effort Parameter, MCP (basic) |
|
|
15
20
|
| **Cline CLI** | `cline` | Near-Full (Tier 2) | Subagents, MCP, 12+ providers |
|
|
16
21
|
| **Aider** | `aider` | Degraded | Sequential, 18+ providers |
|
|
22
|
+
| **opencode** | `opencode` | Sequential | MCP, 75+ providers, model-agnostic |
|
|
23
|
+
|
|
24
|
+
When `LOKI_PROVIDER` is unset, providers auto-detect in priority order:
|
|
25
|
+
`claude > cline > codex > aider > opencode`. An explicit choice always wins and
|
|
26
|
+
is never silently substituted.
|
|
27
|
+
|
|
28
|
+
Gemini was removed as a provider.
|
|
17
29
|
|
|
18
30
|
---
|
|
19
31
|
|
|
@@ -27,6 +39,7 @@ providers/
|
|
|
27
39
|
codex.sh # Degraded mode, effort parameter
|
|
28
40
|
cline.sh # Near-full mode, 12+ providers (Tier 2)
|
|
29
41
|
aider.sh # Degraded mode, 18+ providers
|
|
42
|
+
opencode.sh # Sequential, 75+ providers, model-agnostic
|
|
30
43
|
loader.sh # Provider loader utility
|
|
31
44
|
```
|
|
32
45
|
|
|
@@ -58,12 +71,21 @@ PROVIDER_MAX_PARALLEL=10 # Maximum concurrent agents
|
|
|
58
71
|
```
|
|
59
72
|
|
|
60
73
|
#### Model Configuration
|
|
74
|
+
|
|
75
|
+
Models are NOT hardcoded model IDs. Each tier resolves through env indirection
|
|
76
|
+
to a tier alias (`providers/claude.sh:70-72`):
|
|
77
|
+
|
|
61
78
|
```bash
|
|
62
|
-
PROVIDER_MODEL_PLANNING="
|
|
63
|
-
PROVIDER_MODEL_DEVELOPMENT="
|
|
64
|
-
PROVIDER_MODEL_FAST="
|
|
79
|
+
PROVIDER_MODEL_PLANNING="${LOKI_CLAUDE_MODEL_PLANNING:-${LOKI_MODEL_PLANNING:-$CLAUDE_DEFAULT_PLANNING}}"
|
|
80
|
+
PROVIDER_MODEL_DEVELOPMENT="${LOKI_CLAUDE_MODEL_DEVELOPMENT:-${LOKI_MODEL_DEVELOPMENT:-$CLAUDE_DEFAULT_DEVELOPMENT}}"
|
|
81
|
+
PROVIDER_MODEL_FAST="${LOKI_CLAUDE_MODEL_FAST:-${LOKI_MODEL_FAST:-$CLAUDE_DEFAULT_FAST}}"
|
|
65
82
|
```
|
|
66
83
|
|
|
84
|
+
Defaults are all `sonnet` (`providers/claude.sh:60-62`). Setting
|
|
85
|
+
`LOKI_ALLOW_HAIKU` flips the fast tier to `haiku` (`providers/claude.sh:66`).
|
|
86
|
+
Do not cite pinned model IDs here; they drift. Read `providers/claude.sh` and
|
|
87
|
+
`providers/model_catalog.json` for current values.
|
|
88
|
+
|
|
67
89
|
#### Rate Limiting
|
|
68
90
|
```bash
|
|
69
91
|
PROVIDER_RATE_LIMIT_RPM=50 # Requests per minute
|
package/skills/healing.md
CHANGED
|
@@ -478,10 +478,12 @@ no-op, never a false-green. (Deterministic shell hooks, not LLM calls.)
|
|
|
478
478
|
|----------|---------|---------|
|
|
479
479
|
| `LOKI_HEAL_MODE` | `false` | Enable healing mode |
|
|
480
480
|
| `LOKI_HEAL_PHASE` | `archaeology` | Current healing phase |
|
|
481
|
-
| `LOKI_HEAL_PRESERVE_FRICTION` | `true` | Warn before removing friction points |
|
|
482
|
-
| `LOKI_HEAL_BASELINE_DIR` | `.loki/healing/behavioral-baseline/` | Pre-healing snapshots |
|
|
483
481
|
| `LOKI_HEAL_STRICT` | `false` | Block ALL behavioral changes without approval |
|
|
484
482
|
|
|
483
|
+
Friction preservation is unconditional and the baseline path is hardcoded to
|
|
484
|
+
`.loki/healing/behavioral-baseline/` (`autonomy/hooks/migration-hooks.sh:776`).
|
|
485
|
+
Neither is configurable.
|
|
486
|
+
|
|
485
487
|
---
|
|
486
488
|
|
|
487
489
|
## Known Limitations
|
|
@@ -0,0 +1,488 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Find documentation claims the repo itself contradicts.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. This repo's standing rule is "verify, then document", and it
|
|
5
|
+
was written down because documenting first kept shipping. A doc claim rots
|
|
6
|
+
silently: nothing imports it, no test runs it, and the only reader who can tell
|
|
7
|
+
it went stale is a user who has already followed it and lost an hour. Meanwhile
|
|
8
|
+
every other honesty surface here -- the receipt, the cost figure, the tool
|
|
9
|
+
index -- is machine-checked. Prose was the one place a false claim could sit
|
|
10
|
+
indefinitely and still look like documentation.
|
|
11
|
+
|
|
12
|
+
Three claim kinds are checkable against the repo without running anything:
|
|
13
|
+
|
|
14
|
+
A `tools/<name>` PATH either exists or it does not.
|
|
15
|
+
A documented `LOKI_*` VAR either occurs in source or it does not.
|
|
16
|
+
A CURRENT-VERSION string either matches VERSION or it does not.
|
|
17
|
+
|
|
18
|
+
THE THREE-STATE RULE, which is the whole design. A claim this tool cannot check
|
|
19
|
+
is reported UNCHECKABLE -- never as passing, and never as a finding. Collapsing
|
|
20
|
+
that third state in either direction is a lie in one of the two available
|
|
21
|
+
directions:
|
|
22
|
+
|
|
23
|
+
Call it PASSING and the audit becomes a rubber stamp: `LOKI_JIRA_` "verified",
|
|
24
|
+
meaning only that nobody looked.
|
|
25
|
+
Call it a FINDING and the output fills with noise until a reader stops
|
|
26
|
+
reading, which loses the real findings too.
|
|
27
|
+
|
|
28
|
+
So UNCHECKABLE is a first-class, counted, printed outcome.
|
|
29
|
+
|
|
30
|
+
WHERE EACH CHECK MUST FAIL. The env check is the one with an arms race in it,
|
|
31
|
+
and this deliberately does not enter it. A read-pattern allowlist ($VAR,
|
|
32
|
+
${VAR}, env["VAR"], getenv("VAR"), os.environ, export VAR=) is used ONLY to
|
|
33
|
+
promote a var to PASS. It is NEVER used to demote one to a finding, because a
|
|
34
|
+
read form the allowlist has not seen yet is a bug in the allowlist, not a
|
|
35
|
+
defect in the docs. A finding requires the strictly stronger evidence of ZERO
|
|
36
|
+
occurrences of the token anywhere in source -- no read syntax, no mention, no
|
|
37
|
+
test. Everything between those two poles is UNCHECKABLE.
|
|
38
|
+
|
|
39
|
+
That asymmetry also handles documented PREFIXES for free. `LOKI_SDK_COUNCIL` is
|
|
40
|
+
not a variable; the real ones are `LOKI_SDK_COUNCIL_V2` and
|
|
41
|
+
`LOKI_SDK_COUNCIL_VOTE`. Substring containment sees those, so the prefix is not
|
|
42
|
+
a finding. An extracted-token comparison would have called all eight prefix
|
|
43
|
+
families in this repo false. That was measured, not imagined.
|
|
44
|
+
|
|
45
|
+
WHAT IS NOT SCANNED, and why. CHANGELOG.md and artifacts/ are HISTORICAL
|
|
46
|
+
RECORD, not claims about today. "v8.40.0 fixed the dist check" is true forever
|
|
47
|
+
and its version string disagrees with VERSION by construction. Auditing them
|
|
48
|
+
would produce hundreds of findings that are all correct statements. The
|
|
49
|
+
exclusion list is printed with the output, because an exclusion nobody can see
|
|
50
|
+
is one nobody can challenge.
|
|
51
|
+
|
|
52
|
+
VACUITY. The scanned-file count is printed on every run, and an empty docs set
|
|
53
|
+
exits 3 rather than 0. Scanning nothing is not a clean bill of health -- it is
|
|
54
|
+
an absent measurement, and this repo has paid four releases to learn that a
|
|
55
|
+
substring search over an empty listing reports nothing missing.
|
|
56
|
+
|
|
57
|
+
Exit codes follow the tools/ convention:
|
|
58
|
+
0 checked, every claim held
|
|
59
|
+
1 checked, at least one claim is FALSE
|
|
60
|
+
2 the scan itself could not run
|
|
61
|
+
3 nothing to check (no docs, or no checkable claims in them)
|
|
62
|
+
64 usage error
|
|
63
|
+
66 the given docs root does not exist
|
|
64
|
+
|
|
65
|
+
Reads the filesystem only. Starts nothing, spends nothing, contacts nothing.
|
|
66
|
+
|
|
67
|
+
Usage:
|
|
68
|
+
tools/audit-docs.py [docs-root] [--json]
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
import argparse
|
|
72
|
+
import json
|
|
73
|
+
import os
|
|
74
|
+
import re
|
|
75
|
+
import sys
|
|
76
|
+
|
|
77
|
+
sys.dont_write_bytecode = True
|
|
78
|
+
|
|
79
|
+
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
80
|
+
_ROOT = os.path.dirname(_HERE)
|
|
81
|
+
|
|
82
|
+
# Where evidence that a LOKI_* var is read may live. The list is deliberately
|
|
83
|
+
# WIDE, because every directory missing from it pushes a var toward FALSE --
|
|
84
|
+
# the unsafe direction. An omission here does not weaken a check, it invents a
|
|
85
|
+
# finding.
|
|
86
|
+
#
|
|
87
|
+
# That is not hypothetical. An earlier draft listed eleven directories and
|
|
88
|
+
# omitted `src/` and `deploy/`. It reported 116 distinct vars as undocumented-
|
|
89
|
+
# and-unread; searching the WHOLE repo with no extension filter found 22 of
|
|
90
|
+
# them present, including `LOKI_SERVICE_NAME` at src/observability/otel.js:425
|
|
91
|
+
# under a plain `process.env` read. 19% of the findings were the scan's own
|
|
92
|
+
# blind spot. The evidence line says "0 occurrences across ..." and it must
|
|
93
|
+
# mean it, so the list is enumerated and printed with every run.
|
|
94
|
+
#
|
|
95
|
+
# tests/ is included ON PURPOSE: a test doing env["LOKI_MAX_TIER"] = "low" is
|
|
96
|
+
# evidence some runtime reads it. Counting it can only turn a finding into
|
|
97
|
+
# UNCHECKABLE, never the reverse.
|
|
98
|
+
SOURCE_DIRS = ("autonomy", "loki-ts/src", "loki-ts/test", "loki-ts/tests",
|
|
99
|
+
"src", "tools", "dashboard", "dashboard-ui/src", "mcp",
|
|
100
|
+
"memory", "events", "bin", "providers", "scripts", "deploy",
|
|
101
|
+
"tests", "benchmarks", "web-app/src", ".github")
|
|
102
|
+
|
|
103
|
+
SOURCE_EXTS = (".sh", ".py", ".ts", ".js", ".mjs", ".cjs", ".tsx", ".jsx",
|
|
104
|
+
".json", ".yml", ".yaml", ".toml", ".env", ".example",
|
|
105
|
+
".conf", ".cfg", ".ini", ".txt", "Dockerfile")
|
|
106
|
+
|
|
107
|
+
# EXTENSIONLESS EXECUTABLES ARE SOURCE. The main CLI is `autonomy/loki` --
|
|
108
|
+
# 32,000 lines, no extension. An extension allowlist skipped it entirely, and
|
|
109
|
+
# six variables it genuinely reads (LOKI_NOTIFY_CHANNELS at autonomy/loki:12887
|
|
110
|
+
# among them) were reported as false doc claims. Anything with a shebang counts.
|
|
111
|
+
_SHEBANG = "#!"
|
|
112
|
+
|
|
113
|
+
# .loki/ is THIS repo's own run state, and it holds a bash audit log that
|
|
114
|
+
# records every command anyone ran -- including greps for the very variable
|
|
115
|
+
# names being audited. Counting it would let the tool's own investigation
|
|
116
|
+
# supply the evidence that clears a claim. Self-contamination, not a consumer.
|
|
117
|
+
SKIP_DIRS = {"node_modules", ".git", "dist", "build", "__pycache__",
|
|
118
|
+
".venv", "venv", "coverage", ".loki"}
|
|
119
|
+
|
|
120
|
+
# Repo-root files that are not under any SOURCE_DIRS entry but do read env.
|
|
121
|
+
SOURCE_ROOT_FILES = ("docker-compose.yml", "Dockerfile", "Dockerfile.sandbox",
|
|
122
|
+
"package.json", "server.json")
|
|
123
|
+
|
|
124
|
+
# Markdown trees excluded from the scan, each with the reason a reader can
|
|
125
|
+
# argue with. Printed on every run.
|
|
126
|
+
DOC_EXCLUSIONS = (
|
|
127
|
+
("CHANGELOG.md", "release history; its version strings are true forever"),
|
|
128
|
+
("artifacts/", "frozen point-in-time reports, not claims about today"),
|
|
129
|
+
("node_modules/", "third-party"),
|
|
130
|
+
("wiki/_Footer.md", "generated navigation chrome"),
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
# Files that carry a CURRENT-version claim, per the Release Workflow section of
|
|
134
|
+
# CLAUDE.md. Scoping the version check to this list is what separates a claim
|
|
135
|
+
# ("Current Version: 8.0.0") from history ("fixed in v8.40.0") without needing
|
|
136
|
+
# a classifier for English tense.
|
|
137
|
+
VERSION_CLAIM_FILES = (
|
|
138
|
+
"SKILL.md", "CLAUDE.md", "README.md", "docs/INSTALLATION.md",
|
|
139
|
+
"wiki/Home.md", "wiki/_Sidebar.md", "wiki/API-Reference.md",
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# A version string is a CLAIM ABOUT NOW only in these exact shapes. Everything
|
|
143
|
+
# else in the same file is history ("DEPRECATED in v7.2.0", "zero config
|
|
144
|
+
# (v7.45.0)") and is skipped outright rather than bucketed UNCHECKABLE -- if
|
|
145
|
+
# history landed in that bucket it would fill with hundreds of true statements
|
|
146
|
+
# and the bucket would stop meaning anything.
|
|
147
|
+
#
|
|
148
|
+
# Each pattern was derived from a real line, not guessed:
|
|
149
|
+
# SKILL.md:6 "# Loki Mode v9.8.1"
|
|
150
|
+
# SKILL.md:472 "**v9.8.1 | [Autonomi]..."
|
|
151
|
+
# CLAUDE.md:338 "- Current: v9.8.1 (see [CHANGELOG.md]...)"
|
|
152
|
+
# wiki/Home.md "Current Version: **8.0.0** ([CHANGELOG]...)"
|
|
153
|
+
#
|
|
154
|
+
# A loose cue was tried first and produced three false findings: an IP address
|
|
155
|
+
# (127.0.0.1 in "http://127.0.0.1:57374") and two historical mentions on lines
|
|
156
|
+
# beginning with "#" inside fenced shell blocks. Anchoring on the claim phrase
|
|
157
|
+
# rather than on line shape removed all three.
|
|
158
|
+
_VERSION_CLAIMS = (
|
|
159
|
+
re.compile(r"current\s+version[^0-9\n]{0,20}v?(\d+\.\d+\.\d+)", re.I),
|
|
160
|
+
re.compile(r"current:\s*v?(\d+\.\d+\.\d+)", re.I),
|
|
161
|
+
re.compile(r"^#\s+Loki Mode\s+v?(\d+\.\d+\.\d+)", re.I),
|
|
162
|
+
re.compile(r"^\*\*v?(\d+\.\d+\.\d+)\s*\|"),
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
_TOOL_PATH = re.compile(r"\btools/([A-Za-z0-9_.<>*\[\]{}-]+\.(?:py|sh))")
|
|
166
|
+
_ENV_VAR = re.compile(r"\b(LOKI_[A-Z0-9_]*)")
|
|
167
|
+
|
|
168
|
+
# A placeholder, not a real path: tools/<name>.py, tools/*.py, tools/{x}.py.
|
|
169
|
+
_PLACEHOLDER = re.compile(r"[<>*{}\[\]]")
|
|
170
|
+
|
|
171
|
+
# Promotes a var to PASS. Never demotes one to a finding -- see the docstring.
|
|
172
|
+
_READ = re.compile(
|
|
173
|
+
r"\$\{?(LOKI_[A-Z0-9_]+)" # $VAR ${VAR}
|
|
174
|
+
r"|env(?:iron)?(?:\.get)?[\[\(]\s*[\"'](LOKI_[A-Z0-9_]+)" # env["VAR"]
|
|
175
|
+
r"|getenv\(\s*[\"'](LOKI_[A-Z0-9_]+)" # getenv("VAR")
|
|
176
|
+
r"|[\"'](LOKI_[A-Z0-9_]+)[\"']\s*," # helper(e,"VAR",d)
|
|
177
|
+
r"|^[ \t]*(?:export[ \t]+|local[ \t]+)?(LOKI_[A-Z0-9_]+)=", # VAR=
|
|
178
|
+
re.M)
|
|
179
|
+
|
|
180
|
+
PASS, FALSE, UNCHECKABLE = "pass", "false", "uncheckable"
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
class ScanError(Exception):
|
|
184
|
+
"""The scan could not run. Exit 2, never a verdict."""
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
class _Parser(argparse.ArgumentParser):
|
|
188
|
+
"""argparse exits 2 on a usage error, and 2 already means something else.
|
|
189
|
+
|
|
190
|
+
In this convention 2 is "could NOT check" -- a real answer about the docs.
|
|
191
|
+
A typo in a flag is not that; it is 64. Left alone, `--jsno` would report
|
|
192
|
+
as a failed scan and a CI job could not tell the two apart.
|
|
193
|
+
"""
|
|
194
|
+
|
|
195
|
+
def error(self, message):
|
|
196
|
+
self.print_usage(sys.stderr)
|
|
197
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
198
|
+
raise SystemExit(64)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _read(path):
|
|
202
|
+
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
|
203
|
+
return fh.read()
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _excluded(rel):
|
|
207
|
+
for prefix, _reason in DOC_EXCLUSIONS:
|
|
208
|
+
if rel == prefix or rel.startswith(prefix):
|
|
209
|
+
return True
|
|
210
|
+
return False
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def find_docs(root):
|
|
214
|
+
"""Every markdown file under root, minus the documented exclusions."""
|
|
215
|
+
out = []
|
|
216
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
217
|
+
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
|
218
|
+
for name in sorted(filenames):
|
|
219
|
+
if not name.endswith(".md"):
|
|
220
|
+
continue
|
|
221
|
+
full = os.path.join(dirpath, name)
|
|
222
|
+
rel = os.path.relpath(full, root).replace(os.sep, "/")
|
|
223
|
+
if not _excluded(rel):
|
|
224
|
+
out.append((rel, full))
|
|
225
|
+
return sorted(out)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def source_corpus(root):
|
|
229
|
+
"""Concatenated source text, used only for occurrence containment.
|
|
230
|
+
|
|
231
|
+
One walk, not one grep per variable. 279 documented vars against 300 source
|
|
232
|
+
files is 83,700 greps the naive shape would have run.
|
|
233
|
+
"""
|
|
234
|
+
chunks = []
|
|
235
|
+
for rel in SOURCE_DIRS:
|
|
236
|
+
base = os.path.join(root, rel)
|
|
237
|
+
if not os.path.isdir(base):
|
|
238
|
+
continue
|
|
239
|
+
for dirpath, dirnames, filenames in os.walk(base):
|
|
240
|
+
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
|
241
|
+
for name in filenames:
|
|
242
|
+
full = os.path.join(dirpath, name)
|
|
243
|
+
try:
|
|
244
|
+
if name.endswith(SOURCE_EXTS):
|
|
245
|
+
chunks.append(_read(full))
|
|
246
|
+
elif "." not in name:
|
|
247
|
+
# Extensionless: read it only if it is a script. This
|
|
248
|
+
# is how autonomy/loki (the main CLI) gets counted.
|
|
249
|
+
body = _read(full)
|
|
250
|
+
if body.startswith(_SHEBANG):
|
|
251
|
+
chunks.append(body)
|
|
252
|
+
except OSError:
|
|
253
|
+
continue # unreadable file is not evidence of absence
|
|
254
|
+
for name in SOURCE_ROOT_FILES:
|
|
255
|
+
full = os.path.join(root, name)
|
|
256
|
+
if os.path.isfile(full):
|
|
257
|
+
try:
|
|
258
|
+
chunks.append(_read(full))
|
|
259
|
+
except OSError:
|
|
260
|
+
continue
|
|
261
|
+
text = "\n".join(chunks)
|
|
262
|
+
reads = set()
|
|
263
|
+
for match in _READ.finditer(text):
|
|
264
|
+
reads.add(next(g for g in match.groups() if g))
|
|
265
|
+
return text, reads
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def repo_version(root):
|
|
269
|
+
"""The VERSION file, or None when it cannot be read.
|
|
270
|
+
|
|
271
|
+
None means the version check reports UNCHECKABLE for every version claim.
|
|
272
|
+
A missing baseline is not evidence that the docs are right.
|
|
273
|
+
"""
|
|
274
|
+
try:
|
|
275
|
+
value = _read(os.path.join(root, "VERSION")).strip()
|
|
276
|
+
except OSError:
|
|
277
|
+
return None
|
|
278
|
+
return value or None
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _finding(kind, doc, line, claim, status, evidence):
|
|
282
|
+
return {"kind": kind, "file": doc, "line": line, "claim": claim,
|
|
283
|
+
"status": status, "evidence": evidence}
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def check_tool_paths(root, doc, text):
|
|
287
|
+
for lineno, line in enumerate(text.splitlines(), 1):
|
|
288
|
+
for match in _TOOL_PATH.finditer(line):
|
|
289
|
+
name = match.group(1)
|
|
290
|
+
claim = "tools/" + name
|
|
291
|
+
if _PLACEHOLDER.search(name):
|
|
292
|
+
yield _finding(
|
|
293
|
+
"tool_path", doc, lineno, claim, UNCHECKABLE,
|
|
294
|
+
"placeholder or glob, not a literal path")
|
|
295
|
+
continue
|
|
296
|
+
full = os.path.join(root, "tools", name)
|
|
297
|
+
if os.path.exists(full):
|
|
298
|
+
yield _finding("tool_path", doc, lineno, claim, PASS,
|
|
299
|
+
"file exists: " + claim)
|
|
300
|
+
else:
|
|
301
|
+
yield _finding("tool_path", doc, lineno, claim, FALSE,
|
|
302
|
+
"no such file: " + claim)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def check_env_vars(doc, text, corpus, reads):
|
|
306
|
+
"""A var is FALSE only on zero occurrences anywhere in source.
|
|
307
|
+
|
|
308
|
+
Three outcomes, and the middle one is the point:
|
|
309
|
+
read syntax found -> PASS
|
|
310
|
+
token occurs, no read syntax -> UNCHECKABLE (a mention is not a consumer,
|
|
311
|
+
but neither is it proof of absence)
|
|
312
|
+
token occurs nowhere at all -> FALSE
|
|
313
|
+
"""
|
|
314
|
+
for lineno, line in enumerate(text.splitlines(), 1):
|
|
315
|
+
# One claim per (line, var). A var named twice on one line -- common in
|
|
316
|
+
# `export LOKI_X=${LOKI_X:-0}` -- is a single claim, and counting it
|
|
317
|
+
# twice inflates the headline number a reader will quote.
|
|
318
|
+
seen = set()
|
|
319
|
+
for match in _ENV_VAR.finditer(line):
|
|
320
|
+
name = match.group(1)
|
|
321
|
+
if name in seen:
|
|
322
|
+
continue
|
|
323
|
+
seen.add(name)
|
|
324
|
+
# LOKI_ or LOKI_JIRA_ is an extraction artifact of a prefix family,
|
|
325
|
+
# not a variable anyone can set. Never a claim.
|
|
326
|
+
if name.endswith("_") or name == "LOKI":
|
|
327
|
+
yield _finding("env_var", doc, lineno, name, UNCHECKABLE,
|
|
328
|
+
"prefix family, not a concrete variable name")
|
|
329
|
+
continue
|
|
330
|
+
if name in reads:
|
|
331
|
+
yield _finding("env_var", doc, lineno, name, PASS,
|
|
332
|
+
"read by source (env read syntax found)")
|
|
333
|
+
elif name in corpus:
|
|
334
|
+
yield _finding(
|
|
335
|
+
"env_var", doc, lineno, name, UNCHECKABLE,
|
|
336
|
+
"occurs in source but under no recognised read syntax; "
|
|
337
|
+
"the allowlist may be incomplete")
|
|
338
|
+
else:
|
|
339
|
+
yield _finding(
|
|
340
|
+
"env_var", doc, lineno, name, FALSE,
|
|
341
|
+
"0 occurrences across " + ", ".join(SOURCE_DIRS))
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def check_versions(doc, text, version):
|
|
345
|
+
if doc not in VERSION_CLAIM_FILES:
|
|
346
|
+
return
|
|
347
|
+
for lineno, line in enumerate(text.splitlines(), 1):
|
|
348
|
+
for pattern in _VERSION_CLAIMS:
|
|
349
|
+
match = pattern.search(line)
|
|
350
|
+
if not match:
|
|
351
|
+
continue # history or unrelated number, not a claim about now
|
|
352
|
+
found = match.group(1)
|
|
353
|
+
if version is None:
|
|
354
|
+
yield _finding("version", doc, lineno, found, UNCHECKABLE,
|
|
355
|
+
"VERSION file unreadable; no baseline to "
|
|
356
|
+
"compare against")
|
|
357
|
+
elif found == version:
|
|
358
|
+
yield _finding("version", doc, lineno, found, PASS,
|
|
359
|
+
"matches VERSION (" + version + ")")
|
|
360
|
+
else:
|
|
361
|
+
yield _finding("version", doc, lineno, found, FALSE,
|
|
362
|
+
"VERSION says " + version + ", doc says "
|
|
363
|
+
+ found)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def audit(root, docs_root):
|
|
367
|
+
if not os.path.isdir(docs_root):
|
|
368
|
+
raise ScanError("docs root does not exist: " + docs_root)
|
|
369
|
+
docs = find_docs(docs_root)
|
|
370
|
+
corpus, reads = source_corpus(root)
|
|
371
|
+
if not corpus:
|
|
372
|
+
raise ScanError(
|
|
373
|
+
"no source files found under " + root + "; every env-var claim "
|
|
374
|
+
"would read as false against an empty corpus")
|
|
375
|
+
version = repo_version(root)
|
|
376
|
+
|
|
377
|
+
results = []
|
|
378
|
+
for rel, full in docs:
|
|
379
|
+
try:
|
|
380
|
+
text = _read(full)
|
|
381
|
+
except OSError as exc:
|
|
382
|
+
results.append(_finding("file", rel, 0, rel, UNCHECKABLE,
|
|
383
|
+
"unreadable: %s" % exc))
|
|
384
|
+
continue
|
|
385
|
+
results.extend(check_tool_paths(root, rel, text))
|
|
386
|
+
results.extend(check_env_vars(rel, text, corpus, reads))
|
|
387
|
+
results.extend(check_versions(rel, text, version))
|
|
388
|
+
return docs, results
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _summary(docs, results):
|
|
392
|
+
return {
|
|
393
|
+
"docs_scanned": len(docs),
|
|
394
|
+
"claims_checked": len(results),
|
|
395
|
+
"false": sum(1 for r in results if r["status"] == FALSE),
|
|
396
|
+
"passed": sum(1 for r in results if r["status"] == PASS),
|
|
397
|
+
"uncheckable": sum(1 for r in results if r["status"] == UNCHECKABLE),
|
|
398
|
+
"excluded": [{"path": p, "reason": why} for p, why in DOC_EXCLUSIONS],
|
|
399
|
+
"source_dirs": list(SOURCE_DIRS),
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _exit_code(docs, results):
|
|
404
|
+
if not docs:
|
|
405
|
+
return 3
|
|
406
|
+
if not results:
|
|
407
|
+
return 3 # docs present, nothing checkable: still an absent measurement
|
|
408
|
+
return 1 if any(r["status"] == FALSE for r in results) else 0
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def _render(summary, results, code):
|
|
412
|
+
lines = ["DOC AUDIT"]
|
|
413
|
+
lines.append(" docs scanned: %d" % summary["docs_scanned"])
|
|
414
|
+
lines.append(" claims checked: %d (false %d, passed %d, uncheckable %d)"
|
|
415
|
+
% (summary["claims_checked"], summary["false"],
|
|
416
|
+
summary["passed"], summary["uncheckable"]))
|
|
417
|
+
for item in summary["excluded"]:
|
|
418
|
+
lines.append(" excluded: %-16s %s" % (item["path"], item["reason"]))
|
|
419
|
+
|
|
420
|
+
false = [r for r in results if r["status"] == FALSE]
|
|
421
|
+
if false:
|
|
422
|
+
lines.append("")
|
|
423
|
+
lines.append("FALSE CLAIMS (%d)" % len(false))
|
|
424
|
+
for r in false:
|
|
425
|
+
lines.append(" %s:%d %s" % (r["file"], r["line"], r["claim"]))
|
|
426
|
+
lines.append(" evidence: %s" % r["evidence"])
|
|
427
|
+
|
|
428
|
+
unchecked = [r for r in results if r["status"] == UNCHECKABLE]
|
|
429
|
+
if unchecked:
|
|
430
|
+
lines.append("")
|
|
431
|
+
lines.append("UNCHECKABLE (%d) -- not passing, not failing"
|
|
432
|
+
% len(unchecked))
|
|
433
|
+
for r in unchecked[:20]:
|
|
434
|
+
lines.append(" %s:%d %s -- %s"
|
|
435
|
+
% (r["file"], r["line"], r["claim"], r["evidence"]))
|
|
436
|
+
if len(unchecked) > 20:
|
|
437
|
+
lines.append(" ... %d more (use --json for all)"
|
|
438
|
+
% (len(unchecked) - 20))
|
|
439
|
+
|
|
440
|
+
if code == 3:
|
|
441
|
+
lines.append("")
|
|
442
|
+
lines.append("NOTHING TO CHECK -- scanning nothing is not a clean bill.")
|
|
443
|
+
elif not false:
|
|
444
|
+
lines.append("")
|
|
445
|
+
lines.append("No false claim found.")
|
|
446
|
+
return "\n".join(lines)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def main(argv=None):
|
|
450
|
+
parser = _Parser(
|
|
451
|
+
description="Find documentation claims the repo contradicts.")
|
|
452
|
+
parser.add_argument("docs_root", nargs="?", default=None,
|
|
453
|
+
help="directory of markdown to audit (default: repo root)")
|
|
454
|
+
parser.add_argument("--json", action="store_true",
|
|
455
|
+
help="emit machine-readable output")
|
|
456
|
+
args = parser.parse_args(argv)
|
|
457
|
+
|
|
458
|
+
docs_root = args.docs_root or _ROOT
|
|
459
|
+
if args.docs_root is not None and not os.path.exists(args.docs_root):
|
|
460
|
+
# 66 input missing. Emitted as JSON under --json: a consumer that asked
|
|
461
|
+
# for machine output must not get a bare line it cannot parse.
|
|
462
|
+
payload = {"status": "input_missing", "exit_code": 66,
|
|
463
|
+
"error": "no such path: " + args.docs_root}
|
|
464
|
+
print(json.dumps(payload, indent=2) if args.json
|
|
465
|
+
else "INPUT MISSING -- no such path: " + args.docs_root)
|
|
466
|
+
return 66
|
|
467
|
+
|
|
468
|
+
try:
|
|
469
|
+
docs, results = audit(_ROOT, docs_root)
|
|
470
|
+
except ScanError as exc:
|
|
471
|
+
payload = {"status": "scan_failed", "exit_code": 2, "error": str(exc)}
|
|
472
|
+
print(json.dumps(payload, indent=2) if args.json
|
|
473
|
+
else "CANNOT SCAN -- " + str(exc))
|
|
474
|
+
return 2
|
|
475
|
+
|
|
476
|
+
code = _exit_code(docs, results)
|
|
477
|
+
summary = _summary(docs, results)
|
|
478
|
+
if args.json:
|
|
479
|
+
print(json.dumps({"status": "no_claims" if code == 3 else "audited",
|
|
480
|
+
"exit_code": code, "summary": summary,
|
|
481
|
+
"findings": results}, indent=2))
|
|
482
|
+
else:
|
|
483
|
+
print(_render(summary, results, code))
|
|
484
|
+
return code
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
if __name__ == "__main__":
|
|
488
|
+
sys.exit(main())
|