ecoportal-api 0.10.16 → 0.10.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of ecoportal-api might be problematic. Click here for more details.

Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.ai-assistance/.gitignore +2 -0
  3. data/.ai-assistance/bridge/.gitignore +10 -0
  4. data/.ai-assistance/bridge/CLAUDE.md +96 -0
  5. data/.ai-assistance/bridge/archive/.gitkeep +0 -0
  6. data/.ai-assistance/bridge/inbox/.gitkeep +0 -0
  7. data/.ai-assistance/bridge/outbox/.gitkeep +0 -0
  8. data/.ai-assistance/capabilities/assumptions-log.md +23 -0
  9. data/.ai-assistance/scripts/bridge-inbox-check.sh +119 -0
  10. data/.ai-assistance/scripts/bridge-init.sh +86 -0
  11. data/.ai-assistance/scripts/confine-to-subtree.sh +58 -0
  12. data/.ai-assistance/scripts/dirty-tree-guard.sh +96 -0
  13. data/.ai-assistance/scripts/distill_procedural.py +602 -0
  14. data/.ai-assistance/scripts/log-mcp-access.sh +24 -0
  15. data/.ai-assistance/scripts/log-skill-usage.sh +79 -0
  16. data/.ai-assistance/scripts/log_mcp_access.py +158 -0
  17. data/.ai-assistance/scripts/observe-session.sh +13 -0
  18. data/.ai-assistance/scripts/observe_session.py +287 -0
  19. data/.ai-assistance/scripts/protect-host-paths.sh +135 -0
  20. data/.ai-assistance/scripts/scrub.py +1149 -0
  21. data/.ai-assistance/scripts/scrub.py.sha256 +6 -0
  22. data/.ai-assistance/scripts/surface-procedural.sh +9 -0
  23. data/.ai-assistance/scripts/surface_procedural.py +101 -0
  24. data/.ai-assistance/skills/ep-ai-manager/SKILL.md +519 -0
  25. data/.ai-assistance/skills/project-self-docs/SKILL.md +259 -0
  26. data/.ai-assistance/skills/project-self-docs/scripts/self_docs_scan.py +378 -0
  27. data/.ai-assistance/standards-version.json +12 -0
  28. data/.ai-assistance/version.json +8 -0
  29. data/.claude/.gitignore +2 -0
  30. data/.claude/settings.json +128 -0
  31. data/CHANGELOG.md +8 -5
  32. data/CLAUDE.md +95 -71
  33. data/docs/self-docs/ARCHITECTURE.md +145 -0
  34. data/docs/self-docs/CHANGES.jsonl +7 -0
  35. data/docs/self-docs/COMPLIANCE.md +66 -0
  36. data/docs/self-docs/CONVENTIONS.md +74 -0
  37. data/docs/self-docs/INTEGRATIONS.md +62 -0
  38. data/docs/self-docs/OPERATIONS.md +64 -0
  39. data/docs/self-docs/OVERVIEW.md +61 -0
  40. data/docs/self-docs/STATUS.md +71 -0
  41. data/docs/self-docs/self-docs-index.json +51 -0
  42. data/docs/worklog.md +48 -0
  43. data/lib/ecoportal/api/common/client/with_retry.rb +6 -0
  44. data/lib/ecoportal/api/version.rb +1 -1
  45. metadata +40 -1
@@ -0,0 +1,79 @@
1
+ #!/usr/bin/env bash
2
+ # log-skill-usage.sh -- PreToolUse hook (matcher: Skill) that DETERMINISTICALLY
3
+ # records every Skill-tool invocation to the ISO-week usage log. This replaces
4
+ # the old per-skill self-report instruction (which never fired -- 575 fleet
5
+ # records were ALL the scheduler heartbeat, ZERO skill/* events), matching the
6
+ # token-frugality rule: hard-wire the deterministic part, no inline inference.
7
+ #
8
+ # Standard: standards/kpi/usage-tracking.md ("Who writes usage records").
9
+ #
10
+ # Wired in .claude/settings.json:
11
+ # "PreToolUse": [{ "matcher": "Skill",
12
+ # "hooks": [{ "type": "command",
13
+ # "command": "bash .ai-assistance/scripts/log-skill-usage.sh" }] }]
14
+ #
15
+ # Reads: stdin -- the PreToolUse JSON (tool_name, tool_input.skill, session_id, cwd)
16
+ # Writes: .ai-assistance/local/kpi/usage-<YYYY-WNN>.jsonl (one appended line)
17
+ #
18
+ # NOTE: uses `python -c` (NOT a heredoc) so the hook's stdin -- the PreToolUse
19
+ # JSON -- reaches python. A heredoc would replace stdin with the script itself.
20
+ # Fails OPEN and SILENT on any error: a KPI logger must never wedge or slow a
21
+ # session, and PreToolUse must not deny the tool. Always exits 0 with no stdout.
22
+ PY=python3
23
+ command -v python3 >/dev/null 2>&1 || PY=python
24
+ exec "$PY" -c '
25
+ import sys, json, os, datetime
26
+
27
+ def week_id(now):
28
+ y, w, _ = now.isocalendar()
29
+ return "%04d-W%02d" % (y, w)
30
+
31
+ try:
32
+ raw = sys.stdin.buffer.read()
33
+ # Tolerate cp1252/latin-1 on Windows shells; never choke on odd bytes.
34
+ try:
35
+ text = raw.decode("utf-8")
36
+ except Exception:
37
+ text = raw.decode("cp1252", "replace")
38
+ data = json.loads(text)
39
+ except Exception:
40
+ sys.exit(0) # cannot parse -> record nothing, allow the tool (fail open)
41
+
42
+ if data.get("tool_name") != "Skill":
43
+ sys.exit(0)
44
+
45
+ tool_input = data.get("tool_input") or {}
46
+ skill = tool_input.get("skill") or "unknown"
47
+ # Slash-form or plugin-namespaced skills: keep the leaf name, kebab-safe.
48
+ skill = str(skill).strip().lstrip("/")
49
+ if ":" in skill: # plugin:skill -> skill
50
+ skill = skill.split(":")[-1]
51
+ skill = skill.replace(" ", "-")
52
+
53
+ session_id = data.get("session_id") or ""
54
+ cwd = data.get("cwd") or os.getcwd()
55
+ repo = os.path.basename(os.path.normpath(cwd)) if cwd else ""
56
+
57
+ now = datetime.datetime.now(datetime.timezone.utc)
58
+ ts = now.strftime("%Y-%m-%dT%H:%M:%SZ")
59
+ week = week_id(now)
60
+
61
+ log_dir = os.path.join(".ai-assistance", "local", "kpi")
62
+ log_file = os.path.join(log_dir, "usage-%s.jsonl" % week)
63
+ record = {
64
+ "component": "skill/%s" % skill,
65
+ "action": "invoked",
66
+ "detail": "Skill tool invoked",
67
+ "ts": ts,
68
+ "session_id": session_id,
69
+ "repo": repo,
70
+ }
71
+ try:
72
+ os.makedirs(log_dir, exist_ok=True)
73
+ with open(log_file, "a", encoding="utf-8") as fh:
74
+ fh.write(json.dumps(record) + "\n")
75
+ except Exception:
76
+ pass # never fail the session over a KPI write
77
+
78
+ sys.exit(0)
79
+ '
@@ -0,0 +1,158 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ log_mcp_access.py -- MCP access-audit capture (PostToolUse hook, matcher mcp__.*).
4
+
5
+ Standard: standards/security/access-audit-ledger.md ("Record schema" -- the fields this
6
+ script writes are exactly the ones marked CAPTURED there; nothing else is fabricated).
7
+
8
+ Hook mode (no flags, the only mode used in production): reads the PostToolUse JSON from
9
+ stdin and, IF tool_name looks like an MCP tool call (mcp__<server>__<tool>), appends ONE
10
+ JSON record to the v0 local ledger:
11
+ .ai-assistance/local/audit/access-<ISO-week>.jsonl
12
+
13
+ CAPTURE ALLOW-LIST -- only what is cheaply and honestly derivable from the hook payload:
14
+ - tool_name (direct from the hook JSON)
15
+ - session_id (direct from the hook JSON, "" if absent)
16
+ - repo (basename of cwd, same convention as
17
+ log-skill-usage.sh)
18
+ - ep_ai_standards_version (see standard doc: standards-version.json first,
19
+ this repo's own CHANGELOG.md as a fallback)
20
+ - tool_input_summary (top-level ARGUMENT KEY NAMES only -- sorted list
21
+ + count -- never a single argument VALUE)
22
+ - result ("ok" / "error" / "unknown" from
23
+ tool_response.isError when it is a real bool)
24
+ - ts (wall clock at capture time, UTC ISO 8601)
25
+
26
+ NEVER captured, ever, by design (PII/secret risk -- see the standard's "PII / secret risk"
27
+ section): tool_input VALUES, tool_response body/output/stdout/stderr content, prompt or
28
+ response text, file contents. Only shapes (key names, booleans, counts) are recorded.
29
+
30
+ Fields the owner's spec asked for that are NOT written here because they are not honestly
31
+ capturable from this hook payload today (see the standard doc for the full PLANNED list
32
+ and why): related_mr_task_id, model, reason, associated_cost, triggering_skill,
33
+ skill_version, api_key_sha256, latency. None of them are written as null/empty guesses --
34
+ they are simply absent from the record.
35
+
36
+ Never blocks a session (hook mode): any error -> exit 0, write nothing. ASCII-only output
37
+ (json.dumps with ensure_ascii=True) regardless of what bytes arrive on stdin.
38
+ """
39
+ import json
40
+ import os
41
+ import re
42
+ import sys
43
+ from datetime import datetime, timezone
44
+ from pathlib import Path
45
+
46
+ _MCP_TOOL_RE = re.compile(r"^mcp__")
47
+ _CHANGELOG_VERSION_RE = re.compile(r"^##\s*\[(\d+\.\d+\.\d+)\]", re.MULTILINE)
48
+
49
+
50
+ def _iso_now():
51
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
52
+
53
+
54
+ def _iso_week(now=None):
55
+ now = now or datetime.now(timezone.utc)
56
+ y, w, _ = now.isocalendar()
57
+ return "%04d-W%02d" % (y, w)
58
+
59
+
60
+ def _ep_ai_standards_version(root):
61
+ """Best-effort, cheap, no network. See standard doc for the two sources tried."""
62
+ vf = root / ".ai-assistance" / "standards-version.json"
63
+ try:
64
+ if vf.is_file():
65
+ data = json.loads(vf.read_text(encoding="utf-8"))
66
+ v = data.get("ep-ai-standards-version")
67
+ if isinstance(v, str) and v:
68
+ return v
69
+ except Exception:
70
+ pass
71
+ changelog = root / "CHANGELOG.md"
72
+ try:
73
+ if changelog.is_file():
74
+ text = changelog.read_text(encoding="utf-8", errors="replace")
75
+ m = _CHANGELOG_VERSION_RE.search(text)
76
+ if m:
77
+ return m.group(1)
78
+ except Exception:
79
+ pass
80
+ return None
81
+
82
+
83
+ def _tool_input_summary(tool_input):
84
+ """Top-level argument KEY NAMES only -- never a value. Tolerates any shape."""
85
+ if not isinstance(tool_input, dict):
86
+ return {"n_keys": 0, "keys": []}
87
+ keys = sorted(str(k) for k in tool_input.keys())
88
+ return {"n_keys": len(keys), "keys": keys}
89
+
90
+
91
+ def _result_status(tool_response):
92
+ if isinstance(tool_response, dict) and isinstance(tool_response.get("isError"), bool):
93
+ return "error" if tool_response["isError"] else "ok"
94
+ return "unknown"
95
+
96
+
97
+ def build_record(data):
98
+ """Pure: given a parsed PostToolUse payload dict, return a record dict, or None if
99
+ this event is not an MCP tool call (nothing to log). Never raises -- any malformed
100
+ shape inside `data` degrades a field to its safe default instead of erroring out."""
101
+ if not isinstance(data, dict):
102
+ return None
103
+ tool_name = data.get("tool_name")
104
+ if not isinstance(tool_name, str) or not _MCP_TOOL_RE.match(tool_name):
105
+ return None # not an MCP tool call -- nothing for this ledger
106
+
107
+ cwd = data.get("cwd") or os.getcwd()
108
+ root = Path(cwd)
109
+ repo = os.path.basename(os.path.normpath(str(cwd))) if cwd else ""
110
+ session_id = data.get("session_id") or ""
111
+
112
+ record = {
113
+ "ts": _iso_now(),
114
+ "session_id": session_id,
115
+ "repo": repo,
116
+ "tool_name": tool_name,
117
+ "tool_input_summary": _tool_input_summary(data.get("tool_input")),
118
+ "result": _result_status(data.get("tool_response")),
119
+ }
120
+ version = _ep_ai_standards_version(root)
121
+ if version:
122
+ record["ep_ai_standards_version"] = version
123
+ return record
124
+
125
+
126
+ def _write_record(root, record):
127
+ week = _iso_week()
128
+ log_dir = root / ".ai-assistance" / "local" / "audit"
129
+ log_file = log_dir / ("access-%s.jsonl" % week)
130
+ log_dir.mkdir(parents=True, exist_ok=True)
131
+ with log_file.open("a", encoding="utf-8") as fh:
132
+ fh.write(json.dumps(record, ensure_ascii=True) + "\n")
133
+
134
+
135
+ def main(argv=None): # noqa: ARG001 -- no flags; argv kept for test-call symmetry
136
+ try:
137
+ raw = sys.stdin.buffer.read()
138
+ try:
139
+ text = raw.decode("utf-8")
140
+ except Exception:
141
+ text = raw.decode("cp1252", "replace")
142
+ data = json.loads(text) if text.strip() else {}
143
+ except Exception:
144
+ return 0 # malformed payload -> allow the tool, log nothing (fail open)
145
+
146
+ try:
147
+ record = build_record(data)
148
+ if record is None:
149
+ return 0
150
+ cwd = data.get("cwd") or os.getcwd()
151
+ _write_record(Path(cwd), record)
152
+ except Exception:
153
+ pass # an audit logger must never wedge or fail a session
154
+ return 0
155
+
156
+
157
+ if __name__ == "__main__":
158
+ sys.exit(main())
@@ -0,0 +1,13 @@
1
+ #!/usr/bin/env bash
2
+ # observe-session.sh -- thin Stop/SessionEnd wrapper for the procedural-memory OBSERVE stage.
3
+ # Opt-in (off by default): does nothing unless .ai-assistance/local/procedural-observe.enabled
4
+ # exists or PROCEDURAL_OBSERVE=1 is set. Enable/disable/status (human-run, one command):
5
+ # python .ai-assistance/scripts/observe_session.py --enable | --disable | --status
6
+ # Writes to the dedicated stream .ai-assistance/local/observe-<ISO-week>.jsonl (separate from
7
+ # the kpi/usage heartbeat stream). Never blocks a session (always exits 0).
8
+ # Logic + privacy allow-list + scrub-before-write live in observe_session.py.
9
+ # Standard: standards/workflows/procedural-memory.md (D2).
10
+ PY=python3
11
+ command -v python3 >/dev/null 2>&1 || PY=python
12
+ "$PY" "$(dirname "$0")/observe_session.py" || true
13
+ exit 0
@@ -0,0 +1,287 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ observe_session.py -- Procedural-memory OBSERVE stage (Stop / SessionEnd hook).
4
+
5
+ Standard: standards/workflows/procedural-memory.md (D2 -- logging privacy).
6
+
7
+ Hook mode (no flags): reads the Stop-hook JSON from stdin, and IF the developer has opted
8
+ in, appends ONE low-sensitivity session-summary record to the DEDICATED observe stream
9
+ .ai-assistance/local/observe-<ISO-week>.jsonl. The observe stream is deliberately separate
10
+ from the KPI stream (.ai-assistance/local/kpi/usage-*.jsonl), which is dominated by
11
+ scheduler heartbeats -- mixing the two buried the behavioral signal (2026-07-20 inventory).
12
+
13
+ CLI mode (one-command opt-in management; run by a HUMAN, never by automation):
14
+ python observe_session.py --status # is Observe enabled here? what has it captured?
15
+ python observe_session.py --enable # create the consent flag (explicit human opt-in)
16
+ python observe_session.py --disable # remove the consent flag (streams are kept)
17
+
18
+ CONSENT (opt-in, off by default). Observe runs only when the developer has opted in, by
19
+ either:
20
+ - creating the marker file .ai-assistance/local/procedural-observe.enabled
21
+ (preferred; `--enable` above is the one-command way), or
22
+ - exporting PROCEDURAL_OBSERVE=1 in the environment.
23
+ If neither is set, this script writes nothing and exits 0. The distill pass applies the same
24
+ gate (refuses to read the log without consent). Nothing in the toolchain may create the flag
25
+ automatically -- enabling is a human decision (consent is load-bearing).
26
+
27
+ CAPTURE ALLOW-LIST (D2) -- only structured, low-sensitivity signal:
28
+ - tool/command NAMES (the verb, e.g. Edit, Bash, git) -- never arguments
29
+ - file paths + extensions touched, RELATIVE to the repo root -- never file contents/diffs
30
+ - coarse counts (n tools, n files) and a duration bucket
31
+ - session id, ISO week, timestamp
32
+ NEVER captured: command arguments, file contents/diffs, prompt/response text, the literal
33
+ text of corrections/redirects, secrets/tokens/credentials.
34
+
35
+ REDACTION: every string that could carry a path is passed through scripts/lib/scrub.py
36
+ BEFORE it is written. If scrub reports it could not run cleanly, the field is dropped.
37
+
38
+ Never blocks a session (hook mode): any error -> exit 0, write nothing.
39
+ """
40
+ import argparse
41
+ import json
42
+ import os
43
+ import sys
44
+ from pathlib import Path
45
+
46
+ # ISO-week + timestamp without Date.now()-style nondeterminism concerns: this runs on the
47
+ # host at session end, so real wall-clock is correct and expected here.
48
+ from datetime import datetime, timezone
49
+
50
+ _ALLOWED_FILE_TOOLS = {"Read", "Edit", "Write", "NotebookEdit", "MultiEdit"}
51
+ _DUR_BUCKETS = [(300, "short"), (1800, "medium"), (7200, "long")] # secs -> label; else "xlong"
52
+
53
+ FLAG_NAME = "procedural-observe.enabled"
54
+
55
+
56
+ def _iso_now():
57
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
58
+
59
+
60
+ def _iso_week():
61
+ y, w, _ = datetime.now(timezone.utc).isocalendar()
62
+ return f"{y}-W{w:02d}"
63
+
64
+
65
+ def _load_scrub(root):
66
+ """Import scrub() from scrub.py if available; else None.
67
+
68
+ Returns None when the redactor cannot be loaded. Privacy FAILS CLOSED: the caller must
69
+ write NOTHING when scrub is unavailable, rather than logging unscrubbed signal. (A prior
70
+ version fell back to an identity function -- that would log file paths, which can carry a
71
+ username, without redaction on a repo where scrub.py was not deployed.)"""
72
+ for cand in (root / ".ai-assistance" / "scripts", root / "scripts" / "lib"):
73
+ p = cand / "scrub.py"
74
+ if p.exists():
75
+ sys.path.insert(0, str(cand))
76
+ try:
77
+ import scrub # type: ignore
78
+ return scrub.scrub
79
+ except Exception:
80
+ pass
81
+ return None
82
+
83
+
84
+ def _consent(local_dir):
85
+ if os.environ.get("PROCEDURAL_OBSERVE", "").lower() in ("1", "true", "yes"):
86
+ return True
87
+ return (local_dir / FLAG_NAME).exists()
88
+
89
+
90
+ # --- opt-in management CLI (--enable / --disable / --status) -------------------
91
+
92
+ def _cli_enable(local_dir):
93
+ """Create the consent flag. This command exists to make the HUMAN opt-in trivial;
94
+ it must only ever be run by a developer, never by a hook/scheduler/agent."""
95
+ flag = local_dir / FLAG_NAME
96
+ if flag.exists():
97
+ print(f"[observe] already enabled ({flag})")
98
+ return 0
99
+ local_dir.mkdir(parents=True, exist_ok=True)
100
+ flag.write_text(
101
+ "Procedural-memory Observe opt-in (standard D2 consent flag).\n"
102
+ "Created by observe_session.py --enable at the developer's request.\n"
103
+ f"Enabled: {_iso_now()}\n",
104
+ encoding="utf-8",
105
+ )
106
+ print(f"[observe] ENABLED -- consent flag created: {flag}")
107
+ print("[observe] capture allow-list (D2): tool names, repo-relative paths, coarse counts.")
108
+ print("[observe] never captured: arguments, file contents, diffs, prompt/response text.")
109
+ print("[observe] stream: .ai-assistance/local/observe-<ISO-week>.jsonl (machine-local, gitignored).")
110
+ print("[observe] disable any time: observe_session.py --disable (streams kept; delete them freely).")
111
+ return 0
112
+
113
+
114
+ def _cli_disable(local_dir):
115
+ flag = local_dir / FLAG_NAME
116
+ if not flag.exists():
117
+ print("[observe] already disabled (no consent flag)")
118
+ return 0
119
+ flag.unlink()
120
+ print(f"[observe] DISABLED -- consent flag removed: {flag}")
121
+ print("[observe] existing observe-*.jsonl streams were kept; delete them if you want them gone.")
122
+ return 0
123
+
124
+
125
+ def _cli_status(local_dir):
126
+ flag = local_dir / FLAG_NAME
127
+ env_on = os.environ.get("PROCEDURAL_OBSERVE", "").lower() in ("1", "true", "yes")
128
+ if flag.exists():
129
+ state = f"ENABLED (flag: {flag})"
130
+ elif env_on:
131
+ state = "ENABLED (PROCEDURAL_OBSERVE env var; no flag file)"
132
+ else:
133
+ state = "disabled (no consent flag; enable with: observe_session.py --enable)"
134
+ print(f"[observe] status: {state}")
135
+ streams = sorted(local_dir.glob("observe-*.jsonl")) if local_dir.is_dir() else []
136
+ if not streams:
137
+ print("[observe] streams: none captured yet")
138
+ for s in streams:
139
+ try:
140
+ n = sum(1 for ln in s.read_text(encoding="utf-8", errors="replace").splitlines() if ln.strip())
141
+ except OSError:
142
+ n = "?"
143
+ print(f"[observe] stream: {s.name} ({n} record(s))")
144
+ return 0
145
+
146
+
147
+ def _duration_bucket(secs):
148
+ try:
149
+ secs = float(secs)
150
+ except (TypeError, ValueError):
151
+ return "unknown"
152
+ for lim, label in _DUR_BUCKETS:
153
+ if secs <= lim:
154
+ return label
155
+ return "xlong"
156
+
157
+
158
+ def _extract_from_transcript(path, root, scrub):
159
+ """Return (tool_counts, files) reading ONLY tool-use verbs + file paths from the
160
+ transcript. Never reads prompt/response text or tool arguments beyond a file path."""
161
+ tool_counts = {}
162
+ files = []
163
+ seen = set()
164
+ try:
165
+ lines = Path(path).read_text(encoding="utf-8", errors="replace").splitlines()
166
+ except OSError:
167
+ return tool_counts, files
168
+ for line in lines:
169
+ line = line.strip()
170
+ if not line:
171
+ continue
172
+ try:
173
+ rec = json.loads(line)
174
+ except ValueError:
175
+ continue
176
+ # Anthropic transcript: assistant messages carry content blocks; tool_use blocks
177
+ # have {"type":"tool_use","name":...,"input":{...}}. We read name + a file path only.
178
+ msg = rec.get("message") or rec
179
+ content = msg.get("content") if isinstance(msg, dict) else None
180
+ if not isinstance(content, list):
181
+ continue
182
+ for block in content:
183
+ if not isinstance(block, dict) or block.get("type") != "tool_use":
184
+ continue
185
+ name = block.get("name") or "unknown"
186
+ tool_counts[name] = tool_counts.get(name, 0) + 1
187
+ if name in _ALLOWED_FILE_TOOLS:
188
+ inp = block.get("input") or {}
189
+ fp = inp.get("file_path") or inp.get("path") or inp.get("notebook_path")
190
+ if not fp:
191
+ continue
192
+ # Make repo-relative; keep only path + extension, then scrub.
193
+ try:
194
+ rel = os.path.relpath(fp, str(root))
195
+ except ValueError:
196
+ rel = fp
197
+ rel = rel.replace("\\", "/")
198
+ scrubbed, findings = scrub(rel)
199
+ if any(f.get("type") == "error" for f in findings):
200
+ continue # scrub could not run cleanly -> drop
201
+ ext = os.path.splitext(scrubbed)[1].lower()
202
+ key = (ext, scrubbed)
203
+ if key not in seen:
204
+ seen.add(key)
205
+ files.append({"ext": ext, "path": scrubbed})
206
+ return tool_counts, files
207
+
208
+
209
+ def _hook_mode():
210
+ try:
211
+ raw = sys.stdin.read()
212
+ data = json.loads(raw) if raw.strip() else {}
213
+ except Exception:
214
+ return 0
215
+
216
+ cwd = data.get("cwd") or os.getcwd()
217
+ root = Path(cwd)
218
+ local_dir = root / ".ai-assistance" / "local"
219
+ if not _consent(local_dir):
220
+ return 0 # not opted in -> observe nothing
221
+
222
+ scrub = _load_scrub(root)
223
+ if scrub is None:
224
+ # Fail CLOSED: redactor unavailable -> do not log anything (never write unscrubbed).
225
+ sys.stderr.write("[observe] scrub.py not found; skipping session log (privacy fail-closed).\n")
226
+ return 0
227
+ session_id = data.get("session_id") or data.get("sessionId") or ""
228
+ transcript = data.get("transcript_path") or data.get("transcriptPath")
229
+
230
+ tool_counts, files = ({}, [])
231
+ if transcript:
232
+ tool_counts, files = _extract_from_transcript(transcript, root, scrub)
233
+
234
+ record = {
235
+ "component": "procedural/observe",
236
+ "action": "session-summary",
237
+ "ts": _iso_now(),
238
+ "session_id": session_id,
239
+ "week": _iso_week(),
240
+ "detail": {
241
+ "tools": tool_counts,
242
+ "files": files,
243
+ "n_tools": sum(tool_counts.values()),
244
+ "n_files": len(files),
245
+ "duration_bucket": _duration_bucket(data.get("duration_seconds")),
246
+ },
247
+ }
248
+
249
+ # Dedicated observe stream -- NOT the kpi/usage-*.jsonl heartbeat stream.
250
+ try:
251
+ local_dir.mkdir(parents=True, exist_ok=True)
252
+ out = local_dir / f"observe-{record['week']}.jsonl"
253
+ with out.open("a", encoding="utf-8") as fh:
254
+ fh.write(json.dumps(record) + "\n")
255
+ except OSError:
256
+ return 0
257
+ return 0
258
+
259
+
260
+ def main(argv=None):
261
+ ap = argparse.ArgumentParser(
262
+ description="Procedural-memory Observe stage: Stop-hook logger (default) "
263
+ "+ human opt-in management (--enable/--disable/--status).")
264
+ ap.add_argument("--enable", action="store_true",
265
+ help="create the consent flag (HUMAN opt-in -- run this yourself)")
266
+ ap.add_argument("--disable", action="store_true", help="remove the consent flag")
267
+ ap.add_argument("--status", action="store_true",
268
+ help="show whether Observe is enabled and what it has captured")
269
+ ap.add_argument("--cwd", default=os.getcwd(),
270
+ help="repo root for CLI mode (default: current directory)")
271
+ a = ap.parse_args(argv)
272
+
273
+ if sum((a.enable, a.disable, a.status)) > 1:
274
+ print("[observe] pick ONE of --enable / --disable / --status")
275
+ return 2
276
+ local_dir = Path(a.cwd) / ".ai-assistance" / "local"
277
+ if a.enable:
278
+ return _cli_enable(local_dir)
279
+ if a.disable:
280
+ return _cli_disable(local_dir)
281
+ if a.status:
282
+ return _cli_status(local_dir)
283
+ return _hook_mode()
284
+
285
+
286
+ if __name__ == "__main__":
287
+ sys.exit(main())
@@ -0,0 +1,135 @@
1
+ #!/usr/bin/env bash
2
+ # protect-host-paths.sh -- PreToolUse hook (matcher: Bash) that blocks commands
3
+ # writing to / deleting host-OS system paths. Defense-in-depth: this decision is
4
+ # enforced regardless of permission mode, so it holds even under
5
+ # --dangerously-skip-permissions (where the deny-list prompts are off).
6
+ # Standard: standards/tooling/host-protection.md (Layer 4).
7
+ #
8
+ # Emits a structured deny decision on stdout + exit 0 (NOT exit 2, whose behavior
9
+ # in dangerous mode is unclear). Uses python3 (a documented prereq) -- no jq needed.
10
+ # Fails OPEN on any internal error (a guard must never wedge the session): if it
11
+ # cannot parse, it allows and lets the normal permission flow decide.
12
+ #
13
+ # NOTE: uses `python -c` (NOT a `<<HEREDOC`) so the hook's stdin -- the PreToolUse
14
+ # JSON -- reaches python. A heredoc would replace stdin with the script itself.
15
+ PY=python3
16
+ command -v python3 >/dev/null 2>&1 || PY=python
17
+ exec "$PY" -c '
18
+ import sys, json, re
19
+ try:
20
+ data = json.loads(sys.stdin.read())
21
+ except Exception:
22
+ sys.exit(0) # cannot parse -> allow (fail open); normal permissions still apply
23
+ if data.get("tool_name") != "Bash":
24
+ sys.exit(0)
25
+ cmd = (data.get("tool_input") or {}).get("command", "") or ""
26
+
27
+ verbs = r"(?:\brm\b|\brmdir\b|\bmv\b|\bcp\b|\bdd\b|\btee\b|\btruncate\b|\bchown\b|\bchmod\b|>>?|\bmkfs\S*|\bdiskutil\b)"
28
+
29
+ # Anchor: a root only counts as a true filesystem root when the leading `/`, `~`,
30
+ # or drive-letter is at a path boundary -- start-of-token, whitespace, quote, `=`,
31
+ # or start-of-string -- and NOT preceded by a path character. Without this, an
32
+ # unanchored `/lib` etc. matches mid-path substrings like `scripts/lib/x.py` or
33
+ # `tmp/var/x`, producing false positives (proven 2026-07-22). The negative
34
+ # lookbehind rules out being preceded by a word char, `.`, `/`, `\`, or `-`.
35
+ BOUNDARY = r"(?<![\w./\\-])"
36
+ roots = [
37
+ BOUNDARY + r"/System(?:/|\b)", BOUNDARY + r"/Library(?:/|\b)", BOUNDARY + r"~/Library(?:/|\b)",
38
+ BOUNDARY + r"/usr(?:/|\b)", BOUNDARY + r"/bin(?:/|\b)", BOUNDARY + r"/sbin(?:/|\b)", BOUNDARY + r"/etc(?:/|\b)",
39
+ BOUNDARY + r"/boot(?:/|\b)", BOUNDARY + r"/lib(?:/|\b)", BOUNDARY + r"/opt(?:/|\b)", BOUNDARY + r"/var(?:/|\b)",
40
+ BOUNDARY + r"[A-Za-z]:\\+Windows", BOUNDARY + r"[A-Za-z]:\\+Program Files",
41
+ BOUNDARY + r"/[A-Za-z]/Windows", BOUNDARY + r"/[A-Za-z]/Program Files",
42
+ # User-level persistence / credential targets (literal ~ form; shell-expanded forms
43
+ # can still bypass -- this is a heuristic layer, not a sandbox). NOT ~/.config (too broad).
44
+ BOUNDARY + r"~/\.ssh(?:/|\b)", BOUNDARY + r"~/\.aws(?:/|\b)", BOUNDARY + r"~/\.gnupg(?:/|\b)",
45
+ BOUNDARY + r"~/\.bashrc\b", BOUNDARY + r"~/\.zshrc\b", BOUNDARY + r"~/\.bash_profile\b", BOUNDARY + r"~/\.zprofile\b",
46
+ BOUNDARY + r"~/\.profile\b", BOUNDARY + r"~/\.gitconfig\b",
47
+ ]
48
+ danger = [
49
+ r"\brm\s+-rf\s+/(?:\s|$)",
50
+ r"\brm\s+-rf\s+~(?:/|\s|$)",
51
+ r"\bsudo\b",
52
+ r":\(\)\s*\{",
53
+ ]
54
+
55
+ # Strip commit-message DATA before scanning for danger verbs -- a `git commit
56
+ # -m "..."` (or `--message`) argument, or a message piped in via a quoted
57
+ # heredoc, is text handed to git as the commit message; it is never executed
58
+ # by the shell. Without this, a message merely *mentioning* a danger word --
59
+ # e.g. `git commit -m "docs: mention sudo usage"` -- was denied even though
60
+ # nothing runs. Proven 2026-07-22.
61
+ #
62
+ # Kept conservative on purpose (never widen this to "any quoted string" --
63
+ # a `bash -c "..."` payload IS executed and must keep scanning):
64
+ # - single-quoted -m/--message values: single quotes disable ALL shell
65
+ # expansion, so the content can never itself be executed code -> always
66
+ # safe to strip.
67
+ # - double-quoted -m/--message values: only stripped if they contain no
68
+ # `$(` / backtick command substitution (plain literal text), OR are
69
+ # exactly the standard multi-line idiom, e.g. $(cat <<EOF with a QUOTED
70
+ # delimiter ... EOF) -- in which case only the inert heredoc body is
71
+ # stripped, so something like -m "$(sudo rm -rf /)" (a real, executed
72
+ # command substitution smuggled into a -m value) is left untouched and
73
+ # still blocked.
74
+ # - git commit ... -F - <<EOF (with a QUOTED delimiter) ... EOF: message
75
+ # read directly from a quoted (non-expanding) heredoc attached to
76
+ # `git commit` -- also inert text. An UNQUOTED heredoc delimiter (plain
77
+ # <<EOF, no quotes around it) is deliberately left alone: its body still
78
+ # undergoes shell expansion, so it is not guaranteed to be inert.
79
+ def _strip_commit_message_data(cmd):
80
+ out = cmd
81
+
82
+ out = re.sub(
83
+ r"(-m|--message)(=|\s+)\x27[^\x27]*\x27",
84
+ lambda m: m.group(1) + m.group(2) + "\x27\x27",
85
+ out,
86
+ )
87
+
88
+ def _strip_dq_message(m):
89
+ prefix, sep, body = m.group(1), m.group(2), m.group(3)
90
+ idiom = re.match(
91
+ r"^\$\(\s*cat\s+<<[-~]?\s*([\x27\"])(\w+)\1\s*\r?\n(.*?)\r?\n[ \t]*\2\s*\)\s*$",
92
+ body, re.DOTALL,
93
+ )
94
+ if idiom:
95
+ quote, delim = idiom.group(1), idiom.group(2)
96
+ return prefix + sep + "\"$(cat <<" + quote + delim + quote + "\n\n" + delim + ")\""
97
+ if "$(" not in body and "`" not in body:
98
+ return prefix + sep + "\"\""
99
+ return m.group(0) # contains a real substitution -- leave it, keep scanning
100
+
101
+ out = re.sub(
102
+ r"(-m|--message)(=|\s+)\"((?:\\.|[^\"\\])*)\"",
103
+ _strip_dq_message,
104
+ out,
105
+ flags=re.DOTALL,
106
+ )
107
+
108
+ out = re.sub(
109
+ r"(git\s+commit\b[^\n]*<<[-~]?\s*([\x27\"])(\w+)\2[^\n]*\n)(.*?)(\n[ \t]*\3\b)",
110
+ lambda m: m.group(1) + m.group(5),
111
+ out,
112
+ flags=re.DOTALL,
113
+ )
114
+
115
+ return out
116
+
117
+ def blocked(reason):
118
+ print(json.dumps({"hookSpecificOutput": {
119
+ "hookEventName": "PreToolUse",
120
+ "permissionDecision": "deny",
121
+ "permissionDecisionReason": reason}}))
122
+ sys.exit(0)
123
+
124
+ scan_cmd = _strip_commit_message_data(cmd)
125
+
126
+ for d in danger:
127
+ if re.search(d, scan_cmd):
128
+ blocked("Blocked: destructive/privileged command (host-protection hook).")
129
+
130
+ root_re = "(?:" + "|".join(roots) + ")"
131
+ if re.search(verbs + r"[^\n]*" + root_re, scan_cmd) or re.search(root_re + r"[^\n]*" + verbs, scan_cmd):
132
+ blocked("Blocked: command writes/deletes a host-OS system path (host-protection hook).")
133
+
134
+ sys.exit(0)
135
+ '