ecoportal-api 0.10.16 → 0.10.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of ecoportal-api might be problematic. Click here for more details.
- checksums.yaml +4 -4
- data/.ai-assistance/.gitignore +2 -0
- data/.ai-assistance/bridge/.gitignore +10 -0
- data/.ai-assistance/bridge/CLAUDE.md +96 -0
- data/.ai-assistance/bridge/archive/.gitkeep +0 -0
- data/.ai-assistance/bridge/inbox/.gitkeep +0 -0
- data/.ai-assistance/bridge/outbox/.gitkeep +0 -0
- data/.ai-assistance/capabilities/assumptions-log.md +23 -0
- data/.ai-assistance/scripts/bridge-inbox-check.sh +119 -0
- data/.ai-assistance/scripts/bridge-init.sh +86 -0
- data/.ai-assistance/scripts/confine-to-subtree.sh +58 -0
- data/.ai-assistance/scripts/dirty-tree-guard.sh +96 -0
- data/.ai-assistance/scripts/distill_procedural.py +602 -0
- data/.ai-assistance/scripts/log-mcp-access.sh +24 -0
- data/.ai-assistance/scripts/log-skill-usage.sh +79 -0
- data/.ai-assistance/scripts/log_mcp_access.py +158 -0
- data/.ai-assistance/scripts/observe-session.sh +13 -0
- data/.ai-assistance/scripts/observe_session.py +287 -0
- data/.ai-assistance/scripts/protect-host-paths.sh +135 -0
- data/.ai-assistance/scripts/scrub.py +1149 -0
- data/.ai-assistance/scripts/scrub.py.sha256 +6 -0
- data/.ai-assistance/scripts/surface-procedural.sh +9 -0
- data/.ai-assistance/scripts/surface_procedural.py +101 -0
- data/.ai-assistance/skills/ep-ai-manager/SKILL.md +519 -0
- data/.ai-assistance/skills/project-self-docs/SKILL.md +259 -0
- data/.ai-assistance/skills/project-self-docs/scripts/self_docs_scan.py +378 -0
- data/.ai-assistance/standards-version.json +12 -0
- data/.ai-assistance/version.json +8 -0
- data/.claude/.gitignore +2 -0
- data/.claude/settings.json +128 -0
- data/CHANGELOG.md +8 -5
- data/CLAUDE.md +95 -71
- data/docs/self-docs/ARCHITECTURE.md +145 -0
- data/docs/self-docs/CHANGES.jsonl +7 -0
- data/docs/self-docs/COMPLIANCE.md +66 -0
- data/docs/self-docs/CONVENTIONS.md +74 -0
- data/docs/self-docs/INTEGRATIONS.md +62 -0
- data/docs/self-docs/OPERATIONS.md +64 -0
- data/docs/self-docs/OVERVIEW.md +61 -0
- data/docs/self-docs/STATUS.md +71 -0
- data/docs/self-docs/self-docs-index.json +51 -0
- data/docs/worklog.md +48 -0
- data/lib/ecoportal/api/common/client/with_retry.rb +6 -0
- data/lib/ecoportal/api/version.rb +1 -1
- metadata +40 -1
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# log-skill-usage.sh -- PreToolUse hook (matcher: Skill) that DETERMINISTICALLY
|
|
3
|
+
# records every Skill-tool invocation to the ISO-week usage log. This replaces
|
|
4
|
+
# the old per-skill self-report instruction (which never fired -- 575 fleet
|
|
5
|
+
# records were ALL the scheduler heartbeat, ZERO skill/* events), matching the
|
|
6
|
+
# token-frugality rule: hard-wire the deterministic part, no inline inference.
|
|
7
|
+
#
|
|
8
|
+
# Standard: standards/kpi/usage-tracking.md ("Who writes usage records").
|
|
9
|
+
#
|
|
10
|
+
# Wired in .claude/settings.json:
|
|
11
|
+
# "PreToolUse": [{ "matcher": "Skill",
|
|
12
|
+
# "hooks": [{ "type": "command",
|
|
13
|
+
# "command": "bash .ai-assistance/scripts/log-skill-usage.sh" }] }]
|
|
14
|
+
#
|
|
15
|
+
# Reads: stdin -- the PreToolUse JSON (tool_name, tool_input.skill, session_id, cwd)
|
|
16
|
+
# Writes: .ai-assistance/local/kpi/usage-<YYYY-WNN>.jsonl (one appended line)
|
|
17
|
+
#
|
|
18
|
+
# NOTE: uses `python -c` (NOT a heredoc) so the hook's stdin -- the PreToolUse
|
|
19
|
+
# JSON -- reaches python. A heredoc would replace stdin with the script itself.
|
|
20
|
+
# Fails OPEN and SILENT on any error: a KPI logger must never wedge or slow a
|
|
21
|
+
# session, and PreToolUse must not deny the tool. Always exits 0 with no stdout.
|
|
22
|
+
PY=python3
|
|
23
|
+
command -v python3 >/dev/null 2>&1 || PY=python
|
|
24
|
+
exec "$PY" -c '
|
|
25
|
+
import sys, json, os, datetime
|
|
26
|
+
|
|
27
|
+
def week_id(now):
|
|
28
|
+
y, w, _ = now.isocalendar()
|
|
29
|
+
return "%04d-W%02d" % (y, w)
|
|
30
|
+
|
|
31
|
+
try:
|
|
32
|
+
raw = sys.stdin.buffer.read()
|
|
33
|
+
# Tolerate cp1252/latin-1 on Windows shells; never choke on odd bytes.
|
|
34
|
+
try:
|
|
35
|
+
text = raw.decode("utf-8")
|
|
36
|
+
except Exception:
|
|
37
|
+
text = raw.decode("cp1252", "replace")
|
|
38
|
+
data = json.loads(text)
|
|
39
|
+
except Exception:
|
|
40
|
+
sys.exit(0) # cannot parse -> record nothing, allow the tool (fail open)
|
|
41
|
+
|
|
42
|
+
if data.get("tool_name") != "Skill":
|
|
43
|
+
sys.exit(0)
|
|
44
|
+
|
|
45
|
+
tool_input = data.get("tool_input") or {}
|
|
46
|
+
skill = tool_input.get("skill") or "unknown"
|
|
47
|
+
# Slash-form or plugin-namespaced skills: keep the leaf name, kebab-safe.
|
|
48
|
+
skill = str(skill).strip().lstrip("/")
|
|
49
|
+
if ":" in skill: # plugin:skill -> skill
|
|
50
|
+
skill = skill.split(":")[-1]
|
|
51
|
+
skill = skill.replace(" ", "-")
|
|
52
|
+
|
|
53
|
+
session_id = data.get("session_id") or ""
|
|
54
|
+
cwd = data.get("cwd") or os.getcwd()
|
|
55
|
+
repo = os.path.basename(os.path.normpath(cwd)) if cwd else ""
|
|
56
|
+
|
|
57
|
+
now = datetime.datetime.now(datetime.timezone.utc)
|
|
58
|
+
ts = now.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
59
|
+
week = week_id(now)
|
|
60
|
+
|
|
61
|
+
log_dir = os.path.join(".ai-assistance", "local", "kpi")
|
|
62
|
+
log_file = os.path.join(log_dir, "usage-%s.jsonl" % week)
|
|
63
|
+
record = {
|
|
64
|
+
"component": "skill/%s" % skill,
|
|
65
|
+
"action": "invoked",
|
|
66
|
+
"detail": "Skill tool invoked",
|
|
67
|
+
"ts": ts,
|
|
68
|
+
"session_id": session_id,
|
|
69
|
+
"repo": repo,
|
|
70
|
+
}
|
|
71
|
+
try:
|
|
72
|
+
os.makedirs(log_dir, exist_ok=True)
|
|
73
|
+
with open(log_file, "a", encoding="utf-8") as fh:
|
|
74
|
+
fh.write(json.dumps(record) + "\n")
|
|
75
|
+
except Exception:
|
|
76
|
+
pass # never fail the session over a KPI write
|
|
77
|
+
|
|
78
|
+
sys.exit(0)
|
|
79
|
+
'
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
log_mcp_access.py -- MCP access-audit capture (PostToolUse hook, matcher mcp__.*).
|
|
4
|
+
|
|
5
|
+
Standard: standards/security/access-audit-ledger.md ("Record schema" -- the fields this
|
|
6
|
+
script writes are exactly the ones marked CAPTURED there; nothing else is fabricated).
|
|
7
|
+
|
|
8
|
+
Hook mode (no flags, the only mode used in production): reads the PostToolUse JSON from
|
|
9
|
+
stdin and, IF tool_name looks like an MCP tool call (mcp__<server>__<tool>), appends ONE
|
|
10
|
+
JSON record to the v0 local ledger:
|
|
11
|
+
.ai-assistance/local/audit/access-<ISO-week>.jsonl
|
|
12
|
+
|
|
13
|
+
CAPTURE ALLOW-LIST -- only what is cheaply and honestly derivable from the hook payload:
|
|
14
|
+
- tool_name (direct from the hook JSON)
|
|
15
|
+
- session_id (direct from the hook JSON, "" if absent)
|
|
16
|
+
- repo (basename of cwd, same convention as
|
|
17
|
+
log-skill-usage.sh)
|
|
18
|
+
- ep_ai_standards_version (see standard doc: standards-version.json first,
|
|
19
|
+
this repo's own CHANGELOG.md as a fallback)
|
|
20
|
+
- tool_input_summary (top-level ARGUMENT KEY NAMES only -- sorted list
|
|
21
|
+
+ count -- never a single argument VALUE)
|
|
22
|
+
- result ("ok" / "error" / "unknown" from
|
|
23
|
+
tool_response.isError when it is a real bool)
|
|
24
|
+
- ts (wall clock at capture time, UTC ISO 8601)
|
|
25
|
+
|
|
26
|
+
NEVER captured, ever, by design (PII/secret risk -- see the standard's "PII / secret risk"
|
|
27
|
+
section): tool_input VALUES, tool_response body/output/stdout/stderr content, prompt or
|
|
28
|
+
response text, file contents. Only shapes (key names, booleans, counts) are recorded.
|
|
29
|
+
|
|
30
|
+
Fields the owner's spec asked for that are NOT written here because they are not honestly
|
|
31
|
+
capturable from this hook payload today (see the standard doc for the full PLANNED list
|
|
32
|
+
and why): related_mr_task_id, model, reason, associated_cost, triggering_skill,
|
|
33
|
+
skill_version, api_key_sha256, latency. None of them are written as null/empty guesses --
|
|
34
|
+
they are simply absent from the record.
|
|
35
|
+
|
|
36
|
+
Never blocks a session (hook mode): any error -> exit 0, write nothing. ASCII-only output
|
|
37
|
+
(json.dumps with ensure_ascii=True) regardless of what bytes arrive on stdin.
|
|
38
|
+
"""
|
|
39
|
+
import json
|
|
40
|
+
import os
|
|
41
|
+
import re
|
|
42
|
+
import sys
|
|
43
|
+
from datetime import datetime, timezone
|
|
44
|
+
from pathlib import Path
|
|
45
|
+
|
|
46
|
+
_MCP_TOOL_RE = re.compile(r"^mcp__")
|
|
47
|
+
_CHANGELOG_VERSION_RE = re.compile(r"^##\s*\[(\d+\.\d+\.\d+)\]", re.MULTILINE)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _iso_now():
|
|
51
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _iso_week(now=None):
|
|
55
|
+
now = now or datetime.now(timezone.utc)
|
|
56
|
+
y, w, _ = now.isocalendar()
|
|
57
|
+
return "%04d-W%02d" % (y, w)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _ep_ai_standards_version(root):
|
|
61
|
+
"""Best-effort, cheap, no network. See standard doc for the two sources tried."""
|
|
62
|
+
vf = root / ".ai-assistance" / "standards-version.json"
|
|
63
|
+
try:
|
|
64
|
+
if vf.is_file():
|
|
65
|
+
data = json.loads(vf.read_text(encoding="utf-8"))
|
|
66
|
+
v = data.get("ep-ai-standards-version")
|
|
67
|
+
if isinstance(v, str) and v:
|
|
68
|
+
return v
|
|
69
|
+
except Exception:
|
|
70
|
+
pass
|
|
71
|
+
changelog = root / "CHANGELOG.md"
|
|
72
|
+
try:
|
|
73
|
+
if changelog.is_file():
|
|
74
|
+
text = changelog.read_text(encoding="utf-8", errors="replace")
|
|
75
|
+
m = _CHANGELOG_VERSION_RE.search(text)
|
|
76
|
+
if m:
|
|
77
|
+
return m.group(1)
|
|
78
|
+
except Exception:
|
|
79
|
+
pass
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _tool_input_summary(tool_input):
|
|
84
|
+
"""Top-level argument KEY NAMES only -- never a value. Tolerates any shape."""
|
|
85
|
+
if not isinstance(tool_input, dict):
|
|
86
|
+
return {"n_keys": 0, "keys": []}
|
|
87
|
+
keys = sorted(str(k) for k in tool_input.keys())
|
|
88
|
+
return {"n_keys": len(keys), "keys": keys}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _result_status(tool_response):
|
|
92
|
+
if isinstance(tool_response, dict) and isinstance(tool_response.get("isError"), bool):
|
|
93
|
+
return "error" if tool_response["isError"] else "ok"
|
|
94
|
+
return "unknown"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def build_record(data):
|
|
98
|
+
"""Pure: given a parsed PostToolUse payload dict, return a record dict, or None if
|
|
99
|
+
this event is not an MCP tool call (nothing to log). Never raises -- any malformed
|
|
100
|
+
shape inside `data` degrades a field to its safe default instead of erroring out."""
|
|
101
|
+
if not isinstance(data, dict):
|
|
102
|
+
return None
|
|
103
|
+
tool_name = data.get("tool_name")
|
|
104
|
+
if not isinstance(tool_name, str) or not _MCP_TOOL_RE.match(tool_name):
|
|
105
|
+
return None # not an MCP tool call -- nothing for this ledger
|
|
106
|
+
|
|
107
|
+
cwd = data.get("cwd") or os.getcwd()
|
|
108
|
+
root = Path(cwd)
|
|
109
|
+
repo = os.path.basename(os.path.normpath(str(cwd))) if cwd else ""
|
|
110
|
+
session_id = data.get("session_id") or ""
|
|
111
|
+
|
|
112
|
+
record = {
|
|
113
|
+
"ts": _iso_now(),
|
|
114
|
+
"session_id": session_id,
|
|
115
|
+
"repo": repo,
|
|
116
|
+
"tool_name": tool_name,
|
|
117
|
+
"tool_input_summary": _tool_input_summary(data.get("tool_input")),
|
|
118
|
+
"result": _result_status(data.get("tool_response")),
|
|
119
|
+
}
|
|
120
|
+
version = _ep_ai_standards_version(root)
|
|
121
|
+
if version:
|
|
122
|
+
record["ep_ai_standards_version"] = version
|
|
123
|
+
return record
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _write_record(root, record):
|
|
127
|
+
week = _iso_week()
|
|
128
|
+
log_dir = root / ".ai-assistance" / "local" / "audit"
|
|
129
|
+
log_file = log_dir / ("access-%s.jsonl" % week)
|
|
130
|
+
log_dir.mkdir(parents=True, exist_ok=True)
|
|
131
|
+
with log_file.open("a", encoding="utf-8") as fh:
|
|
132
|
+
fh.write(json.dumps(record, ensure_ascii=True) + "\n")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def main(argv=None): # noqa: ARG001 -- no flags; argv kept for test-call symmetry
|
|
136
|
+
try:
|
|
137
|
+
raw = sys.stdin.buffer.read()
|
|
138
|
+
try:
|
|
139
|
+
text = raw.decode("utf-8")
|
|
140
|
+
except Exception:
|
|
141
|
+
text = raw.decode("cp1252", "replace")
|
|
142
|
+
data = json.loads(text) if text.strip() else {}
|
|
143
|
+
except Exception:
|
|
144
|
+
return 0 # malformed payload -> allow the tool, log nothing (fail open)
|
|
145
|
+
|
|
146
|
+
try:
|
|
147
|
+
record = build_record(data)
|
|
148
|
+
if record is None:
|
|
149
|
+
return 0
|
|
150
|
+
cwd = data.get("cwd") or os.getcwd()
|
|
151
|
+
_write_record(Path(cwd), record)
|
|
152
|
+
except Exception:
|
|
153
|
+
pass # an audit logger must never wedge or fail a session
|
|
154
|
+
return 0
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
if __name__ == "__main__":
|
|
158
|
+
sys.exit(main())
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# observe-session.sh -- thin Stop/SessionEnd wrapper for the procedural-memory OBSERVE stage.
|
|
3
|
+
# Opt-in (off by default): does nothing unless .ai-assistance/local/procedural-observe.enabled
|
|
4
|
+
# exists or PROCEDURAL_OBSERVE=1 is set. Enable/disable/status (human-run, one command):
|
|
5
|
+
# python .ai-assistance/scripts/observe_session.py --enable | --disable | --status
|
|
6
|
+
# Writes to the dedicated stream .ai-assistance/local/observe-<ISO-week>.jsonl (separate from
|
|
7
|
+
# the kpi/usage heartbeat stream). Never blocks a session (always exits 0).
|
|
8
|
+
# Logic + privacy allow-list + scrub-before-write live in observe_session.py.
|
|
9
|
+
# Standard: standards/workflows/procedural-memory.md (D2).
|
|
10
|
+
PY=python3
|
|
11
|
+
command -v python3 >/dev/null 2>&1 || PY=python
|
|
12
|
+
"$PY" "$(dirname "$0")/observe_session.py" || true
|
|
13
|
+
exit 0
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
observe_session.py -- Procedural-memory OBSERVE stage (Stop / SessionEnd hook).
|
|
4
|
+
|
|
5
|
+
Standard: standards/workflows/procedural-memory.md (D2 -- logging privacy).
|
|
6
|
+
|
|
7
|
+
Hook mode (no flags): reads the Stop-hook JSON from stdin, and IF the developer has opted
|
|
8
|
+
in, appends ONE low-sensitivity session-summary record to the DEDICATED observe stream
|
|
9
|
+
.ai-assistance/local/observe-<ISO-week>.jsonl. The observe stream is deliberately separate
|
|
10
|
+
from the KPI stream (.ai-assistance/local/kpi/usage-*.jsonl), which is dominated by
|
|
11
|
+
scheduler heartbeats -- mixing the two buried the behavioral signal (2026-07-20 inventory).
|
|
12
|
+
|
|
13
|
+
CLI mode (one-command opt-in management; run by a HUMAN, never by automation):
|
|
14
|
+
python observe_session.py --status # is Observe enabled here? what has it captured?
|
|
15
|
+
python observe_session.py --enable # create the consent flag (explicit human opt-in)
|
|
16
|
+
python observe_session.py --disable # remove the consent flag (streams are kept)
|
|
17
|
+
|
|
18
|
+
CONSENT (opt-in, off by default). Observe runs only when the developer has opted in, by
|
|
19
|
+
either:
|
|
20
|
+
- creating the marker file .ai-assistance/local/procedural-observe.enabled
|
|
21
|
+
(preferred; `--enable` above is the one-command way), or
|
|
22
|
+
- exporting PROCEDURAL_OBSERVE=1 in the environment.
|
|
23
|
+
If neither is set, this script writes nothing and exits 0. The distill pass applies the same
|
|
24
|
+
gate (refuses to read the log without consent). Nothing in the toolchain may create the flag
|
|
25
|
+
automatically -- enabling is a human decision (consent is load-bearing).
|
|
26
|
+
|
|
27
|
+
CAPTURE ALLOW-LIST (D2) -- only structured, low-sensitivity signal:
|
|
28
|
+
- tool/command NAMES (the verb, e.g. Edit, Bash, git) -- never arguments
|
|
29
|
+
- file paths + extensions touched, RELATIVE to the repo root -- never file contents/diffs
|
|
30
|
+
- coarse counts (n tools, n files) and a duration bucket
|
|
31
|
+
- session id, ISO week, timestamp
|
|
32
|
+
NEVER captured: command arguments, file contents/diffs, prompt/response text, the literal
|
|
33
|
+
text of corrections/redirects, secrets/tokens/credentials.
|
|
34
|
+
|
|
35
|
+
REDACTION: every string that could carry a path is passed through scripts/lib/scrub.py
|
|
36
|
+
BEFORE it is written. If scrub reports it could not run cleanly, the field is dropped.
|
|
37
|
+
|
|
38
|
+
Never blocks a session (hook mode): any error -> exit 0, write nothing.
|
|
39
|
+
"""
|
|
40
|
+
import argparse
|
|
41
|
+
import json
|
|
42
|
+
import os
|
|
43
|
+
import sys
|
|
44
|
+
from pathlib import Path
|
|
45
|
+
|
|
46
|
+
# ISO-week + timestamp without Date.now()-style nondeterminism concerns: this runs on the
|
|
47
|
+
# host at session end, so real wall-clock is correct and expected here.
|
|
48
|
+
from datetime import datetime, timezone
|
|
49
|
+
|
|
50
|
+
_ALLOWED_FILE_TOOLS = {"Read", "Edit", "Write", "NotebookEdit", "MultiEdit"}
|
|
51
|
+
_DUR_BUCKETS = [(300, "short"), (1800, "medium"), (7200, "long")] # secs -> label; else "xlong"
|
|
52
|
+
|
|
53
|
+
FLAG_NAME = "procedural-observe.enabled"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _iso_now():
|
|
57
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _iso_week():
|
|
61
|
+
y, w, _ = datetime.now(timezone.utc).isocalendar()
|
|
62
|
+
return f"{y}-W{w:02d}"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _load_scrub(root):
|
|
66
|
+
"""Import scrub() from scrub.py if available; else None.
|
|
67
|
+
|
|
68
|
+
Returns None when the redactor cannot be loaded. Privacy FAILS CLOSED: the caller must
|
|
69
|
+
write NOTHING when scrub is unavailable, rather than logging unscrubbed signal. (A prior
|
|
70
|
+
version fell back to an identity function -- that would log file paths, which can carry a
|
|
71
|
+
username, without redaction on a repo where scrub.py was not deployed.)"""
|
|
72
|
+
for cand in (root / ".ai-assistance" / "scripts", root / "scripts" / "lib"):
|
|
73
|
+
p = cand / "scrub.py"
|
|
74
|
+
if p.exists():
|
|
75
|
+
sys.path.insert(0, str(cand))
|
|
76
|
+
try:
|
|
77
|
+
import scrub # type: ignore
|
|
78
|
+
return scrub.scrub
|
|
79
|
+
except Exception:
|
|
80
|
+
pass
|
|
81
|
+
return None
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _consent(local_dir):
|
|
85
|
+
if os.environ.get("PROCEDURAL_OBSERVE", "").lower() in ("1", "true", "yes"):
|
|
86
|
+
return True
|
|
87
|
+
return (local_dir / FLAG_NAME).exists()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# --- opt-in management CLI (--enable / --disable / --status) -------------------
|
|
91
|
+
|
|
92
|
+
def _cli_enable(local_dir):
|
|
93
|
+
"""Create the consent flag. This command exists to make the HUMAN opt-in trivial;
|
|
94
|
+
it must only ever be run by a developer, never by a hook/scheduler/agent."""
|
|
95
|
+
flag = local_dir / FLAG_NAME
|
|
96
|
+
if flag.exists():
|
|
97
|
+
print(f"[observe] already enabled ({flag})")
|
|
98
|
+
return 0
|
|
99
|
+
local_dir.mkdir(parents=True, exist_ok=True)
|
|
100
|
+
flag.write_text(
|
|
101
|
+
"Procedural-memory Observe opt-in (standard D2 consent flag).\n"
|
|
102
|
+
"Created by observe_session.py --enable at the developer's request.\n"
|
|
103
|
+
f"Enabled: {_iso_now()}\n",
|
|
104
|
+
encoding="utf-8",
|
|
105
|
+
)
|
|
106
|
+
print(f"[observe] ENABLED -- consent flag created: {flag}")
|
|
107
|
+
print("[observe] capture allow-list (D2): tool names, repo-relative paths, coarse counts.")
|
|
108
|
+
print("[observe] never captured: arguments, file contents, diffs, prompt/response text.")
|
|
109
|
+
print("[observe] stream: .ai-assistance/local/observe-<ISO-week>.jsonl (machine-local, gitignored).")
|
|
110
|
+
print("[observe] disable any time: observe_session.py --disable (streams kept; delete them freely).")
|
|
111
|
+
return 0
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _cli_disable(local_dir):
|
|
115
|
+
flag = local_dir / FLAG_NAME
|
|
116
|
+
if not flag.exists():
|
|
117
|
+
print("[observe] already disabled (no consent flag)")
|
|
118
|
+
return 0
|
|
119
|
+
flag.unlink()
|
|
120
|
+
print(f"[observe] DISABLED -- consent flag removed: {flag}")
|
|
121
|
+
print("[observe] existing observe-*.jsonl streams were kept; delete them if you want them gone.")
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _cli_status(local_dir):
|
|
126
|
+
flag = local_dir / FLAG_NAME
|
|
127
|
+
env_on = os.environ.get("PROCEDURAL_OBSERVE", "").lower() in ("1", "true", "yes")
|
|
128
|
+
if flag.exists():
|
|
129
|
+
state = f"ENABLED (flag: {flag})"
|
|
130
|
+
elif env_on:
|
|
131
|
+
state = "ENABLED (PROCEDURAL_OBSERVE env var; no flag file)"
|
|
132
|
+
else:
|
|
133
|
+
state = "disabled (no consent flag; enable with: observe_session.py --enable)"
|
|
134
|
+
print(f"[observe] status: {state}")
|
|
135
|
+
streams = sorted(local_dir.glob("observe-*.jsonl")) if local_dir.is_dir() else []
|
|
136
|
+
if not streams:
|
|
137
|
+
print("[observe] streams: none captured yet")
|
|
138
|
+
for s in streams:
|
|
139
|
+
try:
|
|
140
|
+
n = sum(1 for ln in s.read_text(encoding="utf-8", errors="replace").splitlines() if ln.strip())
|
|
141
|
+
except OSError:
|
|
142
|
+
n = "?"
|
|
143
|
+
print(f"[observe] stream: {s.name} ({n} record(s))")
|
|
144
|
+
return 0
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _duration_bucket(secs):
|
|
148
|
+
try:
|
|
149
|
+
secs = float(secs)
|
|
150
|
+
except (TypeError, ValueError):
|
|
151
|
+
return "unknown"
|
|
152
|
+
for lim, label in _DUR_BUCKETS:
|
|
153
|
+
if secs <= lim:
|
|
154
|
+
return label
|
|
155
|
+
return "xlong"
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _extract_from_transcript(path, root, scrub):
|
|
159
|
+
"""Return (tool_counts, files) reading ONLY tool-use verbs + file paths from the
|
|
160
|
+
transcript. Never reads prompt/response text or tool arguments beyond a file path."""
|
|
161
|
+
tool_counts = {}
|
|
162
|
+
files = []
|
|
163
|
+
seen = set()
|
|
164
|
+
try:
|
|
165
|
+
lines = Path(path).read_text(encoding="utf-8", errors="replace").splitlines()
|
|
166
|
+
except OSError:
|
|
167
|
+
return tool_counts, files
|
|
168
|
+
for line in lines:
|
|
169
|
+
line = line.strip()
|
|
170
|
+
if not line:
|
|
171
|
+
continue
|
|
172
|
+
try:
|
|
173
|
+
rec = json.loads(line)
|
|
174
|
+
except ValueError:
|
|
175
|
+
continue
|
|
176
|
+
# Anthropic transcript: assistant messages carry content blocks; tool_use blocks
|
|
177
|
+
# have {"type":"tool_use","name":...,"input":{...}}. We read name + a file path only.
|
|
178
|
+
msg = rec.get("message") or rec
|
|
179
|
+
content = msg.get("content") if isinstance(msg, dict) else None
|
|
180
|
+
if not isinstance(content, list):
|
|
181
|
+
continue
|
|
182
|
+
for block in content:
|
|
183
|
+
if not isinstance(block, dict) or block.get("type") != "tool_use":
|
|
184
|
+
continue
|
|
185
|
+
name = block.get("name") or "unknown"
|
|
186
|
+
tool_counts[name] = tool_counts.get(name, 0) + 1
|
|
187
|
+
if name in _ALLOWED_FILE_TOOLS:
|
|
188
|
+
inp = block.get("input") or {}
|
|
189
|
+
fp = inp.get("file_path") or inp.get("path") or inp.get("notebook_path")
|
|
190
|
+
if not fp:
|
|
191
|
+
continue
|
|
192
|
+
# Make repo-relative; keep only path + extension, then scrub.
|
|
193
|
+
try:
|
|
194
|
+
rel = os.path.relpath(fp, str(root))
|
|
195
|
+
except ValueError:
|
|
196
|
+
rel = fp
|
|
197
|
+
rel = rel.replace("\\", "/")
|
|
198
|
+
scrubbed, findings = scrub(rel)
|
|
199
|
+
if any(f.get("type") == "error" for f in findings):
|
|
200
|
+
continue # scrub could not run cleanly -> drop
|
|
201
|
+
ext = os.path.splitext(scrubbed)[1].lower()
|
|
202
|
+
key = (ext, scrubbed)
|
|
203
|
+
if key not in seen:
|
|
204
|
+
seen.add(key)
|
|
205
|
+
files.append({"ext": ext, "path": scrubbed})
|
|
206
|
+
return tool_counts, files
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _hook_mode():
|
|
210
|
+
try:
|
|
211
|
+
raw = sys.stdin.read()
|
|
212
|
+
data = json.loads(raw) if raw.strip() else {}
|
|
213
|
+
except Exception:
|
|
214
|
+
return 0
|
|
215
|
+
|
|
216
|
+
cwd = data.get("cwd") or os.getcwd()
|
|
217
|
+
root = Path(cwd)
|
|
218
|
+
local_dir = root / ".ai-assistance" / "local"
|
|
219
|
+
if not _consent(local_dir):
|
|
220
|
+
return 0 # not opted in -> observe nothing
|
|
221
|
+
|
|
222
|
+
scrub = _load_scrub(root)
|
|
223
|
+
if scrub is None:
|
|
224
|
+
# Fail CLOSED: redactor unavailable -> do not log anything (never write unscrubbed).
|
|
225
|
+
sys.stderr.write("[observe] scrub.py not found; skipping session log (privacy fail-closed).\n")
|
|
226
|
+
return 0
|
|
227
|
+
session_id = data.get("session_id") or data.get("sessionId") or ""
|
|
228
|
+
transcript = data.get("transcript_path") or data.get("transcriptPath")
|
|
229
|
+
|
|
230
|
+
tool_counts, files = ({}, [])
|
|
231
|
+
if transcript:
|
|
232
|
+
tool_counts, files = _extract_from_transcript(transcript, root, scrub)
|
|
233
|
+
|
|
234
|
+
record = {
|
|
235
|
+
"component": "procedural/observe",
|
|
236
|
+
"action": "session-summary",
|
|
237
|
+
"ts": _iso_now(),
|
|
238
|
+
"session_id": session_id,
|
|
239
|
+
"week": _iso_week(),
|
|
240
|
+
"detail": {
|
|
241
|
+
"tools": tool_counts,
|
|
242
|
+
"files": files,
|
|
243
|
+
"n_tools": sum(tool_counts.values()),
|
|
244
|
+
"n_files": len(files),
|
|
245
|
+
"duration_bucket": _duration_bucket(data.get("duration_seconds")),
|
|
246
|
+
},
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
# Dedicated observe stream -- NOT the kpi/usage-*.jsonl heartbeat stream.
|
|
250
|
+
try:
|
|
251
|
+
local_dir.mkdir(parents=True, exist_ok=True)
|
|
252
|
+
out = local_dir / f"observe-{record['week']}.jsonl"
|
|
253
|
+
with out.open("a", encoding="utf-8") as fh:
|
|
254
|
+
fh.write(json.dumps(record) + "\n")
|
|
255
|
+
except OSError:
|
|
256
|
+
return 0
|
|
257
|
+
return 0
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def main(argv=None):
|
|
261
|
+
ap = argparse.ArgumentParser(
|
|
262
|
+
description="Procedural-memory Observe stage: Stop-hook logger (default) "
|
|
263
|
+
"+ human opt-in management (--enable/--disable/--status).")
|
|
264
|
+
ap.add_argument("--enable", action="store_true",
|
|
265
|
+
help="create the consent flag (HUMAN opt-in -- run this yourself)")
|
|
266
|
+
ap.add_argument("--disable", action="store_true", help="remove the consent flag")
|
|
267
|
+
ap.add_argument("--status", action="store_true",
|
|
268
|
+
help="show whether Observe is enabled and what it has captured")
|
|
269
|
+
ap.add_argument("--cwd", default=os.getcwd(),
|
|
270
|
+
help="repo root for CLI mode (default: current directory)")
|
|
271
|
+
a = ap.parse_args(argv)
|
|
272
|
+
|
|
273
|
+
if sum((a.enable, a.disable, a.status)) > 1:
|
|
274
|
+
print("[observe] pick ONE of --enable / --disable / --status")
|
|
275
|
+
return 2
|
|
276
|
+
local_dir = Path(a.cwd) / ".ai-assistance" / "local"
|
|
277
|
+
if a.enable:
|
|
278
|
+
return _cli_enable(local_dir)
|
|
279
|
+
if a.disable:
|
|
280
|
+
return _cli_disable(local_dir)
|
|
281
|
+
if a.status:
|
|
282
|
+
return _cli_status(local_dir)
|
|
283
|
+
return _hook_mode()
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
if __name__ == "__main__":
|
|
287
|
+
sys.exit(main())
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# protect-host-paths.sh -- PreToolUse hook (matcher: Bash) that blocks commands
|
|
3
|
+
# writing to / deleting host-OS system paths. Defense-in-depth: this decision is
|
|
4
|
+
# enforced regardless of permission mode, so it holds even under
|
|
5
|
+
# --dangerously-skip-permissions (where the deny-list prompts are off).
|
|
6
|
+
# Standard: standards/tooling/host-protection.md (Layer 4).
|
|
7
|
+
#
|
|
8
|
+
# Emits a structured deny decision on stdout + exit 0 (NOT exit 2, whose behavior
|
|
9
|
+
# in dangerous mode is unclear). Uses python3 (a documented prereq) -- no jq needed.
|
|
10
|
+
# Fails OPEN on any internal error (a guard must never wedge the session): if it
|
|
11
|
+
# cannot parse, it allows and lets the normal permission flow decide.
|
|
12
|
+
#
|
|
13
|
+
# NOTE: uses `python -c` (NOT a `<<HEREDOC`) so the hook's stdin -- the PreToolUse
|
|
14
|
+
# JSON -- reaches python. A heredoc would replace stdin with the script itself.
|
|
15
|
+
PY=python3
|
|
16
|
+
command -v python3 >/dev/null 2>&1 || PY=python
|
|
17
|
+
exec "$PY" -c '
|
|
18
|
+
import sys, json, re
|
|
19
|
+
try:
|
|
20
|
+
data = json.loads(sys.stdin.read())
|
|
21
|
+
except Exception:
|
|
22
|
+
sys.exit(0) # cannot parse -> allow (fail open); normal permissions still apply
|
|
23
|
+
if data.get("tool_name") != "Bash":
|
|
24
|
+
sys.exit(0)
|
|
25
|
+
cmd = (data.get("tool_input") or {}).get("command", "") or ""
|
|
26
|
+
|
|
27
|
+
verbs = r"(?:\brm\b|\brmdir\b|\bmv\b|\bcp\b|\bdd\b|\btee\b|\btruncate\b|\bchown\b|\bchmod\b|>>?|\bmkfs\S*|\bdiskutil\b)"
|
|
28
|
+
|
|
29
|
+
# Anchor: a root only counts as a true filesystem root when the leading `/`, `~`,
|
|
30
|
+
# or drive-letter is at a path boundary -- start-of-token, whitespace, quote, `=`,
|
|
31
|
+
# or start-of-string -- and NOT preceded by a path character. Without this, an
|
|
32
|
+
# unanchored `/lib` etc. matches mid-path substrings like `scripts/lib/x.py` or
|
|
33
|
+
# `tmp/var/x`, producing false positives (proven 2026-07-22). The negative
|
|
34
|
+
# lookbehind rules out being preceded by a word char, `.`, `/`, `\`, or `-`.
|
|
35
|
+
BOUNDARY = r"(?<![\w./\\-])"
|
|
36
|
+
roots = [
|
|
37
|
+
BOUNDARY + r"/System(?:/|\b)", BOUNDARY + r"/Library(?:/|\b)", BOUNDARY + r"~/Library(?:/|\b)",
|
|
38
|
+
BOUNDARY + r"/usr(?:/|\b)", BOUNDARY + r"/bin(?:/|\b)", BOUNDARY + r"/sbin(?:/|\b)", BOUNDARY + r"/etc(?:/|\b)",
|
|
39
|
+
BOUNDARY + r"/boot(?:/|\b)", BOUNDARY + r"/lib(?:/|\b)", BOUNDARY + r"/opt(?:/|\b)", BOUNDARY + r"/var(?:/|\b)",
|
|
40
|
+
BOUNDARY + r"[A-Za-z]:\\+Windows", BOUNDARY + r"[A-Za-z]:\\+Program Files",
|
|
41
|
+
BOUNDARY + r"/[A-Za-z]/Windows", BOUNDARY + r"/[A-Za-z]/Program Files",
|
|
42
|
+
# User-level persistence / credential targets (literal ~ form; shell-expanded forms
|
|
43
|
+
# can still bypass -- this is a heuristic layer, not a sandbox). NOT ~/.config (too broad).
|
|
44
|
+
BOUNDARY + r"~/\.ssh(?:/|\b)", BOUNDARY + r"~/\.aws(?:/|\b)", BOUNDARY + r"~/\.gnupg(?:/|\b)",
|
|
45
|
+
BOUNDARY + r"~/\.bashrc\b", BOUNDARY + r"~/\.zshrc\b", BOUNDARY + r"~/\.bash_profile\b", BOUNDARY + r"~/\.zprofile\b",
|
|
46
|
+
BOUNDARY + r"~/\.profile\b", BOUNDARY + r"~/\.gitconfig\b",
|
|
47
|
+
]
|
|
48
|
+
danger = [
|
|
49
|
+
r"\brm\s+-rf\s+/(?:\s|$)",
|
|
50
|
+
r"\brm\s+-rf\s+~(?:/|\s|$)",
|
|
51
|
+
r"\bsudo\b",
|
|
52
|
+
r":\(\)\s*\{",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
# Strip commit-message DATA before scanning for danger verbs -- a `git commit
|
|
56
|
+
# -m "..."` (or `--message`) argument, or a message piped in via a quoted
|
|
57
|
+
# heredoc, is text handed to git as the commit message; it is never executed
|
|
58
|
+
# by the shell. Without this, a message merely *mentioning* a danger word --
|
|
59
|
+
# e.g. `git commit -m "docs: mention sudo usage"` -- was denied even though
|
|
60
|
+
# nothing runs. Proven 2026-07-22.
|
|
61
|
+
#
|
|
62
|
+
# Kept conservative on purpose (never widen this to "any quoted string" --
|
|
63
|
+
# a `bash -c "..."` payload IS executed and must keep scanning):
|
|
64
|
+
# - single-quoted -m/--message values: single quotes disable ALL shell
|
|
65
|
+
# expansion, so the content can never itself be executed code -> always
|
|
66
|
+
# safe to strip.
|
|
67
|
+
# - double-quoted -m/--message values: only stripped if they contain no
|
|
68
|
+
# `$(` / backtick command substitution (plain literal text), OR are
|
|
69
|
+
# exactly the standard multi-line idiom, e.g. $(cat <<EOF with a QUOTED
|
|
70
|
+
# delimiter ... EOF) -- in which case only the inert heredoc body is
|
|
71
|
+
# stripped, so something like -m "$(sudo rm -rf /)" (a real, executed
|
|
72
|
+
# command substitution smuggled into a -m value) is left untouched and
|
|
73
|
+
# still blocked.
|
|
74
|
+
# - git commit ... -F - <<EOF (with a QUOTED delimiter) ... EOF: message
|
|
75
|
+
# read directly from a quoted (non-expanding) heredoc attached to
|
|
76
|
+
# `git commit` -- also inert text. An UNQUOTED heredoc delimiter (plain
|
|
77
|
+
# <<EOF, no quotes around it) is deliberately left alone: its body still
|
|
78
|
+
# undergoes shell expansion, so it is not guaranteed to be inert.
|
|
79
|
+
def _strip_commit_message_data(cmd):
|
|
80
|
+
out = cmd
|
|
81
|
+
|
|
82
|
+
out = re.sub(
|
|
83
|
+
r"(-m|--message)(=|\s+)\x27[^\x27]*\x27",
|
|
84
|
+
lambda m: m.group(1) + m.group(2) + "\x27\x27",
|
|
85
|
+
out,
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def _strip_dq_message(m):
|
|
89
|
+
prefix, sep, body = m.group(1), m.group(2), m.group(3)
|
|
90
|
+
idiom = re.match(
|
|
91
|
+
r"^\$\(\s*cat\s+<<[-~]?\s*([\x27\"])(\w+)\1\s*\r?\n(.*?)\r?\n[ \t]*\2\s*\)\s*$",
|
|
92
|
+
body, re.DOTALL,
|
|
93
|
+
)
|
|
94
|
+
if idiom:
|
|
95
|
+
quote, delim = idiom.group(1), idiom.group(2)
|
|
96
|
+
return prefix + sep + "\"$(cat <<" + quote + delim + quote + "\n\n" + delim + ")\""
|
|
97
|
+
if "$(" not in body and "`" not in body:
|
|
98
|
+
return prefix + sep + "\"\""
|
|
99
|
+
return m.group(0) # contains a real substitution -- leave it, keep scanning
|
|
100
|
+
|
|
101
|
+
out = re.sub(
|
|
102
|
+
r"(-m|--message)(=|\s+)\"((?:\\.|[^\"\\])*)\"",
|
|
103
|
+
_strip_dq_message,
|
|
104
|
+
out,
|
|
105
|
+
flags=re.DOTALL,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
out = re.sub(
|
|
109
|
+
r"(git\s+commit\b[^\n]*<<[-~]?\s*([\x27\"])(\w+)\2[^\n]*\n)(.*?)(\n[ \t]*\3\b)",
|
|
110
|
+
lambda m: m.group(1) + m.group(5),
|
|
111
|
+
out,
|
|
112
|
+
flags=re.DOTALL,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
return out
|
|
116
|
+
|
|
117
|
+
def blocked(reason):
|
|
118
|
+
print(json.dumps({"hookSpecificOutput": {
|
|
119
|
+
"hookEventName": "PreToolUse",
|
|
120
|
+
"permissionDecision": "deny",
|
|
121
|
+
"permissionDecisionReason": reason}}))
|
|
122
|
+
sys.exit(0)
|
|
123
|
+
|
|
124
|
+
scan_cmd = _strip_commit_message_data(cmd)
|
|
125
|
+
|
|
126
|
+
for d in danger:
|
|
127
|
+
if re.search(d, scan_cmd):
|
|
128
|
+
blocked("Blocked: destructive/privileged command (host-protection hook).")
|
|
129
|
+
|
|
130
|
+
root_re = "(?:" + "|".join(roots) + ")"
|
|
131
|
+
if re.search(verbs + r"[^\n]*" + root_re, scan_cmd) or re.search(root_re + r"[^\n]*" + verbs, scan_cmd):
|
|
132
|
+
blocked("Blocked: command writes/deletes a host-OS system path (host-protection hook).")
|
|
133
|
+
|
|
134
|
+
sys.exit(0)
|
|
135
|
+
'
|