switchroom 0.20.10 → 0.20.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +68 -3
- package/dist/auth-broker/index.js +211 -46
- package/dist/cli/notion-write-pretool.mjs +68 -3
- package/dist/cli/self-improve-apply-guard-pretool.mjs +357 -92
- package/dist/cli/self-improve-stop.mjs +889 -7
- package/dist/cli/skill-validate-pretool.mjs +82 -3
- package/dist/cli/switchroom.js +5063 -2961
- package/dist/host-control/main.js +93 -27
- package/dist/vault/approvals/kernel-server.js +92 -26
- package/dist/vault/broker/server.js +92 -26
- package/examples/personal-google-workspace-mcp/compose.yaml +1 -1
- package/package.json +1 -1
- package/profiles/_shared/agent-self-service.md.hbs +15 -22
- package/profiles/_shared/delegation-golden-rule.md.hbs +1 -1
- package/profiles/_shared/dev-protocol.md.hbs +1 -1
- package/profiles/_shared/execution-discipline.md.hbs +4 -4
- package/profiles/_shared/vault-protocol.md.hbs +2 -18
- package/profiles/default/CLAUDE.md.hbs +3 -5
- package/skills/switchroom-architecture/telegram.md +0 -1
- package/skills/switchroom-cli/SKILL.md +0 -1
- package/telegram-plugin/README.md +2 -11
- package/telegram-plugin/auto-fallback-fleet.ts +37 -2
- package/telegram-plugin/bridge/bridge.ts +0 -12
- package/telegram-plugin/chat-lock.ts +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +0 -12
- package/telegram-plugin/dist/gateway/gateway.js +1454 -976
- package/telegram-plugin/dist/server.js +0 -12
- package/telegram-plugin/fallback-card-collapse.ts +1 -0
- package/telegram-plugin/gateway/auth-command.ts +11 -1
- package/telegram-plugin/gateway/callback-query-handlers.ts +100 -0
- package/telegram-plugin/gateway/eval-case-proposal-card.ts +86 -0
- package/telegram-plugin/gateway/fleet-fallback-notice-cooldown.test.ts +74 -0
- package/telegram-plugin/gateway/fleet-fallback-notice-cooldown.ts +71 -0
- package/telegram-plugin/gateway/gateway.ts +109 -149
- package/telegram-plugin/gateway/ipc-protocol.ts +43 -0
- package/telegram-plugin/gateway/ipc-server.ts +28 -0
- package/telegram-plugin/gateway/liveness-wiring.ts +6 -1
- package/telegram-plugin/gateway/narrative-lane.ts +33 -2
- package/telegram-plugin/gateway/privacy-reset.test.ts +216 -0
- package/telegram-plugin/gateway/privacy-reset.ts +87 -0
- package/telegram-plugin/gateway/privacy-state.test.ts +165 -0
- package/telegram-plugin/gateway/privacy-state.ts +206 -0
- package/telegram-plugin/gateway/self-improve-proposal-wiring.ts +176 -0
- package/telegram-plugin/gateway/stale-pin-sweep-wiring.ts +24 -14
- package/telegram-plugin/gateway/stale-pin-sweep.test.ts +123 -26
- package/telegram-plugin/gateway/stale-pin-sweep.ts +52 -35
- package/telegram-plugin/gateway/status-pin-store.ts +10 -9
- package/telegram-plugin/gateway/stream-render.ts +4 -4
- package/telegram-plugin/gateway/throttle-tier-wiring.ts +15 -4
- package/telegram-plugin/gateway/turn-record-status.ts +32 -1
- package/telegram-plugin/hooks/hooks.json +13 -12
- package/telegram-plugin/hooks/narration-classify.mjs +1 -2
- package/telegram-plugin/hooks/silent-end-scan.mjs +1 -1
- package/telegram-plugin/slot-banner-driver.ts +42 -5
- package/telegram-plugin/status-pin.ts +2 -5
- package/telegram-plugin/tests/auto-fallback-fleet.test.ts +24 -0
- package/telegram-plugin/tests/backstop-exactly-once.test.ts +8 -2
- package/telegram-plugin/tests/framework-fallback-duration-guard.test.ts +125 -0
- package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +2 -0
- package/telegram-plugin/tests/narrative-lane-golden.test.ts +97 -0
- package/telegram-plugin/tests/pin-message-tool-retired.test.ts +64 -0
- package/telegram-plugin/tests/privacy-reset-call-sites.test.ts +120 -0
- package/telegram-plugin/tests/status-pin-boot-recovery.test.ts +38 -0
- package/telegram-plugin/tests/status-pin-store.test.ts +25 -0
- package/telegram-plugin/tests/throttle-tier.test.ts +16 -0
- package/telegram-plugin/tests/turn-flush-safety.test.ts +67 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +40 -1
- package/telegram-plugin/throttle-tier.ts +12 -3
- package/telegram-plugin/turn-flush-safety.ts +97 -0
- package/telegram-plugin/worker-activity-feed.ts +1 -1
- package/vendor/hindsight-memory/scripts/recall.py +140 -0
- package/vendor/hindsight-memory/scripts/retain.py +306 -0
- package/vendor/hindsight-memory/scripts/subagent_retain.py +29 -1
- package/vendor/hindsight-memory/scripts/tests/test_private_mode.py +415 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_latency_instrumentation.py +277 -0
- package/vendor/hindsight-memory/scripts/tests/test_self_improve_correction_tag.py +167 -0
|
@@ -800,7 +800,7 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
|
|
|
800
800
|
const floodWaitRemainingMs = opts.floodWaitRemainingMs ?? (() => 0)
|
|
801
801
|
const minEditInterval = opts.minEditIntervalMs ?? 2500
|
|
802
802
|
const elapsedRefreshMs = Math.max(minEditInterval, Math.floor(opts.elapsedRefreshMs ?? 15000))
|
|
803
|
-
const firstPaintMin = opts.firstPaintMinMs ??
|
|
803
|
+
const firstPaintMin = opts.firstPaintMinMs ?? 4000
|
|
804
804
|
const heartbeatTickMs = opts.heartbeatTickMs ?? 6000
|
|
805
805
|
const maxRows = Math.max(1, Math.floor(opts.maxRows ?? 8))
|
|
806
806
|
const staleWorkerTtlMs = Math.max(1, Math.floor(opts.staleWorkerTtlMs ?? 50 * 60_000))
|
|
@@ -1037,6 +1037,116 @@ def _write_recall_log(entry: dict) -> None:
|
|
|
1037
1037
|
pass
|
|
1038
1038
|
|
|
1039
1039
|
|
|
1040
|
+
# Switchroom recall-latency instrumentation — full-hook wall-time.
|
|
1041
|
+
#
|
|
1042
|
+
# recall.py is the UserPromptSubmit hook that sits in front of EVERY reply
|
|
1043
|
+
# (pre-first-token), yet it was the one hook with no wall-time record: it is a
|
|
1044
|
+
# DIRECT Claude Code plugin hook (hooks/hooks.json), NOT wrapped by
|
|
1045
|
+
# bin/run-hook.sh, so it never emitted a `hook-timings-<Ddd>.log` row the way
|
|
1046
|
+
# every wrapped hook does, and `recall_log.jsonl` measured only the recall
|
|
1047
|
+
# critical path (`total_elapsed_ms`, from `recall_start_monotonic`) — never the
|
|
1048
|
+
# hook's own import + stdin + cache-check + gate overhead.
|
|
1049
|
+
#
|
|
1050
|
+
# Routing it through run-hook.sh was rejected as the mechanism: the vendored
|
|
1051
|
+
# hooks.json is re-copied verbatim into every agent's plugin dir on `switchroom
|
|
1052
|
+
# apply`, and run-hook.sh lives in the switchroom repo's bin/, not under
|
|
1053
|
+
# CLAUDE_PLUGIN_ROOT — coupling the vendor snapshot to switchroom's bin layout is
|
|
1054
|
+
# fragile, and the os._exit(0) fast-path (which skips atexit / thread-join to
|
|
1055
|
+
# return control the instant stdout is flushed) would have to be reconciled with
|
|
1056
|
+
# the wrapper. Instead the hook emits the SAME JSON line, into the SAME weekday-
|
|
1057
|
+
# ring file, honouring the SAME env knobs, from inside the process at exit — so
|
|
1058
|
+
# `grep duration_ms hook-timings-*.log` sees this hook next to every other one.
|
|
1059
|
+
HOOK_TIMING_SOURCE = "hook:hindsight-recall"
|
|
1060
|
+
HOOK_TIMING_CODE = "recall.py"
|
|
1061
|
+
|
|
1062
|
+
|
|
1063
|
+
def _hook_duration_ms() -> int:
|
|
1064
|
+
"""Milliseconds since this hook process began doing work.
|
|
1065
|
+
|
|
1066
|
+
Anchored at import start (`_IMPORT_START_MONOTONIC`, taken before any of this
|
|
1067
|
+
hook's own imports ran — recall.py line ~48), so the number spans the WHOLE
|
|
1068
|
+
hook: dependency import, the stdin read, the cache check, the gate
|
|
1069
|
+
short-circuits, and — when it got that far — the recall round-trips. That is
|
|
1070
|
+
deliberately WIDER than `total_elapsed_ms` (the recall critical path only,
|
|
1071
|
+
measured from `recall_start_monotonic`): `duration_ms - total_elapsed_ms` is
|
|
1072
|
+
therefore the pre-recall LOCAL overhead, and the per-bank
|
|
1073
|
+
`bank_timings[].elapsed_ms` remain the Hindsight server round-trips — so the
|
|
1074
|
+
log already separates server time from local overhead without a new field.
|
|
1075
|
+
"""
|
|
1076
|
+
return int((time.monotonic() - _IMPORT_START_MONOTONIC) * 1000)
|
|
1077
|
+
|
|
1078
|
+
|
|
1079
|
+
def _emit_hook_timing_log(duration_ms: int, status: int) -> None:
|
|
1080
|
+
"""Append one timing line to `hook-timings-<Ddd>.log`, matching run-hook.sh.
|
|
1081
|
+
|
|
1082
|
+
STDOUT-SAFE by construction: writes only to a log file, NEVER to the hook's
|
|
1083
|
+
stdout contract that Claude Code consumes. Failure-tolerant — any error is
|
|
1084
|
+
swallowed so instrumentation can never take the hook down. Honours the same
|
|
1085
|
+
env knobs as bin/run-hook.sh (`SWITCHROOM_HOOK_TIMING`,
|
|
1086
|
+
`SWITCHROOM_HOOK_TIMING_DIR`, `SWITCHROOM_HOOK_TIMING_MIN_MS`) and reproduces
|
|
1087
|
+
its 7-day self-truncating weekday ring and JSON line shape exactly, so a
|
|
1088
|
+
consumer cannot tell this row apart from a wrapped hook's.
|
|
1089
|
+
|
|
1090
|
+
NOTE on the 12s ceiling: if Claude Code kills this hook at the
|
|
1091
|
+
UserPromptSubmit timeout, the process is terminated before any exit path runs
|
|
1092
|
+
and NO timing line (nor recall_log row) is written — the MISSING line is the
|
|
1093
|
+
breach signal, consistent with the two-signal baseline documented at the
|
|
1094
|
+
recall_log write. This records every invocation that returns under budget,
|
|
1095
|
+
including a fully-degraded/timed-out recall that still exits cleanly.
|
|
1096
|
+
"""
|
|
1097
|
+
try:
|
|
1098
|
+
if os.environ.get("SWITCHROOM_HOOK_TIMING", "1") == "0":
|
|
1099
|
+
return
|
|
1100
|
+
timing_dir = os.environ.get("SWITCHROOM_HOOK_TIMING_DIR") or os.environ.get(
|
|
1101
|
+
"TELEGRAM_STATE_DIR", ""
|
|
1102
|
+
)
|
|
1103
|
+
if not timing_dir or not os.path.isdir(timing_dir):
|
|
1104
|
+
return
|
|
1105
|
+
try:
|
|
1106
|
+
duration_ms = int(duration_ms)
|
|
1107
|
+
except (TypeError, ValueError):
|
|
1108
|
+
return
|
|
1109
|
+
if duration_ms < 0:
|
|
1110
|
+
duration_ms = 0
|
|
1111
|
+
try:
|
|
1112
|
+
min_ms = int(os.environ.get("SWITCHROOM_HOOK_TIMING_MIN_MS", "0"))
|
|
1113
|
+
except (TypeError, ValueError):
|
|
1114
|
+
min_ms = 0
|
|
1115
|
+
if duration_ms < min_ms:
|
|
1116
|
+
return
|
|
1117
|
+
# Local wall clock, matching run-hook.sh's builtin `%(...)T` formatting so
|
|
1118
|
+
# both writers agree on which weekday-ring file today lands in.
|
|
1119
|
+
now = time.localtime()
|
|
1120
|
+
today = time.strftime("%Y-%m-%d", now)
|
|
1121
|
+
dow = time.strftime("%a", now)
|
|
1122
|
+
ts = time.strftime("%Y-%m-%dT%H:%M:%S%z", now)
|
|
1123
|
+
logfile = os.path.join(timing_dir, f"hook-timings-{dow}.log")
|
|
1124
|
+
# 7-day self-truncating ring: the weekday-named file is either today's or
|
|
1125
|
+
# exactly a week stale. If its first line does not carry today's date,
|
|
1126
|
+
# reset it before appending (same rule as run-hook.sh).
|
|
1127
|
+
try:
|
|
1128
|
+
if os.path.getsize(logfile) > 0:
|
|
1129
|
+
with open(logfile, encoding="utf-8") as f:
|
|
1130
|
+
first = f.readline()
|
|
1131
|
+
if f'"date":"{today}"' not in first:
|
|
1132
|
+
open(logfile, "w", encoding="utf-8").close()
|
|
1133
|
+
except OSError:
|
|
1134
|
+
pass
|
|
1135
|
+
# HOOK_TIMING_SOURCE / HOOK_TIMING_CODE are fixed constants with no JSON
|
|
1136
|
+
# metacharacters, so no escape pass is needed (unlike run-hook.sh, whose
|
|
1137
|
+
# source/code are caller-supplied).
|
|
1138
|
+
line = (
|
|
1139
|
+
'{"ts":"%s","date":"%s","source":"%s","code":"%s",'
|
|
1140
|
+
'"duration_ms":%d,"status":%d}\n'
|
|
1141
|
+
% (ts, today, HOOK_TIMING_SOURCE, HOOK_TIMING_CODE, duration_ms, status)
|
|
1142
|
+
)
|
|
1143
|
+
with open(logfile, "a", encoding="utf-8") as f:
|
|
1144
|
+
f.write(line)
|
|
1145
|
+
except Exception:
|
|
1146
|
+
# Instrumentation is never load-bearing — swallow everything.
|
|
1147
|
+
pass
|
|
1148
|
+
|
|
1149
|
+
|
|
1040
1150
|
def _read_transcript_lines(transcript_path: str, tail_bytes: int):
|
|
1041
1151
|
"""Yield the transcript's trailing lines, byte-bounded.
|
|
1042
1152
|
|
|
@@ -1824,6 +1934,12 @@ def main():
|
|
|
1824
1934
|
# timed out"; `deadline_hit is None` means "no banks ran"
|
|
1825
1935
|
# (review finding 3).
|
|
1826
1936
|
"total_elapsed_ms": None,
|
|
1937
|
+
# Switchroom recall-latency instrumentation — full-hook wall time
|
|
1938
|
+
# (import + stdin + cache check), measured to this log write. A
|
|
1939
|
+
# cache hit issues no bank HTTP, so `total_elapsed_ms` is None and
|
|
1940
|
+
# this is pure local overhead — the cheap path this cache exists
|
|
1941
|
+
# to create, now visible per-row.
|
|
1942
|
+
"duration_ms": _hook_duration_ms(),
|
|
1827
1943
|
"directives_elapsed_ms": None,
|
|
1828
1944
|
"bank_timings": [],
|
|
1829
1945
|
"deadline_hit": None,
|
|
@@ -2542,6 +2658,16 @@ def main():
|
|
|
2542
2658
|
# parallelism change measures against; the 17-26% figure it replaces is
|
|
2543
2659
|
# the stale 2026-05-24 pre-fix audit.
|
|
2544
2660
|
"total_elapsed_ms": int((time.monotonic() - recall_start_monotonic) * 1000),
|
|
2661
|
+
# Switchroom recall-latency instrumentation — FULL-HOOK wall time to this
|
|
2662
|
+
# log write: dependency import + stdin read + cache check + the gate
|
|
2663
|
+
# short-circuits + the whole recall critical path. `total_elapsed_ms`
|
|
2664
|
+
# above is the recall critical path ONLY, so `duration_ms -
|
|
2665
|
+
# total_elapsed_ms` is the pre-recall LOCAL overhead this row could not
|
|
2666
|
+
# see before, while `bank_timings[].elapsed_ms` stay the server round-
|
|
2667
|
+
# trips — the log now separates server time from local overhead. Written
|
|
2668
|
+
# here (with the rest of the row, before the empty-block return) so even a
|
|
2669
|
+
# fully-timed-out / degraded recall that reaches this line gets a duration.
|
|
2670
|
+
"duration_ms": _hook_duration_ms(),
|
|
2545
2671
|
"directives_elapsed_ms": directives_elapsed_ms,
|
|
2546
2672
|
"bank_timings": bank_timings,
|
|
2547
2673
|
# Switchroom hindsight-leverage A3 — FINALIZED `deadline_hit` semantics
|
|
@@ -2795,6 +2921,12 @@ if __name__ == "__main__":
|
|
|
2795
2921
|
sys.stdout.flush()
|
|
2796
2922
|
except Exception:
|
|
2797
2923
|
pass
|
|
2924
|
+
# Switchroom recall-latency instrumentation — emit the full-hook timing
|
|
2925
|
+
# line AFTER stdout is flushed (Claude Code already has the bytes) and
|
|
2926
|
+
# BEFORE os._exit, which skips atexit and would otherwise drop it. Adds
|
|
2927
|
+
# ~0.5ms (one file append) before the process exits — the same budget
|
|
2928
|
+
# run-hook.sh spends per wrapped hook. Stdout is untouched.
|
|
2929
|
+
_emit_hook_timing_log(_hook_duration_ms(), 0)
|
|
2798
2930
|
os._exit(0)
|
|
2799
2931
|
except Exception as e:
|
|
2800
2932
|
# Switchroom #1070 (redo per #1085 review).
|
|
@@ -2844,6 +2976,10 @@ if __name__ == "__main__":
|
|
|
2844
2976
|
import traceback
|
|
2845
2977
|
|
|
2846
2978
|
traceback.print_exc(file=sys.stderr)
|
|
2979
|
+
# Instrumentation: record the failed invocation's wall time too
|
|
2980
|
+
# (status 2, the debug-mode block behaviour) so a crash-looping
|
|
2981
|
+
# hook is visible in the timing log, not just a silent gap.
|
|
2982
|
+
_emit_hook_timing_log(_hook_duration_ms(), 2)
|
|
2847
2983
|
# Debug-mode exit 2 is intentional and unchanged —
|
|
2848
2984
|
# operators with HINDSIGHT_DEBUG=1 are chasing a broken
|
|
2849
2985
|
# recall and want the hook to surface its failure.
|
|
@@ -2853,4 +2989,8 @@ if __name__ == "__main__":
|
|
|
2853
2989
|
# 0 with no stdout (agent's prompt assembly treats absent
|
|
2854
2990
|
# additionalContext as "no recall this turn").
|
|
2855
2991
|
_record_issue_safely(_detail, _class)
|
|
2992
|
+
# Instrumentation: the non-debug exit code is 0 (the safe-empty stdout
|
|
2993
|
+
# posture), so log status 0 — the accompanying issue-sink record is where
|
|
2994
|
+
# the failure detail lives; this row just makes the latency observable.
|
|
2995
|
+
_emit_hook_timing_log(_hook_duration_ms(), 0)
|
|
2856
2996
|
sys.exit(0)
|
|
@@ -22,9 +22,21 @@ import json
|
|
|
22
22
|
import os
|
|
23
23
|
import sys
|
|
24
24
|
import time
|
|
25
|
+
from datetime import datetime, timezone
|
|
25
26
|
|
|
26
27
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
27
28
|
|
|
29
|
+
# Switchroom self-improve PR4 (slice 4a) — per-turn correction tagging.
|
|
30
|
+
# The self-improve Stop hook (a SEPARATE process — src/cli/self-improve-stop.ts)
|
|
31
|
+
# drops SELF_IMPROVE_CORRECTION_PENDING_FILE into the shared agent state dir when
|
|
32
|
+
# the deterministic gate fires on an operator-correction. This hook reads-and-
|
|
33
|
+
# clears it and stamps SELF_IMPROVE_CORRECTION_TAG on that turn's retain so PR5's
|
|
34
|
+
# failure-synthesis cron can recall correction turns cheaply. Both constants are
|
|
35
|
+
# pinned to their TS source (src/memory/hindsight-retain-provenance.ts) by
|
|
36
|
+
# tests/scaffold.retain-provenance.test.ts — they must not drift.
|
|
37
|
+
SELF_IMPROVE_CORRECTION_PENDING_FILE = "self-improve-correction-pending"
|
|
38
|
+
SELF_IMPROVE_CORRECTION_TAG = "self-improve:correction"
|
|
39
|
+
|
|
28
40
|
from lib import watermark
|
|
29
41
|
from lib.bank import derive_bank_id, ensure_bank_mission
|
|
30
42
|
from lib.client import HindsightClient
|
|
@@ -38,6 +50,242 @@ from lib.pacing import inflight_lock
|
|
|
38
50
|
from lib.state import increment_turn_count, track_retention
|
|
39
51
|
|
|
40
52
|
|
|
53
|
+
def _self_improve_state_dir() -> str:
|
|
54
|
+
"""State dir the self-improve Stop hook drops its correction sentinel into.
|
|
55
|
+
|
|
56
|
+
Mirrors ``resolveStateDir()`` in ``src/cli/self-improve-stop.ts``:
|
|
57
|
+
``TELEGRAM_STATE_DIR`` when set, else ``~/.claude/channels/telegram``. Both
|
|
58
|
+
hooks run in the same agent process env, so they agree on this path.
|
|
59
|
+
"""
|
|
60
|
+
env = os.environ.get("TELEGRAM_STATE_DIR")
|
|
61
|
+
if env:
|
|
62
|
+
return env
|
|
63
|
+
return os.path.join(os.path.expanduser("~"), ".claude", "channels", "telegram")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def read_and_clear_correction_pending(state_dir: str | None = None) -> list:
|
|
67
|
+
"""Read-and-clear the per-turn operator-correction sentinel (PR4 slice 4a).
|
|
68
|
+
|
|
69
|
+
Returns ``[SELF_IMPROVE_CORRECTION_TAG]`` when the sentinel is present (and
|
|
70
|
+
REMOVES it, so exactly one retain carries the tag — read-once), else ``[]``.
|
|
71
|
+
Called from ``run_retain`` only (the live Stop-hook entry), NOT from the
|
|
72
|
+
network-free ``build_retain_payload`` seam, so backfill / reconcile /
|
|
73
|
+
subagent retains never consume a live turn's marker. Best-effort and NEVER
|
|
74
|
+
raises — a triage marker must not be able to fail a retain.
|
|
75
|
+
"""
|
|
76
|
+
try:
|
|
77
|
+
d = state_dir if state_dir is not None else _self_improve_state_dir()
|
|
78
|
+
path = os.path.join(d, SELF_IMPROVE_CORRECTION_PENDING_FILE)
|
|
79
|
+
if not os.path.exists(path):
|
|
80
|
+
return []
|
|
81
|
+
try:
|
|
82
|
+
os.remove(path)
|
|
83
|
+
except OSError:
|
|
84
|
+
pass
|
|
85
|
+
return [SELF_IMPROVE_CORRECTION_TAG]
|
|
86
|
+
except Exception:
|
|
87
|
+
return []
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
PRIVACY_STATE_FILE = "privacy-state.json"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _parse_iso(value) -> "datetime | None":
|
|
94
|
+
"""Best-effort parse of an ISO-8601 timestamp to an aware ``datetime``.
|
|
95
|
+
|
|
96
|
+
Accepts the ``...Z`` (UTC) suffix the gateway writes and the offset forms
|
|
97
|
+
``datetime.fromisoformat`` understands. Returns ``None`` on anything
|
|
98
|
+
unparseable — never raises. A naive result (no tz) is coerced to UTC so all
|
|
99
|
+
comparisons in ``exclude_private_ranges`` are between aware datetimes.
|
|
100
|
+
"""
|
|
101
|
+
if not isinstance(value, str) or not value:
|
|
102
|
+
return None
|
|
103
|
+
try:
|
|
104
|
+
s = value.strip()
|
|
105
|
+
if s.endswith("Z"):
|
|
106
|
+
s = s[:-1] + "+00:00"
|
|
107
|
+
dt = datetime.fromisoformat(s)
|
|
108
|
+
if dt.tzinfo is None:
|
|
109
|
+
dt = dt.replace(tzinfo=timezone.utc)
|
|
110
|
+
return dt
|
|
111
|
+
except (ValueError, TypeError):
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
PRIVACY_STATE_CORRUPT_KEY = "__corrupt__"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def read_privacy_state(state_dir: str | None = None) -> list:
|
|
119
|
+
"""Read the shared /private-mode interval list (switchroom privacy PR1).
|
|
120
|
+
|
|
121
|
+
Contract with the Telegram gateway's ``/private`` / ``/public`` commands:
|
|
122
|
+
the gateway maintains ``${TELEGRAM_STATE_DIR}/privacy-state.json`` (same dir
|
|
123
|
+
fallback as ``_self_improve_state_dir``) shaped::
|
|
124
|
+
|
|
125
|
+
{"version": 1, "intervals": [
|
|
126
|
+
{"start": "<iso>", "end": "<iso>"}, # a closed private window
|
|
127
|
+
{"start": "<iso>", "end": null} # the OPEN window = private NOW
|
|
128
|
+
]}
|
|
129
|
+
|
|
130
|
+
Returns the ``intervals`` list (list of ``{"start", "end"}`` dicts).
|
|
131
|
+
|
|
132
|
+
ABSENT vs CORRUPT are deliberately DISTINCT (review MAJOR 1 —
|
|
133
|
+
defense-in-depth, independent of the gateway's write atomicity):
|
|
134
|
+
|
|
135
|
+
- A **missing** file, or a valid JSON object with no ``intervals`` key,
|
|
136
|
+
yields ``[]`` — the default = fully public.
|
|
137
|
+
- A file that **exists but cannot be parsed** (truncated mid-rewrite,
|
|
138
|
+
not-a-JSON-object, ``intervals`` not a list) must NOT collapse to the same
|
|
139
|
+
"public" ``[]`` — that would leak the just-completed private turn if a Stop
|
|
140
|
+
hook fired during a non-atomic rewrite. It returns a single CORRUPT
|
|
141
|
+
sentinel interval and logs loudly; downstream this fails TOWARD privacy
|
|
142
|
+
(``_has_open_interval`` → True, ``exclude_private_ranges`` → drop all).
|
|
143
|
+
|
|
144
|
+
Best-effort — NEVER raises: a privacy-state read must not be able to fail a
|
|
145
|
+
retain.
|
|
146
|
+
"""
|
|
147
|
+
try:
|
|
148
|
+
d = state_dir if state_dir is not None else _self_improve_state_dir()
|
|
149
|
+
path = os.path.join(d, PRIVACY_STATE_FILE)
|
|
150
|
+
if not os.path.isfile(path):
|
|
151
|
+
return [] # absent = public (the default)
|
|
152
|
+
except Exception:
|
|
153
|
+
# Could not even resolve/stat the path — treat as absent (public).
|
|
154
|
+
return []
|
|
155
|
+
# The file EXISTS. From here a parse/read failure fails TOWARD privacy.
|
|
156
|
+
try:
|
|
157
|
+
with open(path, encoding="utf-8") as f:
|
|
158
|
+
data = json.load(f)
|
|
159
|
+
if not isinstance(data, dict):
|
|
160
|
+
raise ValueError("privacy-state.json is not a JSON object")
|
|
161
|
+
intervals = data.get("intervals")
|
|
162
|
+
if intervals is None:
|
|
163
|
+
return [] # valid JSON, no intervals declared = public
|
|
164
|
+
if not isinstance(intervals, list):
|
|
165
|
+
raise ValueError("privacy-state.json .intervals is not a list")
|
|
166
|
+
return [iv for iv in intervals if isinstance(iv, dict)]
|
|
167
|
+
except Exception as e:
|
|
168
|
+
print(
|
|
169
|
+
f"[Hindsight] privacy-state.json present but unreadable ({e!r}); "
|
|
170
|
+
f"failing TOWARD privacy — treating this session as private-now",
|
|
171
|
+
file=sys.stderr,
|
|
172
|
+
)
|
|
173
|
+
return [{PRIVACY_STATE_CORRUPT_KEY: True}]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _classify_privacy_intervals(intervals: list) -> tuple:
|
|
177
|
+
"""Parse raw intervals into ``(parsed, fail_private)`` — the single source of
|
|
178
|
+
truth both ``_has_open_interval`` and ``exclude_private_ranges`` use so they
|
|
179
|
+
can NEVER disagree on what "open" means (review MINOR).
|
|
180
|
+
|
|
181
|
+
- ``parsed``: list of ``(start_dt, end_dt_or_None)`` aware-datetime tuples;
|
|
182
|
+
``end_dt is None`` means an OPEN ``[start, ∞)`` window.
|
|
183
|
+
- ``fail_private``: True when the state must be treated as private-now,
|
|
184
|
+
unbounded — a CORRUPT sentinel, or an interval whose ``start`` is present
|
|
185
|
+
but unparseable (we cannot place the window, so we refuse to un-redact).
|
|
186
|
+
|
|
187
|
+
A malformed (present-but-unparseable) ``end`` degrades to OPEN ``[start, ∞)``
|
|
188
|
+
in BOTH consumers — and is LOGGED, not silently applied. Best-effort; never
|
|
189
|
+
raises.
|
|
190
|
+
"""
|
|
191
|
+
parsed = []
|
|
192
|
+
fail_private = False
|
|
193
|
+
for iv in intervals:
|
|
194
|
+
if not isinstance(iv, dict):
|
|
195
|
+
continue
|
|
196
|
+
if iv.get(PRIVACY_STATE_CORRUPT_KEY):
|
|
197
|
+
fail_private = True
|
|
198
|
+
continue
|
|
199
|
+
start_raw = iv.get("start")
|
|
200
|
+
start_dt = _parse_iso(start_raw)
|
|
201
|
+
if start_dt is None:
|
|
202
|
+
if start_raw is not None:
|
|
203
|
+
# Present but unparseable start: cannot place the window — fail
|
|
204
|
+
# toward privacy rather than silently ignoring a private range.
|
|
205
|
+
print(
|
|
206
|
+
f"[Hindsight] privacy interval has unparseable start "
|
|
207
|
+
f"{start_raw!r}; failing TOWARD privacy (private-now)",
|
|
208
|
+
file=sys.stderr,
|
|
209
|
+
)
|
|
210
|
+
fail_private = True
|
|
211
|
+
continue # start missing entirely = an empty entry, covers nothing
|
|
212
|
+
end_raw = iv.get("end")
|
|
213
|
+
if end_raw is None:
|
|
214
|
+
parsed.append((start_dt, None))
|
|
215
|
+
continue
|
|
216
|
+
end_dt = _parse_iso(end_raw)
|
|
217
|
+
if end_dt is None:
|
|
218
|
+
# Present but unparseable end: degrade to OPEN [start, ∞) AND surface
|
|
219
|
+
# it, so the guard and the redactor agree (no silent unbounded wipe).
|
|
220
|
+
print(
|
|
221
|
+
f"[Hindsight] privacy interval has unparseable end {end_raw!r}; "
|
|
222
|
+
f"treating as OPEN [start, ∞) — check the gateway writer",
|
|
223
|
+
file=sys.stderr,
|
|
224
|
+
)
|
|
225
|
+
parsed.append((start_dt, None))
|
|
226
|
+
continue
|
|
227
|
+
parsed.append((start_dt, end_dt))
|
|
228
|
+
return parsed, fail_private
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _has_open_interval(intervals: list) -> bool:
|
|
232
|
+
"""True when the state is private-right-now: some OPEN ``[start, ∞)`` window,
|
|
233
|
+
OR a corrupt/unplaceable state that fails toward privacy. Shares the
|
|
234
|
+
classifier with ``exclude_private_ranges`` so the two never diverge."""
|
|
235
|
+
parsed, fail_private = _classify_privacy_intervals(intervals)
|
|
236
|
+
if fail_private:
|
|
237
|
+
return True
|
|
238
|
+
return any(end is None for _, end in parsed)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def exclude_private_ranges(messages: list, intervals: list) -> list:
|
|
242
|
+
"""Drop any message whose ``timestamp`` falls inside a private interval.
|
|
243
|
+
|
|
244
|
+
An interval ``{"start": s, "end": e}`` covers ``[s, e]`` (inclusive, toward
|
|
245
|
+
privacy); an OPEN interval (``end`` is null, or a malformed ``end``) covers
|
|
246
|
+
``[s, ∞)``. A message whose ``timestamp`` is missing or unparseable is
|
|
247
|
+
dropped ONLY when some interval is open (conservative — we cannot place it,
|
|
248
|
+
and private-now is live), otherwise kept.
|
|
249
|
+
|
|
250
|
+
A CORRUPT state or an unplaceable ``start`` (``fail_private``) drops ALL
|
|
251
|
+
messages — failing toward privacy rather than leaking on a bad file.
|
|
252
|
+
Best-effort and total: the function never raises.
|
|
253
|
+
"""
|
|
254
|
+
if not intervals:
|
|
255
|
+
return list(messages)
|
|
256
|
+
|
|
257
|
+
parsed, fail_private = _classify_privacy_intervals(intervals)
|
|
258
|
+
if fail_private:
|
|
259
|
+
# Corrupt / unplaceable state: refuse to retain anything this pass.
|
|
260
|
+
return []
|
|
261
|
+
if not parsed:
|
|
262
|
+
return list(messages)
|
|
263
|
+
|
|
264
|
+
any_open = any(end is None for _, end in parsed)
|
|
265
|
+
|
|
266
|
+
def _in_any_range(ts_dt) -> bool:
|
|
267
|
+
for start_dt, end_dt in parsed:
|
|
268
|
+
if ts_dt < start_dt:
|
|
269
|
+
continue
|
|
270
|
+
if end_dt is None or ts_dt <= end_dt:
|
|
271
|
+
return True
|
|
272
|
+
return False
|
|
273
|
+
|
|
274
|
+
kept = []
|
|
275
|
+
for m in messages:
|
|
276
|
+
ts_dt = _parse_iso(m.get("timestamp")) if isinstance(m, dict) else None
|
|
277
|
+
if ts_dt is None:
|
|
278
|
+
# No placeable timestamp: drop only if a private window is open now.
|
|
279
|
+
if any_open:
|
|
280
|
+
continue
|
|
281
|
+
kept.append(m)
|
|
282
|
+
continue
|
|
283
|
+
if _in_any_range(ts_dt):
|
|
284
|
+
continue
|
|
285
|
+
kept.append(m)
|
|
286
|
+
return kept
|
|
287
|
+
|
|
288
|
+
|
|
41
289
|
def read_transcript(transcript_path: str, max_bytes: int | None = None) -> list:
|
|
42
290
|
"""Read a JSONL transcript file and return list of message dicts.
|
|
43
291
|
|
|
@@ -92,6 +340,15 @@ def read_transcript(transcript_path: str, max_bytes: int | None = None) -> list:
|
|
|
92
340
|
if uid is not None and "uuid" not in msg:
|
|
93
341
|
msg = dict(msg)
|
|
94
342
|
msg["uuid"] = uid
|
|
343
|
+
# Surface the transcript-entry timestamp the same
|
|
344
|
+
# way (nested format carries it on the OUTER entry).
|
|
345
|
+
# The /private-mode redaction filters on it
|
|
346
|
+
# (exclude_private_ranges); without surfacing it here
|
|
347
|
+
# every message would look timestamp-less.
|
|
348
|
+
ts = entry.get("timestamp")
|
|
349
|
+
if ts is not None and "timestamp" not in msg:
|
|
350
|
+
msg = dict(msg)
|
|
351
|
+
msg["timestamp"] = ts
|
|
95
352
|
messages.append(msg)
|
|
96
353
|
# Flat format (testing / future compatibility)
|
|
97
354
|
elif "role" in entry and "content" in entry:
|
|
@@ -254,6 +511,7 @@ def build_retain_payload(
|
|
|
254
511
|
api_token,
|
|
255
512
|
retain_full_window: bool = True,
|
|
256
513
|
document_id=None,
|
|
514
|
+
extra_tags=None,
|
|
257
515
|
) -> dict | None:
|
|
258
516
|
"""Pure transcript → retain payload + deterministic id (switchroom #3244 §1.4).
|
|
259
517
|
|
|
@@ -323,6 +581,23 @@ def build_retain_payload(
|
|
|
323
581
|
merged.append(lt)
|
|
324
582
|
tags = merged
|
|
325
583
|
|
|
584
|
+
# Switchroom self-improve PR4 (slice 4a) — per-turn correction tag. The
|
|
585
|
+
# caller (``run_retain``) reads-and-clears the operator-correction sentinel
|
|
586
|
+
# and threads the resulting tag(s) in here; this seam stays network-free and
|
|
587
|
+
# state-free (no sentinel IO), so it composes with lesson tags and rides the
|
|
588
|
+
# SAME payload the pending-retains queue persists. ``self-improve:correction``
|
|
589
|
+
# is STABLE by contract (it must NOT match DEFAULT_VOLATILE_SCOPE_PATTERNS),
|
|
590
|
+
# so on a correction turn the retain lands in its own
|
|
591
|
+
# ``[["self-improve:correction"]]`` consolidation scope — exactly the
|
|
592
|
+
# partition PR5's failure-synthesis cron works over. Absent ⇒ tag absent ⇒
|
|
593
|
+
# scope stays byte-identical to a normal turn's ``"shared"``.
|
|
594
|
+
if extra_tags:
|
|
595
|
+
merged = list(tags) if tags else []
|
|
596
|
+
for et in extra_tags:
|
|
597
|
+
if isinstance(et, str) and et and et not in merged:
|
|
598
|
+
merged.append(et)
|
|
599
|
+
tags = merged or None
|
|
600
|
+
|
|
326
601
|
metadata = {
|
|
327
602
|
"retained_at": template_vars["timestamp"],
|
|
328
603
|
"message_count": str(message_count),
|
|
@@ -438,6 +713,17 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
438
713
|
``lib/pending.py``). ``status="skipped"`` is the normal early-exit
|
|
439
714
|
cases (auto-retain disabled, empty transcript, throttled chunk).
|
|
440
715
|
"""
|
|
716
|
+
# /private-mode enforcement (switchroom privacy PR1) — FIRST, before
|
|
717
|
+
# load_config(). An OPEN private interval means the operator is discussing
|
|
718
|
+
# confidential material right now, so a normal (non-forced) auto-retain must
|
|
719
|
+
# not fire at all. Placed above load_config() deliberately so a
|
|
720
|
+
# HINDSIGHT_AUTO_RETAIN=true env pin (the autoRetain gate below) cannot
|
|
721
|
+
# override the privacy guarantee. A FORCED SessionEnd sweep is EXEMPT here on
|
|
722
|
+
# purpose — it must still flush the PUBLIC portion of the session — and
|
|
723
|
+
# relies instead on exclude_private_ranges() dropping the private turns.
|
|
724
|
+
if not force and _has_open_interval(read_privacy_state()):
|
|
725
|
+
return {"status": "skipped", "reason": "private-mode"}
|
|
726
|
+
|
|
441
727
|
config = load_config()
|
|
442
728
|
|
|
443
729
|
if not config.get("autoRetain"):
|
|
@@ -472,6 +758,17 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
472
758
|
debug_log(config, "No messages in transcript, skipping retain")
|
|
473
759
|
return {"status": "skipped", "reason": "empty transcript"}
|
|
474
760
|
|
|
761
|
+
# /private-mode redaction (switchroom privacy PR1) — drop any message that
|
|
762
|
+
# falls inside a private interval BEFORE window selection. This single
|
|
763
|
+
# insertion covers BOTH the chunked sliding window AND the force/full-session
|
|
764
|
+
# slice (select_retain_window returns list(all_messages) for force=True), so
|
|
765
|
+
# a forced SessionEnd sweep flushes only the PUBLIC portion and an
|
|
766
|
+
# open-at-session-end private range is never force-swept into the bank.
|
|
767
|
+
all_messages = exclude_private_ranges(all_messages, read_privacy_state())
|
|
768
|
+
if not all_messages:
|
|
769
|
+
debug_log(config, "All messages fell inside private ranges, skipping retain")
|
|
770
|
+
return {"status": "skipped", "reason": "private-mode"}
|
|
771
|
+
|
|
475
772
|
debug_log(config, f"Read {len(all_messages)} messages from transcript")
|
|
476
773
|
|
|
477
774
|
# Retention mode: full session (vendor default) or chunked. Switchroom
|
|
@@ -564,6 +861,14 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
564
861
|
# chunk 0 → plain session_id (backwards compatible with existing docs)
|
|
565
862
|
document_id = session_id if chunk_index == 0 else f"{session_id}-c{chunk_index}"
|
|
566
863
|
|
|
864
|
+
# Switchroom self-improve PR4 (slice 4a) — read-and-clear the per-turn
|
|
865
|
+
# operator-correction sentinel here, in the live Stop path only, AFTER the
|
|
866
|
+
# throttle / re-fire / empty-transcript early-returns above so only a retain
|
|
867
|
+
# that actually FIRES consumes the marker (a throttled turn leaves it for the
|
|
868
|
+
# next firing retain, whose overlapping window still contains the correction).
|
|
869
|
+
# Read-once: the tag rides exactly one retain. Best-effort; never raises.
|
|
870
|
+
extra_tags = read_and_clear_correction_pending()
|
|
871
|
+
|
|
567
872
|
# Build the payload via the shared, network-free seam (§1.4). This carries
|
|
568
873
|
# the deterministic id, connection info, formatted transcript, tags and
|
|
569
874
|
# metadata — sufficient to retry from a different process (drain_pending).
|
|
@@ -577,6 +882,7 @@ def run_retain(hook_input: dict, force: bool = False) -> dict:
|
|
|
577
882
|
api_token=api_token,
|
|
578
883
|
retain_full_window=retain_full_window,
|
|
579
884
|
document_id=document_id,
|
|
885
|
+
extra_tags=extra_tags,
|
|
580
886
|
)
|
|
581
887
|
if built is None:
|
|
582
888
|
debug_log(config, "Empty transcript after formatting, skipping retain")
|
|
@@ -73,7 +73,13 @@ from lib.pacing import inflight_lock
|
|
|
73
73
|
# retain.py owns the transcript reader, the deterministic-id recipe and the
|
|
74
74
|
# network-free payload builder; reuse them wholesale so the sidechain path
|
|
75
75
|
# stays byte-identical to the main path where it matters (dedup ids, formatting).
|
|
76
|
-
from retain import
|
|
76
|
+
from retain import (
|
|
77
|
+
_has_open_interval,
|
|
78
|
+
build_retain_payload,
|
|
79
|
+
exclude_private_ranges,
|
|
80
|
+
read_privacy_state,
|
|
81
|
+
read_transcript,
|
|
82
|
+
)
|
|
77
83
|
|
|
78
84
|
# Retain the last N human turns of the sidechain. The window is formatted on the
|
|
79
85
|
# TEXT-ONLY path (``retainToolCalls`` is forced False for the sidechain — see
|
|
@@ -333,6 +339,15 @@ def run_subagent_retain(hook_input: dict) -> dict:
|
|
|
333
339
|
"payload": {...}, # only when status == "failed" (for enqueue)
|
|
334
340
|
"error": Exception} # only when status == "failed"
|
|
335
341
|
"""
|
|
342
|
+
# /private-mode enforcement (switchroom privacy PR1) — FIRST, before
|
|
343
|
+
# load_config(). Subagents have no toggle of their own; they honor the
|
|
344
|
+
# parent session's privacy state file (same env, same path). An OPEN private
|
|
345
|
+
# interval means confidential material is being discussed right now, so the
|
|
346
|
+
# sidechain retain must not fire — placed above load_config() so a
|
|
347
|
+
# HINDSIGHT_AUTO_RETAIN=true env pin cannot override the privacy guarantee.
|
|
348
|
+
if _has_open_interval(read_privacy_state()):
|
|
349
|
+
return {"status": "skipped", "reason": "private-mode"}
|
|
350
|
+
|
|
336
351
|
config = load_config()
|
|
337
352
|
|
|
338
353
|
if not config.get("autoRetain"):
|
|
@@ -379,6 +394,19 @@ def run_subagent_retain(hook_input: dict) -> dict:
|
|
|
379
394
|
debug_log(config, f"SubagentStop: empty sidechain transcript {transcript_path}")
|
|
380
395
|
return {"status": "skipped", "reason": "empty transcript"}
|
|
381
396
|
|
|
397
|
+
# /private-mode redaction (review MAJOR 2) — the open-interval early-skip
|
|
398
|
+
# above only catches a range that is STILL open at SubagentStop. But a
|
|
399
|
+
# backgrounded sub-agent can be dispatched under /private, carry that
|
|
400
|
+
# confidential context, and finish AFTER /public closed the interval — no
|
|
401
|
+
# interval open here, yet its window still holds private-window material.
|
|
402
|
+
# Apply the same range redaction the parent Stop path uses so a now-CLOSED
|
|
403
|
+
# private range is excluded from the sidechain retain too. Runs BEFORE the
|
|
404
|
+
# volume gate so the gate measures the redacted content.
|
|
405
|
+
all_messages = exclude_private_ranges(all_messages, read_privacy_state())
|
|
406
|
+
if not all_messages:
|
|
407
|
+
debug_log(config, "SubagentStop: all sidechain messages fell inside private ranges")
|
|
408
|
+
return {"status": "skipped", "reason": "private-mode"}
|
|
409
|
+
|
|
382
410
|
# Volume gate — skip trivial forks, log the skip for coverage auditing.
|
|
383
411
|
passed, turns, chars = passes_volume_gate(all_messages, config)
|
|
384
412
|
if not passed:
|