instar 1.3.983 → 1.3.985
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/CredentialManualLevers.js +0 -0
- package/dist/core/CredentialManualLevers.js.map +1 -1
- package/dist/core/DeliverMessageHandler.js +0 -0
- package/dist/core/DeliverMessageHandler.js.map +1 -1
- package/dist/core/PeerEndpointResolver.d.ts +0 -0
- package/dist/core/PeerEndpointResolver.d.ts.map +1 -1
- package/dist/core/PeerEndpointResolver.js +0 -0
- package/dist/core/PeerEndpointResolver.js.map +1 -1
- package/dist/core/PostUpdateMigrator.d.ts +17 -1
- package/dist/core/PostUpdateMigrator.d.ts.map +1 -1
- package/dist/core/PostUpdateMigrator.js +48 -1
- package/dist/core/PostUpdateMigrator.js.map +1 -1
- package/dist/core/QueueDrainLoop.js +0 -0
- package/dist/core/QueueDrainLoop.js.map +1 -1
- package/dist/core/ScopeAccretionSweep.js +1 -1
- package/dist/core/ScopeAccretionSweep.js.map +1 -1
- package/dist/core/SessionOwnership.js +1 -1
- package/dist/core/SessionOwnership.js.map +1 -1
- package/dist/core/StandardsEnforcementAuditor.js +1 -1
- package/dist/core/StandardsEnforcementAuditor.js.map +1 -1
- package/dist/core/StoreSnapshot.js +2 -2
- package/dist/core/StoreSnapshot.js.map +1 -1
- package/dist/core/UnionReader.js +0 -0
- package/dist/core/UnionReader.js.map +1 -1
- package/dist/core/benchmarkDivergenceCore.js +1 -1
- package/dist/core/benchmarkDivergenceCore.js.map +1 -1
- package/dist/core/cartographerSummary.js +0 -0
- package/dist/core/cartographerSummary.js.map +1 -1
- package/dist/core/routingPriceAuthority.d.ts +0 -0
- package/dist/core/routingPriceAuthority.d.ts.map +1 -1
- package/dist/core/routingPriceAuthority.js +0 -0
- package/dist/core/routingPriceAuthority.js.map +1 -1
- package/dist/messaging/relayContentDedup.js +0 -0
- package/dist/messaging/relayContentDedup.js.map +1 -1
- package/dist/monitoring/BenchmarkDivergenceAnalyzer.js +10 -10
- package/dist/monitoring/BenchmarkDivergenceAnalyzer.js.map +1 -1
- package/dist/monitoring/ExternalHogArmMarker.js +0 -0
- package/dist/monitoring/ExternalHogArmMarker.js.map +1 -1
- package/dist/monitoring/ExternalHogClassifier.js +0 -0
- package/dist/monitoring/ExternalHogClassifier.js.map +1 -1
- package/dist/monitoring/ExternalHogSampler.js +0 -0
- package/dist/monitoring/ExternalHogSampler.js.map +1 -1
- package/dist/monitoring/FeatureMetricsLedger.js +1 -1
- package/dist/monitoring/FeatureMetricsLedger.js.map +1 -1
- package/dist/monitoring/GreenPrAutoMerger.js +1 -1
- package/dist/monitoring/GreenPrAutoMerger.js.map +1 -1
- package/dist/monitoring/PermissionPromptAutoResolver.js +1 -1
- package/dist/monitoring/PermissionPromptAutoResolver.js.map +1 -1
- package/dist/monitoring/ProviderCostReportStore.js +0 -0
- package/dist/monitoring/ProviderCostReportStore.js.map +1 -1
- package/dist/monitoring/blockerSettleAuthority.js +0 -0
- package/dist/monitoring/blockerSettleAuthority.js.map +1 -1
- package/package.json +1 -1
- package/src/data/builtin-manifest.json +20 -20
- package/src/templates/scripts/telegram-reply.sh +61 -0
- package/upgrades/1.3.982.md +17 -2
- package/upgrades/1.3.984.md +116 -0
- package/upgrades/1.3.985.md +109 -0
- package/upgrades/side-effects/behavioural-promise-unverifiable.md +17 -2
- package/upgrades/side-effects/misplaced-flag-sent-as-message-text.md +193 -0
- package/upgrades/side-effects/raw-nul-bytes-hide-source-from-grep.md +199 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "./builtin-manifest.schema.json",
|
|
3
3
|
"schemaVersion": 1,
|
|
4
|
-
"generatedAt": "2026-07-
|
|
5
|
-
"instarVersion": "1.3.
|
|
4
|
+
"generatedAt": "2026-07-26T17:59:55.648Z",
|
|
5
|
+
"instarVersion": "1.3.985",
|
|
6
6
|
"entryCount": 202,
|
|
7
7
|
"entries": {
|
|
8
8
|
"hook:session-start": {
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
"domain": "identity",
|
|
12
12
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
13
13
|
"installedPath": ".instar/hooks/instar/session-start.sh",
|
|
14
|
-
"contentHash": "
|
|
14
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
15
15
|
"since": "2025-01-01"
|
|
16
16
|
},
|
|
17
17
|
"hook:dangerous-command-guard": {
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
"domain": "safety",
|
|
21
21
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
22
22
|
"installedPath": ".instar/hooks/instar/dangerous-command-guard.sh",
|
|
23
|
-
"contentHash": "
|
|
23
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
24
24
|
"since": "2025-01-01"
|
|
25
25
|
},
|
|
26
26
|
"hook:grounding-before-messaging": {
|
|
@@ -29,7 +29,7 @@
|
|
|
29
29
|
"domain": "safety",
|
|
30
30
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
31
31
|
"installedPath": ".instar/hooks/instar/grounding-before-messaging.sh",
|
|
32
|
-
"contentHash": "
|
|
32
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
33
33
|
"since": "2025-01-01"
|
|
34
34
|
},
|
|
35
35
|
"hook:compaction-recovery": {
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
"domain": "identity",
|
|
39
39
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
40
40
|
"installedPath": ".instar/hooks/instar/compaction-recovery.sh",
|
|
41
|
-
"contentHash": "
|
|
41
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
42
42
|
"since": "2025-01-01"
|
|
43
43
|
},
|
|
44
44
|
"hook:external-operation-gate": {
|
|
@@ -47,7 +47,7 @@
|
|
|
47
47
|
"domain": "safety",
|
|
48
48
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
49
49
|
"installedPath": ".instar/hooks/instar/external-operation-gate.js",
|
|
50
|
-
"contentHash": "
|
|
50
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
51
51
|
"since": "2025-01-01"
|
|
52
52
|
},
|
|
53
53
|
"hook:deferral-detector": {
|
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
"domain": "safety",
|
|
57
57
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
58
58
|
"installedPath": ".instar/hooks/instar/deferral-detector.js",
|
|
59
|
-
"contentHash": "
|
|
59
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
60
60
|
"since": "2025-01-01"
|
|
61
61
|
},
|
|
62
62
|
"hook:self-stop-guard": {
|
|
@@ -65,7 +65,7 @@
|
|
|
65
65
|
"domain": "coherence",
|
|
66
66
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
67
67
|
"installedPath": ".instar/hooks/instar/self-stop-guard.js",
|
|
68
|
-
"contentHash": "
|
|
68
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
69
69
|
"since": "2025-01-01"
|
|
70
70
|
},
|
|
71
71
|
"hook:post-action-reflection": {
|
|
@@ -74,7 +74,7 @@
|
|
|
74
74
|
"domain": "evolution",
|
|
75
75
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
76
76
|
"installedPath": ".instar/hooks/instar/post-action-reflection.js",
|
|
77
|
-
"contentHash": "
|
|
77
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
78
78
|
"since": "2025-01-01"
|
|
79
79
|
},
|
|
80
80
|
"hook:external-communication-guard": {
|
|
@@ -83,7 +83,7 @@
|
|
|
83
83
|
"domain": "safety",
|
|
84
84
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
85
85
|
"installedPath": ".instar/hooks/instar/external-communication-guard.js",
|
|
86
|
-
"contentHash": "
|
|
86
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
87
87
|
"since": "2025-01-01"
|
|
88
88
|
},
|
|
89
89
|
"hook:scope-coherence-collector": {
|
|
@@ -92,7 +92,7 @@
|
|
|
92
92
|
"domain": "coherence",
|
|
93
93
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
94
94
|
"installedPath": ".instar/hooks/instar/scope-coherence-collector.js",
|
|
95
|
-
"contentHash": "
|
|
95
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
96
96
|
"since": "2025-01-01"
|
|
97
97
|
},
|
|
98
98
|
"hook:scope-coherence-checkpoint": {
|
|
@@ -101,7 +101,7 @@
|
|
|
101
101
|
"domain": "coherence",
|
|
102
102
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
103
103
|
"installedPath": ".instar/hooks/instar/scope-coherence-checkpoint.js",
|
|
104
|
-
"contentHash": "
|
|
104
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
105
105
|
"since": "2025-01-01"
|
|
106
106
|
},
|
|
107
107
|
"hook:free-text-guard": {
|
|
@@ -110,7 +110,7 @@
|
|
|
110
110
|
"domain": "safety",
|
|
111
111
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
112
112
|
"installedPath": ".instar/hooks/instar/free-text-guard.sh",
|
|
113
|
-
"contentHash": "
|
|
113
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
114
114
|
"since": "2025-01-01"
|
|
115
115
|
},
|
|
116
116
|
"hook:claim-intercept": {
|
|
@@ -119,7 +119,7 @@
|
|
|
119
119
|
"domain": "coherence",
|
|
120
120
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
121
121
|
"installedPath": ".instar/hooks/instar/claim-intercept.js",
|
|
122
|
-
"contentHash": "
|
|
122
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
123
123
|
"since": "2025-01-01"
|
|
124
124
|
},
|
|
125
125
|
"hook:claim-intercept-response": {
|
|
@@ -128,7 +128,7 @@
|
|
|
128
128
|
"domain": "coherence",
|
|
129
129
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
130
130
|
"installedPath": ".instar/hooks/instar/claim-intercept-response.js",
|
|
131
|
-
"contentHash": "
|
|
131
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
132
132
|
"since": "2025-01-01"
|
|
133
133
|
},
|
|
134
134
|
"hook:stop-gate-router": {
|
|
@@ -137,7 +137,7 @@
|
|
|
137
137
|
"domain": "safety",
|
|
138
138
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
139
139
|
"installedPath": ".instar/hooks/instar/stop-gate-router.js",
|
|
140
|
-
"contentHash": "
|
|
140
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
141
141
|
"since": "2025-01-01"
|
|
142
142
|
},
|
|
143
143
|
"hook:auto-approve-permissions": {
|
|
@@ -146,7 +146,7 @@
|
|
|
146
146
|
"domain": "safety",
|
|
147
147
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
148
148
|
"installedPath": ".instar/hooks/instar/auto-approve-permissions.js",
|
|
149
|
-
"contentHash": "
|
|
149
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
150
150
|
"since": "2025-01-01"
|
|
151
151
|
},
|
|
152
152
|
"job:health-check": {
|
|
@@ -1242,7 +1242,7 @@
|
|
|
1242
1242
|
"type": "template",
|
|
1243
1243
|
"domain": "operations",
|
|
1244
1244
|
"sourcePath": "src/templates/scripts/telegram-reply.sh",
|
|
1245
|
-
"contentHash": "
|
|
1245
|
+
"contentHash": "4464581188f5c736a62edac5e6a2edecfcfcd365557a18e514b741731bed6e0b",
|
|
1246
1246
|
"since": "2025-01-01"
|
|
1247
1247
|
},
|
|
1248
1248
|
"template:whatsapp-reply.sh": {
|
|
@@ -1562,7 +1562,7 @@
|
|
|
1562
1562
|
"type": "subsystem",
|
|
1563
1563
|
"domain": "updates",
|
|
1564
1564
|
"sourcePath": "src/core/PostUpdateMigrator.ts",
|
|
1565
|
-
"contentHash": "
|
|
1565
|
+
"contentHash": "59f279be8e27e1d3f4a19c810f39645b9d63dacfbba3b1cc800a4ffeec67d57d",
|
|
1566
1566
|
"since": "2025-01-01"
|
|
1567
1567
|
},
|
|
1568
1568
|
"subsystem:scheduler": {
|
|
@@ -4,6 +4,12 @@
|
|
|
4
4
|
# Usage:
|
|
5
5
|
# ./telegram-reply.sh TOPIC_ID "message text"
|
|
6
6
|
# ./telegram-reply.sh --format markdown TOPIC_ID "**bold**"
|
|
7
|
+
#
|
|
8
|
+
# EVERY FLAG GOES BEFORE THE TOPIC ID. Flag parsing stops at the topic id, so
|
|
9
|
+
# anything after it is message text. A flag in the wrong position used to be
|
|
10
|
+
# sent to the user as literal text with its effect silently dropped; it is now
|
|
11
|
+
# refused. Correct:
|
|
12
|
+
# ./telegram-reply.sh --tone-ack B2_FILE_PATH --tone-reason "why" TOPIC_ID "msg"
|
|
7
13
|
# echo "message text" | ./telegram-reply.sh TOPIC_ID
|
|
8
14
|
# cat <<'EOF' | ./telegram-reply.sh TOPIC_ID
|
|
9
15
|
# Multi-line message here
|
|
@@ -25,6 +31,24 @@
|
|
|
25
31
|
# a blanket pre-ack that silently disables the inform
|
|
26
32
|
# layer; spec outbound-jargon-filepath-gap §2.4(4)).
|
|
27
33
|
#
|
|
34
|
+
# Tone-gate advisory reactions (answering a 422 `tone-gate-advisory`). BOTH
|
|
35
|
+
# forms are recorded as evidence that tunes the gate; neither is optional
|
|
36
|
+
# once you have been handed a decisionRef.
|
|
37
|
+
# --tone-complied <RULE> You agreed with the nudge and revised the text.
|
|
38
|
+
# Grades the check `right`.
|
|
39
|
+
# --tone-ack <RULE> You disagree and are sending unchanged. Grades
|
|
40
|
+
# the check `wrong`. REQUIRES --tone-reason: a
|
|
41
|
+
# reasonless ack is refused and nothing sends,
|
|
42
|
+
# because that reason IS the tuning evidence.
|
|
43
|
+
# --tone-reason "<why>" Why the nudge is wrong in this case.
|
|
44
|
+
# --tone-decision-ref <REF> The decisionRef from the 422, so the reaction
|
|
45
|
+
# joins the verdict it answers.
|
|
46
|
+
#
|
|
47
|
+
# Example (note the ordering — flags first, topic id last):
|
|
48
|
+
# ./telegram-reply.sh --tone-ack B2_FILE_PATH \
|
|
49
|
+
# --tone-reason "the operator asked for the path explicitly" \
|
|
50
|
+
# --tone-decision-ref DQ-1234 29723 "the message"
|
|
51
|
+
#
|
|
28
52
|
# Outbound advisory preflight (inform-only — spec outbound-jargon-filepath-gap §2.4):
|
|
29
53
|
# When this send comes from an automated LLM job session (the scheduler
|
|
30
54
|
# stamps INSTAR_MESSAGE_KIND=automated + INSTAR_SENDER_CLASS=llm-session
|
|
@@ -133,6 +157,43 @@ if [ -z "$TOPIC_ID" ]; then
|
|
|
133
157
|
exit 1
|
|
134
158
|
fi
|
|
135
159
|
|
|
160
|
+
# A flag placed AFTER the topic id was silently swallowed into the message.
|
|
161
|
+
#
|
|
162
|
+
# The parse loop above stops at the first non-flag argument (the topic id), so
|
|
163
|
+
# everything from there on becomes message text via `MSG="$*"` below. A
|
|
164
|
+
# misplaced `--tone-ack` was therefore SENT TO THE USER as visible message body
|
|
165
|
+
# while the override it was meant to carry never reached the server — and,
|
|
166
|
+
# because the flags never applied, the tone gate re-reviewed the send and its
|
|
167
|
+
# verdict was then misread as absurd. That is how a CORRECT check came to be
|
|
168
|
+
# graded `wrong` in the decision-quality data on 2026-07-26. The record is
|
|
169
|
+
# durable; the cause was two characters of argument order.
|
|
170
|
+
#
|
|
171
|
+
# The asymmetry is the defect: a flag-shaped token BEFORE the topic id is fatal
|
|
172
|
+
# ("Unknown flag", above), but after it the script was maximally permissive.
|
|
173
|
+
# Refuse flag-shaped tokens in both positions. This also catches a TYPO'd flag
|
|
174
|
+
# (`--tone-akc`), which is the realistic case and was equally silent.
|
|
175
|
+
#
|
|
176
|
+
# Deliberately strict: any `--*` argument. A message that genuinely needs such a
|
|
177
|
+
# token as literal text goes through stdin, which is the documented primary path
|
|
178
|
+
# and is unaffected by this check ($# is 0 there).
|
|
179
|
+
for _arg in "$@"; do
|
|
180
|
+
case "$_arg" in
|
|
181
|
+
--*)
|
|
182
|
+
echo "Refused: '$_arg' appears AFTER the topic id." >&2
|
|
183
|
+
echo "" >&2
|
|
184
|
+
echo "Flags are only parsed BEFORE the topic id. Placed here it would have been" >&2
|
|
185
|
+
echo "sent to the user as literal message text, and its effect silently dropped." >&2
|
|
186
|
+
echo "" >&2
|
|
187
|
+
echo " Correct: telegram-reply.sh --tone-ack RULE --tone-reason \"why\" $TOPIC_ID \"message\"" >&2
|
|
188
|
+
echo " You ran: telegram-reply.sh $TOPIC_ID ... $_arg ..." >&2
|
|
189
|
+
echo "" >&2
|
|
190
|
+
echo "If '$_arg' is genuinely part of the message text, pipe the message on stdin:" >&2
|
|
191
|
+
echo " cat <<'EOF' | telegram-reply.sh $TOPIC_ID" >&2
|
|
192
|
+
exit 1
|
|
193
|
+
;;
|
|
194
|
+
esac
|
|
195
|
+
done
|
|
196
|
+
|
|
136
197
|
# Read message from args or stdin
|
|
137
198
|
if [ $# -gt 0 ]; then
|
|
138
199
|
MSG="$*"
|
package/upgrades/1.3.982.md
CHANGED
|
@@ -42,6 +42,17 @@ genuinely reads live state.
|
|
|
42
42
|
|
|
43
43
|
## What to Tell Your User
|
|
44
44
|
|
|
45
|
+
> **CORRECTION (2026-07-26, after release).** The paragraph below over-claimed and is kept
|
|
46
|
+
> visible rather than quietly rewritten. **The change is forward-only.** Promises about
|
|
47
|
+
> behaviour recorded from now on stay open, and the runaway tally stopped for all of them —
|
|
48
|
+
> but promises that already existed **keep the verdict they were last stamped with**, because
|
|
49
|
+
> the fix works by declining to reach a verdict, which means it never revisits one. Measured
|
|
50
|
+
> on a live install after deploy: 98 behavioural records still showed 74 "kept" and 24
|
|
51
|
+
> "broken", unchanged. So if you had behavioural commitments before upgrading, their old and
|
|
52
|
+
> untrustworthy verdicts are still on screen, and the health line will still count those 74 as
|
|
53
|
+
> verified. Clearing them needs a deliberate decision (rewriting stored records, or having the
|
|
54
|
+
> display refuse a verdict for anything unverifiable) and has not been made.
|
|
55
|
+
|
|
45
56
|
If you track commitments, promises about behaviour now show as **pending** rather than
|
|
46
57
|
verified or violated, with both counters at zero. That is more honest, not a regression:
|
|
47
58
|
nothing was ever watching whether those promises were kept, so "verified" was reassurance
|
|
@@ -69,10 +80,14 @@ commitment"`. Live store corroboration: 74/74 with the field `verified`, 24/24 w
|
|
|
69
80
|
|
|
70
81
|
**Observed after,** same scenarios against the built code:
|
|
71
82
|
|
|
83
|
+
**Scope note added 2026-07-26:** every row below was measured on a **newly created** commitment.
|
|
84
|
+
A row that already existed before the upgrade keeps its stored verdict — see the correction above.
|
|
85
|
+
|
|
72
86
|
| scenario | before | after (25 sweeps) |
|
|
73
87
|
|---|---|---|
|
|
74
|
-
| behavioural, `behavioralRule` set | `verified`, +1 tick per sweep | `pending`, verifications 0, violations 0 |
|
|
75
|
-
| behavioural, field absent | `violated`, +1 tick per sweep | `pending`, verifications 0, violations 0 |
|
|
88
|
+
| behavioural, `behavioralRule` set (NEW row) | `verified`, +1 tick per sweep | `pending`, verifications 0, violations 0 |
|
|
89
|
+
| behavioural, field absent (NEW row) | `violated`, +1 tick per sweep | `pending`, verifications 0, violations 0 |
|
|
90
|
+
| behavioural row that PREDATES the change | `verified` or `violated`, ticking | ticking stops; **stored verdict unchanged** |
|
|
76
91
|
| rules file deleted | produced a verdict | no verdict; file self-heals; status unchanged |
|
|
77
92
|
| health, 1 behavioural row | `"1 commitment(s) tracked, all verified"` | `"1 commitment(s) tracked — 0 verified, 1 not automatically verifiable (no violations)"`, status `healthy` |
|
|
78
93
|
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# Upgrade Guide — vNEXT
|
|
2
|
+
|
|
3
|
+
<!-- assembled-by: assemble-next-md -->
|
|
4
|
+
<!-- bump: patch -->
|
|
5
|
+
|
|
6
|
+
## What Changed
|
|
7
|
+
|
|
8
|
+
**Thirty source files were invisible to `grep`. A search across them returned nothing,
|
|
9
|
+
and nothing read as "there is nothing there."**
|
|
10
|
+
|
|
11
|
+
`grep` classifies a file as binary if it contains a NUL (0x00) byte, and on a binary file
|
|
12
|
+
it emits **nothing at all** — not a match, not a `Binary file X matches` line, not even a
|
|
13
|
+
`0` under `-c`. Thirty tracked text files each carried one, so every grep-based audit over
|
|
14
|
+
`src/` silently skipped them. A search that examined two-thirds of the tree was
|
|
15
|
+
byte-for-byte indistinguishable from one that examined all of it and found nothing wrong.
|
|
16
|
+
|
|
17
|
+
Twenty-two were live source. Among them: `blockerSettleAuthority.ts` (the gate that
|
|
18
|
+
decides whether a blocker is genuinely unresolvable), `SessionOwnership.ts`,
|
|
19
|
+
`GreenPrAutoMerger.ts`, `PermissionPromptAutoResolver.ts` (an always-on safety floor), all
|
|
20
|
+
three `ExternalHog*` modules — and `StandardsEnforcementAuditor.ts`, the module that
|
|
21
|
+
audits whether our standards carry structural guards. The auditor of guarantees was itself
|
|
22
|
+
invisible to the standard search instrument.
|
|
23
|
+
|
|
24
|
+
**A second consequence, worse in kind.** Git applies the same rule but only sniffs the
|
|
25
|
+
first 8000 bytes. For the **11 files** whose NUL fell inside that window, `git diff`
|
|
26
|
+
rendered `Bin 5407 -> 5412 bytes` instead of a line diff — so pull requests touching
|
|
27
|
+
safety-critical authority code were reviewed **without the reviewer being shown the
|
|
28
|
+
changed lines**. For the other 19, git saw text while `grep` did not. The two instruments
|
|
29
|
+
disagreed on the same tree, which is exactly why this survived so long: whichever one you
|
|
30
|
+
happened to reach for decided what you believed.
|
|
31
|
+
|
|
32
|
+
**None of it was corruption.** Every byte was a deliberate composite-key or hash
|
|
33
|
+
separator — the classic collision-proof choice, since a NUL cannot occur inside a model
|
|
34
|
+
name or a framework name:
|
|
35
|
+
|
|
36
|
+
```ts
|
|
37
|
+
const key = `${row.model}<a literal 0x00 byte>${row.framework}`;
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
The delimiter is correct. Writing it as the byte rather than its escape is the entire
|
|
41
|
+
defect. All thirty now use the six-character escape, which denotes the identical
|
|
42
|
+
character — so runtime behaviour is unchanged, and every hash derived from these
|
|
43
|
+
separators is byte-identical. Four sites feed them into digests
|
|
44
|
+
(`relayContentDedup`, `blockerSettleAuthority`, `UnionReader`, `ExternalHogArmMarker`);
|
|
45
|
+
this was the primary risk and it was verified rather than assumed. No migration, no cache
|
|
46
|
+
invalidation, no re-keying.
|
|
47
|
+
|
|
48
|
+
A new ratchet, `tests/unit/no-raw-nul-bytes-in-source.test.ts`, fails the build if one
|
|
49
|
+
returns. It reads bytes via `readFileSync` and never shells out to a search tool, so it
|
|
50
|
+
cannot be blinded by the defect it detects. It carries no exemption list, because no case
|
|
51
|
+
needs one: any runtime NUL a program wants is expressible as an escape.
|
|
52
|
+
|
|
53
|
+
**What this does not do:** it fixes one silent instrument, not the class. Other raw
|
|
54
|
+
control bytes (ESC, BEL) remain in a few hostile-input test fixtures — deliberately, after
|
|
55
|
+
verifying empirically that they do **not** cause the grep skip. Only 0x00 does. A lint
|
|
56
|
+
should enforce exactly the failure it is named for, and widening this one for tidiness
|
|
57
|
+
would have added churn while claiming safety it does not provide.
|
|
58
|
+
|
|
59
|
+
## What to Tell Your User
|
|
60
|
+
|
|
61
|
+
Nothing changes in how the agent behaves — this is a source-text and build-time fix with
|
|
62
|
+
no route, setting, or visible surface.
|
|
63
|
+
|
|
64
|
+
What it does change is trustworthiness of process: for months, searching the codebase for
|
|
65
|
+
a given pattern could quietly skip thirty files and report a clean result, and for eleven
|
|
66
|
+
of those files a code review showed "binary file changed" instead of the actual lines. Any
|
|
67
|
+
past conclusion of the form "we don't do X anywhere" may have been drawn from an
|
|
68
|
+
instrument that wasn't looking. Those files are readable again, and a check now fails the
|
|
69
|
+
build if it recurs.
|
|
70
|
+
|
|
71
|
+
## Summary of New Capabilities
|
|
72
|
+
|
|
73
|
+
- Thirty text files that were invisible to `grep` — twenty-two of them live source,
|
|
74
|
+
including several safety-critical authority modules — are searchable again.
|
|
75
|
+
- Eleven files that rendered as binary diffs in code review now show reviewable lines.
|
|
76
|
+
- A build-time ratchet refuses any text file containing a raw NUL byte, naming the file
|
|
77
|
+
and the consequence rather than emitting a bare assertion failure.
|
|
78
|
+
- The ratchet proves its own detector on every run, so a check that has gone dead is
|
|
79
|
+
distinguishable from one that has nothing to report.
|
|
80
|
+
|
|
81
|
+
## Evidence
|
|
82
|
+
|
|
83
|
+
**Reproduction (before):** `grep -c export src/monitoring/blockerSettleAuthority.ts`
|
|
84
|
+
printed nothing — no count, no error, exit status indistinguishable from a clean miss.
|
|
85
|
+
`file` reported the same TypeScript source as `data`, not text. Three one-line fixtures
|
|
86
|
+
isolate the cause: a clean file and a file containing ESC + BEL both return 1 match; a
|
|
87
|
+
file differing only by one NUL byte returns nothing at all.
|
|
88
|
+
|
|
89
|
+
**Observed after,** same scenarios against the fixed tree:
|
|
90
|
+
|
|
91
|
+
| scenario | before | after |
|
|
92
|
+
|---|---|---|
|
|
93
|
+
| `grep -c export` on `blockerSettleAuthority.ts` | *(silent)* | `1` |
|
|
94
|
+
| `grep -c export` on `StandardsEnforcementAuditor.ts` | *(silent)* | `13` |
|
|
95
|
+
| `grep -c export` on `FeatureMetricsLedger.ts` | *(silent)* | `20` |
|
|
96
|
+
| `file` on those sources | `data` | `ASCII text` |
|
|
97
|
+
| `git diff` on the 11 early-NUL files | `Bin NNNN -> NNNN bytes` | reviewable line diff |
|
|
98
|
+
| runtime value of the separator | U+0000 | U+0000 (unchanged) |
|
|
99
|
+
| digests derived from the separator | — | byte-identical, verified |
|
|
100
|
+
| ratchet against a reintroduced NUL | *(no check existed)* | fails, names the file |
|
|
101
|
+
|
|
102
|
+
**Tests:** `tests/unit/no-raw-nul-bytes-in-source.test.ts` (4 new). One asserts the
|
|
103
|
+
detector flags a raw-byte file and clears an escaped one — a dead-check guard, since a
|
|
104
|
+
lint that has never objected is output-identical to a lint that cannot object. One asserts
|
|
105
|
+
the escape decodes to exactly the raw byte, so the behaviour-preserving claim is checked
|
|
106
|
+
rather than stated. One scans the tree. One asserts the scan examined more than 500 files,
|
|
107
|
+
so a scan that walked the wrong roots cannot pass forever by finding nothing.
|
|
108
|
+
|
|
109
|
+
Verified by reintroducing a raw NUL into `FeatureMetricsLedger.ts`: the lint exits 1,
|
|
110
|
+
names the file, and states the consequence. `npx tsc --noEmit` clean; the full unit suite
|
|
111
|
+
runs 39,046 tests with no failure attributable to this change.
|
|
112
|
+
|
|
113
|
+
**Provenance note.** This was found only because a *different* investigation produced a
|
|
114
|
+
suspiciously empty search result that was about to be written up as "no such mechanism
|
|
115
|
+
exists in the codebase." That search had skipped twenty-two files. The finding was
|
|
116
|
+
manufactured by the defect, and has been retracted and superseded rather than deleted.
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
# Upgrade Guide — vNEXT
|
|
2
|
+
|
|
3
|
+
<!-- assembled-by: assemble-next-md -->
|
|
4
|
+
<!-- bump: patch -->
|
|
5
|
+
|
|
6
|
+
## What Changed
|
|
7
|
+
|
|
8
|
+
**A flag placed after the topic id was sent to the user as literal message text, and the
|
|
9
|
+
setting it carried was silently dropped.**
|
|
10
|
+
|
|
11
|
+
`telegram-reply.sh` parses flags in a loop that breaks at the first non-flag argument — the
|
|
12
|
+
topic id. Everything after that becomes the message via `MSG="$*"`. So
|
|
13
|
+
`telegram-reply.sh 29723 --tone-ack B15 --tone-reason "why"` sent the user the literal text
|
|
14
|
+
`--tone-ack B15 --tone-reason why`, while the tone-advisory override never reached the
|
|
15
|
+
server. Both failures were silent; the script exited 0.
|
|
16
|
+
|
|
17
|
+
This is measured, not inferred. The pre-fix template is captured verbatim as a test fixture
|
|
18
|
+
(at the same SHA now registered in the migrator's shipped-SHA allowlist) and run against a
|
|
19
|
+
stub of the reply route: exit 0, received text containing `--tone-ack`, and no
|
|
20
|
+
`toneAdvisoryAck` in the metadata.
|
|
21
|
+
|
|
22
|
+
**The cost was a corrupted measurement.** Because the override never applied, the tone gate
|
|
23
|
+
re-reviewed the send as an ordinary message whose text now began with option noise. The
|
|
24
|
+
verdict looked absurd, was read as a malfunction, and a **correct** check was graded `wrong`
|
|
25
|
+
in the decision-quality data — the data used to decide which checks to trust. That record is
|
|
26
|
+
durable and is not retracted by this change.
|
|
27
|
+
|
|
28
|
+
**The root cause was not the script.** These flags were documented nowhere agent-facing:
|
|
29
|
+
not in the script's own usage header (which covers `--format` and `--stdin-base64`), and not
|
|
30
|
+
in the agent instructions, which document the HTTP `metadata.*` fields while simultaneously
|
|
31
|
+
mandating "ALWAYS the relay script, never a hand-rolled curl". The bridge between the two
|
|
32
|
+
did not exist, so the invocation had to be guessed — and the script accepted the guess
|
|
33
|
+
without complaint. A capability shipped without instructions, plus a tool that stayed silent
|
|
34
|
+
on the only usage an uninstructed caller would try.
|
|
35
|
+
|
|
36
|
+
An inconsistency underneath made the fix obvious once seen: the script **already** treated an
|
|
37
|
+
unrecognised flag-shaped token as fatal when it appeared *before* the topic id. It was strict
|
|
38
|
+
about nonsense in one position and fully permissive about a real flag in the other. Both
|
|
39
|
+
positions are now treated the same.
|
|
40
|
+
|
|
41
|
+
Three parts, because the chain had three links:
|
|
42
|
+
|
|
43
|
+
1. **The script refuses** a `--*` argument after the topic id, printing the correct ordering
|
|
44
|
+
and the corrected command. Nothing is sent.
|
|
45
|
+
2. **A typo'd flag is caught too** (`--tone-akc`) — the realistic mistake, and equally silent
|
|
46
|
+
before. The guard matches the *shape*, not a list of known names, which is what makes it
|
|
47
|
+
useful rather than decorative.
|
|
48
|
+
3. **The flags are documented** — in the script's usage header and in the agent instructions,
|
|
49
|
+
with a worked example showing flags before the topic id.
|
|
50
|
+
|
|
51
|
+
**What this does not do:** it does not retract the false grade from 2026-07-26. Whether a
|
|
52
|
+
mistaken grade can be corrected at all — and whether correcting one erases the evidence that
|
|
53
|
+
it was ever made, which would be the opposite defect — is a separate open question, and is
|
|
54
|
+
deliberately not answered here. It also fixes only this script; `slack-reply.sh` and
|
|
55
|
+
`whatsapp-reply.sh` share the parse-then-break shape but carry no tone flags today, so they
|
|
56
|
+
are named rather than swept in.
|
|
57
|
+
|
|
58
|
+
## What to Tell Your User
|
|
59
|
+
|
|
60
|
+
If a stray double-dash option ever appeared in the middle of one of my messages to you, that
|
|
61
|
+
was this: a setting that belonged in the plumbing ended up in the text, and whatever it was
|
|
62
|
+
supposed to do quietly didn't happen. It can't reach you as text any more — the send is
|
|
63
|
+
refused and corrected instead.
|
|
64
|
+
|
|
65
|
+
The part worth knowing: one of those settings is how I record that I disagree with one of my
|
|
66
|
+
own safety checks. When it silently failed to apply, I misread the result and marked a check
|
|
67
|
+
as faulty when it had actually been right. So this wasn't only cosmetic — it put a wrong
|
|
68
|
+
entry in the records used to judge whether those checks are any good. That entry is still
|
|
69
|
+
there; this stops more from being created.
|
|
70
|
+
|
|
71
|
+
## Summary of New Capabilities
|
|
72
|
+
|
|
73
|
+
- A flag placed after the topic id is refused with an actionable message naming the correct
|
|
74
|
+
ordering, instead of being sent to the user as literal text with its effect dropped.
|
|
75
|
+
- A misspelled flag after the topic id is caught by the same guard.
|
|
76
|
+
- The tone-advisory reaction flags (`--tone-ack`, `--tone-reason`, `--tone-complied`,
|
|
77
|
+
`--tone-decision-ref`) are documented in the script's usage header and in the agent
|
|
78
|
+
instructions, with a worked example.
|
|
79
|
+
- Existing agents receive both the new script and the new documentation — the doc block is
|
|
80
|
+
sniffed on its own marker so an agent that already carries the surrounding section is not
|
|
81
|
+
skipped.
|
|
82
|
+
|
|
83
|
+
## Evidence
|
|
84
|
+
|
|
85
|
+
**Reproduction (before),** the pre-fix template against a stub reply route:
|
|
86
|
+
|
|
87
|
+
| invocation | exit | text received by the user | override applied |
|
|
88
|
+
|---|---|---|---|
|
|
89
|
+
| `… 4242 --tone-ack B15 --tone-reason because` | `0` | `--tone-ack B15 --tone-reason because` | no |
|
|
90
|
+
|
|
91
|
+
**Observed after,** same stub:
|
|
92
|
+
|
|
93
|
+
| invocation | before | after |
|
|
94
|
+
|---|---|---|
|
|
95
|
+
| flag after the topic id | sent as message text, exit 0 | refused, exit 1, **zero requests sent** |
|
|
96
|
+
| typo'd flag after the topic id | sent as message text, exit 0 | refused, exit 1, zero requests |
|
|
97
|
+
| flags before the topic id | worked | works — text clean, `toneAdvisoryAck` + reason present |
|
|
98
|
+
| message on stdin containing a flag-shaped token | sent verbatim | sent verbatim (stdin is not inspected) |
|
|
99
|
+
| unknown flag before the topic id | `Unknown flag`, exit 1 | unchanged |
|
|
100
|
+
|
|
101
|
+
**Tests:** `tests/integration/telegram-reply-misplaced-flag.test.ts` (6) runs the real
|
|
102
|
+
template — and the pre-fix fixture — against a stub route, so "was it sent as text?" is
|
|
103
|
+
answered by the received payload rather than by reading the script.
|
|
104
|
+
`tests/unit/PostUpdateMigrator-toneAdvisoryFlagPosition.test.ts` (3) covers migration parity,
|
|
105
|
+
including the load-bearing case: an agent that already carries the tone-advisory section must
|
|
106
|
+
still receive the invocation guidance. That test was verified to FAIL when the second
|
|
107
|
+
migration block is disabled, so it cannot pass whether or not the migration exists. The full
|
|
108
|
+
telegram-reply and template-SHA surface (10 files, 51 tests) passes unchanged;
|
|
109
|
+
`npx tsc --noEmit` clean.
|
|
@@ -100,8 +100,23 @@ explicit in code: absence of a verdict is reported as absence, never as either o
|
|
|
100
100
|
between them causes no change in expiry behaviour — checked before editing.
|
|
101
101
|
- **PromiseBeacon / overdue surfacing** operate on open commitments; `pending` keeps them
|
|
102
102
|
visible, which is the stated intent of the precedent ("should NAG, not vanish").
|
|
103
|
-
- No persistence change, no migration, no config key.
|
|
104
|
-
|
|
103
|
+
- No persistence change, no migration, no config key.
|
|
104
|
+
- ~~Existing rows re-evaluate to `pending` on the next sweep without a data rewrite.~~
|
|
105
|
+
**CORRECTED 2026-07-26 — this was FALSE and is struck rather than deleted.** The
|
|
106
|
+
behavioural branch returns BEFORE the `switch` in `verifyOne`, so `mutateSync` never runs
|
|
107
|
+
and a pre-existing row is never touched. **The change is FORWARD-ONLY.**
|
|
108
|
+
Verified on the live store after deploy: 98 behavioural rows still read **74 `verified` /
|
|
109
|
+
24 `violated`**, unchanged. What the change DID do, measured over two reads 75s apart:
|
|
110
|
+
tick accumulation stopped dead (one row frozen at 163,135 where it had been climbing every
|
|
111
|
+
minute). What it did NOT do: clear the 98 stale verdicts — so the false comfort this change
|
|
112
|
+
set out to remove is still displayed for every row that predates it, and `getHealth`, which
|
|
113
|
+
now counts actual `verified` rows, will report those 74 stale stamps as verified.
|
|
114
|
+
Remedy is an open operator decision, not a reflex: either a migration resetting behavioural
|
|
115
|
+
rows to `pending` (a data rewrite this very section claimed was unnecessary) or a read
|
|
116
|
+
surface that refuses a verdict for a never-verifiable row whatever is stored.
|
|
117
|
+
**Found by reading the live surface for the exact case the change fixed** — no test caught
|
|
118
|
+
it, and the suite was green. An artifact that over-claims is the same defect class as the
|
|
119
|
+
bug it documents, which is why the wrong sentence stays visible above.
|
|
105
120
|
|
|
106
121
|
## 6. External surfaces
|
|
107
122
|
|