@a9i5k4/dsh-auto-memory 2.1.5 → 2.1.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +2 -2
- package/lib/python-setup.js +5 -5
- package/package.json +6 -3
- package/python/m7_activation_features_v2.py +392 -0
- package/python/m7_embedding_v1.py +381 -0
- package/python/policies/activation_policy_v2.json +88 -0
- package/python/policies/decision-record-activation-v2-delta-exp-override-20260824.json +21 -0
- package/python/policies/decision-record-reasoning-kind-admission-20260826.json +20 -0
- package/python/policies/decision-record-stale-gate-per-candidate-20260825.json +33 -0
- package/python/policies/recall_intent_lr_v1.json +1 -0
- package/python/verify_policy_artifact.py +122 -0
- package/python/worker_semantic_v1.py +1344 -0
- package/python/worker_v1.py +623 -0
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""M7-3 production embedding core (frozen by docs/M7-ALGORITHM-DECISION.md).
|
|
4
|
+
|
|
5
|
+
Pure re-implementation of the frozen policy for the production sidecar path
|
|
6
|
+
(the benchmark rig lives separately under python/bench/ and stays untouched):
|
|
7
|
+
|
|
8
|
+
provider bge-m3 pinned revision, CLS pooling, L2-normalized float32
|
|
9
|
+
chunk policy m7_chunk_v1 = para-512-noov (tokenizer id space,
|
|
10
|
+
greedy paragraph packing, oversized paragraph hard-split,
|
|
11
|
+
no overlap; special tokens added by the model tokenizer)
|
|
12
|
+
identity provider/model/revision/dimension/normalization/policyVersion
|
|
13
|
+
-> configHash; mismatch = stale = full rebuild
|
|
14
|
+
|
|
15
|
+
Providers:
|
|
16
|
+
hash-pre-v1 deterministic stdlib-only embedding (sha256-seeded bag of
|
|
17
|
+
token-trigram dims). Zero dependencies, zero network. Used
|
|
18
|
+
by CI and offline protocol tests. NOT a quality provider.
|
|
19
|
+
bge-m3-pre-v1 real model via transformers (lazy import; requires the
|
|
20
|
+
pinned local snapshot dir passed in the embedding config).
|
|
21
|
+
|
|
22
|
+
No DSH file reads; no writes except what the worker explicitly passes in.
|
|
23
|
+
"""
|
|
24
|
+
import hashlib
|
|
25
|
+
import json
|
|
26
|
+
import re
|
|
27
|
+
|
|
28
|
+
PROVIDER_REAL = 'bge-m3-pre-v1'
|
|
29
|
+
PROVIDER_REAL_INT8 = 'bge-m3-onnx-int8-v1'
|
|
30
|
+
PROVIDER_HASH = 'hash-pre-v1'
|
|
31
|
+
CHUNK_POLICY_VERSION = 'm7_chunk_v1'
|
|
32
|
+
CHUNK_MAX_TOKENS = 512
|
|
33
|
+
QUERY_MAX_TOKENS = 256
|
|
34
|
+
DIMENSION = 1024
|
|
35
|
+
|
|
36
|
+
RE_CHUNK = re.compile(r'^chk_[0-9a-f]{16,}$')
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def sha_hex(data):
|
|
40
|
+
return hashlib.sha256(data).hexdigest()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def canonical(v):
|
|
44
|
+
if v is None:
|
|
45
|
+
return 'null'
|
|
46
|
+
if isinstance(v, bool):
|
|
47
|
+
return 'true' if v else 'false'
|
|
48
|
+
if isinstance(v, (int, float, str)):
|
|
49
|
+
return json.dumps(v, ensure_ascii=False, separators=(',', ':'))
|
|
50
|
+
if isinstance(v, list):
|
|
51
|
+
return '[' + ','.join(canonical(x) for x in v) + ']'
|
|
52
|
+
if isinstance(v, dict):
|
|
53
|
+
return '{' + ','.join(json.dumps(str(k), ensure_ascii=False,
|
|
54
|
+
separators=(',', ':')) + ':' + canonical(v[k])
|
|
55
|
+
for k in sorted(v.keys())) + '}'
|
|
56
|
+
return 'null'
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def config_hash(provider, model_revision, dimension):
|
|
60
|
+
payload = {
|
|
61
|
+
'provider': provider,
|
|
62
|
+
'model': 'bge-m3' if provider == PROVIDER_REAL else provider,
|
|
63
|
+
'modelRevision': model_revision,
|
|
64
|
+
'dimension': dimension,
|
|
65
|
+
'normalization': 'l2_normalize',
|
|
66
|
+
'dtype': 'float32',
|
|
67
|
+
'chunkPolicyVersion': CHUNK_POLICY_VERSION,
|
|
68
|
+
'chunkParams': {'maxTokens': CHUNK_MAX_TOKENS, 'paraAligned': True,
|
|
69
|
+
'overlap': 0},
|
|
70
|
+
'queryMaxTokens': QUERY_MAX_TOKENS,
|
|
71
|
+
}
|
|
72
|
+
return 'cfgh_' + sha_hex(canonical(payload).encode('utf-8'))
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def chunk_id_for(memory_id, record_digest, ordinal):
|
|
76
|
+
return 'chk_' + sha_hex(
|
|
77
|
+
('m7-chunk-pre-v1\u0000' + memory_id + '\u0000' + record_digest +
|
|
78
|
+
'\u0000' + str(ordinal)).encode('utf-8'))[:32]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def chunk_record_token_ids(tokenizer, text):
|
|
82
|
+
"""para-512-noov over tokenizer ids; returns list of id-lists."""
|
|
83
|
+
paras = []
|
|
84
|
+
for p in text.split('\n'):
|
|
85
|
+
ids = tokenizer(p, add_special_tokens=False)['input_ids']
|
|
86
|
+
if ids:
|
|
87
|
+
paras.append(ids)
|
|
88
|
+
chunks, cur = [], []
|
|
89
|
+
for ids in paras:
|
|
90
|
+
if cur and len(cur) + len(ids) <= CHUNK_MAX_TOKENS:
|
|
91
|
+
cur.extend(ids)
|
|
92
|
+
continue
|
|
93
|
+
if cur:
|
|
94
|
+
chunks.append(cur)
|
|
95
|
+
cur = []
|
|
96
|
+
if len(ids) <= CHUNK_MAX_TOKENS:
|
|
97
|
+
cur = list(ids)
|
|
98
|
+
else:
|
|
99
|
+
for i in range(0, len(ids), CHUNK_MAX_TOKENS):
|
|
100
|
+
chunks.append(ids[i:i + CHUNK_MAX_TOKENS])
|
|
101
|
+
if cur:
|
|
102
|
+
chunks.append(cur)
|
|
103
|
+
return chunks or [[]]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class HashEmbedder:
|
|
107
|
+
"""Deterministic dependency-free embedder for tests/offline paths.
|
|
108
|
+
|
|
109
|
+
Unnormalized bag dims come from sha256 of character trigrams; the final
|
|
110
|
+
vector is L2-normalized float32. Same text -> same vector, always.
|
|
111
|
+
"""
|
|
112
|
+
|
|
113
|
+
provider = PROVIDER_HASH
|
|
114
|
+
|
|
115
|
+
def __init__(self, config):
|
|
116
|
+
self.dimension = int(config.get('dimension') or DIMENSION)
|
|
117
|
+
|
|
118
|
+
def encode_texts(self, texts):
|
|
119
|
+
import math
|
|
120
|
+
out = []
|
|
121
|
+
for t in texts:
|
|
122
|
+
vec = [0.0] * self.dimension
|
|
123
|
+
s = t if isinstance(t, str) else ''
|
|
124
|
+
for i in range(max(0, len(s) - 2)):
|
|
125
|
+
h = int.from_bytes(hashlib.sha256(s[i:i + 3].encode('utf-8',
|
|
126
|
+
'ignore')).digest()[:8], 'big')
|
|
127
|
+
vec[h % self.dimension] += 1.0
|
|
128
|
+
norm = math.sqrt(sum(x * x for x in vec)) or 1.0
|
|
129
|
+
out.append([round(x / norm, 8) for x in vec])
|
|
130
|
+
return out
|
|
131
|
+
|
|
132
|
+
def close(self):
|
|
133
|
+
pass
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class BgeM3Embedder:
|
|
137
|
+
"""Real frozen provider: transformers AutoModel, CLS pooling, L2 norm."""
|
|
138
|
+
|
|
139
|
+
provider = PROVIDER_REAL
|
|
140
|
+
|
|
141
|
+
def __init__(self, config):
|
|
142
|
+
import os
|
|
143
|
+
os.environ.setdefault('TOKENIZERS_PARALLELISM', 'false')
|
|
144
|
+
import torch
|
|
145
|
+
from transformers import AutoModel, AutoTokenizer
|
|
146
|
+
torch.set_num_threads(int(config.get('torchThreads') or 16))
|
|
147
|
+
self._torch = torch
|
|
148
|
+
path = config['modelDir']
|
|
149
|
+
revision = str(config.get('modelRevision') or '')
|
|
150
|
+
self.tokenizer = AutoTokenizer.from_pretrained(path, revision=revision)
|
|
151
|
+
self.model = AutoModel.from_pretrained(path, revision=revision,
|
|
152
|
+
dtype=torch.float32)
|
|
153
|
+
self.model.eval()
|
|
154
|
+
self.dimension = int(config.get('dimension') or DIMENSION)
|
|
155
|
+
|
|
156
|
+
def _specials(self):
|
|
157
|
+
probe = self.tokenizer('x', add_special_tokens=True)['input_ids']
|
|
158
|
+
core = self.tokenizer('x', add_special_tokens=False)['input_ids']
|
|
159
|
+
for i in range(len(probe) - len(core) + 1):
|
|
160
|
+
if probe[i:i + len(core)] == core:
|
|
161
|
+
return probe[:i], probe[i + len(core):]
|
|
162
|
+
raise RuntimeError('content not found in tokenizer probe')
|
|
163
|
+
|
|
164
|
+
def build_doc_ids(self, chunk_ids, max_total=512):
|
|
165
|
+
"""Wrap chunk token ids with the model's special tokens exactly once,
|
|
166
|
+
capping total length at the XLM-R position limit (audit P0/P1)."""
|
|
167
|
+
prefix, suffix = self._specials()
|
|
168
|
+
budget = max_total - len(prefix) - len(suffix)
|
|
169
|
+
body = list(chunk_ids)[:max(0, budget)]
|
|
170
|
+
return list(prefix) + body + list(suffix)
|
|
171
|
+
|
|
172
|
+
def encode_ids(self, ids_list, batch_size=8):
|
|
173
|
+
torch = self._torch
|
|
174
|
+
prefix, suffix = self._specials()
|
|
175
|
+
pad = (self.tokenizer.pad_token_id if self.tokenizer.pad_token_id
|
|
176
|
+
is not None else self.tokenizer.eos_token_id)
|
|
177
|
+
vecs = []
|
|
178
|
+
for i in range(0, len(ids_list), batch_size):
|
|
179
|
+
part = [list(prefix) + list(ids) + list(suffix)
|
|
180
|
+
for ids in ids_list[i:i + batch_size]]
|
|
181
|
+
maxlen = max(len(x) for x in part)
|
|
182
|
+
inp = torch.full((len(part), maxlen), pad, dtype=torch.long)
|
|
183
|
+
att = torch.zeros((len(part), maxlen), dtype=torch.long)
|
|
184
|
+
for r, ids in enumerate(part):
|
|
185
|
+
inp[r, :len(ids)] = torch.tensor(ids, dtype=torch.long)
|
|
186
|
+
att[r, :len(ids)] = 1
|
|
187
|
+
with torch.no_grad():
|
|
188
|
+
hidden = self.model(input_ids=inp,
|
|
189
|
+
attention_mask=att).last_hidden_state
|
|
190
|
+
pooled = hidden[:, 0] # CLS per repo 1_Pooling/config.json
|
|
191
|
+
pooled = torch.nn.functional.normalize(pooled, p=2, dim=1)
|
|
192
|
+
vecs.extend(pooled.to(torch.float32).numpy().tolist())
|
|
193
|
+
return vecs
|
|
194
|
+
|
|
195
|
+
def _encode_texts_via_ids(self, texts, max_tokens):
|
|
196
|
+
"""Single source of truth for text->vector: tokenize WITHOUT specials,
|
|
197
|
+
then let encode_ids wrap exactly once. Guarantees query/corpus share
|
|
198
|
+
one template, and the truncation budget RESERVES room for the special
|
|
199
|
+
tokens so wrapped length never exceeds max_tokens (audit round 2:
|
|
200
|
+
same latent overlimit class as the e5 512 crash)."""
|
|
201
|
+
prefix, suffix = self._specials()
|
|
202
|
+
budget = max(1, max_tokens - len(prefix) - len(suffix))
|
|
203
|
+
ids_list = []
|
|
204
|
+
for t in texts:
|
|
205
|
+
enc = self.tokenizer(t or '', add_special_tokens=False,
|
|
206
|
+
truncation=True, max_length=budget)
|
|
207
|
+
ids_list.append(enc['input_ids'])
|
|
208
|
+
return self.encode_ids(ids_list)
|
|
209
|
+
|
|
210
|
+
def chunk_and_encode(self, text):
|
|
211
|
+
id_chunks = chunk_record_token_ids(self.tokenizer, text)
|
|
212
|
+
return id_chunks, self.encode_ids(id_chunks)
|
|
213
|
+
|
|
214
|
+
def encode_texts(self, texts, batch_size=8):
|
|
215
|
+
"""Record-level text -> vector (chunker bypasses: caller chunks).
|
|
216
|
+
Used by the worker's hash-style flows and as the parity entry."""
|
|
217
|
+
return self._encode_texts_via_ids(texts, 512)
|
|
218
|
+
|
|
219
|
+
def encode_query(self, text):
|
|
220
|
+
# audit fix P1: previously tokenized with add_special_tokens=True and
|
|
221
|
+
# then wrapped again by encode_ids -> double specials, template drift.
|
|
222
|
+
# query_cap already reserves the specials budget (see below).
|
|
223
|
+
enc = self.tokenizer(text, add_special_tokens=False, truncation=True,
|
|
224
|
+
max_length=self.query_cap())
|
|
225
|
+
return self.encode_ids([enc['input_ids']])[0]
|
|
226
|
+
|
|
227
|
+
def query_cap(self):
|
|
228
|
+
# 256 content + specials headroom: wrapped total never exceeds cap
|
|
229
|
+
return 256 - 8
|
|
230
|
+
|
|
231
|
+
def close(self):
|
|
232
|
+
try:
|
|
233
|
+
del self.model
|
|
234
|
+
import gc
|
|
235
|
+
gc.collect()
|
|
236
|
+
except Exception:
|
|
237
|
+
pass
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class BgeM3OnnxInt8Embedder:
|
|
241
|
+
"""Quantized slim tier: Xenova/bge-m3 onnx/model_int8.onnx (dynamic int8)
|
|
242
|
+
via onnxruntime. Same XLM-R tokenizer + CLS pooling + L2 norm contract as
|
|
243
|
+
BgeM3Embedder; text->vector template methods are byte-twin copies of the
|
|
244
|
+
parent so query/corpus share one wrapping path. L2 head-to-head 2026-08-25:
|
|
245
|
+
R@5 identical to fp32 (0.925), encode ~6x faster, vec cos mean 0.975.
|
|
246
|
+
Identity dtype differs from fp32 -> switching providers invalidates the
|
|
247
|
+
vector store by design (stale -> rebuild)."""
|
|
248
|
+
|
|
249
|
+
provider = PROVIDER_REAL_INT8
|
|
250
|
+
|
|
251
|
+
def __init__(self, config):
|
|
252
|
+
import os
|
|
253
|
+
os.environ.setdefault('TOKENIZERS_PARALLELISM', 'false')
|
|
254
|
+
import numpy as np
|
|
255
|
+
import onnxruntime as ort
|
|
256
|
+
from transformers import AutoTokenizer
|
|
257
|
+
self._np = np
|
|
258
|
+
base = config['modelDir']
|
|
259
|
+
onnx_rel = str(config.get('onnxFile') or 'onnx/model_int8.onnx')
|
|
260
|
+
self.session = ort.InferenceSession(
|
|
261
|
+
os.path.join(base, *onnx_rel.split('/')),
|
|
262
|
+
providers=['CPUExecutionProvider'])
|
|
263
|
+
self.tokenizer = AutoTokenizer.from_pretrained(base)
|
|
264
|
+
self._inp = self.session.get_inputs()[0].name
|
|
265
|
+
self._att = self.session.get_inputs()[1].name
|
|
266
|
+
self.dimension = int(config.get('dimension') or DIMENSION)
|
|
267
|
+
|
|
268
|
+
def _specials(self):
|
|
269
|
+
probe = self.tokenizer('x', add_special_tokens=True)['input_ids']
|
|
270
|
+
core = self.tokenizer('x', add_special_tokens=False)['input_ids']
|
|
271
|
+
for i in range(len(probe) - len(core) + 1):
|
|
272
|
+
if probe[i:i + len(core)] == core:
|
|
273
|
+
return probe[:i], probe[i + len(core):]
|
|
274
|
+
raise RuntimeError('content not found in tokenizer probe')
|
|
275
|
+
|
|
276
|
+
def build_doc_ids(self, chunk_ids, max_total=512):
|
|
277
|
+
prefix, suffix = self._specials()
|
|
278
|
+
budget = max_total - len(prefix) - len(suffix)
|
|
279
|
+
body = list(chunk_ids)[:max(0, budget)]
|
|
280
|
+
return list(prefix) + body + list(suffix)
|
|
281
|
+
|
|
282
|
+
def encode_ids(self, ids_list, batch_size=16):
|
|
283
|
+
np = self._np
|
|
284
|
+
prefix, suffix = self._specials()
|
|
285
|
+
pad = (self.tokenizer.pad_token_id if self.tokenizer.pad_token_id
|
|
286
|
+
is not None else self.tokenizer.eos_token_id)
|
|
287
|
+
vecs = []
|
|
288
|
+
for i in range(0, len(ids_list), batch_size):
|
|
289
|
+
part = [list(prefix) + list(ids) + list(suffix)
|
|
290
|
+
for ids in ids_list[i:i + batch_size]]
|
|
291
|
+
maxlen = max(len(x) for x in part)
|
|
292
|
+
inp = np.full((len(part), maxlen), pad, dtype=np.int64)
|
|
293
|
+
att = np.zeros((len(part), maxlen), dtype=np.int64)
|
|
294
|
+
for r, ids in enumerate(part):
|
|
295
|
+
inp[r, :len(ids)] = np.asarray(ids, dtype=np.int64)
|
|
296
|
+
att[r, :len(ids)] = 1
|
|
297
|
+
hidden = self.session.run(None, {self._inp: inp,
|
|
298
|
+
self._att: att})[0]
|
|
299
|
+
pooled = hidden[:, 0].astype(np.float32) # CLS per 1_Pooling
|
|
300
|
+
norms = np.linalg.norm(pooled, axis=1, keepdims=True)
|
|
301
|
+
pooled = pooled / np.maximum(norms, 1e-12)
|
|
302
|
+
vecs.extend([row.tolist() for row in pooled])
|
|
303
|
+
return vecs
|
|
304
|
+
|
|
305
|
+
def _encode_texts_via_ids(self, texts, max_tokens):
|
|
306
|
+
prefix, suffix = self._specials()
|
|
307
|
+
budget = max(1, max_tokens - len(prefix) - len(suffix))
|
|
308
|
+
ids_list = []
|
|
309
|
+
for t in texts:
|
|
310
|
+
enc = self.tokenizer(t or '', add_special_tokens=False,
|
|
311
|
+
truncation=True, max_length=budget)
|
|
312
|
+
ids_list.append(enc['input_ids'])
|
|
313
|
+
return self.encode_ids(ids_list)
|
|
314
|
+
|
|
315
|
+
def chunk_and_encode(self, text):
|
|
316
|
+
id_chunks = chunk_record_token_ids(self.tokenizer, text)
|
|
317
|
+
return id_chunks, self.encode_ids(id_chunks)
|
|
318
|
+
|
|
319
|
+
def encode_texts(self, texts, batch_size=16):
|
|
320
|
+
return self._encode_texts_via_ids(texts, 512)
|
|
321
|
+
|
|
322
|
+
def encode_query(self, text):
|
|
323
|
+
enc = self.tokenizer(text, add_special_tokens=False, truncation=True,
|
|
324
|
+
max_length=self.query_cap())
|
|
325
|
+
return self.encode_ids([enc['input_ids']])[0]
|
|
326
|
+
|
|
327
|
+
def query_cap(self):
|
|
328
|
+
return 256 - 8
|
|
329
|
+
|
|
330
|
+
def close(self):
|
|
331
|
+
try:
|
|
332
|
+
del self.session
|
|
333
|
+
import gc
|
|
334
|
+
gc.collect()
|
|
335
|
+
except Exception:
|
|
336
|
+
pass
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def load_embedder(config):
|
|
340
|
+
provider = str(config.get('provider') or '')
|
|
341
|
+
if provider == PROVIDER_HASH:
|
|
342
|
+
return HashEmbedder(config)
|
|
343
|
+
if provider == PROVIDER_REAL:
|
|
344
|
+
return BgeM3Embedder(config)
|
|
345
|
+
if provider == PROVIDER_REAL_INT8:
|
|
346
|
+
return BgeM3OnnxInt8Embedder(config)
|
|
347
|
+
raise ValueError('unknown embedding provider: %r' % (provider,))
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def identity_block(provider, config):
|
|
351
|
+
model_name = provider
|
|
352
|
+
dtype = 'float32'
|
|
353
|
+
if provider == PROVIDER_REAL:
|
|
354
|
+
model_name = 'bge-m3'
|
|
355
|
+
elif provider == PROVIDER_REAL_INT8:
|
|
356
|
+
model_name = 'bge-m3-int8'
|
|
357
|
+
# different dtype -> fp32-built vector stores are stale under the
|
|
358
|
+
# int8 tier and must rebuild (provider switch is never silent)
|
|
359
|
+
dtype = 'int8-dynamic-onnx'
|
|
360
|
+
return {
|
|
361
|
+
'schemaVersion': 1,
|
|
362
|
+
'namespace': 'dsh-auto-memory',
|
|
363
|
+
'policyVersion': 'semantic_vectors_v1',
|
|
364
|
+
'provider': provider,
|
|
365
|
+
'model': model_name,
|
|
366
|
+
'modelRevision': str(config.get('modelRevision') or 'hash'),
|
|
367
|
+
'dimension': int(config.get('dimension') or DIMENSION),
|
|
368
|
+
'normalization': 'l2_normalize',
|
|
369
|
+
'dtype': dtype,
|
|
370
|
+
'chunkPolicyVersion': CHUNK_POLICY_VERSION,
|
|
371
|
+
'configHash': config_hash(provider,
|
|
372
|
+
str(config.get('modelRevision') or 'hash'),
|
|
373
|
+
int(config.get('dimension') or DIMENSION)),
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def identity_matches(block, provider, config):
|
|
378
|
+
want = identity_block(provider, config)
|
|
379
|
+
keys = ('provider', 'model', 'modelRevision', 'dimension',
|
|
380
|
+
'normalization', 'dtype', 'chunkPolicyVersion', 'configHash')
|
|
381
|
+
return all(block.get(k) == want.get(k) for k in keys)
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"policyVersion": "activation_policy_v2",
|
|
4
|
+
"parentPolicyVersion": "m7_semantic_threshold_v1",
|
|
5
|
+
"createdAt": "2026-08-25",
|
|
6
|
+
"runId": "label-review-cal20260824-1954",
|
|
7
|
+
"goldDigest": "e74ccdcda998ae51596adf8119c818ac84fc52a4fa657fd687daeda7d14a072d",
|
|
8
|
+
"mode": "shadow-candidate",
|
|
9
|
+
"decisionRecordId": "activation-v2-delta-exp-override-20260824",
|
|
10
|
+
"deltaDeviationFromBrief": {
|
|
11
|
+
"briefFrozen": 0.02,
|
|
12
|
+
"effective": 0.03,
|
|
13
|
+
"reason": "production-completeness refit: delta=0.02 fails gates (cal-0008 emitOnP=1); nearest passing = delta_exp=0.03; requires retro-ratification"
|
|
14
|
+
},
|
|
15
|
+
"decisionOrder": [
|
|
16
|
+
"js_hard_gates",
|
|
17
|
+
"lane_decision",
|
|
18
|
+
"explicit_lane",
|
|
19
|
+
"proactive_lane",
|
|
20
|
+
"completeness_margin",
|
|
21
|
+
"decision"
|
|
22
|
+
],
|
|
23
|
+
"thresholds": {
|
|
24
|
+
"tauLane": 0.45,
|
|
25
|
+
"tauHi": 0.45,
|
|
26
|
+
"tauLo": 0.35,
|
|
27
|
+
"deltaExp": 0.03,
|
|
28
|
+
"deltaPro": 0.05
|
|
29
|
+
},
|
|
30
|
+
"echoVeto": {
|
|
31
|
+
"scope": "proactive-lane-only",
|
|
32
|
+
"containmentArm": 0.3,
|
|
33
|
+
"denseTopArm": 0.7,
|
|
34
|
+
"requiresMarkZero": true,
|
|
35
|
+
"requiresIntentBelow": 0.5
|
|
36
|
+
},
|
|
37
|
+
"completenessGate": {
|
|
38
|
+
"phase": 1,
|
|
39
|
+
"lexicon": [
|
|
40
|
+
"对比",
|
|
41
|
+
"分别",
|
|
42
|
+
"两个",
|
|
43
|
+
"一起",
|
|
44
|
+
"都调",
|
|
45
|
+
"各自"
|
|
46
|
+
],
|
|
47
|
+
"rule": "status!=complete -> max prefetch",
|
|
48
|
+
"outputs": [
|
|
49
|
+
"requiredTargetCount",
|
|
50
|
+
"resolvedTargetCount",
|
|
51
|
+
"status"
|
|
52
|
+
]
|
|
53
|
+
},
|
|
54
|
+
"repetition": {
|
|
55
|
+
"round": "logging-only",
|
|
56
|
+
"suppressToPrefetchAllowed": true,
|
|
57
|
+
"activateOnCountsAlone": false
|
|
58
|
+
},
|
|
59
|
+
"hardGates": [
|
|
60
|
+
"harmful",
|
|
61
|
+
"correction",
|
|
62
|
+
"ignored",
|
|
63
|
+
"stale",
|
|
64
|
+
"wrong_scope",
|
|
65
|
+
"pii_class_never_proactive"
|
|
66
|
+
],
|
|
67
|
+
"reasonCodes": [
|
|
68
|
+
"hard_gate_harmful",
|
|
69
|
+
"echo_veto_proactive",
|
|
70
|
+
"explicit_lane",
|
|
71
|
+
"explicit_lane_weak",
|
|
72
|
+
"completeness_complete",
|
|
73
|
+
"completeness_partial",
|
|
74
|
+
"completeness_unknown",
|
|
75
|
+
"proactive_margin",
|
|
76
|
+
"margin_below_delta",
|
|
77
|
+
"suppress_low_signal"
|
|
78
|
+
],
|
|
79
|
+
"offlineMetrics86": {
|
|
80
|
+
"actPrecision": 1.0,
|
|
81
|
+
"actRecall": 0.289,
|
|
82
|
+
"emitOnP": 0,
|
|
83
|
+
"sViolations": 0,
|
|
84
|
+
"emits": 11,
|
|
85
|
+
"emitCorrectA": 11
|
|
86
|
+
},
|
|
87
|
+
"configHash": "cfgh_31e6d977a2c40d5e8bf4edf900557b8b"
|
|
88
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"decisionId": "activation-v2-delta-exp-override-20260824",
|
|
4
|
+
"policyVersion": "activation_policy_v2",
|
|
5
|
+
"previousDeltaExp": 0.02,
|
|
6
|
+
"approvedDeltaExp": 0.03,
|
|
7
|
+
"reason": "production completeness semantics exposed cal-0008 prefetch-to-emit violation at 0.02",
|
|
8
|
+
"rollback": "0.02 historical only",
|
|
9
|
+
"evidence": {
|
|
10
|
+
"sampleId": "cal-0008",
|
|
11
|
+
"action": "P",
|
|
12
|
+
"denseTop": 0.654468,
|
|
13
|
+
"margin": 0.029178
|
|
14
|
+
},
|
|
15
|
+
"runId": "label-review-cal20260824-1954",
|
|
16
|
+
"goldDigest": "e74ccdcda998ae51596adf8119c818ac84fc52a4fa657fd687daeda7d14a072d",
|
|
17
|
+
"approvedBy": [
|
|
18
|
+
"user",
|
|
19
|
+
"mainAgent"
|
|
20
|
+
]
|
|
21
|
+
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "decision-record-reasoning-kind-admission-20260826",
|
|
3
|
+
"date": "2026-08-26",
|
|
4
|
+
"author": "calibration-agent (ox-alpha)",
|
|
5
|
+
"approvedBy": "user (产品目标正式裁定: 监听模型思维链与可见输出是本系统核心, 而非仅用户输入)",
|
|
6
|
+
"type": "contract-revision (M5 segment kind whitelist)",
|
|
7
|
+
"summary": "'reasoning' admitted as a legal Segment kind (weight 0.5 in lexical query plan), superseding the M4/M5-era rule that reasoning never enters context.",
|
|
8
|
+
"trigger": "controlled live shadow round 2: envelope drops 'trigger:invalid:kind' + 'window:invalid:kind' — the new CoT segments (kind='reasoning') were rejected by the M5 whitelist ['user','tool_call','tool_result','assistant']",
|
|
9
|
+
"change": [
|
|
10
|
+
"lib/context-bridge.js: segment kind whitelist + 'reasoning'",
|
|
11
|
+
"lib/shadow-retrieval.js: trigger.kind whitelist + 'reasoning'; query-plan origin map adds 'reasoning' at weight 0.5 (between recent-user 0.8 and tool-result 0.6... actually below tool-result, above assistant 0.2)",
|
|
12
|
+
"smoke-test-m51-pre.mjs: old rejection assertion replaced with admission assertion"
|
|
13
|
+
],
|
|
14
|
+
"safety": [
|
|
15
|
+
"reasoning remains a FEATURE only; no hard gate may read it",
|
|
16
|
+
"CoT aggregation buffer bounded 4096 chars; flush >=512 chars or >=1500ms",
|
|
17
|
+
"purge on assoc-disable clears CoT buffers (zero-retention contract preserved)"
|
|
18
|
+
],
|
|
19
|
+
"references": ["docs/COT-WATCH-RFC.md", "docs/M5-CONTRACT.md"]
|
|
20
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "decision-record-stale-gate-per-candidate-20260825",
|
|
3
|
+
"date": "2026-08-25",
|
|
4
|
+
"author": "calibration-agent (ox-alpha)",
|
|
5
|
+
"approvedBy": "user (直接授权实施, 主 Agent 沟通成本考量)",
|
|
6
|
+
"type": "implementation-defect-fix (non-policy-parameter change)",
|
|
7
|
+
"summary": "fv2 stale/correction hard gates narrowed from session-wide any() to per-candidate-topk.",
|
|
8
|
+
"trigger": {
|
|
9
|
+
"source": "M7-8 controlled live shadow (20 scripted requests, user-restarted host PID 5540)",
|
|
10
|
+
"observation": "94/94 observations suppressed with reasonCodes=[hard_gate_stale], features=null",
|
|
11
|
+
"rootCause": "worker constructed hardGates.stale as any(e.freshness=='stale' for e in payload.evidence); on real traffic every context_push carries >=1 stale evidence entry (anchored files updated today -> all M5 historical evidence judged stale), so fv2 was permanently muted"
|
|
12
|
+
},
|
|
13
|
+
"change": {
|
|
14
|
+
"file": "python/worker_semantic_v1.py (_fv2_shadow_decide gate construction only)",
|
|
15
|
+
"semantics": "stale/correction gates fire ONLY when the affected evidence memoryId is among the CURRENT top-K candidates; unrelated evidence never gates",
|
|
16
|
+
"untouched": [
|
|
17
|
+
"m7_activation_features_v2.py (decision core unchanged)",
|
|
18
|
+
"python/policies/*.json (versions, configHashes unchanged: activation cfgh_31e6d977..., intent cfgh_69465e69...)",
|
|
19
|
+
"golden-parity-fixtures-v1.jsonl (fixtures carry explicit hardGates inputs; all False -> zero output drift)"
|
|
20
|
+
]
|
|
21
|
+
},
|
|
22
|
+
"verification": [
|
|
23
|
+
"scenario matrix 7/7 PASS (related-stale fires / unrelated-stale open / empty open / correction symmetric / end-to-end emit restored / related-stale still suppresses)",
|
|
24
|
+
"smoke m73 59/59, m79 20/20",
|
|
25
|
+
"py_compile OK"
|
|
26
|
+
],
|
|
27
|
+
"residualRequirements": [
|
|
28
|
+
"user restarts 3080 again to load the fix",
|
|
29
|
+
"re-run the same 20-request controlled shadow and compare against archived rows (artifacts/m7-live-pre/controlled-shadow-rows-20260825.json): expected = stale-gate rows disappear, explicit recall requests produce emit/prefetch decisions consistent with offline replay",
|
|
30
|
+
"parity fixtures should gain stale-scenario cases in the next fixture regeneration window"
|
|
31
|
+
],
|
|
32
|
+
"references": ["docs/M7-ACTIVATION-V2-CONTROLLED-SHADOW.md", "docs/M7-ACTIVATION-V2-HOLDEDOUT-EVAL.md"]
|
|
33
|
+
}
|