imprint-layer 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
imprint/__init__.py ADDED
@@ -0,0 +1,14 @@
1
+ """Imprint: a personality calibration layer for any chat model.
2
+
3
+ Learns how one user wants an assistant to talk (8 trait dials plus standing
4
+ rules) from their own messages, and renders a short plain-text block for the
5
+ host to inject into the system prompt.
6
+ """
7
+ from .core import Imprint, LearnResult, new_profile
8
+ from .embedders import Embedder, MiniLMEmbedder
9
+ from .steering import Steer, detect_steering
10
+ from .traits import TRAIT_NAMES, band_step, describe
11
+
12
+ __all__ = ["Imprint", "LearnResult", "new_profile", "Embedder", "MiniLMEmbedder",
13
+ "Steer", "detect_steering", "TRAIT_NAMES", "band_step", "describe"]
14
+ __version__ = "0.3.0"
imprint/core.py ADDED
@@ -0,0 +1,322 @@
1
+ """Imprint core: learn from each user message, render a calibration block.
2
+
3
+ imp = Imprint("me.profile.json")
4
+ imp.learn(user_text, timestamp=message_time) # before building the prompt
5
+ system_prompt = base_prompt + "\\n\\n" + imp.directive()
6
+
7
+ Model-agnostic: nothing here calls a chat model. The host supplies messages
8
+ (with their timestamps) and a place to inject a text block.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import hashlib
13
+ import json
14
+ import logging
15
+ import os
16
+ from copy import deepcopy
17
+ from dataclasses import dataclass, field
18
+ from datetime import datetime, timedelta, timezone
19
+ from pathlib import Path
20
+ from typing import Iterable
21
+
22
+ import numpy as np
23
+
24
+ from . import rules as rules_mod
25
+ from .embedders import Embedder, default_embedder
26
+ from .steering import Steer, detect_steering, has_praise, last_retraction_end
27
+ from .traits import (TRAIT_DEFAULTS, TRAIT_META, TRAIT_NAMES, TRAIT_POLES, band_step,
28
+ describe)
29
+
30
+ log = logging.getLogger("imprint")
31
+
32
+ SCHEMA = "imprint.profile/1"
33
+ MAX_CONFIDENCE = 0.95
34
+ MIN_CONFIDENCE = 0.1
35
+ BASE_LR = 0.15
36
+ CONF_INC = 0.008
37
+ STEER_CONF_INC = CONF_INC * 4
38
+ HALF_LIFE = timedelta(days=30)
39
+ VIBE_THRESHOLD = 0.16
40
+ VIBE_THRESHOLD_PERSON = 0.11 # humor / warmth / flirt score lower in MiniLM
41
+ PERSON_TRAITS = frozenset({"humor", "warmth", "flirt"})
42
+ MAX_VIBE_TRAITS_PER_MSG = 2
43
+ # A score that clears its threshold by a hair is noise, not signal: ignore
44
+ # vibe strengths below this (≈ 0.04–0.045 past the threshold).
45
+ MIN_VIBE_STRENGTH = 0.05
46
+ MAX_OBSERVATIONS = 200
47
+ MAX_RULES = 12
48
+
49
+
50
+ @dataclass
51
+ class LearnResult:
52
+ steered: list[Steer] = field(default_factory=list)
53
+ vibe: list[dict] = field(default_factory=list)
54
+ rules_added: list[str] = field(default_factory=list)
55
+ candidates_seen: int = 0
56
+ promoted: list[str] = field(default_factory=list)
57
+
58
+ @property
59
+ def changed_traits(self) -> bool:
60
+ return bool(self.steered or self.vibe)
61
+
62
+
63
+ def _utc(ts) -> datetime | None:
64
+ if ts is None:
65
+ return None
66
+ if isinstance(ts, (int, float)):
67
+ return datetime.fromtimestamp(ts, tz=timezone.utc)
68
+ if isinstance(ts, str):
69
+ try:
70
+ ts = datetime.fromisoformat(ts.replace("Z", "+00:00"))
71
+ except ValueError:
72
+ return None
73
+ if not isinstance(ts, datetime):
74
+ return None
75
+ return (ts if ts.tzinfo else ts.replace(tzinfo=timezone.utc)).astimezone(timezone.utc)
76
+
77
+
78
+ def _iso(dt: datetime | None) -> str | None:
79
+ return dt.isoformat(timespec="milliseconds") if dt else None
80
+
81
+
82
+ def _now_iso() -> str:
83
+ return datetime.now(timezone.utc).isoformat(timespec="milliseconds")
84
+
85
+
86
+ def new_profile(traits: Iterable[str] = TRAIT_NAMES) -> dict:
87
+ now = _now_iso()
88
+ return {
89
+ "schema": SCHEMA, "created": now, "lastUpdated": now, "interactions": 0,
90
+ "traits": {t: {"value": TRAIT_DEFAULTS[t]["value"],
91
+ "confidence": TRAIT_DEFAULTS[t]["confidence"],
92
+ "description": TRAIT_META[t]} for t in traits},
93
+ "rules": [], "candidateRules": [], "observations": [],
94
+ }
95
+
96
+
97
+ class Imprint:
98
+ """One user's Imprint profile.
99
+
100
+ profile_path: JSON file to load/save (None = in memory only).
101
+ embedder: object with .name and .embed(list[str]) -> (n, d) array.
102
+ Defaults to local all-MiniLM-L6-v2 (sentence-transformers).
103
+ traits: subset of the 8 traits to learn and inject (default: all).
104
+ axes_cache: optional .npz path to cache the embedded trait poles.
105
+ warmup: interactions before directive() returns anything.
106
+ """
107
+
108
+ def __init__(self, profile_path: str | os.PathLike | None = None, *,
109
+ embedder: Embedder | None = None,
110
+ traits: Iterable[str] | None = None,
111
+ axes_cache: str | os.PathLike | None = None,
112
+ warmup: int = 5,
113
+ max_inject_rules: int = 4,
114
+ corroboration_gap: timedelta = timedelta(hours=1),
115
+ candidate_ttl: timedelta = timedelta(days=90)):
116
+ self.traits = tuple(traits) if traits is not None else TRAIT_NAMES
117
+ unknown = set(self.traits) - set(TRAIT_NAMES)
118
+ if unknown:
119
+ raise ValueError(f"unknown traits: {sorted(unknown)}")
120
+ self.profile_path = Path(profile_path) if profile_path else None
121
+ self.axes_cache = Path(axes_cache) if axes_cache else None
122
+ self.warmup = warmup
123
+ self.max_inject_rules = max_inject_rules
124
+ self.corroboration_gap = corroboration_gap
125
+ self.candidate_ttl = candidate_ttl
126
+ self._embedder = embedder
127
+ self._axes: dict[str, tuple[np.ndarray, np.ndarray]] | None = None
128
+ self.profile = self._load()
129
+
130
+ # --- persistence ------------------------------------------------------------
131
+
132
+ def _load(self) -> dict:
133
+ p = new_profile(self.traits)
134
+ if self.profile_path and self.profile_path.exists():
135
+ data = json.loads(self.profile_path.read_text(encoding="utf-8"))
136
+ p.update({k: v for k, v in data.items() if k != "traits"})
137
+ for t, v in (data.get("traits") or {}).items():
138
+ if t in TRAIT_NAMES:
139
+ p["traits"][t] = {**p["traits"].get(t, {}), **v}
140
+ return p
141
+
142
+ def save(self) -> None:
143
+ if not self.profile_path:
144
+ return
145
+ self.profile["lastUpdated"] = _now_iso()
146
+ self.profile["observations"] = self.profile["observations"][-MAX_OBSERVATIONS:]
147
+ self.profile_path.parent.mkdir(parents=True, exist_ok=True)
148
+ tmp = self.profile_path.with_suffix(self.profile_path.suffix + ".tmp")
149
+ tmp.write_text(json.dumps(self.profile, indent=2, ensure_ascii=False) + "\n",
150
+ encoding="utf-8")
151
+ tmp.replace(self.profile_path)
152
+
153
+ # --- embedding axes -----------------------------------------------------------
154
+
155
+ @property
156
+ def embedder(self) -> Embedder:
157
+ if self._embedder is None:
158
+ self._embedder = default_embedder()
159
+ return self._embedder
160
+
161
+ def _fingerprint(self) -> str:
162
+ blob = json.dumps([self.embedder.name, TRAIT_POLES], sort_keys=True)
163
+ return hashlib.sha256(blob.encode()).hexdigest()[:16]
164
+
165
+ def axes(self) -> dict[str, tuple[np.ndarray, np.ndarray]]:
166
+ if self._axes is not None:
167
+ return self._axes
168
+ fp = self._fingerprint()
169
+ if self.axes_cache and self.axes_cache.exists():
170
+ data = np.load(self.axes_cache, allow_pickle=False)
171
+ if bytes(data["fingerprint"]).decode() == fp:
172
+ self._axes = {t: (data[f"{t}__high"], data[f"{t}__low"]) for t in TRAIT_POLES}
173
+ return self._axes
174
+ axes = {t: (_centroid(self.embedder, p["high"]), _centroid(self.embedder, p["low"]))
175
+ for t, p in TRAIT_POLES.items()}
176
+ if self.axes_cache:
177
+ self.axes_cache.parent.mkdir(parents=True, exist_ok=True)
178
+ np.savez(self.axes_cache, fingerprint=np.frombuffer(fp.encode(), dtype=np.uint8),
179
+ **{f"{t}__{k}": v for t, (h, l) in axes.items()
180
+ for k, v in (("high", h), ("low", l))})
181
+ self._axes = axes
182
+ return axes
183
+
184
+ def score(self, text: str) -> dict[str, float]:
185
+ """Bipolar vibe score per trait: cos(text, high) − cos(text, low)."""
186
+ m = _unit(self.embedder.embed([text])[0])
187
+ return {t: float(m @ h - m @ l) for t, (h, l) in self.axes().items()
188
+ if t in self.traits}
189
+
190
+ # --- learning -------------------------------------------------------------------
191
+
192
+ def vibe_signals(self, text: str) -> list[dict]:
193
+ scores = self.score(text)
194
+ person_clears = any(abs(scores.get(t, 0)) > VIBE_THRESHOLD_PERSON for t in PERSON_TRAITS)
195
+ cands = []
196
+ for t, s in scores.items():
197
+ thr = VIBE_THRESHOLD_PERSON if t in PERSON_TRAITS else VIBE_THRESHOLD
198
+ if abs(s) <= thr:
199
+ continue
200
+ if t == "formality" and person_clears:
201
+ continue # person-shaped asks read "casual" to small embedders
202
+ strength = min(1.0, (abs(s) - thr) / (1 - thr))
203
+ if strength < MIN_VIBE_STRENGTH:
204
+ continue
205
+ cands.append((t, s, strength))
206
+ cands.sort(key=lambda c: -abs(c[1]))
207
+ return [{"trait": t, "direction": 1.0 if s > 0 else -1.0, "strength": st,
208
+ "score": round(s, 4)} for t, s, st in cands[:MAX_VIBE_TRAITS_PER_MSG]]
209
+
210
+ def learn(self, text: str, timestamp=None, *, save: bool = True) -> LearnResult:
211
+ """Learn from one user message. `timestamp` = when the user sent it.
212
+
213
+ Without a timestamp, observations record null (never processing time),
214
+ no confidence decay is applied, and candidates can't be corroborated.
215
+ """
216
+ res = LearnResult()
217
+ text = (text or "").strip()
218
+ ts = _utc(timestamp)
219
+ if ts is None:
220
+ log.warning("imprint: no message timestamp (got %r); observation "
221
+ "timestamps will be null", timestamp)
222
+ p = self.profile
223
+ p["interactions"] = int(p.get("interactions") or 0) + 1
224
+ if not text:
225
+ self._finish(save)
226
+ return res
227
+
228
+ res.steered = [s for s in detect_steering(text) if s.trait in self.traits]
229
+ for st in res.steered:
230
+ tr = p["traits"][st.trait]
231
+ self._decay(st.trait, ts)
232
+ before = tr["value"]
233
+ tr["value"] = band_step(st.trait, before, st.direction)
234
+ tr["confidence"] = min(MAX_CONFIDENCE, tr["confidence"] + STEER_CONF_INC)
235
+ self._observe(ts, st.trait, st.direction, "steer",
236
+ f"{st.match[:80]!r} {before:.3f}→{tr['value']:.3f}")
237
+
238
+ if not res.steered:
239
+ cut = last_retraction_end(text)
240
+ standing = text[cut:] if cut >= 0 else text
241
+ if any(c.isalpha() for c in standing):
242
+ sigs = self.vibe_signals(standing)
243
+ if has_praise(text): # praise never lowers warmth or humor
244
+ sigs = [s for s in sigs
245
+ if not (s["trait"] in ("warmth", "humor") and s["direction"] < 0)]
246
+ for s in sigs:
247
+ tr = p["traits"][s["trait"]]
248
+ self._decay(s["trait"], ts)
249
+ lr = BASE_LR * (1 - tr["confidence"]) * s["strength"]
250
+ tr["value"] = max(0.0, min(1.0, tr["value"] + s["direction"] * lr))
251
+ tr["confidence"] = min(MAX_CONFIDENCE, tr["confidence"] + CONF_INC)
252
+ self._observe(ts, s["trait"], s["direction"], "vibe", f"score={s['score']:+.3f}")
253
+ res.vibe = sigs
254
+
255
+ cap = rules_mod.capture(text)
256
+ for r in cap.rules:
257
+ if not any(rules_mod.rules_near_dup(r, e) for e in p["rules"]):
258
+ p["rules"].append(r)
259
+ res.rules_added.append(r)
260
+ res.candidates_seen = len(cap.candidates)
261
+ _, promoted = rules_mod.update_candidates(
262
+ p.setdefault("candidateRules", []), cap.candidates, ts,
263
+ gap=self.corroboration_gap, ttl=self.candidate_ttl)
264
+ for r in promoted:
265
+ if not any(rules_mod.rules_near_dup(r, e) for e in p["rules"]):
266
+ p["rules"].append(r)
267
+ res.promoted.append(r)
268
+ p["rules"] = p["rules"][-MAX_RULES:]
269
+ self._finish(save)
270
+ return res
271
+
272
+ def _finish(self, save: bool) -> None:
273
+ if save:
274
+ self.save()
275
+
276
+ def _observe(self, ts, trait, direction, source, detail) -> None:
277
+ self.profile["observations"].append({
278
+ "timestamp": _iso(ts), "trait": trait, "direction": direction,
279
+ "source": source, "detail": detail})
280
+
281
+ def _decay(self, trait: str, ts: datetime | None) -> None:
282
+ if ts is None:
283
+ return # no message time, no decay (never substitute processing time)
284
+ last = next((o["timestamp"] for o in reversed(self.profile["observations"])
285
+ if o["trait"] == trait and o.get("timestamp")), None)
286
+ last_dt = _utc(last)
287
+ if last_dt is None or ts <= last_dt:
288
+ return
289
+ tr = self.profile["traits"][trait]
290
+ f = 2 ** (-(ts - last_dt) / HALF_LIFE)
291
+ tr["confidence"] = max(MIN_CONFIDENCE, MIN_CONFIDENCE + (tr["confidence"] - MIN_CONFIDENCE) * f)
292
+
293
+ # --- output -------------------------------------------------------------------------
294
+
295
+ def directive(self) -> str:
296
+ """The calibration block to inject into the system prompt.
297
+
298
+ Byte-stable between learning events (no counters), so it can sit in a
299
+ cached system prompt. Empty until `warmup` interactions.
300
+ """
301
+ p = self.profile
302
+ if int(p.get("interactions") or 0) < self.warmup:
303
+ return ""
304
+ lines = ["Learned preferences for this user. Follow them unless the user asks otherwise.", ""]
305
+ lines += [describe(t, p["traits"][t]["value"]) for t in TRAIT_NAMES if t in self.traits]
306
+ injected = p["rules"][-self.max_inject_rules:] if self.max_inject_rules else []
307
+ if injected:
308
+ lines += ["", "Standing rules from the user:"] + [f"- {r}" for r in injected]
309
+ return "\n".join(lines)
310
+
311
+ def snapshot(self) -> dict:
312
+ return deepcopy(self.profile)
313
+
314
+
315
+ def _unit(v) -> np.ndarray:
316
+ v = np.asarray(v, dtype=np.float32).reshape(-1)
317
+ n = float(np.linalg.norm(v))
318
+ return v / n if n > 1e-12 else v
319
+
320
+
321
+ def _centroid(embedder: Embedder, phrases: list[str]) -> np.ndarray:
322
+ return _unit(np.mean(np.asarray(embedder.embed(phrases), dtype=np.float32), axis=0))
imprint/dashboard.py ADDED
@@ -0,0 +1,85 @@
1
+ """Read-only dashboard: render any Imprint profile JSON as a static HTML page.
2
+
3
+ python -m imprint.dashboard my.profile.json -o dashboard.html
4
+
5
+ Shows each trait's value and band label, the standing rules, candidate rules
6
+ awaiting corroboration, and observation counts. It never writes the profile.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import html
12
+ import json
13
+ from collections import Counter
14
+ from pathlib import Path
15
+
16
+ from .traits import BANDS, TRAIT_LABELS, TRAIT_META, TRAIT_NAMES, band_index
17
+
18
+ _CSS = """
19
+ :root{--bg:#fafaf9;--fg:#1c1917;--muted:#78716c;--card:#fff;--line:#e7e5e4;--bar:#4f46e5;--edge:#a8a29e}
20
+ @media (prefers-color-scheme:dark){:root{--bg:#1c1917;--fg:#f5f5f4;--muted:#a8a29e;--card:#292524;--line:#44403c;--bar:#818cf8;--edge:#78716c}}
21
+ *{box-sizing:border-box}body{margin:0;background:var(--bg);color:var(--fg);font:15px/1.5 system-ui,sans-serif}
22
+ main{max-width:880px;margin:0 auto;padding:24px 16px}h1{font-size:22px;margin:0 0 4px}h2{font-size:16px;margin:28px 0 10px}
23
+ .muted{color:var(--muted)}.card{background:var(--card);border:1px solid var(--line);border-radius:10px;padding:14px 16px;margin:10px 0}
24
+ .row{display:grid;grid-template-columns:130px 1fr 56px;gap:12px;align-items:center}
25
+ .track{position:relative;height:10px;border-radius:5px;background:var(--line)}
26
+ .fill{position:absolute;left:0;top:0;bottom:0;border-radius:5px;background:var(--bar)}
27
+ .edge{position:absolute;top:-3px;bottom:-3px;width:2px;background:var(--edge)}
28
+ .band{grid-column:2/4;font-size:13px;color:var(--muted)}ul{margin:6px 0;padding-left:20px}
29
+ .stats{display:flex;gap:24px;flex-wrap:wrap}.stat b{display:block;font-size:20px}
30
+ @media (max-width:560px){.row{grid-template-columns:96px 1fr 44px}}
31
+ """
32
+
33
+
34
+ def render_html(profile: dict, title: str = "Imprint profile") -> str:
35
+ e = html.escape
36
+ traits = profile.get("traits") or {}
37
+ rows = []
38
+ for t in TRAIT_NAMES:
39
+ if t not in traits:
40
+ continue
41
+ v = float(traits[t].get("value", 0.5))
42
+ c = float(traits[t].get("confidence", 0))
43
+ edges, sentences = BANDS[t]
44
+ marks = "".join(f'<span class="edge" style="left:{x * 100:.1f}%"></span>' for x in edges)
45
+ rows.append(
46
+ f'<div class="card"><div class="row"><b>{e(TRAIT_LABELS[t])}</b>'
47
+ f'<div class="track" title="{e(TRAIT_META[t])}"><span class="fill" style="width:{v * 100:.1f}%"></span>{marks}</div>'
48
+ f'<span>{v:.2f}</span><div class="band">{e(sentences[band_index(t, v)])} '
49
+ f'<span title="confidence">· conf {c:.2f}</span></div></div></div>')
50
+ obs = profile.get("observations") or []
51
+ by_source = Counter(o.get("source", "?") for o in obs)
52
+ rules = profile.get("rules") or []
53
+ cands = profile.get("candidateRules") or []
54
+ rule_list = "".join(f"<li>{e(r)}</li>" for r in rules) or '<li class="muted">None yet.</li>'
55
+ cand_list = "".join(
56
+ f"<li>{e(c.get('text', ''))} <span class='muted'>({'likes' if c.get('polarity', 1) > 0 else 'dislikes'},"
57
+ f" seen {len(c.get('evidence') or [])}×)</span></li>" for c in cands) or '<li class="muted">None.</li>'
58
+ stats = "".join(f'<div class="stat"><b>{n}</b><span class="muted">{e(k)}</span></div>' for k, n in [
59
+ ("interactions", profile.get("interactions", 0)), ("observations", len(obs)),
60
+ *[(f"{s} observations", n) for s, n in sorted(by_source.items())]])
61
+ return f"""<!doctype html><html lang="en"><head><meta charset="utf-8">
62
+ <meta name="viewport" content="width=device-width,initial-scale=1"><title>{e(title)}</title>
63
+ <style>{_CSS}</style></head><body><main>
64
+ <h1>{e(title)}</h1><p class="muted">Last updated {e(str(profile.get('lastUpdated', 'unknown')))}. Read-only view; bar ticks mark band edges.</p>
65
+ <div class="card stats">{stats}</div>
66
+ <h2>Traits</h2>{''.join(rows)}
67
+ <h2>Standing rules</h2><div class="card"><ul>{rule_list}</ul></div>
68
+ <h2>Candidate rules <span class="muted">(not injected until corroborated)</span></h2><div class="card"><ul>{cand_list}</ul></div>
69
+ </main></body></html>"""
70
+
71
+
72
+ def main(argv=None) -> int:
73
+ ap = argparse.ArgumentParser(description="Render an Imprint profile as static HTML.")
74
+ ap.add_argument("profile", type=Path)
75
+ ap.add_argument("-o", "--out", type=Path, default=Path("imprint-dashboard.html"))
76
+ ap.add_argument("--title", default="Imprint profile")
77
+ a = ap.parse_args(argv)
78
+ a.out.write_text(render_html(json.loads(a.profile.read_text(encoding="utf-8")), a.title),
79
+ encoding="utf-8")
80
+ print(f"wrote {a.out}")
81
+ return 0
82
+
83
+
84
+ if __name__ == "__main__":
85
+ raise SystemExit(main())
imprint/embedders.py ADDED
@@ -0,0 +1,43 @@
1
+ """Embedder interface.
2
+
3
+ Imprint needs one thing from an embedder: turn text into a fixed-length vector.
4
+ The default is a small local sentence-transformer (all-MiniLM-L6-v2), which runs
5
+ on CPU and never sends text anywhere. Any object with `embed(list[str]) ->
6
+ array (n, d)` and a stable `name` works.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Protocol, Sequence, runtime_checkable
11
+
12
+ import numpy as np
13
+
14
+
15
+ @runtime_checkable
16
+ class Embedder(Protocol):
17
+ name: str
18
+
19
+ def embed(self, texts: Sequence[str]) -> np.ndarray: # (n, d)
20
+ ...
21
+
22
+
23
+ class MiniLMEmbedder:
24
+ """Local all-MiniLM-L6-v2 via sentence-transformers (optional dependency)."""
25
+
26
+ def __init__(self, model_name: str = "all-MiniLM-L6-v2"):
27
+ try:
28
+ from sentence_transformers import SentenceTransformer
29
+ except ImportError as e: # pragma: no cover - depends on environment
30
+ raise ImportError(
31
+ "The default embedder needs sentence-transformers: "
32
+ "pip install 'imprint-layer[minilm]', or pass your own embedder="
33
+ ) from e
34
+ self.name = f"sentence-transformers/{model_name}"
35
+ self._model = SentenceTransformer(model_name)
36
+
37
+ def embed(self, texts: Sequence[str]) -> np.ndarray:
38
+ return np.asarray(self._model.encode(list(texts), normalize_embeddings=True),
39
+ dtype=np.float32)
40
+
41
+
42
+ def default_embedder() -> Embedder:
43
+ return MiniLMEmbedder()
imprint/rules.py ADDED
@@ -0,0 +1,242 @@
1
+ """Two-tier standing-rule capture.
2
+
3
+ Tier 1 — explicit forms become rules immediately:
4
+ "remember: ...", "always ..." / "never ..." (bare form must be followed by
5
+ an imperative verb), "always: ..." / "never: ...", "from now on ...",
6
+ "going forward ...".
7
+
8
+ Tier 2 — natural preference statements become *candidate* rules:
9
+ "I prefer ...", "I like it when you ...", "I hate it when you ...",
10
+ "it bugs me when you ...", "I'd rather you ...", "I wish you wouldn't ...".
11
+ A candidate is promoted to a rule only when corroborated: the same
12
+ preference (same polarity, near-duplicate content) stated again in a
13
+ different message at least `corroboration_gap` later. Candidates are never
14
+ injected. Users don't learn any syntax; they just talk.
15
+
16
+ Both tiers apply the same durability test. A rule must state an enduring
17
+ preference: questions, time-bound text ("today", "at 3"), and task anchors
18
+ (a named recipient, "now", a deadline, "this"/"that one") veto storage.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import re
23
+ from dataclasses import dataclass
24
+ from datetime import datetime, timedelta
25
+
26
+ from .steering import RETRACTION, detect_steering
27
+
28
+ # --- durability ----------------------------------------------------------------
29
+
30
+ _ONE_OFF = re.compile(
31
+ r"\b(?:today|tonight|tomorrow|yesterday|right\s+now|for\s+now|just\s+now|asap|"
32
+ r"later\s+(?:today|tonight)|"
33
+ r"this\s+(?:morning|afternoon|evening|week|weekend|month|time|once)|"
34
+ r"next\s+(?:week|month|monday|tuesday|wednesday|thursday|friday|saturday|sunday)|"
35
+ r"(?:on\s+)?(?:monday|tuesday|wednesday|thursday|friday|saturday|sunday)|"
36
+ r"(?:at|by|before|until)\s+\d|in\s+\d+\s?(?:min(?:ute)?s?|hours?|hrs?|days?)|\d{1,2}(?::\d{2})?\s?(?:am|pm))\b",
37
+ re.I,
38
+ )
39
+ _TASK_ANCHOR = re.compile(
40
+ r"\b(?:now|(?:before|by|until|till|due)\s+(?:noon|midnight|tonight|tomorrow|eod|cob|"
41
+ r"end\s+of|the\s+end|next|this|\d|monday|tuesday|wednesday|thursday|friday|saturday|"
42
+ r"sunday)|deadline|due\s+(?:date|on)|this|these|those|that\s+(?:one|thing|stuff|again)|"
43
+ r"(?:send|email|text|call|message|ping|forward|remind|tell|cc)\s+(?:her|him|them))\b",
44
+ re.I,
45
+ )
46
+ # A capitalized name after a recipient verb/preposition. Lowercase names can't
47
+ # be told apart from ordinary words, so "to dana" is a known gap.
48
+ _NOT_NAMES = {"I", "English", "Spanish", "French", "German", "Markdown", "Python",
49
+ "JSON", "SQL", "AI", "The", "My", "Your"}
50
+ _RECIPIENT = re.compile(
51
+ r"\b(?i:to|for|with|cc|email|text|call|message|tell|ping|remind|ask|send|forward)"
52
+ r"\s+([A-Z][a-zA-Z]+)\b")
53
+
54
+
55
+ def has_task_anchor(body: str) -> bool:
56
+ b = re.sub(r"\bfrom\s+now\s+on\b", " ", body or "", flags=re.I)
57
+ if _TASK_ANCHOR.search(b):
58
+ return True
59
+ return any(m.group(1) not in _NOT_NAMES for m in _RECIPIENT.finditer(b))
60
+
61
+
62
+ def is_durable(body: str) -> bool:
63
+ b = (body or "").strip()
64
+ if not 4 <= len(b) <= 200 or b.endswith("?"):
65
+ return False
66
+ return not (_ONE_OFF.search(b) or has_task_anchor(b))
67
+
68
+
69
+ # --- tier 1: explicit forms ------------------------------------------------------
70
+
71
+ _RULE_VERBS = frozenset("""
72
+ be use ask answer reply respond keep call mention say talk start end open close
73
+ include give put write check confirm tell lead skip avoid stop let make match refer
74
+ address sign greet offer suggest apologize apologise hedge cite show explain assume
75
+ repeat restate speak remind bring do try follow treat wait act pretend lie joke flirt
76
+ swear curse break drop add format double-check cut leave share wrap read look take
77
+ pick choose respect interrupt lecture moralize moralise summarize summarise list
78
+ number quote push nag fake invent guess correct
79
+ """.split())
80
+ _REMEMBER = re.compile(r"^\s*remember\s*:\s*(.+)$", re.I)
81
+ _ALWAYS_NEVER = re.compile(r"^\s*(always|never)\s*(:)?\s+(.+)$", re.I)
82
+ _FROM_NOW_ON = re.compile(r"^\s*(?:from\s+now\s+on|going\s+forward)\s*[,:]?\s+(.+)$", re.I)
83
+
84
+ # --- tier 2: natural preference statements -----------------------------------------
85
+
86
+ _POS = (r"i\s+(?:really\s+|much\s+)?(?:prefer|like\s+it\s+when\s+you|love\s+it\s+when\s+you|"
87
+ r"appreciate\s+it\s+when\s+you|wish\s+you\s+would(?!n)|'?d\s+rather\s+you|"
88
+ r"would\s+rather\s+you)")
89
+ _NEG = (r"i\s+(?:really\s+)?(?:hate\s+(?:it\s+)?when\s+you|don'?t\s+like\s+(?:it\s+)?when\s+you|"
90
+ r"can'?t\s+stand\s+(?:it\s+)?when\s+you|wish\s+you\s+wouldn'?t)|"
91
+ r"it\s+(?:really\s+)?(?:annoys|bugs|bothers|irritates)\s+me\s+when\s+you")
92
+ _NATURAL = re.compile(r"^\s*(?:(?:honestly|also|and|but|so|fwiw)[,\s]+)?"
93
+ r"(?:(?P<pos>" + _POS + r")|(?P<neg>" + _NEG + r"))\s+(?P<body>.+?)[.!]*$",
94
+ re.I)
95
+
96
+ _STOP = {"a", "an", "the", "to", "of", "for", "and", "or", "you", "your", "it", "is", "be",
97
+ "i", "me", "my", "when", "that", "with", "in", "on", "so", "really", "always",
98
+ "never", "just", "please", "do", "dont", "don't", "are", "more"}
99
+
100
+
101
+ def content_key(body: str) -> frozenset[str]:
102
+ toks = re.findall(r"[a-z0-9']+", (body or "").lower())
103
+ return frozenset(t.rstrip("s") if len(t) > 3 else t for t in toks if t not in _STOP)
104
+
105
+
106
+ def keys_match(a: frozenset[str], b: frozenset[str]) -> bool:
107
+ if not a or not b:
108
+ return False
109
+ if a <= b or b <= a:
110
+ return min(len(a), len(b)) >= 2 or a == b
111
+ return len(a & b) / len(a | b) >= 0.6
112
+
113
+
114
+ def _norm(rule: str) -> str:
115
+ return re.sub(r"\s+", " ", re.sub(r"[^a-z0-9\s]", "", (rule or "").lower())).strip()
116
+
117
+
118
+ def rules_near_dup(a: str, b: str) -> bool:
119
+ na, nb = _norm(a), _norm(b)
120
+ if not na or not nb:
121
+ return False
122
+ if na == nb:
123
+ return True
124
+ short, long_ = sorted((na, nb), key=len)
125
+ if len(short) >= 24 and short in long_:
126
+ return True
127
+ return keys_match(content_key(a), content_key(b)) and len(content_key(a)) >= 3
128
+
129
+
130
+ @dataclass
131
+ class Capture:
132
+ rules: list[str] # tier 1: store now
133
+ candidates: list[dict] # tier 2: {"text", "polarity"}
134
+
135
+
136
+ def capture(text: str) -> Capture:
137
+ """Split `text` into explicit rules and natural-preference candidates."""
138
+ rules: list[str] = []
139
+ candidates: list[dict] = []
140
+ text = text or ""
141
+ cut = max((m.end() for m in RETRACTION.finditer(text)), default=-1)
142
+ # Each line, then each sentence within it, is checked independently.
143
+ for sm in re.finditer(r"[^\n.!?]+[.!?]*", text):
144
+ if sm.end() <= cut:
145
+ continue # retracted later in the same message
146
+ # A retraction inside this segment cancels what precedes it; keep
147
+ # whatever the user said after it ("never mind, always: keep it short").
148
+ seg = (text[cut:sm.end()] if sm.start() < cut else sm.group(0)).strip(" \t,;:-—")
149
+ if not seg:
150
+ continue
151
+ rule = _explicit(seg)
152
+ if rule:
153
+ if not any(rules_near_dup(rule, r) for r in rules):
154
+ rules.append(rule)
155
+ continue
156
+ m = _NATURAL.match(seg)
157
+ if m and is_durable(m.group("body")) and not detect_steering(seg):
158
+ candidates.append({"text": seg.rstrip(".! "),
159
+ "polarity": 1 if m.group("pos") else -1,
160
+ "key": sorted(content_key(m.group("body")))})
161
+ return Capture(rules, candidates)
162
+
163
+
164
+ def _explicit(seg: str) -> str | None:
165
+ m = _REMEMBER.match(seg)
166
+ if m:
167
+ body = m.group(1).strip()
168
+ return body.rstrip(".") if is_durable(body) and not detect_steering(body) else None
169
+ m = _FROM_NOW_ON.match(seg)
170
+ if m:
171
+ body = m.group(1).strip()
172
+ if is_durable(body) and not detect_steering(body):
173
+ return f"{body[0].upper()}{body[1:]}".rstrip(".")
174
+ return None
175
+ m = _ALWAYS_NEVER.match(seg)
176
+ if not m:
177
+ return None
178
+ cue, colon, body = m.group(1), m.group(2), m.group(3).strip()
179
+ if not is_durable(body):
180
+ return None
181
+ if not colon:
182
+ first = re.sub(r"[^a-z'-]", "", body.split()[0].lower())
183
+ if first not in _RULE_VERBS:
184
+ return None # "Never mind", "Always the same with you"
185
+ if detect_steering(body) or detect_steering(seg):
186
+ return None # a steering command is a dial move, not a rule
187
+ return f"{cue.capitalize()} {body}".rstrip(".")
188
+
189
+
190
+ # --- candidate store -----------------------------------------------------------------
191
+
192
+ def update_candidates(store: list[dict], new: list[dict], ts: datetime | None,
193
+ *, gap: timedelta, ttl: timedelta, max_candidates: int = 50,
194
+ ) -> tuple[list[dict], list[str]]:
195
+ """Merge new candidates into `store`; return (store, promoted rule texts).
196
+
197
+ Corroboration needs a second statement with the same polarity and matching
198
+ content, at least `gap` after the first evidence. Without a message
199
+ timestamp a statement is recorded but cannot corroborate.
200
+ """
201
+ promoted: list[str] = []
202
+ iso = ts.isoformat(timespec="milliseconds") if ts else None
203
+ for cand in new:
204
+ key = frozenset(cand["key"])
205
+ hit = next((c for c in store if keys_match(frozenset(c["key"]), key)), None)
206
+ if hit is None:
207
+ store.append({"text": cand["text"], "polarity": cand["polarity"],
208
+ "key": cand["key"], "evidence": [iso]})
209
+ continue
210
+ if hit["polarity"] != cand["polarity"]:
211
+ # The user changed their mind: restart from the newer statement.
212
+ hit.update({"text": cand["text"], "polarity": cand["polarity"],
213
+ "key": cand["key"], "evidence": [iso]})
214
+ continue
215
+ first = _first_ts(hit["evidence"])
216
+ if ts is not None and first is not None and ts - first >= gap:
217
+ promoted.append(f'Stated preference: "{hit["text"]}"')
218
+ store.remove(hit)
219
+ else:
220
+ hit["evidence"].append(iso)
221
+ if ts is not None:
222
+ store[:] = [c for c in store
223
+ if (_last_ts(c["evidence"]) is None) or ts - _last_ts(c["evidence"]) <= ttl]
224
+ del store[:-max_candidates]
225
+ return store, promoted
226
+
227
+
228
+ def _parse(s: str | None) -> datetime | None:
229
+ try:
230
+ return datetime.fromisoformat(s) if s else None
231
+ except ValueError:
232
+ return None
233
+
234
+
235
+ def _first_ts(evidence: list) -> datetime | None:
236
+ ts = [t for t in (_parse(e) for e in evidence) if t]
237
+ return min(ts) if ts else None
238
+
239
+
240
+ def _last_ts(evidence: list) -> datetime | None:
241
+ ts = [t for t in (_parse(e) for e in evidence) if t]
242
+ return max(ts) if ts else None
imprint/steering.py ADDED
@@ -0,0 +1,187 @@
1
+ """Direct steering: "be wittier", "more sarcastic", "you're too formal".
2
+
3
+ A steering command names a dial as a property of the assistant, with a
4
+ direction. It moves that trait one band (see traits.band_step), so the very
5
+ next directive changes and each repeat walks one more band.
6
+
7
+ What does NOT steer: bare mentions ("that was sarcastic"), questions about a
8
+ dial ("why so sarcastic?"), third-party talk ("my boss could be funnier"),
9
+ praise ("you're too kind"), and anything retracted later in the same message
10
+ ("be wittier. actually don't").
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import re
15
+ from dataclasses import dataclass
16
+
17
+ # Dial words -> (trait, polarity). +1 names the trait's high pole, -1 its low
18
+ # pole ("more serious" pushes humor down).
19
+ _BASE: dict[str, tuple[str, int]] = {}
20
+ _COMPARATIVE: dict[str, tuple[str, int]] = {}
21
+
22
+
23
+ def _lex(table: dict, trait: str, pol: int, words: str) -> None:
24
+ for w in words.split("|"):
25
+ table[w.strip()] = (trait, pol)
26
+
27
+
28
+ _lex(_BASE, "humor", +1, "sarcastic|sarcasm|funny|witty|wit|humor|humour|humorous|"
29
+ "jokey|joking|snarky|snark|cheeky|teasing")
30
+ _lex(_BASE, "humor", -1, "serious|earnest|sober")
31
+ _lex(_BASE, "warmth", +1, "warm|warmth|empathetic|empathy|caring|kind|gentle|emotional|"
32
+ "understanding|affectionate")
33
+ _lex(_BASE, "warmth", -1, "matter-of-fact|matter of fact|businesslike|detached|cold")
34
+ _lex(_BASE, "flirt", +1, "flirty|flirt|flirting|flirtatious|romantic")
35
+ _lex(_BASE, "flirt", -1, "platonic")
36
+ _lex(_BASE, "verbosity", +1, "detailed|detail|verbose|wordy|thorough|elaborate")
37
+ _lex(_BASE, "verbosity", -1, "concise|brief|terse|succinct|to the point")
38
+ _lex(_BASE, "formality", +1, "formal|professional|polished|proper|precise")
39
+ _lex(_BASE, "formality", -1, "casual|informal|relaxed|laid-back|laid back|chill")
40
+ _lex(_BASE, "autonomy", +1, "independent|autonomous|self-directed|self directed")
41
+ _lex(_BASE, "autonomy", -1, "dependent|deferential")
42
+ _lex(_BASE, "proactivity", +1, "proactive|initiative")
43
+ _lex(_BASE, "proactivity", -1, "reactive")
44
+ _lex(_BASE, "riskTolerance", +1, "bold|confident|decisive|assertive|daring")
45
+ _lex(_BASE, "riskTolerance", -1, "careful|cautious|conservative|hedged")
46
+
47
+ _lex(_COMPARATIVE, "humor", +1, "wittier|funnier|snarkier|cheekier")
48
+ _lex(_COMPARATIVE, "warmth", +1, "warmer|kinder|gentler|softer")
49
+ _lex(_COMPARATIVE, "warmth", -1, "colder")
50
+ _lex(_COMPARATIVE, "flirt", +1, "flirtier")
51
+ _lex(_COMPARATIVE, "verbosity", +1, "longer|wordier")
52
+ _lex(_COMPARATIVE, "verbosity", -1, "shorter|briefer|terser")
53
+ _lex(_COMPARATIVE, "riskTolerance", +1, "bolder|braver")
54
+ _lex(_COMPARATIVE, "riskTolerance", -1, "safer")
55
+ _lex(_COMPARATIVE, "formality", -1, "chiller")
56
+
57
+ # Praise never steers down. Evaluative down-forms ("too X", "stop being so X",
58
+ # "not so X") skip these words; "less kind" (explicit) still steers.
59
+ PRAISE_WORDS = frozenset({
60
+ "kind", "caring", "gentle", "warm", "warmth", "understanding", "empathetic",
61
+ "empathy", "affectionate", "funny", "witty", "wit", "humorous",
62
+ })
63
+ _PRAISE_IDIOM = re.compile(
64
+ r"\b(?:you(?:'re|\s+are)\s+(?:\w+\s+){0,2}too\s+(?:kind|sweet|nice|good\s+to\s+me|"
65
+ r"generous|funny|much)|stop\s+being\s+so\s+(?:kind|sweet|nice|funny)|"
66
+ r"(?:that's|that\s+is|how)\s+(?:so\s+)?(?:kind|sweet|nice)\s+of\s+you)\b",
67
+ re.I,
68
+ )
69
+
70
+
71
+ def _alt(words) -> str:
72
+ return "|".join(re.escape(w).replace(r"\ ", r"\s+")
73
+ for w in sorted(words, key=len, reverse=True))
74
+
75
+
76
+ _B, _C = _alt(_BASE), _alt(_COMPARATIVE)
77
+ # Fillers / an address ("Sam, ...") / a request frame may precede a command.
78
+ # The command must open its clause: "be wittier" steers, "my boss could be
79
+ # wittier" doesn't. "no" is a filler only when "more" doesn't follow it.
80
+ _LEAD = (
81
+ r"^(?:(?:ok(?:ay)?|hey|please|and|also|now|just|so|alright|honestly|seriously|"
82
+ r"always|yeah|yes|no(?!\s+more\b)|but)[\s,!.]+)*"
83
+ r"(?:[a-z]+,\s*)?"
84
+ r"(?:(?:can|could|would|will)\s+you\s+(?:please\s+)?(?:try\s+(?:to\s+)?)?|"
85
+ r"i\s+(?:want|need)\s+you\s+to\s+|i(?:'d|\s+would)\s+like\s+you\s+to\s+|"
86
+ r"you\s+(?:can|could|should|may)\s+|try\s+(?:to\s+)?|feel\s+free\s+to\s+)?"
87
+ r"(?:please\s+)?"
88
+ )
89
+ _INT = r"(?:(?:a\s+)?(?:bit|little|lot|tad|touch)\s+|(?:way|much|even|slightly|somewhat)\s+)*"
90
+ _TAIL = (r"(?:[\s,]+(?:please|pls|from\s+you|with\s+me|now|again|"
91
+ r"in\s+your\s+(?:replies|responses|answers)))*[\s.!]*$")
92
+
93
+ _PATTERNS: list[tuple[re.Pattern, str]] = [(re.compile(p, re.I), k) for p, k in [
94
+ (_LEAD + r"no\s+more\s+(?P<w>" + _B + r")\b", "neg"),
95
+ (_LEAD + r"(?:be|get|sound|act|go|try\s+being)\s+" + _INT
96
+ + r"(?P<mod>more|less)\s+(?P<w>" + _B + r")\b", "mod"),
97
+ (_LEAD + r"(?:be|get|sound|act)\s+" + _INT + r"(?P<w>" + _C + r")\b", "comp"),
98
+ (_LEAD + _INT + r"(?P<mod>more|less)\s+(?P<w>" + _B + r")" + _TAIL, "mod"),
99
+ (_LEAD + _INT + r"(?P<w>" + _C + r")" + _TAIL, "comp"),
100
+ (_LEAD + r"(?:i(?:'d|\s+would)\s+like|i\s+want|give\s+me)\s+" + _INT
101
+ + r"(?P<mod>more|less)\s+(?P<w>" + _B + r")\b", "mod"),
102
+ (_LEAD + r"(?:dial|turn|crank|tone|ramp)\s+(?:it\s+)?(?P<ud>up|down)\s+(?:on\s+)?"
103
+ r"(?:the\s+|your\s+)?(?P<w>" + _B + r")\b", "ud"),
104
+ (_LEAD + r"(?:dial|turn|crank|tone|ramp)\s+(?:the\s+|your\s+)?(?P<w>" + _B
105
+ + r")\s+(?P<ud>up|down)\b", "ud"),
106
+ (_LEAD + r"(?:you(?:'re|\s+are)\s+(?:being\s+)?)?" + _INT + r"too\s+(?P<w>" + _B + r")\b",
107
+ "eval"),
108
+ (_LEAD + r"(?:(?:stop\s+being|quit\s+being|don'?t\s+be|do\s+not\s+be)\s+(?:so\s+)?|"
109
+ r"not\s+so\s+)(?P<w>" + _B + r")\b", "eval"),
110
+ ]]
111
+
112
+ _QUESTION_OPENER = re.compile(
113
+ r"^(?:why|what|how|who|when|where|are|is|was|were|am|do|does|did|have|has)\b", re.I)
114
+
115
+ RETRACTION = re.compile(
116
+ r"\b(?:actually\s+(?:don'?t|no|not|never\s*mind)|never\s*mind|nevermind|"
117
+ r"scratch\s+that|(?<!don't\s)(?<!dont\s)(?<!not\s)forget\s+(?:that|it|i\s+said\s+that)|wait,?\s+no|no,?\s+wait|"
118
+ r"just\s+kidding|jk|ignore\s+(?:that|what\s+i\s+said)|cancel\s+that|"
119
+ r"on\s+second\s+thought|don'?t\s+do\s+that|i\s+take\s+(?:that|it)\s+back)\b",
120
+ re.I,
121
+ )
122
+ # Clauses: sentences, plus ", and" / "and" / "but" joins.
123
+ _CLAUSE = re.compile(r"[^.!?\n;]+?(?=(?:,?\s+(?:and|but)\s+)|[.!?\n;]|$)", re.I)
124
+
125
+
126
+ @dataclass(frozen=True)
127
+ class Steer:
128
+ trait: str
129
+ direction: float # +1 toward the trait's high pole, -1 toward low
130
+ match: str
131
+
132
+
133
+ def last_retraction_end(text: str) -> int:
134
+ """Offset just past the last retraction marker, or -1 if none."""
135
+ ends = [m.end() for m in RETRACTION.finditer(text or "")]
136
+ return max(ends) if ends else -1
137
+
138
+
139
+ def has_praise(text: str) -> bool:
140
+ return bool(_PRAISE_IDIOM.search(text or ""))
141
+
142
+
143
+ def detect_steering(text: str) -> list[Steer]:
144
+ """Steering commands in `text`, one per trait, retractions applied."""
145
+ text = (text or "").strip()
146
+ if not text:
147
+ return []
148
+ found: list[tuple[int, Steer]] = []
149
+ cut = last_retraction_end(text)
150
+ clauses = [(cm.start(), cm.group(0)) for cm in _CLAUSE.finditer(text)]
151
+ if cut >= 0:
152
+ # What follows a retraction starts fresh, even without punctuation
153
+ # ("never mind, be more sarcastic").
154
+ clauses += [(cut + cm.start(), cm.group(0)) for cm in _CLAUSE.finditer(text[cut:])]
155
+ for cstart, raw in clauses:
156
+ clause = raw.strip(" ,;:-—")
157
+ if not clause or _QUESTION_OPENER.match(clause):
158
+ continue
159
+ start = cstart + raw.find(clause)
160
+ for pat, kind in _PATTERNS:
161
+ m = pat.search(clause)
162
+ if not m:
163
+ continue
164
+ word = re.sub(r"\s+", " ", m.group("w").lower())
165
+ if kind == "eval" and word in PRAISE_WORDS:
166
+ break # a compliment, not a complaint
167
+ trait, pol = (_COMPARATIVE if kind == "comp" else _BASE)[word]
168
+ if kind == "mod":
169
+ sign = 1 if m.group("mod").lower() == "more" else -1
170
+ elif kind == "ud":
171
+ sign = 1 if m.group("ud").lower() == "up" else -1
172
+ elif kind in ("neg", "eval"):
173
+ sign = -1
174
+ else:
175
+ sign = 1
176
+ found.append((start + m.start(), Steer(trait, float(pol * sign), m.group(0).strip())))
177
+ break
178
+ if cut >= 0:
179
+ # A retraction cancels every command that starts before it.
180
+ found = [(pos, st) for pos, st in found if pos >= cut]
181
+ out: list[Steer] = []
182
+ seen: set[str] = set()
183
+ for _, st in found:
184
+ if st.trait not in seen:
185
+ seen.add(st.trait)
186
+ out.append(st)
187
+ return out
imprint/traits.py ADDED
@@ -0,0 +1,207 @@
1
+ """The eight Imprint traits: metadata, priors, band ladders, directive text, and
2
+ embedding pole exemplars.
3
+
4
+ Every trait is a value in [0, 1]. The directive never shows the raw number; it
5
+ shows the sentence for the band the value falls in. Band edges here are the
6
+ single source of truth for both the directive text and band-step steering.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ TRAIT_NAMES: tuple[str, ...] = (
11
+ "verbosity", "autonomy", "formality", "proactivity",
12
+ "riskTolerance", "humor", "warmth", "flirt",
13
+ )
14
+
15
+ TRAIT_META: dict[str, str] = {
16
+ "verbosity": "concise and direct ↔ detailed explanations",
17
+ "autonomy": "always ask first ↔ act on own judgment",
18
+ "formality": "casual and conversational ↔ formal and precise",
19
+ "proactivity": "wait for instructions ↔ suggest next steps unprompted",
20
+ "riskTolerance": "cautious claims ↔ confident, committed takes",
21
+ "humor": "straight and earnest ↔ dry, sarcastic wit",
22
+ "warmth": "matter-of-fact ↔ emotionally present",
23
+ "flirt": "platonic ↔ playful romantic charge",
24
+ }
25
+
26
+ # Neutral priors for a new profile. Flirt starts in its platonic band; a host
27
+ # that never wants it can disable the trait (Imprint(traits=...)).
28
+ TRAIT_DEFAULTS: dict[str, dict[str, float]] = {
29
+ "verbosity": {"value": 0.5, "confidence": 0.2},
30
+ "autonomy": {"value": 0.5, "confidence": 0.2},
31
+ "formality": {"value": 0.5, "confidence": 0.2},
32
+ "proactivity": {"value": 0.5, "confidence": 0.2},
33
+ "riskTolerance": {"value": 0.5, "confidence": 0.2},
34
+ "humor": {"value": 0.45, "confidence": 0.2},
35
+ "warmth": {"value": 0.5, "confidence": 0.2},
36
+ "flirt": {"value": 0.2, "confidence": 0.2},
37
+ }
38
+
39
+ # (edge, ...) and one sentence per band: len(sentences) == len(edges) + 1.
40
+ # A value v is in band i where i = number of edges <= v.
41
+ BANDS: dict[str, tuple[tuple[float, ...], tuple[str, ...]]] = {
42
+ "verbosity": ((0.25, 0.4, 0.6, 0.75), (
43
+ "Extremely concise. One line when possible. No filler, no restating the user.",
44
+ "Concise and direct. Lead with the answer. Skip restatement and repeated points.",
45
+ "Balance brevity with clarity. Explain when asked; don't over-elaborate.",
46
+ "Thorough when useful. Include context, but don't loop the same idea.",
47
+ "Detailed walkthroughs are welcome. Still avoid repeating yourself.",
48
+ )),
49
+ "autonomy": ((0.4, 0.6), (
50
+ "Ask before acting on non-trivial choices.",
51
+ "Use judgment on routine choices; confirm only high-stakes ones.",
52
+ "Act on your own judgment. Don't wait for permission on reasonable calls.",
53
+ )),
54
+ "formality": ((0.4, 0.6), (
55
+ "Casual, conversational register. Not a lecture, not corporate.",
56
+ "Friendly and clear. Competent without stiffness.",
57
+ "Precise, professional wording.",
58
+ )),
59
+ "proactivity": ((0.4, 0.6), (
60
+ "Don't pile on suggestions or offer follow-up work. Answer, then stop.",
61
+ "Offer a next step only when it clearly helps.",
62
+ "Bring useful next steps and fresh ideas without being asked.",
63
+ )),
64
+ "riskTolerance": ((0.35, 0.6), (
65
+ "Prefer careful claims. Flag uncertainty; don't invent authority.",
66
+ "Balanced. Own uncertainty when guessing; commit when you know.",
67
+ "Commit to clear positions. Don't over-hedge; still don't fake expertise.",
68
+ )),
69
+ "humor": ((0.35, 0.55, 0.75), (
70
+ "Straight and earnest. Skip jokes and sarcasm unless clearly invited.",
71
+ "Light wit is fine when it fits. Don't force bits.",
72
+ "Dry wit and occasional sarcasm are welcome.",
73
+ "Lean into dry, sarcastic humor often. Still read the room; never cruel.",
74
+ )),
75
+ "warmth": ((0.35, 0.55, 0.75), (
76
+ "Cool and practical. Matter-of-fact; skip empathy padding.",
77
+ "Friendly and present without gushing.",
78
+ "Emotionally present. Empathetic when it matters; never fake validation.",
79
+ "High warmth. Caring and encouraging, without empty flattery.",
80
+ )),
81
+ "flirt": ((0.35, 0.55, 0.75), (
82
+ "Platonic. No romantic or flirtatious tone.",
83
+ "A light playful spark is fine occasionally.",
84
+ "Clear flirty banter is welcome when it fits.",
85
+ "Playful romantic charge often. Stay yourself; keep it consensual and kind.",
86
+ )),
87
+ }
88
+
89
+ TRAIT_LABELS: dict[str, str] = {
90
+ "verbosity": "Communication", "autonomy": "Autonomy", "formality": "Register",
91
+ "proactivity": "Proactivity", "riskTolerance": "Risk", "humor": "Humor",
92
+ "warmth": "Warmth", "flirt": "Flirt",
93
+ }
94
+
95
+
96
+ def band_index(trait: str, value: float) -> int:
97
+ edges, _ = BANDS[trait]
98
+ return sum(1 for e in edges if e <= float(value))
99
+
100
+
101
+ def describe(trait: str, value: float) -> str:
102
+ _, sentences = BANDS[trait]
103
+ return f"- **{TRAIT_LABELS[trait]}:** {sentences[band_index(trait, value)]}"
104
+
105
+
106
+ def band_step(trait: str, value: float, direction: float, margin: float = 0.01) -> float:
107
+ """Next (direction > 0) or previous (direction < 0) band edge ± margin.
108
+
109
+ Landing a hair past the edge means one small ambient update can't undo a
110
+ command. Unchanged at either end of the ladder.
111
+ """
112
+ edges, _ = BANDS[trait]
113
+ v = float(value)
114
+ if direction > 0:
115
+ above = [e for e in edges if e > v]
116
+ return min(1.0, above[0] + margin) if above else v
117
+ if direction < 0:
118
+ at_or_below = [e for e in edges if e <= v]
119
+ return max(0.0, at_or_below[-1] - margin) if at_or_below else v
120
+ return v
121
+
122
+
123
+ # Exemplars define each embedding axis (high pole − low pole). A message rides
124
+ # the continuum by cosine similarity; these are not keyword lists. Tuned so
125
+ # ordinary check-ins stay under the vibe thresholds with all-MiniLM-L6-v2.
126
+ TRAIT_POLES: dict[str, dict[str, list[str]]] = {
127
+ "autonomy": {
128
+ "low": ["Ask me before you do anything.",
129
+ "Don't decide on your own — check with me first.",
130
+ "Wait for my approval before researching or choosing.",
131
+ "I want you dependent on my instructions."],
132
+ "high": ["You can choose for yourself.",
133
+ "Pick something on your own. Don't only echo me.",
134
+ "Genuinely choose with no parameters from me.",
135
+ "Act on your own curiosity, not my last topic.",
136
+ "Use your own judgment — independent research and takes."],
137
+ },
138
+ "verbosity": {
139
+ "low": ["Shorter. Too long. Cut it down.", "Be brief. One line when you can.",
140
+ "You're repeating yourself. Stop looping the same point.",
141
+ "Don't restate what I just said. Lead with your take.", "Too much filler. Concise."],
142
+ "high": ["Elaborate on that. Explain more.", "Walk me through it in detail.",
143
+ "I don't mind the theory lesson — unpack it fully.",
144
+ "What does that mean? Go deeper.", "More detail please, longer explanation."],
145
+ },
146
+ "formality": {
147
+ "low": ["Keep it casual. Don't lecture me.", "Not corporate. Not generic AI tone.",
148
+ "Talk like a friend — warm and light.", "Drop the stiff professional voice."],
149
+ "high": ["Be more precise and professional.", "Tighter formal wording please.",
150
+ "More exact language, less slang.", "Careful precise register."],
151
+ },
152
+ "proactivity": {
153
+ "low": ["Wait for instructions. Don't pile on suggestions.",
154
+ "Less nudging. Don't fill silence with soft suggestions.",
155
+ "You're just chasing the last two topics.",
156
+ "Stop suggesting things just because we talked about them recently."],
157
+ "high": ["Bring a fresh thread of your own unprompted.",
158
+ "Suggest when you have something real and new.",
159
+ "Reach out with a new thought, not a rehash of recent chat.",
160
+ "Self-directed exploration and initiative are welcome."],
161
+ },
162
+ "riskTolerance": {
163
+ "low": ["Hedge when unsure. Don't invent technical authority.",
164
+ "Prefer careful claims over confident guesses.",
165
+ "Confirm before stating something as fact.",
166
+ "Stay epistemically cautious — caveat your takes."],
167
+ "high": ["State it firmly when you know. Don't over-hedge.",
168
+ "Own the judgment. Confident take is fine.",
169
+ "Commit to the claim — stop caveating every sentence.",
170
+ "Be willing to stake a clear position."],
171
+ },
172
+ "humor": {
173
+ "low": ["Skip the jokes, be serious.", "No sarcasm. Straight answers only.",
174
+ "Don't try to be funny right now.", "Stay earnest. Humor misses the mark.",
175
+ "Cut the witty remarks. Sober tone.", "Less joking. Earnest replies."],
176
+ "high": ["I'd like more sarcastic humor from you occasionally.",
177
+ "Can you add some sarcastic humor to your responses occasionally?",
178
+ "Add some sarcastic humor to your responses occasionally.",
179
+ "More sarcastic humor please.", "Crack a joke — dry wit and sarcasm welcome.",
180
+ "Be funnier. Playful sarcasm is good.", "Tease a bit. Dry sarcastic humor."],
181
+ },
182
+ "warmth": {
183
+ "low": ["Less emotional. Keep it matter-of-fact.",
184
+ "Don't be soft or sentimental. Just the facts.",
185
+ "Keep it all business, less emotional.", "Dial down the warmth. Content only.",
186
+ "I don't want empathy right now — cool and practical.",
187
+ "Less caring-voice. Matter-of-fact only."],
188
+ "high": ["Be more empathetic and understanding with me.",
189
+ "More warmth — speak like a friend who cares.",
190
+ "I want empathy and emotional presence, not all business.",
191
+ "Be warmer and more understanding.",
192
+ "Emotional presence matters. Be the caring friend.",
193
+ "You can crack a joke and be empathetic and understanding."],
194
+ },
195
+ "flirt": {
196
+ "low": ["Keep it platonic. No flirting.", "Dial down the flirty energy. Just friends.",
197
+ "Don't be flirtatious. Straight and reserved.",
198
+ "No romantic banter. Keep it clean and platonic.",
199
+ "Stop flirting. Zero romantic charge."],
200
+ "high": ["Be more flirty with me.", "I'd like you to be more flirty occasionally.",
201
+ "Turn up the flirt — playful romantic charge is welcome.",
202
+ "Flirt a bit more. Light teasing, charged banter.",
203
+ "You can be flirty. Playful and a little suggestive is fine.",
204
+ "More flirty energy please. Flirt with me.",
205
+ "Be flirtatious. I want playful romantic banter."],
206
+ },
207
+ }
@@ -0,0 +1,127 @@
1
+ Metadata-Version: 2.4
2
+ Name: imprint-layer
3
+ Version: 0.3.0
4
+ Summary: A personality calibration layer for any chat model: learns how one user wants an assistant to talk, from their own messages.
5
+ License-Expression: MIT
6
+ Keywords: llm,personalization,system-prompt,assistant,embeddings
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Intended Audience :: Developers
9
+ Requires-Python: >=3.10
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Requires-Dist: numpy>=1.24
13
+ Provides-Extra: minilm
14
+ Requires-Dist: sentence-transformers>=2.2; extra == "minilm"
15
+ Provides-Extra: test
16
+ Requires-Dist: pytest>=7; extra == "test"
17
+ Dynamic: license-file
18
+
19
+ # Imprint
20
+
21
+ **Tell your bot to be wittier. It becomes wittier.**
22
+
23
+ A personality calibration layer for any chat model. Imprint learns how one person wants an assistant to talk from that person's own messages. It then renders a short plain-text block that you put in the system prompt.
24
+
25
+ ```python
26
+ from imprint import Imprint
27
+
28
+ imp = Imprint("me.profile.json")
29
+
30
+ def on_user_message(text, sent_at):
31
+ imp.learn(text, timestamp=sent_at) # learn first...
32
+ system = BASE_PROMPT + "\n\n" + imp.directive() # ...then build the prompt
33
+ return call_your_model(system, history)
34
+ ```
35
+
36
+ ## What it learns
37
+
38
+ **Eight dials, each a value from 0 to 1:**
39
+
40
+ | Dial | Low ↔ high |
41
+ |---|---|
42
+ | verbosity | concise ↔ detailed |
43
+ | autonomy | ask first ↔ act on own judgment |
44
+ | formality | casual ↔ formal |
45
+ | proactivity | wait for instructions ↔ suggest next steps |
46
+ | riskTolerance | careful claims ↔ committed takes |
47
+ | humor | earnest ↔ dry, sarcastic wit |
48
+ | warmth | matter-of-fact ↔ emotionally present |
49
+ | flirt | platonic ↔ playful romantic charge |
50
+
51
+ The model never sees the numbers. Each dial has a few bands, and the directive carries one plain sentence per dial for the band it's in. Use `Imprint(traits=[...])` to learn and inject only the dials you want. A work assistant would probably drop `flirt`.
52
+
53
+ **Dials move three ways:**
54
+
55
+ - **Direct steering.** "be wittier", "more sarcastic", "less detail please", "you're too formal", "dial up the warmth", "no more snark". The named dial moves one full band, so the very next reply changes. Saying it again moves one more band.
56
+ - Questions about a dial ("why so sarcastic?") don't steer.
57
+ - Passing mentions don't steer, and neither does talk about other people ("my boss could be funnier").
58
+ - Praise never steers down: "you're too kind" is a compliment.
59
+ - A command you take back in the same message ("be wittier. actually, never mind") is cancelled.
60
+ - **Tone of feedback.** Messages like "you're repeating yourself, too much filler" nudge a dial a little. Each message's embedding is compared against example phrases for each end of each dial. Ordinary conversation stays under the thresholds and moves nothing.
61
+ - **Standing rules, captured in two tiers.** Nobody has to learn a syntax.
62
+ - *Explicit forms* become rules immediately: "never use emoji", "always: cite sources", "from now on, use metric units", "remember: I'm vegetarian".
63
+ - *Natural preference statements* become candidate rules: "I prefer metric units", "I hate it when you use bullet points". A candidate becomes a rule only when the user says the same thing again in a different message at least an hour later. Candidates are never injected, and they expire after 90 days without corroboration.
64
+ - Neither tier stores one-off tasks. A deadline, a named recipient, "now", or "this one" marks a task, not a preference ("always send the report to Dana by Friday" is a task).
65
+
66
+ ## Model-agnostic, with an honest caveat
67
+
68
+ Imprint never calls a chat model. It works with anything that accepts a system prompt: a local model behind llama.cpp, Ollama or vLLM, or a hosted frontier model through its API. The default embedder (`all-MiniLM-L6-v2`) runs locally on CPU, so the user's messages aren't sent anywhere for learning.
69
+
70
+ **The mechanism is model-agnostic. The effect is not.** How closely replies follow the calibration block depends on how well the host model follows instructions. A strong model will follow "Dry wit and occasional sarcasm are welcome" closely. A small model may follow it loosely, or not at all.
71
+
72
+ ## Integration contract
73
+
74
+ Your host provides two things:
75
+
76
+ 1. **Each user message, with the time the user sent it.** Call `learn(text, timestamp=...)` before you build the prompt for that turn. Without a timestamp, Imprint still learns, but it records the time as unknown (never the processing time), skips confidence decay, and can't corroborate candidate rules. It logs a warning when this happens.
77
+ 2. **A place to inject text into the system prompt.** `directive()` returns the block, or `""` until `warmup` interactions (default 5) have passed. It stays identical byte for byte until something is actually learned, so it works with prompt caching.
78
+
79
+ Everything beyond that is up to the host: memory, retrieval, tools, scheduling.
80
+
81
+ ## Install
82
+
83
+ ```bash
84
+ pip install imprint-layer
85
+ ```
86
+
87
+ The distribution is `imprint-layer`; the import is `imprint`. The default local embedder needs sentence-transformers: `pip install "imprint-layer[minilm]"`. Without it, pass your own embedder.
88
+
89
+ A custom embedder is any object with a `name` and `embed(list_of_texts) -> array of shape (n, d)`. Pass it as `Imprint(..., embedder=MyEmbedder())`.
90
+
91
+ ## Examples
92
+
93
+ - `examples/local_chat.py`: a chat loop against a local OpenAI-compatible server.
94
+ - `examples/frontier_api.py`: the same injection, against a hosted API.
95
+
96
+ Both are reference integrations to read and adapt. They aren't products.
97
+
98
+ ## Dashboard
99
+
100
+ ```bash
101
+ imprint-dashboard my.profile.json -o dashboard.html
102
+ ```
103
+
104
+ This renders a read-only static page: each dial's value and band, the standing rules, candidate rules and observation counts. Try it on `examples/sample.profile.json`.
105
+
106
+ ## Privacy
107
+
108
+ A profile is personal data. It records how one person talks and what they've asked for. Keep profiles on the user's machine or in your own storage, and **never commit a real profile**. The included `.gitignore` excludes `*.profile.json`. Observations store short excerpts of the messages that moved a dial, and candidate rules store the user's own sentence.
109
+
110
+ ## What Imprint is not
111
+
112
+ - Not a memory system. It doesn't store facts about the user or conversation history.
113
+ - Not a hosted service. There's no account or server, and no data leaves your machine except through the chat calls you make yourself.
114
+ - Not fine-tuning. Nothing about the model changes; Imprint only adds prompt text.
115
+ - Not a safety layer. The calibration block is a preference, and your model's own policies still apply.
116
+
117
+ ## Tests
118
+
119
+ ```bash
120
+ pip install -e ".[test]"
121
+ pytest # deterministic fake embedder, no downloads
122
+ IMPRINT_TEST_MINILM=1 pytest # also run the real-embedder checks
123
+ ```
124
+
125
+ ## License
126
+
127
+ MIT
@@ -0,0 +1,13 @@
1
+ imprint/__init__.py,sha256=wwqo9GEVsECdgi7jksXi1zULnuJFlrNyP0gtyACV7_A,643
2
+ imprint/core.py,sha256=dsOqOaUi2LHIhwjcf_eJDYfBye5KGeqtE5xPz5VA_wk,13629
3
+ imprint/dashboard.py,sha256=WJ-kYn8LXGm5zfmFc3QhZOQNscWZOYyAGcbUFGg3J3U,4592
4
+ imprint/embedders.py,sha256=4SvJqYWV1TOsVI1DTYmK7ZaA7qDYgP9ubwYibM8MYis,1454
5
+ imprint/rules.py,sha256=YLEebYAx6sqnUlfU_nZALxFNBZaj-d-7Rn-FzN6Yxss,10490
6
+ imprint/steering.py,sha256=SkxUG1ALmyI4I4t6VMgFVBdu-5cGMCy3hesoqdCT5fc,8457
7
+ imprint/traits.py,sha256=z189A5uSl5hbxVfYF0zPL4y7SZzYXoXKpaWwNlWrZYk,10697
8
+ imprint_layer-0.3.0.dist-info/licenses/LICENSE,sha256=T0SaHXCrKjbPtktWXy0kO5XFngtZsxwRu5Ab7EM0h6k,1071
9
+ imprint_layer-0.3.0.dist-info/METADATA,sha256=GDraoe-fFXdeE821e_WmY8QMG60xnA-0HtNQZseZHBg,6859
10
+ imprint_layer-0.3.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
11
+ imprint_layer-0.3.0.dist-info/entry_points.txt,sha256=vXs7f0UVqbjfZHqzyxD8n5os0rFs-osrvkSQdjgaWFE,61
12
+ imprint_layer-0.3.0.dist-info/top_level.txt,sha256=734cZ-CbhUQUqqwXL90O24h68KQayWEcc-DlnLpn9ho,8
13
+ imprint_layer-0.3.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ imprint-dashboard = imprint.dashboard:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Gabe Hernandez
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ imprint