cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,760 @@
|
|
|
1
|
+
"""The μ=0 extraction pattern library.
|
|
2
|
+
|
|
3
|
+
High-precision syntactic patterns over Subject-Relation-Value triples:
|
|
4
|
+
first-person (user), third-person (entities), and assistant-turn
|
|
5
|
+
(second-person) forms, with temporal anchors resolved by bridge.dates.
|
|
6
|
+
Zero LLM calls — this is the deterministic perception layer that makes
|
|
7
|
+
BEAM-honest μ=0 ingest possible while competitors burn LLM extraction
|
|
8
|
+
at write time.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from datetime import datetime
|
|
16
|
+
|
|
17
|
+
from cortexm.bridge.dates import find_dates
|
|
18
|
+
|
|
19
|
+
# ---------------------------------------------------------------- helpers
|
|
20
|
+
|
|
21
|
+
# Case-sensitive (scoped — patterns compile with re.I for verbs, but
|
|
22
|
+
# entity values must keep their capitalization semantics): first word
|
|
23
|
+
# capitalized, subsequent words capitalized or internal to the name.
|
|
24
|
+
ORG = r"(?-i:[A-Z][\w&'.,-]*)(?:\s+(?-i:[A-Z])[\w&'.,-]*)*"
|
|
25
|
+
NAME = r"(?-i:[A-Z][a-zA-Z'-]+)(?:\s+(?-i:[A-Z])[a-zA-Z'-]+)?"
|
|
26
|
+
LOWERPH = r"[a-zA-Z][\w' +#.-]{1,44}"
|
|
27
|
+
|
|
28
|
+
PRONOUNS = {"she", "he", "they", "it", "her", "him", "them"}
|
|
29
|
+
FIRST_PRONOUNS = {"i", "we", "me", "us", "my", "our"}
|
|
30
|
+
|
|
31
|
+
ROLE_BLOCK = {
|
|
32
|
+
"tired", "happy", "sad", "sorry", "sure", "okay", "ok", "fine",
|
|
33
|
+
"glad", "ready", "here", "back", "busy", "excited", "curious",
|
|
34
|
+
"confused", "hungry", "free", "done", "good", "great", "well",
|
|
35
|
+
"not", "just", "still", "new", "all", "so", "very", "really",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
FAMILY_MAP = {
|
|
39
|
+
"sister": "sibling", "brother": "sibling", "twin": "sibling",
|
|
40
|
+
"mother": "parent", "mom": "parent", "father": "parent", "dad": "parent",
|
|
41
|
+
"wife": "spouse", "husband": "spouse", "partner": "spouse",
|
|
42
|
+
"daughter": "child", "son": "child", "cousin": "friend",
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
IS_MY_MAP = {
|
|
46
|
+
"sister": "sibling", "brother": "sibling", "mother": "parent",
|
|
47
|
+
"mom": "parent", "father": "parent", "dad": "parent", "wife": "spouse",
|
|
48
|
+
"husband": "spouse", "partner": "spouse", "friend": "friend",
|
|
49
|
+
"colleague": "friend", "teammate": "friend", "cousin": "friend",
|
|
50
|
+
"manager": "reports_to", "boss": "reports_to",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class Candidate:
|
|
56
|
+
subject: str
|
|
57
|
+
relation: str
|
|
58
|
+
value: str
|
|
59
|
+
confidence: float
|
|
60
|
+
pattern: str
|
|
61
|
+
span: tuple[int, int] = (0, 0)
|
|
62
|
+
valid_from: str | None = None
|
|
63
|
+
valid_to: str | None = None
|
|
64
|
+
retraction: bool = False
|
|
65
|
+
note: str = ""
|
|
66
|
+
# Tier-4 fix: track whether this candidate came from a strict
|
|
67
|
+
# trigger match or a Bitap-widened trigger. Bitap widening is
|
|
68
|
+
# essential for OOD recall (catches "wrks" -> "works"), but the
|
|
69
|
+
# wider net admits false positives — a sentence that triggered
|
|
70
|
+
# on a fuzzy match should carry a slightly lower confidence so
|
|
71
|
+
# the writer's min_confidence threshold can filter out noisy
|
|
72
|
+
# extractions. Stays μ=0 — the flag is deterministic.
|
|
73
|
+
trigger_source: str = "strict" # "strict" | "bitap_widened"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass
|
|
77
|
+
class ExtractionContext:
|
|
78
|
+
user_id: str = "default"
|
|
79
|
+
agent_id: str | None = None
|
|
80
|
+
run_id: str | None = None
|
|
81
|
+
ts: datetime | None = None
|
|
82
|
+
speaker: str = "user" # user | assistant | system
|
|
83
|
+
subject_name: str | None = None # learned canonical name of the user
|
|
84
|
+
lexicon: set[str] = field(default_factory=set)
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def subject(self) -> str:
|
|
88
|
+
return self.subject_name or f"user:{self.user_id}"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def clean_value(v: str) -> str:
|
|
92
|
+
v = v.strip().strip('"\',.!?;:')
|
|
93
|
+
v = re.sub(r"^(?:the|a|an)\s+", "", v, flags=re.I)
|
|
94
|
+
v = re.sub(r"\s+", " ", v)
|
|
95
|
+
v = v.strip().rstrip(".,!?;:")
|
|
96
|
+
return v.strip()
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def date_in(text: str, ts: datetime | None) -> str | None:
|
|
100
|
+
if ts is None:
|
|
101
|
+
return None
|
|
102
|
+
ds = find_dates(text, ts)
|
|
103
|
+
return ds[0]["iso"] if ds else None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# ---------------------------------------------------------------- patterns
|
|
107
|
+
# Each entry: (name, regex, handler(match, ctx, sent_span) -> list[Candidate])
|
|
108
|
+
# m.group("val") etc. Handlers return candidates with subject placeholder
|
|
109
|
+
# "SELF" resolved by the extractor to ctx.subject.
|
|
110
|
+
|
|
111
|
+
PATTERNS: list[tuple[str, re.Pattern, object]] = []
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def pattern(name: str, rx: str):
|
|
115
|
+
compiled = re.compile(rx, re.I | re.M)
|
|
116
|
+
|
|
117
|
+
def deco(fn):
|
|
118
|
+
PATTERNS.append((name, compiled, fn))
|
|
119
|
+
return fn
|
|
120
|
+
|
|
121
|
+
return deco
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# --- identity -------------------------------------------------------------
|
|
125
|
+
@pattern("name_intro", rf"\bmy name is\s+(?P<val>{NAME})")
|
|
126
|
+
def _name(m, ctx, sp, ts, sent):
|
|
127
|
+
full = clean_value(m.group("val"))
|
|
128
|
+
out = []
|
|
129
|
+
if ctx.subject_name is None or ctx.subject_name.lower() != full.lower():
|
|
130
|
+
out.append(Candidate("SELF", "name", full, 0.95, "name_intro", (sp[0] + m.start(), sp[0] + m.end())))
|
|
131
|
+
parts = full.split()
|
|
132
|
+
if len(parts) > 1 and len(parts[0]) > 2:
|
|
133
|
+
out.append(Candidate(full, "alias", parts[0], 0.9, "name_intro_alias"))
|
|
134
|
+
return out
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@pattern("called", rf"\b(?:i am called|i'm called|call me|i go by|you can call me)\s+(?P<val>{NAME})")
|
|
138
|
+
def _called(m, ctx, sp, ts, sent):
|
|
139
|
+
v = clean_value(m.group("val"))
|
|
140
|
+
rel = "name" if ctx.subject_name is None else "alias"
|
|
141
|
+
target = ctx.subject_name or "SELF"
|
|
142
|
+
return [Candidate(target, rel, v, 0.9, "called")]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
# --- employment ------------------------------------------------------------
|
|
146
|
+
WORK_AT = rf"(?P<val>{ORG})"
|
|
147
|
+
@pattern("works_at",
|
|
148
|
+
rf"\bi(?:'?m)?\s+(?:(?:now|currently|these days)\s+)?"
|
|
149
|
+
rf"(?:work|worked|working"
|
|
150
|
+
rf"|am\s+(?:(?:now|currently|these days)\s+)?working"
|
|
151
|
+
rf"|now\s+work)\s+(?:at|for)\s+(?:the\s+)?{WORK_AT}")
|
|
152
|
+
def _works(m, ctx, sp, ts, sent):
|
|
153
|
+
v = clean_value(m.group("val"))
|
|
154
|
+
vf = date_in(sent, ts) if re.search(r"\b(joined|started|got a job)\b", sent, re.I) else None
|
|
155
|
+
return [Candidate("SELF", "works_at", v, 0.92, "works_at", valid_from=vf)]
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@pattern("joined_org",
|
|
159
|
+
rf"\bi\s+(?:joined|started(?:\s+working)?\s+at|got a job at|moved to a job at)\s+{WORK_AT}")
|
|
160
|
+
def _joined(m, ctx, sp, ts, sent):
|
|
161
|
+
v = clean_value(m.group("val"))
|
|
162
|
+
return [Candidate("SELF", "works_at", v, 0.92, "joined_org",
|
|
163
|
+
valid_from=date_in(sent, ts))]
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@pattern("at_org", rf"\bi'?m\s+(?:now\s+|currently\s+|these days\s+)?at\s+(?P<val>{ORG})")
|
|
167
|
+
def _at_org(m, ctx, sp, ts, sent):
|
|
168
|
+
v = clean_value(m.group("val"))
|
|
169
|
+
return [Candidate("SELF", "works_at", v, 0.85, "at_org")]
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@pattern("left_org",
|
|
173
|
+
rf"\bi\s+(?:left|quit|resigned from|was laid off from|departed)\s+(?P<val>{ORG})")
|
|
174
|
+
def _left(m, ctx, sp, ts, sent):
|
|
175
|
+
v = clean_value(m.group("val"))
|
|
176
|
+
vf = date_in(sent, ts)
|
|
177
|
+
return [Candidate("SELF", "left", v, 0.9, "left_org", valid_from=vf,
|
|
178
|
+
retraction=True)]
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
@pattern("no_longer", rf"\bi\s+(?:no longer|don'?t|do not)\s+work\s+(?:at|for)\s+(?P<val>{ORG})")
|
|
182
|
+
def _no_longer(m, ctx, sp, ts, sent):
|
|
183
|
+
v = clean_value(m.group("val"))
|
|
184
|
+
return [Candidate("SELF", "left", v, 0.88, "no_longer", retraction=True)]
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@pattern("no_longer_at", rf"\bi'?m\s+no longer\s+(?:at|with)\s+(?P<val>{ORG})")
|
|
188
|
+
def _no_longer_at(m, ctx, sp, ts, sent):
|
|
189
|
+
v = clean_value(m.group("val"))
|
|
190
|
+
return [Candidate("SELF", "left", v, 0.88, "no_longer_at", retraction=True)]
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@pattern("role", rf"\bi\s+work\s+as\s+(?:a|an|the)\s+(?P<val>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)"
|
|
194
|
+
rf"|\bi'?m\s+(?:a|an|the)\s+(?P<val2>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)"
|
|
195
|
+
rf"|\bi\s+am\s+(?:a|an|the)\s+(?P<val3>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)")
|
|
196
|
+
def _role(m, ctx, sp, ts, sent):
|
|
197
|
+
v = clean_value(m.group("val") or m.group("val2") or m.group("val3"))
|
|
198
|
+
if v.lower() in ROLE_BLOCK:
|
|
199
|
+
return []
|
|
200
|
+
return [Candidate("SELF", "role", v, 0.85, "role")]
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
@pattern("role_as", rf"\bas\s+(?:a|an)\s+(?P<val>[a-z][a-z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for)\b|$)")
|
|
204
|
+
def _role_as(m, ctx, sp, ts, sent):
|
|
205
|
+
v = clean_value(m.group("val"))
|
|
206
|
+
if v.lower() in ROLE_BLOCK:
|
|
207
|
+
return []
|
|
208
|
+
return [Candidate("SELF", "role", v, 0.8, "role_as")]
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
@pattern("role_my", rf"\bmy\s+(?:job|role|title|position)\s+is\s+(?:a|an|the)?\s*(?P<val>[a-z][a-z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for)\b|$)")
|
|
212
|
+
def _role_my(m, ctx, sp, ts, sent):
|
|
213
|
+
v = clean_value(m.group("val"))
|
|
214
|
+
return [Candidate("SELF", "role", v, 0.9, "role_my")]
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@pattern("reports_to", rf"\bmy\s+(?:manager|boss|lead|supervisor|team lead)\s+is\s+(?P<val>{NAME})")
|
|
218
|
+
def _mgr(m, ctx, sp, ts, sent):
|
|
219
|
+
return [Candidate("SELF", "reports_to", clean_value(m.group("val")), 0.9, "reports_to")]
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
@pattern("i_report", rf"\bi\s+report\s+to\s+(?P<val>{NAME})")
|
|
223
|
+
def _i_report(m, ctx, sp, ts, sent):
|
|
224
|
+
return [Candidate("SELF", "reports_to", clean_value(m.group("val")), 0.9, "i_report")]
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
@pattern("member_of", rf"\bi'?m\s+(?:now\s+)?(?:on|part of)\s+the\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s*(?:team|group|squad|org)?\b")
|
|
228
|
+
def _member(m, ctx, sp, ts, sent):
|
|
229
|
+
v = clean_value(m.group("val"))
|
|
230
|
+
if not v:
|
|
231
|
+
return []
|
|
232
|
+
team = v if v.lower().endswith(("team", "group", "squad")) else f"{v} team"
|
|
233
|
+
return [Candidate("SELF", "member_of", team, 0.88, "member_of")]
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@pattern("joined_team", rf"\bi\s+joined\s+the\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s+(?:team|group|squad)\b")
|
|
237
|
+
def _joined_team(m, ctx, sp, ts, sent):
|
|
238
|
+
v = clean_value(m.group("val"))
|
|
239
|
+
return [Candidate("SELF", "member_of", f"{v} team", 0.9, "joined_team",
|
|
240
|
+
valid_from=date_in(sent, ts))]
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# --- residence -------------------------------------------------------------
|
|
244
|
+
@pattern("lives_in", rf"\bi\s+(?:live|lived|'m living|am living|'m based|am based|reside)\s+in\s+(?P<val>{ORG})")
|
|
245
|
+
def _lives(m, ctx, sp, ts, sent):
|
|
246
|
+
return [Candidate("SELF", "lives_in", clean_value(m.group("val")), 0.92, "lives_in")]
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
@pattern("im_in", rf"\bi'?m\s+(?:currently\s+|now\s+)?in\s+(?P<val>{ORG})")
|
|
250
|
+
def _im_in(m, ctx, sp, ts, sent):
|
|
251
|
+
v = clean_value(m.group("val"))
|
|
252
|
+
if v.lower() in ("fact", "love", "trouble", "debt", "charge", "love with"):
|
|
253
|
+
return []
|
|
254
|
+
return [Candidate("SELF", "lives_in", v, 0.7, "im_in")]
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
@pattern("moved_to", rf"\b(?:i|we)\s+(?:moved|relocated)\s+to\s+(?P<val>{ORG})")
|
|
258
|
+
def _moved(m, ctx, sp, ts, sent):
|
|
259
|
+
v = clean_value(m.group("val"))
|
|
260
|
+
return [Candidate("SELF", "moved_to", v, 0.92, "moved_to",
|
|
261
|
+
valid_from=date_in(sent, ts))]
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
# --- preferences -----------------------------------------------------------
|
|
265
|
+
LIKE_TAIL = r"(?=[,.!?]|\s+(?:but|though|when|because|and|so|especially)\b|$)"
|
|
266
|
+
@pattern("likes", rf"\bi\s+(?:(?:really|absolutely|totally)\s+)?(?:love|loved|like|liked|enjoy|enjoyed)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?){LIKE_TAIL}")
|
|
267
|
+
def _likes(m, ctx, sp, ts, sent):
|
|
268
|
+
v = clean_value(m.group("val"))
|
|
269
|
+
if v.lower() in ("it", "that", "this", "them", "you"):
|
|
270
|
+
return []
|
|
271
|
+
return [Candidate("SELF", "likes", v, 0.85, "likes")]
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@pattern("fan_of", rf"\bi'?m\s+(?:a|an)\s+(?:big\s+)?fan\s+of\s+(?P<val>{LOWERPH})")
|
|
275
|
+
def _fan(m, ctx, sp, ts, sent):
|
|
276
|
+
return [Candidate("SELF", "likes", clean_value(m.group("val")), 0.85, "fan_of")]
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
@pattern("favorite", rf"\bmy favorite\s+[\w ]{{2,24}}\s+is\s+(?P<val>{LOWERPH})")
|
|
280
|
+
def _favorite(m, ctx, sp, ts, sent):
|
|
281
|
+
return [Candidate("SELF", "likes", clean_value(m.group("val")), 0.88, "favorite")]
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
@pattern("dislikes", rf"\bi\s+(?:hate|hated|dislike|can'?t stand|don'?t like|do not like)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?){LIKE_TAIL}")
|
|
285
|
+
def _dislikes(m, ctx, sp, ts, sent):
|
|
286
|
+
return [Candidate("SELF", "dislikes", clean_value(m.group("val")), 0.85, "dislikes")]
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
@pattern("prefers", rf"\bi\s+(?:'d\s+)?(?:prefer|preferre?d)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?:\s+(?:over|to|than)\s+[\w' -]+)?{LIKE_TAIL}")
|
|
290
|
+
def _prefers(m, ctx, sp, ts, sent):
|
|
291
|
+
v = clean_value(m.group("val"))
|
|
292
|
+
cat = re.search(r"\bfor\s+([a-z][a-z ]{2,20})", sent[m.start():])
|
|
293
|
+
if cat:
|
|
294
|
+
v = f"{v} (for {clean_value(cat.group(1))})"
|
|
295
|
+
return [Candidate("SELF", "prefers", v, 0.9, "prefers")]
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
@pattern("pref_change", rf"\b(?:actually|these days|nowadays|lately|recently)[,.]?\s*(?:i\s+)?(?:prefer|like|love|drink|use|order)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?=[,.!?]|$|\s+(?:but|though|and)\b)")
|
|
299
|
+
def _pref_change(m, ctx, sp, ts, sent):
|
|
300
|
+
return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.88, "pref_change")]
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
@pattern("switched_to", rf"\b(?:i'?ve|i have)\s+(?:switched|moved)\s+to\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?=[,.!?]|$|\s+(?:but|though|and)\b)")
|
|
304
|
+
def _switched(m, ctx, sp, ts, sent):
|
|
305
|
+
return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.88, "switched_to")]
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
@pattern("more_of_a", rf"\bi'?m\s+more\s+of\s+a\s+(?P<val>{LOWERPH})\s+(?:person|guy|girl|fan|drinker|person\s+now)")
|
|
309
|
+
def _more_of(m, ctx, sp, ts, sent):
|
|
310
|
+
return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.82, "more_of_a")]
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
# --- skills & education ----------------------------------------------------
|
|
314
|
+
@pattern("skill_know", rf"\bi\s+(?:know|code in|write|program in|build with)\s+(?P<val>[A-Za-z+#.][\w+#.]*(?:\s+(?:and|&)\s+[A-Za-z+#.][\w+#.]*)*)")
|
|
315
|
+
def _skill(m, ctx, sp, ts, sent):
|
|
316
|
+
out = []
|
|
317
|
+
for part in re.split(r"\s+(?:and|&)\s+", clean_value(m.group("val"))):
|
|
318
|
+
if len(part) > 1 and part.lower() not in ("it", "that", "this", "them"):
|
|
319
|
+
out.append(Candidate("SELF", "has_skill", part, 0.85, "skill_know"))
|
|
320
|
+
return out
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
@pattern("skill_learning", rf"\bi(?:'ve|\s+have)\s+been\s+learning\s+(?P<val>[A-Za-z+#.][\w+#.]*)|\bi'?m\s+learning\s+(?P<val2>[A-Za-z+#.][\w+#.]*)")
|
|
324
|
+
def _skill_learn(m, ctx, sp, ts, sent):
|
|
325
|
+
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
326
|
+
return [Candidate("SELF", "has_skill", v, 0.8, "skill_learning")] if v else []
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
@pattern("skill_prof", rf"\bi'?m\s+(?:proficient|skilled|experienced)\s+in\s+(?P<val>[\w+#. ]{{2,40}})")
|
|
330
|
+
def _skill_prof(m, ctx, sp, ts, sent):
|
|
331
|
+
return [Candidate("SELF", "has_skill", clean_value(m.group("val")), 0.85, "skill_prof")]
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
@pattern("studied_at", rf"\bi\s+studied\s+(?P<major>[a-z][a-z ]{{2,40}}?)\s+at\s+(?P<val>{ORG})")
|
|
335
|
+
def _studied(m, ctx, sp, ts, sent):
|
|
336
|
+
major = clean_value(m.group("major"))
|
|
337
|
+
org = clean_value(m.group("val"))
|
|
338
|
+
return [Candidate("SELF", "studied", f"{major} at {org}", 0.88, "studied_at")]
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
@pattern("majored", rf"\bi\s+majored\s+in\s+(?P<val>[a-z][a-z ]{{2,40}})|\bmy degree is in\s+(?P<val2>[a-z][a-z ]{{2,40}})")
|
|
342
|
+
def _majored(m, ctx, sp, ts, sent):
|
|
343
|
+
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
344
|
+
return [Candidate("SELF", "studied", v, 0.85, "majored")] if v else []
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
@pattern("speaks", rf"\bi\s+speak\s+(?P<val>[A-Za-z]+(?:\s+and\s+[A-Za-z]+)*)")
|
|
348
|
+
def _speaks(m, ctx, sp, ts, sent):
|
|
349
|
+
out = []
|
|
350
|
+
for part in re.split(r"\s+and\s+", clean_value(m.group("val"))):
|
|
351
|
+
out.append(Candidate("SELF", "speaks", part, 0.85, "speaks"))
|
|
352
|
+
return out
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
# --- personal ---------------------------------------------------------------
|
|
356
|
+
@pattern("birthday", rf"\bmy birthday is\s+(?P<val>[^,.!?]{{3,30}})|\bi was born on\s+(?P<val2>[^,.!?]{{3,30}})")
|
|
357
|
+
def _birthday(m, ctx, sp, ts, sent):
|
|
358
|
+
raw = m.group("val") or m.group("val2") or ""
|
|
359
|
+
v = clean_value(raw)
|
|
360
|
+
return [Candidate("SELF", "birthday", v, 0.9, "birthday")]
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
@pattern("age", r"\bi'?m\s+(\d{1,2})\s+years?\s+old")
|
|
364
|
+
def _age(m, ctx, sp, ts, sent):
|
|
365
|
+
return [Candidate("SELF", "age", m.group(1), 0.9, "age")]
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
@pattern("family", rf"\bmy\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|daughter|son|cousin|twin)(?:'s name)?\s+(?:is|is called)\s+(?P<val>{NAME})")
|
|
369
|
+
def _family(m, ctx, sp, ts, sent):
|
|
370
|
+
rel = FAMILY_MAP.get(m.group("rel").lower(), "friend")
|
|
371
|
+
return [Candidate("SELF", rel, clean_value(m.group("val")), 0.9, "family")]
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
@pattern("family2", rf"\bmy\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|daughter|son|cousin|twin)\s+(?P<val>{NAME})\b")
|
|
375
|
+
def _family2(m, ctx, sp, ts, sent):
|
|
376
|
+
rel = FAMILY_MAP.get(m.group("rel").lower(), "friend")
|
|
377
|
+
return [Candidate("SELF", rel, clean_value(m.group("val")), 0.88, "family2")]
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
@pattern("is_my", rf"\b(?P<val>{NAME})\s+is\s+my\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|friend|colleague|teammate|cousin|manager|boss)")
|
|
381
|
+
def _is_my(m, ctx, sp, ts, sent):
|
|
382
|
+
rel = IS_MY_MAP.get(m.group("rel").lower(), "friend")
|
|
383
|
+
return [Candidate("SELF", rel, clean_value(m.group("val")), 0.88, "is_my")]
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
@pattern("pet", rf"\bmy\s+(?P<kind>dog|cat|bird|rabbit)\s+is\s+(?:named\s+|called\s+)?(?P<val>{NAME})")
|
|
387
|
+
def _pet(m, ctx, sp, ts, sent):
|
|
388
|
+
return [Candidate("SELF", "has_pet", f"{m.group('kind')} named {clean_value(m.group('val'))}",
|
|
389
|
+
0.88, "pet")]
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
@pattern("hobby", rf"\bmy hobby is\s+(?P<val>[a-z][a-z ]{{2,40}})|\bin my free time\s+i\s+(?P<val2>[a-z][a-z ]{{2,40}})")
|
|
393
|
+
def _hobby(m, ctx, sp, ts, sent):
|
|
394
|
+
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
395
|
+
return [Candidate("SELF", "hobby", v, 0.8, "hobby")] if v else []
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
@pattern("goal", rf"\bmy goal is to\s+(?P<val>[a-z][a-z ]{{2,50}})|\bi'?m planning to\s+(?P<val2>[a-z][a-z ]{{2,50}})")
|
|
399
|
+
def _goal(m, ctx, sp, ts, sent):
|
|
400
|
+
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
401
|
+
return [Candidate("SELF", "goal", v, 0.75, "goal")] if v else []
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
# --- projects & events -------------------------------------------------------
|
|
405
|
+
@pattern("works_on", rf"\bi'?m\s+(?:currently\s+)?working\s+on\s+(?P<val>{ORG})|\bi\s+work\s+on\s+(?P<val2>{ORG})")
|
|
406
|
+
def _works_on(m, ctx, sp, ts, sent):
|
|
407
|
+
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
408
|
+
return [Candidate("SELF", "works_on", v, 0.88, "works_on")] if v else []
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
@pattern("building", rf"\b(?:we'?re|we are|i'?m|i am)\s+(?:currently\s+)?building\s+(?P<val>{ORG})")
|
|
412
|
+
def _building(m, ctx, sp, ts, sent):
|
|
413
|
+
return [Candidate("SELF", "works_on", clean_value(m.group("val")), 0.85, "building")]
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
@pattern("completed", rf"\b(?:we|i)\s+(?:shipped|launched|finished|completed|released|deployed)\s+(?:the\s+)?(?P<val>{ORG})")
|
|
417
|
+
def _completed(m, ctx, sp, ts, sent):
|
|
418
|
+
v = clean_value(m.group("val"))
|
|
419
|
+
return [Candidate("SELF", "completed", v, 0.88, "completed",
|
|
420
|
+
valid_to=date_in(sent, ts))]
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
@pattern("used_to_work", rf"\bi used to\s+(?:work|working)\s+(?:at|for)\s+(?P<val>{ORG})")
|
|
424
|
+
def _used_work(m, ctx, sp, ts, sent):
|
|
425
|
+
v = clean_value(m.group("val"))
|
|
426
|
+
return [Candidate("SELF", "works_at", v, 0.7, "used_to_work",
|
|
427
|
+
valid_to=date_in(sent, ts) or None)]
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
@pattern("used_to_live", rf"\bi used to\s+live\s+in\s+(?P<val>{ORG})")
|
|
431
|
+
def _used_live(m, ctx, sp, ts, sent):
|
|
432
|
+
v = clean_value(m.group("val"))
|
|
433
|
+
return [Candidate("SELF", "lives_in", v, 0.7, "used_to_live",
|
|
434
|
+
valid_to=date_in(sent, ts) or None)]
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
# --- third person -------------------------------------------------------------
|
|
438
|
+
TP_VERBS = {
|
|
439
|
+
"works at": "works_at", "work at": "works_at", "worked at": "works_at",
|
|
440
|
+
"joined": "works_at", "left": "left", "quit": "left",
|
|
441
|
+
"lives in": "lives_in", "live in": "lives_in", "moved to": "moved_to",
|
|
442
|
+
"manages": "manages", "manage": "manages", "managing": "manages",
|
|
443
|
+
"uses": "uses", "use": "uses", "using": "uses",
|
|
444
|
+
"prefers": "prefers", "prefer": "prefers",
|
|
445
|
+
"likes": "likes", "like": "likes",
|
|
446
|
+
"studied at": "studied_at", "reports to": "reports_to",
|
|
447
|
+
"leads": "manages", "lead": "manages",
|
|
448
|
+
}
|
|
449
|
+
TP_RX = (rf"\b(?P<subj>{NAME})\s+(?P<verb>"
|
|
450
|
+
+ "|".join(re.escape(k) for k in sorted(TP_VERBS, key=len, reverse=True))
|
|
451
|
+
+ rf")\s+(?:the\s+)?(?P<val>{ORG}(?:\s+(?:team|group|squad|org|department|division))?)")
|
|
452
|
+
PATTERNS.append(("third_person", re.compile(TP_RX, re.I),
|
|
453
|
+
lambda m, ctx, sp, ts, sent: [
|
|
454
|
+
Candidate(clean_value(m.group("subj")),
|
|
455
|
+
TP_VERBS[m.group("verb").lower()],
|
|
456
|
+
clean_value(m.group("val")), 0.85, "third_person",
|
|
457
|
+
retraction=TP_VERBS[m.group("verb").lower()] == "left")]))
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
# Mem0/Zep-style migrated summaries: "User prefers oat milk lattes.",
|
|
461
|
+
# "User knows Rust." — competitor exports state facts in exactly this
|
|
462
|
+
# third-person shape, so the migration path needs to catch them.
|
|
463
|
+
SUMMARY_VERBS = {
|
|
464
|
+
"prefers": "prefers", "likes": "likes", "knows": "knows",
|
|
465
|
+
"uses": "uses", "lives in": "lives_in", "works at": "works_at",
|
|
466
|
+
"speaks": "speaks", "owns": "owns", "enjoys": "likes",
|
|
467
|
+
"hates": "dislikes", "dislikes": "dislikes",
|
|
468
|
+
"studied": "studied", "plays": "plays",
|
|
469
|
+
}
|
|
470
|
+
_SUMMARY_RX = (rf"\b(?P<subj>User|[A-Z][a-z]{{2,}})\s+(?P<verb>"
|
|
471
|
+
+ "|".join(re.escape(k) for k in
|
|
472
|
+
sorted(SUMMARY_VERBS, key=len, reverse=True))
|
|
473
|
+
+ rf")\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)"
|
|
474
|
+
+ rf"(?=[.!?]|$|,|\s+(?:but|and|however|so)\b)")
|
|
475
|
+
PATTERNS.append(("user_summary", re.compile(_SUMMARY_RX),
|
|
476
|
+
lambda m, ctx, sp, ts, sent: [
|
|
477
|
+
Candidate(clean_value(m.group("subj")),
|
|
478
|
+
SUMMARY_VERBS[m.group("verb").lower()],
|
|
479
|
+
clean_value(m.group("val")), 0.80,
|
|
480
|
+
"user_summary")]))
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
@pattern("team_uses", rf"\bthe\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s+team\s+uses?\s+(?P<tech>[A-Za-z+#.][\w+#.]*)")
|
|
484
|
+
def _team_uses(m, ctx, sp, ts, sent):
|
|
485
|
+
team = f"{clean_value(m.group('val'))} team"
|
|
486
|
+
return [Candidate(team, "uses", clean_value(m.group("tech")), 0.88, "team_uses")]
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
@pattern("possessive", rf"\b(?P<subj>{NAME})'s\s+(?P<rel>sister|brother|manager|boss|team|birthday|role|job|dog|cat|wife|husband)\s+is\s+(?P<val>[^,.!?]{{2,40}})")
|
|
490
|
+
def _possessive(m, ctx, sp, ts, sent):
|
|
491
|
+
subj = clean_value(m.group("subj"))
|
|
492
|
+
rel = m.group("rel").lower()
|
|
493
|
+
raw = m.group("val")
|
|
494
|
+
rel_map = {"sister": "sibling", "brother": "sibling", "manager": "reports_to",
|
|
495
|
+
"boss": "reports_to", "team": "member_of", "birthday": "birthday",
|
|
496
|
+
"role": "role", "job": "role", "dog": "has_pet", "cat": "has_pet",
|
|
497
|
+
"wife": "spouse", "husband": "spouse"}
|
|
498
|
+
val = date_in(raw, ts) if rel == "birthday" and ts else clean_value(raw)
|
|
499
|
+
return [Candidate(subj, rel_map.get(rel, "mentioned"), val, 0.85, "possessive")]
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
@pattern("role_at_org",
|
|
505
|
+
rf"\bi'?m\s+(?:a|an)\s+(?P<role>[a-z][a-z /-]{{2,40}}?)\s+at\s+(?P<val>{ORG})")
|
|
506
|
+
def _role_at_org(m, ctx, sp, ts, sent):
|
|
507
|
+
out = [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.9,
|
|
508
|
+
"role_at_org")]
|
|
509
|
+
# "I'm a software engineer at Netflix" carries BOTH the org and the
|
|
510
|
+
# occupation — the role must not be silently dropped, or "what does
|
|
511
|
+
# X do for a living?" becomes unanswerable.
|
|
512
|
+
rv = clean_value(m.group("role"))
|
|
513
|
+
if rv.lower() not in ROLE_BLOCK:
|
|
514
|
+
out.append(Candidate("SELF", "role", rv, 0.85, "role_at_org"))
|
|
515
|
+
return out
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
@pattern("been_working", rf"\bi'?ve\s+been\s+(?:working\s+)?at\s+(?P<val>{ORG})")
|
|
519
|
+
def _been_working(m, ctx, sp, ts, sent):
|
|
520
|
+
# "I've been working at X since March 2024" — the since-clause dates
|
|
521
|
+
# the START of the employment interval (not the session date).
|
|
522
|
+
vf = date_in(sent, ts) if re.search(r"\bsince\b", sent, re.I) else None
|
|
523
|
+
return [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.9,
|
|
524
|
+
"been_working", valid_from=vf)]
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
@pattern("instruction", rf"\b(?:please\s+)?(?:always|never)\s+(?P<val>[a-z][a-z ]{{3,60}}?)(?=[.!?]|$)")
|
|
528
|
+
def _instruction(m, ctx, sp, ts, sent):
|
|
529
|
+
v = clean_value(m.group("val"))
|
|
530
|
+
if v.lower() in ROLE_BLOCK:
|
|
531
|
+
return []
|
|
532
|
+
return [Candidate("SELF", "instruction", m.group(0).strip().rstrip(".!").lower(),
|
|
533
|
+
0.78, "instruction")]
|
|
534
|
+
|
|
535
|
+
|
|
536
|
+
@pattern("from_now_on", rf"\bfrom now on[,.]?\s*(?:please\s+)?(?P<val>[a-z][a-z ]{{3,60}}?)(?=[.!?]|$)")
|
|
537
|
+
def _from_now_on(m, ctx, sp, ts, sent):
|
|
538
|
+
return [Candidate("SELF", "instruction", "from now on " + clean_value(m.group("val")),
|
|
539
|
+
0.78, "from_now_on")]
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
# --- assistant (second person about the user) --------------------------------
|
|
543
|
+
@pattern("you_work", rf"\byou\s+(?:mentioned\s+)?(?:work|worked|working)\s+(?:at|for)\s+(?P<val>{ORG})")
|
|
544
|
+
def _you_work(m, ctx, sp, ts, sent):
|
|
545
|
+
return [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.75, "you_work")]
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
@pattern("you_live", rf"\byou\s+(?:live|lived)\s+in\s+(?P<val>{ORG})")
|
|
549
|
+
def _you_live(m, ctx, sp, ts, sent):
|
|
550
|
+
return [Candidate("SELF", "lives_in", clean_value(m.group("val")), 0.75, "you_live")]
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
# --- generic event with date --------------------------------------------------
|
|
554
|
+
EVENT_START = re.compile(r"\b(?:i|we)\s+(?P<vp>[a-z][a-z' -]{2,60})", re.I)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
import re as _re2
|
|
558
|
+
_DATE_GATE = _re2.compile(
|
|
559
|
+
r"\d|jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec|yesterday|today|"
|
|
560
|
+
r"tomorrow|last|ago", _re2.I)
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def extract_events(sent: str, sp: tuple[int, int], ts: datetime | None,
|
|
564
|
+
ctx: ExtractionContext) -> list[Candidate]:
|
|
565
|
+
if ts is None or not _DATE_GATE.search(sent):
|
|
566
|
+
return []
|
|
567
|
+
dates = find_dates(sent, ts)
|
|
568
|
+
if not dates:
|
|
569
|
+
return []
|
|
570
|
+
m = EVENT_START.search(sent)
|
|
571
|
+
if not m:
|
|
572
|
+
return []
|
|
573
|
+
vp = m.group("vp").strip()
|
|
574
|
+
# strip a trailing date surface from the verb phrase
|
|
575
|
+
for d in dates:
|
|
576
|
+
surf = d["surface"].strip()
|
|
577
|
+
if surf and surf.lower() in vp.lower():
|
|
578
|
+
vp = re.sub(re.escape(surf), "", vp, flags=re.I).strip()
|
|
579
|
+
_MONTHS = "january|february|march|april|may|june|july|august|september|october|november|december"
|
|
580
|
+
for _ in range(3):
|
|
581
|
+
vp = re.sub(r"\s+(?:on|in|at|since|ago|back|last|this|next)\s*$", "", vp, flags=re.I).strip()
|
|
582
|
+
vp = re.sub(rf"\s+(?:{_MONTHS})\s*$", "", vp, flags=re.I).strip()
|
|
583
|
+
vp = re.sub(r"\s+\d{1,4}(?:st|nd|rd|th)?\s*$", "", vp).strip()
|
|
584
|
+
vp = re.sub(r"\s+(?:week|month|year|day)s?\s*$", "", vp, flags=re.I).strip()
|
|
585
|
+
vp = re.sub(r"\s+", " ", vp)
|
|
586
|
+
if len(vp) < 4 or vp.split()[0] in ("will", "would", "hope", "want", "plan"):
|
|
587
|
+
return []
|
|
588
|
+
# conversational fluff is not a memorable life event
|
|
589
|
+
_JUNK = re.compile(
|
|
590
|
+
r"^(?:read|reads|watch|watches|watched|see|saw|seen|hear|heard|"
|
|
591
|
+
r"think|thinks|thought|look|looked|talk|talked|remember|"
|
|
592
|
+
r"was|were|am|be|been)\b", re.I)
|
|
593
|
+
if _JUNK.match(vp):
|
|
594
|
+
return []
|
|
595
|
+
return [Candidate("SELF", "event", vp, 0.7, "event",
|
|
596
|
+
valid_from=dates[0]["iso"],
|
|
597
|
+
span=(sp[0] + m.start(), sp[0] + m.end()))]
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
# --- label-value form (BEAM-10M user_profile bullets) ---------------------
|
|
601
|
+
# BEAM-10M's user_profile uses "Name: X / Age: Y / Location: Z / Profession: P"
|
|
602
|
+
# form (label-value pairs separated by a colon). This is a different surface
|
|
603
|
+
# than the conversational "My name is X" form — the user_profile section is
|
|
604
|
+
# a structured profile, not a chat message. The patterns below extract facts
|
|
605
|
+
# from that structured form so the μ=0 extractor can score on BEAM-10M
|
|
606
|
+
# ground-truth facts.
|
|
607
|
+
#
|
|
608
|
+
# Confidence is 0.90 (vs 0.95 for "my name is X") because the profile
|
|
609
|
+
# form doesn't establish first-person attribution as strongly as a
|
|
610
|
+
# user saying "my name is X" in their own chat turn — the profile
|
|
611
|
+
# could be metadata about another user. The writer's quarantine /
|
|
612
|
+
# sandbox layer still applies.
|
|
613
|
+
@pattern("profile_name",
|
|
614
|
+
rf"^\s*•?\s*Name:\s+(?P<val>{NAME})\s*$")
|
|
615
|
+
def _profile_name(m, ctx, sp, ts, sent):
|
|
616
|
+
v = clean_value(m.group("val"))
|
|
617
|
+
target = ctx.subject_name or "SELF"
|
|
618
|
+
rel = "name" if ctx.subject_name is None else "alias"
|
|
619
|
+
return [Candidate(target, rel, v, 0.90, "profile_name")]
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
@pattern("profile_age",
|
|
623
|
+
rf"^\s*•?\s*Age:\s+(?P<val>\d+)\s*(?:years?\s*old)?\s*$")
|
|
624
|
+
def _profile_age(m, ctx, sp, ts, sent):
|
|
625
|
+
v = clean_value(m.group("val"))
|
|
626
|
+
target = ctx.subject_name or "SELF"
|
|
627
|
+
return [Candidate(target, "age", v, 0.88, "profile_age")]
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
@pattern("profile_gender",
|
|
631
|
+
rf"^\s*•?\s*Gender:\s+(?P<val>male|female|non-binary|other)\s*$")
|
|
632
|
+
def _profile_gender(m, ctx, sp, ts, sent):
|
|
633
|
+
v = clean_value(m.group("val"))
|
|
634
|
+
target = ctx.subject_name or "SELF"
|
|
635
|
+
return [Candidate(target, "gender", v, 0.88, "profile_gender")]
|
|
636
|
+
|
|
637
|
+
|
|
638
|
+
@pattern("profile_location",
|
|
639
|
+
rf"^\s*•?\s*Location:\s+(?P<val>[^\n]+?)\s*$")
|
|
640
|
+
def _profile_location(m, ctx, sp, ts, sent):
|
|
641
|
+
v = clean_value(m.group("val"))
|
|
642
|
+
target = ctx.subject_name or "SELF"
|
|
643
|
+
return [Candidate(target, "location", v, 0.88, "profile_location")]
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
@pattern("profile_profession",
|
|
647
|
+
rf"^\s*•?\s*Profession:\s+(?P<val>[^\n]+?)\s*$")
|
|
648
|
+
def _profile_profession(m, ctx, sp, ts, sent):
|
|
649
|
+
v = clean_value(m.group("val"))
|
|
650
|
+
target = ctx.subject_name or "SELF"
|
|
651
|
+
return [Candidate(target, "profession", v, 0.88, "profile_profession")]
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
# --- relationship bullets (BEAM-10M user_relationships) ------------------
|
|
655
|
+
# The user_relationships section uses bullet form like:
|
|
656
|
+
# • Alicia (female, age 80)
|
|
657
|
+
# under a header like "PARENTS & GUARDIANS:" — we match the bullet+name
|
|
658
|
+
# but the section header sets the relation. Since we don't track state
|
|
659
|
+
# across matches in the pattern layer (the ExtractionContext doesn't
|
|
660
|
+
# see the section header), we emit these as "related_to" with the name
|
|
661
|
+
# as the value. A post-process pass (in the writer) could resolve
|
|
662
|
+
# section→relation by re-reading the user_relationships text.
|
|
663
|
+
@pattern("profile_relative",
|
|
664
|
+
rf"^\s*•\s+(?P<val>{NAME})\s*\(.*?\)\s*$")
|
|
665
|
+
def _profile_relative(m, ctx, sp, ts, sent):
|
|
666
|
+
v = clean_value(m.group("val"))
|
|
667
|
+
target = ctx.subject_name or "SELF"
|
|
668
|
+
# we don't know the section here; emit as 'related_to' so the
|
|
669
|
+
# fact is stored and the relation can be re-classified later.
|
|
670
|
+
return [Candidate(target, "related_to", v, 0.75,
|
|
671
|
+
"profile_relative")]
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
# --- section-aware kinship extraction (BEAM-10M user_relationships) -----------
|
|
675
|
+
# The previous `profile_relative` pattern emitted every kinship bullet as
|
|
676
|
+
# `related_to` because it didn't know which section header the bullet was
|
|
677
|
+
# under. The fix is a multi-line pattern that captures the section header
|
|
678
|
+
# + bullet as one regex, so the relation can be derived from the header
|
|
679
|
+
# in the same pass.
|
|
680
|
+
#
|
|
681
|
+
# BEAM-10M user_relationships section looks like:
|
|
682
|
+
# PARENTS & GUARDIANS:
|
|
683
|
+
# • Alicia (female, age 80)
|
|
684
|
+
# • John (male, age 82)
|
|
685
|
+
# ROMANTIC PARTNER:
|
|
686
|
+
# • Chris (35, software engineer)
|
|
687
|
+
# CHILDREN:
|
|
688
|
+
# • Brittany (12)
|
|
689
|
+
# • Maria (8)
|
|
690
|
+
# SIBLINGS:
|
|
691
|
+
# • Sam (28, lawyer)
|
|
692
|
+
# FRIENDS:
|
|
693
|
+
# • Bob (40, teacher)
|
|
694
|
+
#
|
|
695
|
+
# Section→relation map (BEAM-10M canonical headers):
|
|
696
|
+
_KINSHIP_SECTIONS = {
|
|
697
|
+
"PARENTS & GUARDIANS": "parent",
|
|
698
|
+
"PARENTS": "parent",
|
|
699
|
+
"GUARDIANS": "parent",
|
|
700
|
+
"ROMANTIC PARTNER": "partner",
|
|
701
|
+
"ROMANTIC PARTNERS": "partner",
|
|
702
|
+
"PARTNER": "partner",
|
|
703
|
+
"PARTNERS": "partner",
|
|
704
|
+
"SPOUSE": "spouse",
|
|
705
|
+
"CHILDREN": "child",
|
|
706
|
+
"CHILD": "child",
|
|
707
|
+
"SIBLINGS": "sibling",
|
|
708
|
+
"SIBLING": "sibling",
|
|
709
|
+
"FRIENDS": "friend",
|
|
710
|
+
"FRIEND": "friend",
|
|
711
|
+
"COLLEAGUES": "colleague",
|
|
712
|
+
"COLLEAGUE": "colleague",
|
|
713
|
+
"COWORKERS": "colleague",
|
|
714
|
+
"EXTENDED FAMILY": "family",
|
|
715
|
+
"GRANDPARENTS": "parent", # grandparent is a parent role
|
|
716
|
+
"GRANDCHILDREN": "child",
|
|
717
|
+
"NEPHEWS & NIECES": "sibling", # sibling lineage
|
|
718
|
+
"IN-LAWS": "family",
|
|
719
|
+
}
|
|
720
|
+
# Build one big alternation of section headers (longest first so
|
|
721
|
+
# "PARENTS & GUARDIANS" beats "PARENTS")
|
|
722
|
+
_KINSHIP_HEADERS = sorted(_KINSHIP_SECTIONS.keys(), key=len, reverse=True)
|
|
723
|
+
_KINSHIP_HEADERS_RE = "|".join(
|
|
724
|
+
re.escape(h).replace(r"\ ", r"\s+") for h in _KINSHIP_HEADERS)
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
@pattern("profile_kinship_section",
|
|
728
|
+
rf"(?P<section>{_KINSHIP_HEADERS_RE})\s*:\s*\n"
|
|
729
|
+
rf"(?P<bullets>(?:\s*•\s+{NAME}\s*\([^)]*\)\s*\n?)+)")
|
|
730
|
+
def _profile_kinship_section(m, ctx, sp, ts, sent):
|
|
731
|
+
"""Section-aware kinship extraction.
|
|
732
|
+
|
|
733
|
+
Matches a whole section header + bullet block, emits one Candidate
|
|
734
|
+
per bullet with the section-derived relation. This replaces the
|
|
735
|
+
lossy `related_to` fallback that the single-line `profile_relative`
|
|
736
|
+
pattern was emitting.
|
|
737
|
+
"""
|
|
738
|
+
section_raw = m.group("section")
|
|
739
|
+
# normalize header to canonical form (collapse whitespace, uppercase)
|
|
740
|
+
section_key = re.sub(r"\s+", " ", section_raw).strip().upper()
|
|
741
|
+
relation = _KINSHIP_SECTIONS.get(section_key)
|
|
742
|
+
if relation is None:
|
|
743
|
+
# try matching by prefix (e.g. "PARENTS & GUARDIANS" might have
|
|
744
|
+
# spacing differences)
|
|
745
|
+
for h in _KINSHIP_HEADERS:
|
|
746
|
+
if h in section_key:
|
|
747
|
+
relation = _KINSHIP_SECTIONS[h]
|
|
748
|
+
break
|
|
749
|
+
if relation is None:
|
|
750
|
+
return [] # unknown section — leave to single-line pattern
|
|
751
|
+
bullets = m.group("bullets")
|
|
752
|
+
target = ctx.subject_name or "SELF"
|
|
753
|
+
out: list[Candidate] = []
|
|
754
|
+
# iterate bullets within the section
|
|
755
|
+
for bm in re.finditer(rf"•\s+(?P<val>{NAME})\s*\([^)]*\)", bullets):
|
|
756
|
+
v = clean_value(bm.group("val"))
|
|
757
|
+
if v and len(v) >= 2:
|
|
758
|
+
out.append(Candidate(target, relation, v, 0.85,
|
|
759
|
+
"profile_kinship_section"))
|
|
760
|
+
return out
|