cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,760 @@
1
+ """The μ=0 extraction pattern library.
2
+
3
+ High-precision syntactic patterns over Subject-Relation-Value triples:
4
+ first-person (user), third-person (entities), and assistant-turn
5
+ (second-person) forms, with temporal anchors resolved by bridge.dates.
6
+ Zero LLM calls — this is the deterministic perception layer that makes
7
+ BEAM-honest μ=0 ingest possible while competitors burn LLM extraction
8
+ at write time.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass, field
15
+ from datetime import datetime
16
+
17
+ from cortexm.bridge.dates import find_dates
18
+
19
+ # ---------------------------------------------------------------- helpers
20
+
21
+ # Case-sensitive (scoped — patterns compile with re.I for verbs, but
22
+ # entity values must keep their capitalization semantics): first word
23
+ # capitalized, subsequent words capitalized or internal to the name.
24
+ ORG = r"(?-i:[A-Z][\w&'.,-]*)(?:\s+(?-i:[A-Z])[\w&'.,-]*)*"
25
+ NAME = r"(?-i:[A-Z][a-zA-Z'-]+)(?:\s+(?-i:[A-Z])[a-zA-Z'-]+)?"
26
+ LOWERPH = r"[a-zA-Z][\w' +#.-]{1,44}"
27
+
28
+ PRONOUNS = {"she", "he", "they", "it", "her", "him", "them"}
29
+ FIRST_PRONOUNS = {"i", "we", "me", "us", "my", "our"}
30
+
31
+ ROLE_BLOCK = {
32
+ "tired", "happy", "sad", "sorry", "sure", "okay", "ok", "fine",
33
+ "glad", "ready", "here", "back", "busy", "excited", "curious",
34
+ "confused", "hungry", "free", "done", "good", "great", "well",
35
+ "not", "just", "still", "new", "all", "so", "very", "really",
36
+ }
37
+
38
+ FAMILY_MAP = {
39
+ "sister": "sibling", "brother": "sibling", "twin": "sibling",
40
+ "mother": "parent", "mom": "parent", "father": "parent", "dad": "parent",
41
+ "wife": "spouse", "husband": "spouse", "partner": "spouse",
42
+ "daughter": "child", "son": "child", "cousin": "friend",
43
+ }
44
+
45
+ IS_MY_MAP = {
46
+ "sister": "sibling", "brother": "sibling", "mother": "parent",
47
+ "mom": "parent", "father": "parent", "dad": "parent", "wife": "spouse",
48
+ "husband": "spouse", "partner": "spouse", "friend": "friend",
49
+ "colleague": "friend", "teammate": "friend", "cousin": "friend",
50
+ "manager": "reports_to", "boss": "reports_to",
51
+ }
52
+
53
+
54
+ @dataclass
55
+ class Candidate:
56
+ subject: str
57
+ relation: str
58
+ value: str
59
+ confidence: float
60
+ pattern: str
61
+ span: tuple[int, int] = (0, 0)
62
+ valid_from: str | None = None
63
+ valid_to: str | None = None
64
+ retraction: bool = False
65
+ note: str = ""
66
+ # Tier-4 fix: track whether this candidate came from a strict
67
+ # trigger match or a Bitap-widened trigger. Bitap widening is
68
+ # essential for OOD recall (catches "wrks" -> "works"), but the
69
+ # wider net admits false positives — a sentence that triggered
70
+ # on a fuzzy match should carry a slightly lower confidence so
71
+ # the writer's min_confidence threshold can filter out noisy
72
+ # extractions. Stays μ=0 — the flag is deterministic.
73
+ trigger_source: str = "strict" # "strict" | "bitap_widened"
74
+
75
+
76
+ @dataclass
77
+ class ExtractionContext:
78
+ user_id: str = "default"
79
+ agent_id: str | None = None
80
+ run_id: str | None = None
81
+ ts: datetime | None = None
82
+ speaker: str = "user" # user | assistant | system
83
+ subject_name: str | None = None # learned canonical name of the user
84
+ lexicon: set[str] = field(default_factory=set)
85
+
86
+ @property
87
+ def subject(self) -> str:
88
+ return self.subject_name or f"user:{self.user_id}"
89
+
90
+
91
+ def clean_value(v: str) -> str:
92
+ v = v.strip().strip('"\',.!?;:')
93
+ v = re.sub(r"^(?:the|a|an)\s+", "", v, flags=re.I)
94
+ v = re.sub(r"\s+", " ", v)
95
+ v = v.strip().rstrip(".,!?;:")
96
+ return v.strip()
97
+
98
+
99
+ def date_in(text: str, ts: datetime | None) -> str | None:
100
+ if ts is None:
101
+ return None
102
+ ds = find_dates(text, ts)
103
+ return ds[0]["iso"] if ds else None
104
+
105
+
106
+ # ---------------------------------------------------------------- patterns
107
+ # Each entry: (name, regex, handler(match, ctx, sent_span) -> list[Candidate])
108
+ # m.group("val") etc. Handlers return candidates with subject placeholder
109
+ # "SELF" resolved by the extractor to ctx.subject.
110
+
111
+ PATTERNS: list[tuple[str, re.Pattern, object]] = []
112
+
113
+
114
+ def pattern(name: str, rx: str):
115
+ compiled = re.compile(rx, re.I | re.M)
116
+
117
+ def deco(fn):
118
+ PATTERNS.append((name, compiled, fn))
119
+ return fn
120
+
121
+ return deco
122
+
123
+
124
+ # --- identity -------------------------------------------------------------
125
+ @pattern("name_intro", rf"\bmy name is\s+(?P<val>{NAME})")
126
+ def _name(m, ctx, sp, ts, sent):
127
+ full = clean_value(m.group("val"))
128
+ out = []
129
+ if ctx.subject_name is None or ctx.subject_name.lower() != full.lower():
130
+ out.append(Candidate("SELF", "name", full, 0.95, "name_intro", (sp[0] + m.start(), sp[0] + m.end())))
131
+ parts = full.split()
132
+ if len(parts) > 1 and len(parts[0]) > 2:
133
+ out.append(Candidate(full, "alias", parts[0], 0.9, "name_intro_alias"))
134
+ return out
135
+
136
+
137
+ @pattern("called", rf"\b(?:i am called|i'm called|call me|i go by|you can call me)\s+(?P<val>{NAME})")
138
+ def _called(m, ctx, sp, ts, sent):
139
+ v = clean_value(m.group("val"))
140
+ rel = "name" if ctx.subject_name is None else "alias"
141
+ target = ctx.subject_name or "SELF"
142
+ return [Candidate(target, rel, v, 0.9, "called")]
143
+
144
+
145
+ # --- employment ------------------------------------------------------------
146
+ WORK_AT = rf"(?P<val>{ORG})"
147
+ @pattern("works_at",
148
+ rf"\bi(?:'?m)?\s+(?:(?:now|currently|these days)\s+)?"
149
+ rf"(?:work|worked|working"
150
+ rf"|am\s+(?:(?:now|currently|these days)\s+)?working"
151
+ rf"|now\s+work)\s+(?:at|for)\s+(?:the\s+)?{WORK_AT}")
152
+ def _works(m, ctx, sp, ts, sent):
153
+ v = clean_value(m.group("val"))
154
+ vf = date_in(sent, ts) if re.search(r"\b(joined|started|got a job)\b", sent, re.I) else None
155
+ return [Candidate("SELF", "works_at", v, 0.92, "works_at", valid_from=vf)]
156
+
157
+
158
+ @pattern("joined_org",
159
+ rf"\bi\s+(?:joined|started(?:\s+working)?\s+at|got a job at|moved to a job at)\s+{WORK_AT}")
160
+ def _joined(m, ctx, sp, ts, sent):
161
+ v = clean_value(m.group("val"))
162
+ return [Candidate("SELF", "works_at", v, 0.92, "joined_org",
163
+ valid_from=date_in(sent, ts))]
164
+
165
+
166
+ @pattern("at_org", rf"\bi'?m\s+(?:now\s+|currently\s+|these days\s+)?at\s+(?P<val>{ORG})")
167
+ def _at_org(m, ctx, sp, ts, sent):
168
+ v = clean_value(m.group("val"))
169
+ return [Candidate("SELF", "works_at", v, 0.85, "at_org")]
170
+
171
+
172
+ @pattern("left_org",
173
+ rf"\bi\s+(?:left|quit|resigned from|was laid off from|departed)\s+(?P<val>{ORG})")
174
+ def _left(m, ctx, sp, ts, sent):
175
+ v = clean_value(m.group("val"))
176
+ vf = date_in(sent, ts)
177
+ return [Candidate("SELF", "left", v, 0.9, "left_org", valid_from=vf,
178
+ retraction=True)]
179
+
180
+
181
+ @pattern("no_longer", rf"\bi\s+(?:no longer|don'?t|do not)\s+work\s+(?:at|for)\s+(?P<val>{ORG})")
182
+ def _no_longer(m, ctx, sp, ts, sent):
183
+ v = clean_value(m.group("val"))
184
+ return [Candidate("SELF", "left", v, 0.88, "no_longer", retraction=True)]
185
+
186
+
187
+ @pattern("no_longer_at", rf"\bi'?m\s+no longer\s+(?:at|with)\s+(?P<val>{ORG})")
188
+ def _no_longer_at(m, ctx, sp, ts, sent):
189
+ v = clean_value(m.group("val"))
190
+ return [Candidate("SELF", "left", v, 0.88, "no_longer_at", retraction=True)]
191
+
192
+
193
+ @pattern("role", rf"\bi\s+work\s+as\s+(?:a|an|the)\s+(?P<val>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)"
194
+ rf"|\bi'?m\s+(?:a|an|the)\s+(?P<val2>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)"
195
+ rf"|\bi\s+am\s+(?:a|an|the)\s+(?P<val3>[a-zA-Z][a-zA-Z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for|and|but|where)\b|$)")
196
+ def _role(m, ctx, sp, ts, sent):
197
+ v = clean_value(m.group("val") or m.group("val2") or m.group("val3"))
198
+ if v.lower() in ROLE_BLOCK:
199
+ return []
200
+ return [Candidate("SELF", "role", v, 0.85, "role")]
201
+
202
+
203
+ @pattern("role_as", rf"\bas\s+(?:a|an)\s+(?P<val>[a-z][a-z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for)\b|$)")
204
+ def _role_as(m, ctx, sp, ts, sent):
205
+ v = clean_value(m.group("val"))
206
+ if v.lower() in ROLE_BLOCK:
207
+ return []
208
+ return [Candidate("SELF", "role", v, 0.8, "role_as")]
209
+
210
+
211
+ @pattern("role_my", rf"\bmy\s+(?:job|role|title|position)\s+is\s+(?:a|an|the)?\s*(?P<val>[a-z][a-z /-]{{2,40}}?)(?=[,.!?]|\s+(?:at|in|on|with|for)\b|$)")
212
+ def _role_my(m, ctx, sp, ts, sent):
213
+ v = clean_value(m.group("val"))
214
+ return [Candidate("SELF", "role", v, 0.9, "role_my")]
215
+
216
+
217
+ @pattern("reports_to", rf"\bmy\s+(?:manager|boss|lead|supervisor|team lead)\s+is\s+(?P<val>{NAME})")
218
+ def _mgr(m, ctx, sp, ts, sent):
219
+ return [Candidate("SELF", "reports_to", clean_value(m.group("val")), 0.9, "reports_to")]
220
+
221
+
222
+ @pattern("i_report", rf"\bi\s+report\s+to\s+(?P<val>{NAME})")
223
+ def _i_report(m, ctx, sp, ts, sent):
224
+ return [Candidate("SELF", "reports_to", clean_value(m.group("val")), 0.9, "i_report")]
225
+
226
+
227
+ @pattern("member_of", rf"\bi'?m\s+(?:now\s+)?(?:on|part of)\s+the\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s*(?:team|group|squad|org)?\b")
228
+ def _member(m, ctx, sp, ts, sent):
229
+ v = clean_value(m.group("val"))
230
+ if not v:
231
+ return []
232
+ team = v if v.lower().endswith(("team", "group", "squad")) else f"{v} team"
233
+ return [Candidate("SELF", "member_of", team, 0.88, "member_of")]
234
+
235
+
236
+ @pattern("joined_team", rf"\bi\s+joined\s+the\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s+(?:team|group|squad)\b")
237
+ def _joined_team(m, ctx, sp, ts, sent):
238
+ v = clean_value(m.group("val"))
239
+ return [Candidate("SELF", "member_of", f"{v} team", 0.9, "joined_team",
240
+ valid_from=date_in(sent, ts))]
241
+
242
+
243
+ # --- residence -------------------------------------------------------------
244
+ @pattern("lives_in", rf"\bi\s+(?:live|lived|'m living|am living|'m based|am based|reside)\s+in\s+(?P<val>{ORG})")
245
+ def _lives(m, ctx, sp, ts, sent):
246
+ return [Candidate("SELF", "lives_in", clean_value(m.group("val")), 0.92, "lives_in")]
247
+
248
+
249
+ @pattern("im_in", rf"\bi'?m\s+(?:currently\s+|now\s+)?in\s+(?P<val>{ORG})")
250
+ def _im_in(m, ctx, sp, ts, sent):
251
+ v = clean_value(m.group("val"))
252
+ if v.lower() in ("fact", "love", "trouble", "debt", "charge", "love with"):
253
+ return []
254
+ return [Candidate("SELF", "lives_in", v, 0.7, "im_in")]
255
+
256
+
257
+ @pattern("moved_to", rf"\b(?:i|we)\s+(?:moved|relocated)\s+to\s+(?P<val>{ORG})")
258
+ def _moved(m, ctx, sp, ts, sent):
259
+ v = clean_value(m.group("val"))
260
+ return [Candidate("SELF", "moved_to", v, 0.92, "moved_to",
261
+ valid_from=date_in(sent, ts))]
262
+
263
+
264
+ # --- preferences -----------------------------------------------------------
265
+ LIKE_TAIL = r"(?=[,.!?]|\s+(?:but|though|when|because|and|so|especially)\b|$)"
266
+ @pattern("likes", rf"\bi\s+(?:(?:really|absolutely|totally)\s+)?(?:love|loved|like|liked|enjoy|enjoyed)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?){LIKE_TAIL}")
267
+ def _likes(m, ctx, sp, ts, sent):
268
+ v = clean_value(m.group("val"))
269
+ if v.lower() in ("it", "that", "this", "them", "you"):
270
+ return []
271
+ return [Candidate("SELF", "likes", v, 0.85, "likes")]
272
+
273
+
274
+ @pattern("fan_of", rf"\bi'?m\s+(?:a|an)\s+(?:big\s+)?fan\s+of\s+(?P<val>{LOWERPH})")
275
+ def _fan(m, ctx, sp, ts, sent):
276
+ return [Candidate("SELF", "likes", clean_value(m.group("val")), 0.85, "fan_of")]
277
+
278
+
279
+ @pattern("favorite", rf"\bmy favorite\s+[\w ]{{2,24}}\s+is\s+(?P<val>{LOWERPH})")
280
+ def _favorite(m, ctx, sp, ts, sent):
281
+ return [Candidate("SELF", "likes", clean_value(m.group("val")), 0.88, "favorite")]
282
+
283
+
284
+ @pattern("dislikes", rf"\bi\s+(?:hate|hated|dislike|can'?t stand|don'?t like|do not like)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?){LIKE_TAIL}")
285
+ def _dislikes(m, ctx, sp, ts, sent):
286
+ return [Candidate("SELF", "dislikes", clean_value(m.group("val")), 0.85, "dislikes")]
287
+
288
+
289
+ @pattern("prefers", rf"\bi\s+(?:'d\s+)?(?:prefer|preferre?d)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?:\s+(?:over|to|than)\s+[\w' -]+)?{LIKE_TAIL}")
290
+ def _prefers(m, ctx, sp, ts, sent):
291
+ v = clean_value(m.group("val"))
292
+ cat = re.search(r"\bfor\s+([a-z][a-z ]{2,20})", sent[m.start():])
293
+ if cat:
294
+ v = f"{v} (for {clean_value(cat.group(1))})"
295
+ return [Candidate("SELF", "prefers", v, 0.9, "prefers")]
296
+
297
+
298
+ @pattern("pref_change", rf"\b(?:actually|these days|nowadays|lately|recently)[,.]?\s*(?:i\s+)?(?:prefer|like|love|drink|use|order)\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?=[,.!?]|$|\s+(?:but|though|and)\b)")
299
+ def _pref_change(m, ctx, sp, ts, sent):
300
+ return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.88, "pref_change")]
301
+
302
+
303
+ @pattern("switched_to", rf"\b(?:i'?ve|i have)\s+(?:switched|moved)\s+to\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)(?=[,.!?]|$|\s+(?:but|though|and)\b)")
304
+ def _switched(m, ctx, sp, ts, sent):
305
+ return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.88, "switched_to")]
306
+
307
+
308
+ @pattern("more_of_a", rf"\bi'?m\s+more\s+of\s+a\s+(?P<val>{LOWERPH})\s+(?:person|guy|girl|fan|drinker|person\s+now)")
309
+ def _more_of(m, ctx, sp, ts, sent):
310
+ return [Candidate("SELF", "prefers", clean_value(m.group("val")), 0.82, "more_of_a")]
311
+
312
+
313
+ # --- skills & education ----------------------------------------------------
314
+ @pattern("skill_know", rf"\bi\s+(?:know|code in|write|program in|build with)\s+(?P<val>[A-Za-z+#.][\w+#.]*(?:\s+(?:and|&)\s+[A-Za-z+#.][\w+#.]*)*)")
315
+ def _skill(m, ctx, sp, ts, sent):
316
+ out = []
317
+ for part in re.split(r"\s+(?:and|&)\s+", clean_value(m.group("val"))):
318
+ if len(part) > 1 and part.lower() not in ("it", "that", "this", "them"):
319
+ out.append(Candidate("SELF", "has_skill", part, 0.85, "skill_know"))
320
+ return out
321
+
322
+
323
+ @pattern("skill_learning", rf"\bi(?:'ve|\s+have)\s+been\s+learning\s+(?P<val>[A-Za-z+#.][\w+#.]*)|\bi'?m\s+learning\s+(?P<val2>[A-Za-z+#.][\w+#.]*)")
324
+ def _skill_learn(m, ctx, sp, ts, sent):
325
+ v = clean_value(m.group("val") or m.group("val2") or "")
326
+ return [Candidate("SELF", "has_skill", v, 0.8, "skill_learning")] if v else []
327
+
328
+
329
+ @pattern("skill_prof", rf"\bi'?m\s+(?:proficient|skilled|experienced)\s+in\s+(?P<val>[\w+#. ]{{2,40}})")
330
+ def _skill_prof(m, ctx, sp, ts, sent):
331
+ return [Candidate("SELF", "has_skill", clean_value(m.group("val")), 0.85, "skill_prof")]
332
+
333
+
334
+ @pattern("studied_at", rf"\bi\s+studied\s+(?P<major>[a-z][a-z ]{{2,40}}?)\s+at\s+(?P<val>{ORG})")
335
+ def _studied(m, ctx, sp, ts, sent):
336
+ major = clean_value(m.group("major"))
337
+ org = clean_value(m.group("val"))
338
+ return [Candidate("SELF", "studied", f"{major} at {org}", 0.88, "studied_at")]
339
+
340
+
341
+ @pattern("majored", rf"\bi\s+majored\s+in\s+(?P<val>[a-z][a-z ]{{2,40}})|\bmy degree is in\s+(?P<val2>[a-z][a-z ]{{2,40}})")
342
+ def _majored(m, ctx, sp, ts, sent):
343
+ v = clean_value(m.group("val") or m.group("val2") or "")
344
+ return [Candidate("SELF", "studied", v, 0.85, "majored")] if v else []
345
+
346
+
347
+ @pattern("speaks", rf"\bi\s+speak\s+(?P<val>[A-Za-z]+(?:\s+and\s+[A-Za-z]+)*)")
348
+ def _speaks(m, ctx, sp, ts, sent):
349
+ out = []
350
+ for part in re.split(r"\s+and\s+", clean_value(m.group("val"))):
351
+ out.append(Candidate("SELF", "speaks", part, 0.85, "speaks"))
352
+ return out
353
+
354
+
355
+ # --- personal ---------------------------------------------------------------
356
+ @pattern("birthday", rf"\bmy birthday is\s+(?P<val>[^,.!?]{{3,30}})|\bi was born on\s+(?P<val2>[^,.!?]{{3,30}})")
357
+ def _birthday(m, ctx, sp, ts, sent):
358
+ raw = m.group("val") or m.group("val2") or ""
359
+ v = clean_value(raw)
360
+ return [Candidate("SELF", "birthday", v, 0.9, "birthday")]
361
+
362
+
363
+ @pattern("age", r"\bi'?m\s+(\d{1,2})\s+years?\s+old")
364
+ def _age(m, ctx, sp, ts, sent):
365
+ return [Candidate("SELF", "age", m.group(1), 0.9, "age")]
366
+
367
+
368
+ @pattern("family", rf"\bmy\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|daughter|son|cousin|twin)(?:'s name)?\s+(?:is|is called)\s+(?P<val>{NAME})")
369
+ def _family(m, ctx, sp, ts, sent):
370
+ rel = FAMILY_MAP.get(m.group("rel").lower(), "friend")
371
+ return [Candidate("SELF", rel, clean_value(m.group("val")), 0.9, "family")]
372
+
373
+
374
+ @pattern("family2", rf"\bmy\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|daughter|son|cousin|twin)\s+(?P<val>{NAME})\b")
375
+ def _family2(m, ctx, sp, ts, sent):
376
+ rel = FAMILY_MAP.get(m.group("rel").lower(), "friend")
377
+ return [Candidate("SELF", rel, clean_value(m.group("val")), 0.88, "family2")]
378
+
379
+
380
+ @pattern("is_my", rf"\b(?P<val>{NAME})\s+is\s+my\s+(?P<rel>sister|brother|mother|mom|father|dad|wife|husband|partner|friend|colleague|teammate|cousin|manager|boss)")
381
+ def _is_my(m, ctx, sp, ts, sent):
382
+ rel = IS_MY_MAP.get(m.group("rel").lower(), "friend")
383
+ return [Candidate("SELF", rel, clean_value(m.group("val")), 0.88, "is_my")]
384
+
385
+
386
+ @pattern("pet", rf"\bmy\s+(?P<kind>dog|cat|bird|rabbit)\s+is\s+(?:named\s+|called\s+)?(?P<val>{NAME})")
387
+ def _pet(m, ctx, sp, ts, sent):
388
+ return [Candidate("SELF", "has_pet", f"{m.group('kind')} named {clean_value(m.group('val'))}",
389
+ 0.88, "pet")]
390
+
391
+
392
+ @pattern("hobby", rf"\bmy hobby is\s+(?P<val>[a-z][a-z ]{{2,40}})|\bin my free time\s+i\s+(?P<val2>[a-z][a-z ]{{2,40}})")
393
+ def _hobby(m, ctx, sp, ts, sent):
394
+ v = clean_value(m.group("val") or m.group("val2") or "")
395
+ return [Candidate("SELF", "hobby", v, 0.8, "hobby")] if v else []
396
+
397
+
398
+ @pattern("goal", rf"\bmy goal is to\s+(?P<val>[a-z][a-z ]{{2,50}})|\bi'?m planning to\s+(?P<val2>[a-z][a-z ]{{2,50}})")
399
+ def _goal(m, ctx, sp, ts, sent):
400
+ v = clean_value(m.group("val") or m.group("val2") or "")
401
+ return [Candidate("SELF", "goal", v, 0.75, "goal")] if v else []
402
+
403
+
404
+ # --- projects & events -------------------------------------------------------
405
+ @pattern("works_on", rf"\bi'?m\s+(?:currently\s+)?working\s+on\s+(?P<val>{ORG})|\bi\s+work\s+on\s+(?P<val2>{ORG})")
406
+ def _works_on(m, ctx, sp, ts, sent):
407
+ v = clean_value(m.group("val") or m.group("val2") or "")
408
+ return [Candidate("SELF", "works_on", v, 0.88, "works_on")] if v else []
409
+
410
+
411
+ @pattern("building", rf"\b(?:we'?re|we are|i'?m|i am)\s+(?:currently\s+)?building\s+(?P<val>{ORG})")
412
+ def _building(m, ctx, sp, ts, sent):
413
+ return [Candidate("SELF", "works_on", clean_value(m.group("val")), 0.85, "building")]
414
+
415
+
416
+ @pattern("completed", rf"\b(?:we|i)\s+(?:shipped|launched|finished|completed|released|deployed)\s+(?:the\s+)?(?P<val>{ORG})")
417
+ def _completed(m, ctx, sp, ts, sent):
418
+ v = clean_value(m.group("val"))
419
+ return [Candidate("SELF", "completed", v, 0.88, "completed",
420
+ valid_to=date_in(sent, ts))]
421
+
422
+
423
+ @pattern("used_to_work", rf"\bi used to\s+(?:work|working)\s+(?:at|for)\s+(?P<val>{ORG})")
424
+ def _used_work(m, ctx, sp, ts, sent):
425
+ v = clean_value(m.group("val"))
426
+ return [Candidate("SELF", "works_at", v, 0.7, "used_to_work",
427
+ valid_to=date_in(sent, ts) or None)]
428
+
429
+
430
+ @pattern("used_to_live", rf"\bi used to\s+live\s+in\s+(?P<val>{ORG})")
431
+ def _used_live(m, ctx, sp, ts, sent):
432
+ v = clean_value(m.group("val"))
433
+ return [Candidate("SELF", "lives_in", v, 0.7, "used_to_live",
434
+ valid_to=date_in(sent, ts) or None)]
435
+
436
+
437
+ # --- third person -------------------------------------------------------------
438
+ TP_VERBS = {
439
+ "works at": "works_at", "work at": "works_at", "worked at": "works_at",
440
+ "joined": "works_at", "left": "left", "quit": "left",
441
+ "lives in": "lives_in", "live in": "lives_in", "moved to": "moved_to",
442
+ "manages": "manages", "manage": "manages", "managing": "manages",
443
+ "uses": "uses", "use": "uses", "using": "uses",
444
+ "prefers": "prefers", "prefer": "prefers",
445
+ "likes": "likes", "like": "likes",
446
+ "studied at": "studied_at", "reports to": "reports_to",
447
+ "leads": "manages", "lead": "manages",
448
+ }
449
+ TP_RX = (rf"\b(?P<subj>{NAME})\s+(?P<verb>"
450
+ + "|".join(re.escape(k) for k in sorted(TP_VERBS, key=len, reverse=True))
451
+ + rf")\s+(?:the\s+)?(?P<val>{ORG}(?:\s+(?:team|group|squad|org|department|division))?)")
452
+ PATTERNS.append(("third_person", re.compile(TP_RX, re.I),
453
+ lambda m, ctx, sp, ts, sent: [
454
+ Candidate(clean_value(m.group("subj")),
455
+ TP_VERBS[m.group("verb").lower()],
456
+ clean_value(m.group("val")), 0.85, "third_person",
457
+ retraction=TP_VERBS[m.group("verb").lower()] == "left")]))
458
+
459
+
460
+ # Mem0/Zep-style migrated summaries: "User prefers oat milk lattes.",
461
+ # "User knows Rust." — competitor exports state facts in exactly this
462
+ # third-person shape, so the migration path needs to catch them.
463
+ SUMMARY_VERBS = {
464
+ "prefers": "prefers", "likes": "likes", "knows": "knows",
465
+ "uses": "uses", "lives in": "lives_in", "works at": "works_at",
466
+ "speaks": "speaks", "owns": "owns", "enjoys": "likes",
467
+ "hates": "dislikes", "dislikes": "dislikes",
468
+ "studied": "studied", "plays": "plays",
469
+ }
470
+ _SUMMARY_RX = (rf"\b(?P<subj>User|[A-Z][a-z]{{2,}})\s+(?P<verb>"
471
+ + "|".join(re.escape(k) for k in
472
+ sorted(SUMMARY_VERBS, key=len, reverse=True))
473
+ + rf")\s+(?P<val>[a-zA-Z][\w' +#.-]{{1,44}}?)"
474
+ + rf"(?=[.!?]|$|,|\s+(?:but|and|however|so)\b)")
475
+ PATTERNS.append(("user_summary", re.compile(_SUMMARY_RX),
476
+ lambda m, ctx, sp, ts, sent: [
477
+ Candidate(clean_value(m.group("subj")),
478
+ SUMMARY_VERBS[m.group("verb").lower()],
479
+ clean_value(m.group("val")), 0.80,
480
+ "user_summary")]))
481
+
482
+
483
+ @pattern("team_uses", rf"\bthe\s+(?P<val>[A-Z][\w-]*(?:\s+[\w-]+)*?)\s+team\s+uses?\s+(?P<tech>[A-Za-z+#.][\w+#.]*)")
484
+ def _team_uses(m, ctx, sp, ts, sent):
485
+ team = f"{clean_value(m.group('val'))} team"
486
+ return [Candidate(team, "uses", clean_value(m.group("tech")), 0.88, "team_uses")]
487
+
488
+
489
+ @pattern("possessive", rf"\b(?P<subj>{NAME})'s\s+(?P<rel>sister|brother|manager|boss|team|birthday|role|job|dog|cat|wife|husband)\s+is\s+(?P<val>[^,.!?]{{2,40}})")
490
+ def _possessive(m, ctx, sp, ts, sent):
491
+ subj = clean_value(m.group("subj"))
492
+ rel = m.group("rel").lower()
493
+ raw = m.group("val")
494
+ rel_map = {"sister": "sibling", "brother": "sibling", "manager": "reports_to",
495
+ "boss": "reports_to", "team": "member_of", "birthday": "birthday",
496
+ "role": "role", "job": "role", "dog": "has_pet", "cat": "has_pet",
497
+ "wife": "spouse", "husband": "spouse"}
498
+ val = date_in(raw, ts) if rel == "birthday" and ts else clean_value(raw)
499
+ return [Candidate(subj, rel_map.get(rel, "mentioned"), val, 0.85, "possessive")]
500
+
501
+
502
+
503
+
504
+ @pattern("role_at_org",
505
+ rf"\bi'?m\s+(?:a|an)\s+(?P<role>[a-z][a-z /-]{{2,40}}?)\s+at\s+(?P<val>{ORG})")
506
+ def _role_at_org(m, ctx, sp, ts, sent):
507
+ out = [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.9,
508
+ "role_at_org")]
509
+ # "I'm a software engineer at Netflix" carries BOTH the org and the
510
+ # occupation — the role must not be silently dropped, or "what does
511
+ # X do for a living?" becomes unanswerable.
512
+ rv = clean_value(m.group("role"))
513
+ if rv.lower() not in ROLE_BLOCK:
514
+ out.append(Candidate("SELF", "role", rv, 0.85, "role_at_org"))
515
+ return out
516
+
517
+
518
+ @pattern("been_working", rf"\bi'?ve\s+been\s+(?:working\s+)?at\s+(?P<val>{ORG})")
519
+ def _been_working(m, ctx, sp, ts, sent):
520
+ # "I've been working at X since March 2024" — the since-clause dates
521
+ # the START of the employment interval (not the session date).
522
+ vf = date_in(sent, ts) if re.search(r"\bsince\b", sent, re.I) else None
523
+ return [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.9,
524
+ "been_working", valid_from=vf)]
525
+
526
+
527
+ @pattern("instruction", rf"\b(?:please\s+)?(?:always|never)\s+(?P<val>[a-z][a-z ]{{3,60}}?)(?=[.!?]|$)")
528
+ def _instruction(m, ctx, sp, ts, sent):
529
+ v = clean_value(m.group("val"))
530
+ if v.lower() in ROLE_BLOCK:
531
+ return []
532
+ return [Candidate("SELF", "instruction", m.group(0).strip().rstrip(".!").lower(),
533
+ 0.78, "instruction")]
534
+
535
+
536
+ @pattern("from_now_on", rf"\bfrom now on[,.]?\s*(?:please\s+)?(?P<val>[a-z][a-z ]{{3,60}}?)(?=[.!?]|$)")
537
+ def _from_now_on(m, ctx, sp, ts, sent):
538
+ return [Candidate("SELF", "instruction", "from now on " + clean_value(m.group("val")),
539
+ 0.78, "from_now_on")]
540
+
541
+
542
+ # --- assistant (second person about the user) --------------------------------
543
+ @pattern("you_work", rf"\byou\s+(?:mentioned\s+)?(?:work|worked|working)\s+(?:at|for)\s+(?P<val>{ORG})")
544
+ def _you_work(m, ctx, sp, ts, sent):
545
+ return [Candidate("SELF", "works_at", clean_value(m.group("val")), 0.75, "you_work")]
546
+
547
+
548
+ @pattern("you_live", rf"\byou\s+(?:live|lived)\s+in\s+(?P<val>{ORG})")
549
+ def _you_live(m, ctx, sp, ts, sent):
550
+ return [Candidate("SELF", "lives_in", clean_value(m.group("val")), 0.75, "you_live")]
551
+
552
+
553
+ # --- generic event with date --------------------------------------------------
554
+ EVENT_START = re.compile(r"\b(?:i|we)\s+(?P<vp>[a-z][a-z' -]{2,60})", re.I)
555
+
556
+
557
+ import re as _re2
558
+ _DATE_GATE = _re2.compile(
559
+ r"\d|jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec|yesterday|today|"
560
+ r"tomorrow|last|ago", _re2.I)
561
+
562
+
563
+ def extract_events(sent: str, sp: tuple[int, int], ts: datetime | None,
564
+ ctx: ExtractionContext) -> list[Candidate]:
565
+ if ts is None or not _DATE_GATE.search(sent):
566
+ return []
567
+ dates = find_dates(sent, ts)
568
+ if not dates:
569
+ return []
570
+ m = EVENT_START.search(sent)
571
+ if not m:
572
+ return []
573
+ vp = m.group("vp").strip()
574
+ # strip a trailing date surface from the verb phrase
575
+ for d in dates:
576
+ surf = d["surface"].strip()
577
+ if surf and surf.lower() in vp.lower():
578
+ vp = re.sub(re.escape(surf), "", vp, flags=re.I).strip()
579
+ _MONTHS = "january|february|march|april|may|june|july|august|september|october|november|december"
580
+ for _ in range(3):
581
+ vp = re.sub(r"\s+(?:on|in|at|since|ago|back|last|this|next)\s*$", "", vp, flags=re.I).strip()
582
+ vp = re.sub(rf"\s+(?:{_MONTHS})\s*$", "", vp, flags=re.I).strip()
583
+ vp = re.sub(r"\s+\d{1,4}(?:st|nd|rd|th)?\s*$", "", vp).strip()
584
+ vp = re.sub(r"\s+(?:week|month|year|day)s?\s*$", "", vp, flags=re.I).strip()
585
+ vp = re.sub(r"\s+", " ", vp)
586
+ if len(vp) < 4 or vp.split()[0] in ("will", "would", "hope", "want", "plan"):
587
+ return []
588
+ # conversational fluff is not a memorable life event
589
+ _JUNK = re.compile(
590
+ r"^(?:read|reads|watch|watches|watched|see|saw|seen|hear|heard|"
591
+ r"think|thinks|thought|look|looked|talk|talked|remember|"
592
+ r"was|were|am|be|been)\b", re.I)
593
+ if _JUNK.match(vp):
594
+ return []
595
+ return [Candidate("SELF", "event", vp, 0.7, "event",
596
+ valid_from=dates[0]["iso"],
597
+ span=(sp[0] + m.start(), sp[0] + m.end()))]
598
+
599
+
600
+ # --- label-value form (BEAM-10M user_profile bullets) ---------------------
601
+ # BEAM-10M's user_profile uses "Name: X / Age: Y / Location: Z / Profession: P"
602
+ # form (label-value pairs separated by a colon). This is a different surface
603
+ # than the conversational "My name is X" form — the user_profile section is
604
+ # a structured profile, not a chat message. The patterns below extract facts
605
+ # from that structured form so the μ=0 extractor can score on BEAM-10M
606
+ # ground-truth facts.
607
+ #
608
+ # Confidence is 0.90 (vs 0.95 for "my name is X") because the profile
609
+ # form doesn't establish first-person attribution as strongly as a
610
+ # user saying "my name is X" in their own chat turn — the profile
611
+ # could be metadata about another user. The writer's quarantine /
612
+ # sandbox layer still applies.
613
+ @pattern("profile_name",
614
+ rf"^\s*•?\s*Name:\s+(?P<val>{NAME})\s*$")
615
+ def _profile_name(m, ctx, sp, ts, sent):
616
+ v = clean_value(m.group("val"))
617
+ target = ctx.subject_name or "SELF"
618
+ rel = "name" if ctx.subject_name is None else "alias"
619
+ return [Candidate(target, rel, v, 0.90, "profile_name")]
620
+
621
+
622
+ @pattern("profile_age",
623
+ rf"^\s*•?\s*Age:\s+(?P<val>\d+)\s*(?:years?\s*old)?\s*$")
624
+ def _profile_age(m, ctx, sp, ts, sent):
625
+ v = clean_value(m.group("val"))
626
+ target = ctx.subject_name or "SELF"
627
+ return [Candidate(target, "age", v, 0.88, "profile_age")]
628
+
629
+
630
+ @pattern("profile_gender",
631
+ rf"^\s*•?\s*Gender:\s+(?P<val>male|female|non-binary|other)\s*$")
632
+ def _profile_gender(m, ctx, sp, ts, sent):
633
+ v = clean_value(m.group("val"))
634
+ target = ctx.subject_name or "SELF"
635
+ return [Candidate(target, "gender", v, 0.88, "profile_gender")]
636
+
637
+
638
+ @pattern("profile_location",
639
+ rf"^\s*•?\s*Location:\s+(?P<val>[^\n]+?)\s*$")
640
+ def _profile_location(m, ctx, sp, ts, sent):
641
+ v = clean_value(m.group("val"))
642
+ target = ctx.subject_name or "SELF"
643
+ return [Candidate(target, "location", v, 0.88, "profile_location")]
644
+
645
+
646
+ @pattern("profile_profession",
647
+ rf"^\s*•?\s*Profession:\s+(?P<val>[^\n]+?)\s*$")
648
+ def _profile_profession(m, ctx, sp, ts, sent):
649
+ v = clean_value(m.group("val"))
650
+ target = ctx.subject_name or "SELF"
651
+ return [Candidate(target, "profession", v, 0.88, "profile_profession")]
652
+
653
+
654
+ # --- relationship bullets (BEAM-10M user_relationships) ------------------
655
+ # The user_relationships section uses bullet form like:
656
+ # • Alicia (female, age 80)
657
+ # under a header like "PARENTS & GUARDIANS:" — we match the bullet+name
658
+ # but the section header sets the relation. Since we don't track state
659
+ # across matches in the pattern layer (the ExtractionContext doesn't
660
+ # see the section header), we emit these as "related_to" with the name
661
+ # as the value. A post-process pass (in the writer) could resolve
662
+ # section→relation by re-reading the user_relationships text.
663
+ @pattern("profile_relative",
664
+ rf"^\s*•\s+(?P<val>{NAME})\s*\(.*?\)\s*$")
665
+ def _profile_relative(m, ctx, sp, ts, sent):
666
+ v = clean_value(m.group("val"))
667
+ target = ctx.subject_name or "SELF"
668
+ # we don't know the section here; emit as 'related_to' so the
669
+ # fact is stored and the relation can be re-classified later.
670
+ return [Candidate(target, "related_to", v, 0.75,
671
+ "profile_relative")]
672
+
673
+
674
+ # --- section-aware kinship extraction (BEAM-10M user_relationships) -----------
675
+ # The previous `profile_relative` pattern emitted every kinship bullet as
676
+ # `related_to` because it didn't know which section header the bullet was
677
+ # under. The fix is a multi-line pattern that captures the section header
678
+ # + bullet as one regex, so the relation can be derived from the header
679
+ # in the same pass.
680
+ #
681
+ # BEAM-10M user_relationships section looks like:
682
+ # PARENTS & GUARDIANS:
683
+ # • Alicia (female, age 80)
684
+ # • John (male, age 82)
685
+ # ROMANTIC PARTNER:
686
+ # • Chris (35, software engineer)
687
+ # CHILDREN:
688
+ # • Brittany (12)
689
+ # • Maria (8)
690
+ # SIBLINGS:
691
+ # • Sam (28, lawyer)
692
+ # FRIENDS:
693
+ # • Bob (40, teacher)
694
+ #
695
+ # Section→relation map (BEAM-10M canonical headers):
696
+ _KINSHIP_SECTIONS = {
697
+ "PARENTS & GUARDIANS": "parent",
698
+ "PARENTS": "parent",
699
+ "GUARDIANS": "parent",
700
+ "ROMANTIC PARTNER": "partner",
701
+ "ROMANTIC PARTNERS": "partner",
702
+ "PARTNER": "partner",
703
+ "PARTNERS": "partner",
704
+ "SPOUSE": "spouse",
705
+ "CHILDREN": "child",
706
+ "CHILD": "child",
707
+ "SIBLINGS": "sibling",
708
+ "SIBLING": "sibling",
709
+ "FRIENDS": "friend",
710
+ "FRIEND": "friend",
711
+ "COLLEAGUES": "colleague",
712
+ "COLLEAGUE": "colleague",
713
+ "COWORKERS": "colleague",
714
+ "EXTENDED FAMILY": "family",
715
+ "GRANDPARENTS": "parent", # grandparent is a parent role
716
+ "GRANDCHILDREN": "child",
717
+ "NEPHEWS & NIECES": "sibling", # sibling lineage
718
+ "IN-LAWS": "family",
719
+ }
720
+ # Build one big alternation of section headers (longest first so
721
+ # "PARENTS & GUARDIANS" beats "PARENTS")
722
+ _KINSHIP_HEADERS = sorted(_KINSHIP_SECTIONS.keys(), key=len, reverse=True)
723
+ _KINSHIP_HEADERS_RE = "|".join(
724
+ re.escape(h).replace(r"\ ", r"\s+") for h in _KINSHIP_HEADERS)
725
+
726
+
727
+ @pattern("profile_kinship_section",
728
+ rf"(?P<section>{_KINSHIP_HEADERS_RE})\s*:\s*\n"
729
+ rf"(?P<bullets>(?:\s*•\s+{NAME}\s*\([^)]*\)\s*\n?)+)")
730
+ def _profile_kinship_section(m, ctx, sp, ts, sent):
731
+ """Section-aware kinship extraction.
732
+
733
+ Matches a whole section header + bullet block, emits one Candidate
734
+ per bullet with the section-derived relation. This replaces the
735
+ lossy `related_to` fallback that the single-line `profile_relative`
736
+ pattern was emitting.
737
+ """
738
+ section_raw = m.group("section")
739
+ # normalize header to canonical form (collapse whitespace, uppercase)
740
+ section_key = re.sub(r"\s+", " ", section_raw).strip().upper()
741
+ relation = _KINSHIP_SECTIONS.get(section_key)
742
+ if relation is None:
743
+ # try matching by prefix (e.g. "PARENTS & GUARDIANS" might have
744
+ # spacing differences)
745
+ for h in _KINSHIP_HEADERS:
746
+ if h in section_key:
747
+ relation = _KINSHIP_SECTIONS[h]
748
+ break
749
+ if relation is None:
750
+ return [] # unknown section — leave to single-line pattern
751
+ bullets = m.group("bullets")
752
+ target = ctx.subject_name or "SELF"
753
+ out: list[Candidate] = []
754
+ # iterate bullets within the section
755
+ for bm in re.finditer(rf"•\s+(?P<val>{NAME})\s*\([^)]*\)", bullets):
756
+ v = clean_value(bm.group("val"))
757
+ if v and len(v) >= 2:
758
+ out.append(Candidate(target, relation, v, 0.85,
759
+ "profile_kinship_section"))
760
+ return out