perceptkit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- perceptkit/__init__.py +85 -0
- perceptkit/algorithms/__init__.py +40 -0
- perceptkit/algorithms/attribution.py +147 -0
- perceptkit/algorithms/glance.py +236 -0
- perceptkit/algorithms/history.py +663 -0
- perceptkit/algorithms/identity.py +43 -0
- perceptkit/algorithms/observation.py +44 -0
- perceptkit/algorithms/streaks.py +111 -0
- perceptkit/algorithms/trend_models.py +184 -0
- perceptkit/algorithms/wake.py +149 -0
- perceptkit/catalog.py +252 -0
- perceptkit/conformance/__init__.py +28 -0
- perceptkit/conformance/memory.py +364 -0
- perceptkit/conformance/report.py +170 -0
- perceptkit/conformance/suite.py +419 -0
- perceptkit/conformance/wake.py +151 -0
- perceptkit/contracts/__init__.py +97 -0
- perceptkit/contracts/_time.py +89 -0
- perceptkit/contracts/availability.py +77 -0
- perceptkit/contracts/context.py +50 -0
- perceptkit/contracts/delivery.py +167 -0
- perceptkit/contracts/errors.py +22 -0
- perceptkit/contracts/event.py +137 -0
- perceptkit/contracts/observation.py +172 -0
- perceptkit/contracts/receipt.py +129 -0
- perceptkit/contracts/records.py +367 -0
- perceptkit/contracts/report.py +127 -0
- perceptkit/contracts/versioning.py +63 -0
- perceptkit/fields.py +184 -0
- perceptkit/kit.py +223 -0
- perceptkit/manifest/__init__.py +57 -0
- perceptkit/manifest/checks.py +323 -0
- perceptkit/manifest/mapping.py +96 -0
- perceptkit/manifest/minimal.py +1282 -0
- perceptkit/manifest/types.py +211 -0
- perceptkit/manifest/units.py +84 -0
- perceptkit/ports/__init__.py +19 -0
- perceptkit/ports/storage.py +288 -0
- perceptkit/ports/wake.py +43 -0
- perceptkit/processing/__init__.py +49 -0
- perceptkit/processing/aggregate.py +80 -0
- perceptkit/processing/dispatch.py +356 -0
- perceptkit/processing/normalize.py +458 -0
- perceptkit/processing/pipeline.py +406 -0
- perceptkit/processing/recompute.py +170 -0
- perceptkit/processing/recurrence.py +166 -0
- perceptkit/processing/scheduled.py +233 -0
- perceptkit/prompts.py +75 -0
- perceptkit/queries/__init__.py +32 -0
- perceptkit/queries/api.py +457 -0
- perceptkit/retention.py +84 -0
- perceptkit/rules/__init__.py +19 -0
- perceptkit/rules/engine.py +112 -0
- perceptkit/rules/evaluators.py +228 -0
- perceptkit/rules/types.py +236 -0
- perceptkit-0.2.2.dist-info/METADATA +439 -0
- perceptkit-0.2.2.dist-info/RECORD +59 -0
- perceptkit-0.2.2.dist-info/WHEEL +4 -0
- perceptkit-0.2.2.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,457 @@
|
|
|
1
|
+
"""读取侧 —— agent 主动来查。
|
|
2
|
+
|
|
3
|
+
写入侧(``processing``)和读取侧(这里)**方向相反、共用同一份存储**:
|
|
4
|
+
|
|
5
|
+
设备上报 ──▶ 标准化 ──▶ 存 ──▶ 事件 ──▶ 戳醒 agent
|
|
6
|
+
▲
|
|
7
|
+
│
|
|
8
|
+
agent 想知道什么 ──▶ 这八个函数
|
|
9
|
+
|
|
10
|
+
**这八个函数不是转发器。** 直接查数据库的宿主会漏掉四件每个宿主都必须一样的事:
|
|
11
|
+
|
|
12
|
+
TTL 判定 过期的当前值不许冒充现在。让模型说"你电量还有 87%"
|
|
13
|
+
(其实是四小时前的),是这类系统最常见的一种说错话。
|
|
14
|
+
趋势模型 波动 / 漂移 / 周期三种算法,同一句"最近怎么样"要走不同的路,
|
|
15
|
+
选错了结论就是错的。
|
|
16
|
+
缺数据显式化 十四天里两天没戴表,不能当成"睡了 0 分钟" —— 那会把均值直接拉垮。
|
|
17
|
+
隐私投影 精确坐标、BSSID 这类字段永远不给 agent 看,由 manifest 声明,
|
|
18
|
+
不靠每个宿主自觉。
|
|
19
|
+
|
|
20
|
+
**工具层(MCP schema、工具名、中文描述、给谁开)留在宿主。** kit 提供数据和契约,
|
|
21
|
+
宿主决定怎么把它暴露成工具 —— 那一层全是产品决策,换个宿主全不一样。
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from datetime import date, datetime, timezone
|
|
27
|
+
from typing import Any, Mapping, Sequence
|
|
28
|
+
|
|
29
|
+
from ..algorithms import history as _history
|
|
30
|
+
from ..algorithms import trend_models as _trend
|
|
31
|
+
from ..manifest.types import SignalDefinition
|
|
32
|
+
from ..ports.storage import StoragePort
|
|
33
|
+
from ..processing import recurrence as _recurrence
|
|
34
|
+
|
|
35
|
+
#: 所有 list 查询的默认与硬上限。agent 问一句"我这个月都去过哪",
|
|
36
|
+
#: 不设上限就是几千条直接塞进模型上下文。
|
|
37
|
+
DEFAULT_LIMIT = 50
|
|
38
|
+
MAX_LIMIT = 500
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _clamp(limit: int) -> int:
|
|
42
|
+
return max(1, min(int(limit or DEFAULT_LIMIT), MAX_LIMIT))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _page(rows: list, cursor: str | None, limit: int) -> tuple[list, str | None]:
|
|
46
|
+
"""按偏移量分页。
|
|
47
|
+
|
|
48
|
+
**只用在有界集合上**(一个人的日历镜像、待投递的事件)—— 观测时间线
|
|
49
|
+
那种可能上百万行的,用的是 ``(occurred_at, observation_id)`` 键游标,
|
|
50
|
+
因为偏移量分页在中间插入一条迟到数据时会漏掉或重复一条,而且不报错。
|
|
51
|
+
这里的集合不会被中途插入到打乱顺序的程度,用偏移量换取不改端口。
|
|
52
|
+
"""
|
|
53
|
+
start = int(cursor) if cursor else 0
|
|
54
|
+
size = _clamp(limit)
|
|
55
|
+
page = rows[start:start + size]
|
|
56
|
+
return page, (str(start + size) if start + size < len(rows) else None)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def visible_fields(sig: SignalDefinition, *, on_demand: bool = True) -> tuple[str, ...]:
|
|
60
|
+
"""这个信号有哪些字段能给 agent 看。
|
|
61
|
+
|
|
62
|
+
``query_visibility="never"`` 的一律排除 —— 精确坐标、BSSID 这类既不该
|
|
63
|
+
持久化也不该进模型上下文,由 manifest 声明成规则,不靠每个宿主自觉。
|
|
64
|
+
"""
|
|
65
|
+
allowed = {"always"} | ({"on_demand"} if on_demand else set())
|
|
66
|
+
return tuple(f.key for f in sig.fields if f.query_visibility in allowed)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def project(sig: SignalDefinition, value: Mapping[str, Any] | None,
|
|
70
|
+
*, on_demand: bool = True) -> dict[str, Any] | None:
|
|
71
|
+
"""按 manifest 的可见性过滤一份 payload。"""
|
|
72
|
+
if value is None:
|
|
73
|
+
return None
|
|
74
|
+
keep = set(visible_fields(sig, on_demand=on_demand))
|
|
75
|
+
return {k: v for k, v in value.items() if k in keep}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ---------------------------------------------------------------------------
|
|
79
|
+
# 当前值
|
|
80
|
+
# ---------------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class CurrentView:
|
|
84
|
+
"""一个信号的当前状态,以及"它还算不算当前"。"""
|
|
85
|
+
|
|
86
|
+
signal: str
|
|
87
|
+
#: ``fresh`` / ``stale`` / ``unavailable`` / ``no_data``
|
|
88
|
+
state: str
|
|
89
|
+
value: dict[str, Any] | None
|
|
90
|
+
#: 最后一次可靠值。**stale 时也给**,但必须带 ``as_of``,不能冒充当前。
|
|
91
|
+
last_known: dict[str, Any] | None = None
|
|
92
|
+
as_of: str | None = None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def get_current(
|
|
96
|
+
storage: StoragePort, *, subject_id: str, signals: Sequence[str],
|
|
97
|
+
manifest: Mapping[str, SignalDefinition], now: datetime,
|
|
98
|
+
on_demand: bool = True,
|
|
99
|
+
) -> dict[str, CurrentView]:
|
|
100
|
+
"""取当前值,带 TTL 判定和隐私投影。"""
|
|
101
|
+
out: dict[str, CurrentView] = {}
|
|
102
|
+
raw = storage.get_current(subject_id=subject_id, signals=list(signals))
|
|
103
|
+
for signal in signals:
|
|
104
|
+
sig = manifest.get(signal)
|
|
105
|
+
projections = raw.get(signal) or []
|
|
106
|
+
if sig is None or not projections:
|
|
107
|
+
out[signal] = CurrentView(signal, "no_data", None)
|
|
108
|
+
continue
|
|
109
|
+
proj = max(projections, key=lambda p: p.observed_at)
|
|
110
|
+
visible = project(sig, proj.typed_value, on_demand=on_demand)
|
|
111
|
+
fresh = proj.expires_at is None or proj.expires_at > now
|
|
112
|
+
if proj.availability != "observed":
|
|
113
|
+
state = "unavailable"
|
|
114
|
+
else:
|
|
115
|
+
state = "fresh" if fresh else "stale"
|
|
116
|
+
out[signal] = CurrentView(
|
|
117
|
+
signal=signal,
|
|
118
|
+
state=state,
|
|
119
|
+
value=visible if state == "fresh" else None,
|
|
120
|
+
last_known=visible,
|
|
121
|
+
as_of=proj.observed_at.isoformat(),
|
|
122
|
+
)
|
|
123
|
+
return out
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def get_last_known(
|
|
127
|
+
storage: StoragePort, *, subject_id: str, signal: str,
|
|
128
|
+
manifest: Mapping[str, SignalDefinition], on_demand: bool = True,
|
|
129
|
+
) -> CurrentView:
|
|
130
|
+
"""最后一次可靠值,**不判 TTL**。
|
|
131
|
+
|
|
132
|
+
和 ``get_current`` 的区别是意图:这个函数的调用方已经知道自己要的是
|
|
133
|
+
"最后一次",不是"现在"。所以永远带 ``as_of``,永远不说 fresh。
|
|
134
|
+
"""
|
|
135
|
+
sig = manifest.get(signal)
|
|
136
|
+
projections = storage.get_current(subject_id=subject_id, signals=[signal]).get(signal)
|
|
137
|
+
if sig is None or not projections:
|
|
138
|
+
return CurrentView(signal, "no_data", None)
|
|
139
|
+
proj = max(projections, key=lambda p: p.observed_at)
|
|
140
|
+
return CurrentView(
|
|
141
|
+
signal=signal, state="last_known", value=None,
|
|
142
|
+
last_known=project(sig, proj.typed_value, on_demand=on_demand),
|
|
143
|
+
as_of=proj.observed_at.isoformat(),
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# ---------------------------------------------------------------------------
|
|
148
|
+
# 历史
|
|
149
|
+
# ---------------------------------------------------------------------------
|
|
150
|
+
|
|
151
|
+
def list_timeline(
|
|
152
|
+
storage: StoragePort, *, subject_id: str, signal: str,
|
|
153
|
+
manifest: Mapping[str, SignalDefinition],
|
|
154
|
+
start: datetime | None = None, end: datetime | None = None,
|
|
155
|
+
cursor: str | None = None, limit: int = DEFAULT_LIMIT,
|
|
156
|
+
on_demand: bool = True,
|
|
157
|
+
) -> tuple[list[dict[str, Any]], str | None]:
|
|
158
|
+
"""原始观测时间线。分页,有硬上限。"""
|
|
159
|
+
sig = manifest.get(signal)
|
|
160
|
+
rows, nxt = storage.list_observations(
|
|
161
|
+
subject_id=subject_id, signal=signal, start=start, end=end,
|
|
162
|
+
cursor=cursor, limit=_clamp(limit),
|
|
163
|
+
)
|
|
164
|
+
out = [
|
|
165
|
+
{
|
|
166
|
+
"occurred_at": o.occurred_at.isoformat(),
|
|
167
|
+
"availability": o.availability,
|
|
168
|
+
"value": project(sig, o.typed_value, on_demand=on_demand) if sig else None,
|
|
169
|
+
"local_date": o.effective_local_date.isoformat(),
|
|
170
|
+
}
|
|
171
|
+
for o in rows
|
|
172
|
+
]
|
|
173
|
+
return out, nxt
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass(frozen=True)
|
|
177
|
+
class DailyView:
|
|
178
|
+
date: str
|
|
179
|
+
value: dict[str, Any]
|
|
180
|
+
#: 这一天有没有数据。**空缺的日子不补零** —— `no_data` 不是 0。
|
|
181
|
+
has_data: bool = True
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def get_daily_aggregates(
|
|
185
|
+
storage: StoragePort, *, subject_id: str, signal: str,
|
|
186
|
+
start_date: date, end_date: date,
|
|
187
|
+
) -> list[DailyView]:
|
|
188
|
+
"""日聚合。**只返回有数据的日子,不补零。**
|
|
189
|
+
|
|
190
|
+
补零是这类系统最常见的一个静默错误:十四天里两天没戴表,补两个 0 进去,
|
|
191
|
+
平均睡眠时长立刻被拉垮,而且没有任何地方报错。
|
|
192
|
+
"""
|
|
193
|
+
rows = storage.get_aggregate(
|
|
194
|
+
subject_id=subject_id, signal=signal,
|
|
195
|
+
start_date=start_date, end_date=end_date, aggregation_kind="daily",
|
|
196
|
+
)
|
|
197
|
+
return [
|
|
198
|
+
DailyView(date=r.local_date.isoformat(), value=r.typed_aggregate)
|
|
199
|
+
for r in sorted(rows, key=lambda r: r.local_date)
|
|
200
|
+
]
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def get_trend(
|
|
204
|
+
storage: StoragePort, *, subject_id: str, signal: str, field: str,
|
|
205
|
+
manifest: Mapping[str, SignalDefinition],
|
|
206
|
+
start_date: date, end_date: date,
|
|
207
|
+
) -> dict[str, Any]:
|
|
208
|
+
"""趋势。**按 manifest 声明的模型选算法** —— 选错了结论就是错的。
|
|
209
|
+
|
|
210
|
+
返回里一定带 ``days_with_data`` / ``days_missing``:缺了几天必须说出来,
|
|
211
|
+
否则调用方无从判断这个趋势可不可信。
|
|
212
|
+
"""
|
|
213
|
+
sig = manifest.get(signal)
|
|
214
|
+
if sig is None:
|
|
215
|
+
return {"model": "none", "reason": f"manifest 里没有 {signal}"}
|
|
216
|
+
fd = sig.field_map().get(field)
|
|
217
|
+
if fd is None:
|
|
218
|
+
return {"model": "none", "reason": f"{signal} 没有 {field} 这个字段"}
|
|
219
|
+
if fd.query_visibility == "never":
|
|
220
|
+
return {"model": "none", "reason": "这个字段不对 agent 开放"}
|
|
221
|
+
|
|
222
|
+
rows = storage.get_aggregate(
|
|
223
|
+
subject_id=subject_id, signal=signal,
|
|
224
|
+
start_date=start_date, end_date=end_date, aggregation_kind="daily",
|
|
225
|
+
)
|
|
226
|
+
docs = [
|
|
227
|
+
{"date": r.local_date.isoformat(), "doc": r.typed_aggregate}
|
|
228
|
+
for r in sorted(rows, key=lambda r: r.local_date)
|
|
229
|
+
]
|
|
230
|
+
span = (end_date - start_date).days + 1
|
|
231
|
+
coverage = {"days_with_data": len(docs), "days_missing": max(0, span - len(docs))}
|
|
232
|
+
|
|
233
|
+
if not docs:
|
|
234
|
+
return {"model": fd.trend_model, "reason": "这段时间一条数据都没有", **coverage}
|
|
235
|
+
|
|
236
|
+
# 三种模型走三条完全不同的路。旧的按信号名查表那条路(trend_models.model_for)
|
|
237
|
+
# 保留不动;这里按 manifest 走,因为 manifest 是新的单一声明处。
|
|
238
|
+
model = fd.trend_model
|
|
239
|
+
if model == "drifting":
|
|
240
|
+
result = _trend.read_drift(docs, signal, field)
|
|
241
|
+
elif model == "cyclical":
|
|
242
|
+
result = _trend.read_cycles([d["date"] for d in docs],
|
|
243
|
+
today=end_date.isoformat())
|
|
244
|
+
elif model == "fluctuating":
|
|
245
|
+
result = _history.read_trend(docs, signal, field)
|
|
246
|
+
else:
|
|
247
|
+
return {"model": "none", "reason": "这个字段没有声明趋势模型", **coverage}
|
|
248
|
+
|
|
249
|
+
return {"model": model, "unit": fd.unit, **coverage, **result}
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
# ---------------------------------------------------------------------------
|
|
253
|
+
# 来源镜像与事件
|
|
254
|
+
# ---------------------------------------------------------------------------
|
|
255
|
+
|
|
256
|
+
def list_calendar_events(
|
|
257
|
+
storage: StoragePort, *, subject_id: str,
|
|
258
|
+
start: datetime | None = None, end: datetime | None = None,
|
|
259
|
+
cursor: str | None = None, limit: int = DEFAULT_LIMIT,
|
|
260
|
+
) -> tuple[list[dict[str, Any]], str | None]:
|
|
261
|
+
"""当前来源镜像里的日程。
|
|
262
|
+
|
|
263
|
+
⚠️ 同步长期失败时应该显示 stale,而不是继续声称这是最新完整的数据 ——
|
|
264
|
+
调用方要自己去看 ``SourceSyncState.last_successful_sync_at``。
|
|
265
|
+
"""
|
|
266
|
+
# 先按窗口取一批(上限而不是页大小 —— 重复日程展开之后条数会变多,
|
|
267
|
+
# 按页大小取会让展开后不足一页)。
|
|
268
|
+
rows = storage.list_calendar_events(
|
|
269
|
+
subject_id=subject_id, start=start, end=end, limit=MAX_LIMIT,
|
|
270
|
+
)
|
|
271
|
+
out: list[dict[str, Any]] = []
|
|
272
|
+
for e in rows:
|
|
273
|
+
base = {"source_event_id": e.source_event_id, **e.event_fields}
|
|
274
|
+
rule_raw = e.event_fields.get("recurrence")
|
|
275
|
+
if not rule_raw or start is None or end is None:
|
|
276
|
+
# 没有窗口就不展开 —— "把所有重复日程都给我"对一条无限重复的
|
|
277
|
+
# 规则没有答案,只有一个上限截断出来的假象。
|
|
278
|
+
out.append(base)
|
|
279
|
+
continue
|
|
280
|
+
try:
|
|
281
|
+
rule = _recurrence.RecurrenceRule.parse(rule_raw)
|
|
282
|
+
occurrences = _recurrence.expand(
|
|
283
|
+
e.event_fields["start_at"], rule,
|
|
284
|
+
window_start=start.date(), window_end=end.date(),
|
|
285
|
+
)
|
|
286
|
+
except (_recurrence.RecurrenceUnsupported, KeyError, ValueError) as exc:
|
|
287
|
+
# 展不开就把系列本身交出去并说清原因,**不猜日期**。
|
|
288
|
+
# 算错重复日程会让用户准时出现在一个不存在的会议上。
|
|
289
|
+
out.append({**base, "recurrence_expanded": False,
|
|
290
|
+
"recurrence_note": str(exc)})
|
|
291
|
+
continue
|
|
292
|
+
for at in occurrences:
|
|
293
|
+
out.append({**base, "start_at": at, "recurrence_expanded": True,
|
|
294
|
+
"recurrence_identity": e.recurrence_identity
|
|
295
|
+
or e.source_event_id})
|
|
296
|
+
return _page(out, cursor, limit)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def list_reminders(
|
|
300
|
+
storage: StoragePort, *, subject_id: str, include_completed: bool = False,
|
|
301
|
+
cursor: str | None = None, limit: int = DEFAULT_LIMIT,
|
|
302
|
+
) -> tuple[list[dict[str, Any]], str | None]:
|
|
303
|
+
rows = storage.list_reminders(
|
|
304
|
+
subject_id=subject_id, include_completed=include_completed,
|
|
305
|
+
limit=MAX_LIMIT,
|
|
306
|
+
)
|
|
307
|
+
return _page([{"source_reminder_id": r.source_reminder_id, **r.reminder_fields}
|
|
308
|
+
for r in rows], cursor, limit)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def list_events(
|
|
312
|
+
storage: StoragePort, *, subject_id: str, status: str | None = None,
|
|
313
|
+
event_type: str | None = None,
|
|
314
|
+
start: datetime | None = None, end: datetime | None = None,
|
|
315
|
+
cursor: str | None = None, limit: int = DEFAULT_LIMIT,
|
|
316
|
+
) -> tuple[list[dict[str, Any]], str | None]:
|
|
317
|
+
"""这个用户的事件。用于排查「为什么没提醒我」。
|
|
318
|
+
|
|
319
|
+
``status`` 按投递状态筛。**「为什么没提醒我」的答案往往不是 pending,
|
|
320
|
+
而是 suppressed 或 rejected** —— 只能看待投递的话,那些事件在排查时
|
|
321
|
+
根本不出现,看上去就像"压根没产生过"。
|
|
322
|
+
|
|
323
|
+
返回 ``(事件, 下一页游标)``。所有 list 查询都必须分页或有明确上限
|
|
324
|
+
(产品规范 §15)。
|
|
325
|
+
"""
|
|
326
|
+
rows = list(storage.list_pending_events(subject_id=subject_id,
|
|
327
|
+
limit=MAX_LIMIT))
|
|
328
|
+
if status is not None:
|
|
329
|
+
rows = [e for e in rows if e.delivery_state == status]
|
|
330
|
+
if event_type is not None:
|
|
331
|
+
rows = [e for e in rows if e.event_type == event_type]
|
|
332
|
+
if start is not None:
|
|
333
|
+
rows = [e for e in rows if e.occurred_at >= start]
|
|
334
|
+
if end is not None:
|
|
335
|
+
rows = [e for e in rows if e.occurred_at <= end]
|
|
336
|
+
page, nxt = _page(rows, cursor, limit)
|
|
337
|
+
return [
|
|
338
|
+
{
|
|
339
|
+
"event_id": e.event_id, "type": e.event_type,
|
|
340
|
+
"occurred_at": e.occurred_at.isoformat(),
|
|
341
|
+
"delivery_state": e.delivery_state, "attempts": e.attempt_count,
|
|
342
|
+
}
|
|
343
|
+
for e in page
|
|
344
|
+
], nxt
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def list_definitions(
|
|
348
|
+
definitions: Sequence[Any], *, subject_id: str | None = None,
|
|
349
|
+
include_disabled: bool = True,
|
|
350
|
+
) -> list[dict[str, Any]]:
|
|
351
|
+
"""这个用户当前配着哪些规则(产品规范 §15)。
|
|
352
|
+
|
|
353
|
+
**不查存储** —— 规则定义是宿主装配 kit 时传进来的配置,不是 kit 存的数据。
|
|
354
|
+
宿主按用户存规则的话,自己筛完再传进来;``subject_id`` 只是原样带回,
|
|
355
|
+
方便调用方对上是谁的。
|
|
356
|
+
|
|
357
|
+
为什么需要这个查询:用户能自己配规则,就一定会问「我配的那条还在吗、
|
|
358
|
+
是不是被关掉了」。没有它,唯一的排查手段是去读宿主的配置表 ——
|
|
359
|
+
而那正是这个包想让宿主不必自己造的东西。
|
|
360
|
+
"""
|
|
361
|
+
out: list[dict[str, Any]] = []
|
|
362
|
+
for d in definitions:
|
|
363
|
+
enabled = getattr(d, "enabled", True)
|
|
364
|
+
if not include_disabled and not enabled:
|
|
365
|
+
continue
|
|
366
|
+
out.append({
|
|
367
|
+
"definition_id": getattr(d, "definition_id", None),
|
|
368
|
+
"version": getattr(d, "version", None),
|
|
369
|
+
"signal": getattr(d, "signal", None),
|
|
370
|
+
"field": getattr(d, "field_name", None),
|
|
371
|
+
"condition": getattr(d, "condition_type", None),
|
|
372
|
+
"event_type": getattr(d, "event_type", None),
|
|
373
|
+
# 关掉的规则**也列出来并标明** —— 「它没触发」和「它被关了」
|
|
374
|
+
# 是两个完全不同的排查方向。
|
|
375
|
+
"enabled": enabled,
|
|
376
|
+
"subject_id": subject_id,
|
|
377
|
+
})
|
|
378
|
+
return out
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def export_subject(
|
|
382
|
+
storage: StoragePort, *, subject_id: str,
|
|
383
|
+
manifest: Mapping[str, SignalDefinition],
|
|
384
|
+
start: datetime | None = None, end: datetime | None = None,
|
|
385
|
+
per_signal_limit: int = MAX_LIMIT,
|
|
386
|
+
) -> dict[str, Any]:
|
|
387
|
+
"""把一个人的全部感知数据导出来。
|
|
388
|
+
|
|
389
|
+
产品规范 §8 要的是「按 subject 定位、**导出**和删除」。删除有
|
|
390
|
+
``purge_subject``,定位有各个 list —— 导出这一半以前没有。
|
|
391
|
+
没有它,"把我的数据给我"只能靠调用方自己把八个查询拼一遍,
|
|
392
|
+
而**拼漏一个就是少给了用户一部分数据,还没人会发现**。
|
|
393
|
+
|
|
394
|
+
刻意用现有端口方法拼,不新增端口:导出不是热路径,为它增加所有宿主
|
|
395
|
+
都必须实现的一个方法不划算。
|
|
396
|
+
|
|
397
|
+
⚠️ 导出的是 kit 管的东西。宿主自己存的(原始载荷、加密信封、
|
|
398
|
+
它自己的业务表)要由宿主追加进来 —— 返回值里的 ``kit_managed_only``
|
|
399
|
+
就是提醒这件事的。
|
|
400
|
+
"""
|
|
401
|
+
signals = sorted(manifest)
|
|
402
|
+
observations: dict[str, list[dict[str, Any]]] = {}
|
|
403
|
+
for signal in signals:
|
|
404
|
+
rows, _ = list_timeline(
|
|
405
|
+
storage, subject_id=subject_id, signal=signal, manifest=manifest,
|
|
406
|
+
start=start, end=end, limit=per_signal_limit,
|
|
407
|
+
# 导出给用户本人,所以按需字段也要给;
|
|
408
|
+
# `query_visibility="never"` 的仍然不给 —— 那些**根本没存**。
|
|
409
|
+
on_demand=True,
|
|
410
|
+
)
|
|
411
|
+
if rows:
|
|
412
|
+
observations[signal] = rows
|
|
413
|
+
|
|
414
|
+
current = {
|
|
415
|
+
signal: {"state": view.state, "value": view.value,
|
|
416
|
+
"last_known": view.last_known, "as_of": view.as_of}
|
|
417
|
+
for signal, view in get_current(
|
|
418
|
+
storage, subject_id=subject_id, signals=signals,
|
|
419
|
+
manifest=manifest, now=end or _far_future(),
|
|
420
|
+
).items()
|
|
421
|
+
if view.state != "no_data"
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
return {
|
|
425
|
+
"subject_id": subject_id,
|
|
426
|
+
"kit_managed_only": True,
|
|
427
|
+
"current": current,
|
|
428
|
+
"observations": observations,
|
|
429
|
+
# 导出取第一页 —— 一个人的日历镜像是有界的,per_signal_limit 已经够大。
|
|
430
|
+
"calendar_events": list_calendar_events(
|
|
431
|
+
storage, subject_id=subject_id, start=start, end=end,
|
|
432
|
+
limit=per_signal_limit)[0],
|
|
433
|
+
"reminders": list_reminders(storage, subject_id=subject_id,
|
|
434
|
+
include_completed=True,
|
|
435
|
+
limit=per_signal_limit)[0],
|
|
436
|
+
# 导出取第一页就够 —— 待投递的事件不是"用户的数据",
|
|
437
|
+
# 是我们还没送到的东西,列在这里只为完整。
|
|
438
|
+
"pending_events": list_events(storage, subject_id=subject_id,
|
|
439
|
+
limit=per_signal_limit)[0],
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _far_future() -> datetime:
|
|
444
|
+
"""导出不判新鲜度 —— 用户要的是"我有什么数据",不是"哪些还算当前"。
|
|
445
|
+
|
|
446
|
+
这个包不读时钟,所以用一个固定的远期时间,而不是 ``now()``。
|
|
447
|
+
"""
|
|
448
|
+
return datetime(9999, 12, 31, tzinfo=timezone.utc)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
__all__ = [
|
|
452
|
+
"DEFAULT_LIMIT", "MAX_LIMIT", "CurrentView", "DailyView",
|
|
453
|
+
"visible_fields", "project",
|
|
454
|
+
"get_current", "get_last_known", "list_timeline", "get_daily_aggregates",
|
|
455
|
+
"export_subject", "list_definitions",
|
|
456
|
+
"get_trend", "list_calendar_events", "list_reminders", "list_events",
|
|
457
|
+
]
|
perceptkit/retention.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""保留期(存多久才删)与保质期(多久之后不再采信)的声明表。
|
|
2
|
+
|
|
3
|
+
★ 本模块只声明,不含任何删除逻辑 —— 真正的清理任务要接定时器,属于消费方。
|
|
4
|
+
|
|
5
|
+
★ 保留期的判据:这条数据在 N 个月后,还会改变 agent 对这个人的理解吗。
|
|
6
|
+
|
|
7
|
+
★ 保质期这一块只列「改判测量时间之后需要改的」。改判之前它判的是
|
|
8
|
+
「距这次上报多久」,改判之后判「这条数据多久前测的」—— 沿用旧值会把功能
|
|
9
|
+
杀死:体重 24 小时 = 除非今天刚称过否则永远 null。
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
KEEP_FOREVER = None
|
|
14
|
+
|
|
15
|
+
_DAY = 86400.0
|
|
16
|
+
|
|
17
|
+
# signal -> 保留天数;None = 永久;不在表里 = 不进历史表
|
|
18
|
+
RETENTION_DAYS: dict[str, int | None] = {
|
|
19
|
+
# 永久:趋势本身就是价值
|
|
20
|
+
"health_body": KEEP_FOREVER,
|
|
21
|
+
"health_sleep": KEEP_FOREVER,
|
|
22
|
+
"health_vitals": KEEP_FOREVER,
|
|
23
|
+
"health_activity": KEEP_FOREVER,
|
|
24
|
+
"health_workout": KEEP_FOREVER,
|
|
25
|
+
"health_metabolic": KEEP_FOREVER,
|
|
26
|
+
"health_mood": KEEP_FOREVER,
|
|
27
|
+
"health_cycle": KEEP_FOREVER,
|
|
28
|
+
"location_signal": KEEP_FOREVER,
|
|
29
|
+
# 一年:年度口味有价值,再久没人问
|
|
30
|
+
"playback": 365,
|
|
31
|
+
# 90 天:瞬时状态,回看价值掉得快
|
|
32
|
+
"motion_state": 90,
|
|
33
|
+
"focus": 90,
|
|
34
|
+
"audio_route": 90,
|
|
35
|
+
# weather 现在的 SHAPE 是 NUMERIC_DIST(history.py),仍在产生 rollup,
|
|
36
|
+
# 所以现在必须有真实保留期 —— 跟瞬时状态同档。
|
|
37
|
+
# ⚠️ Codex code_review 2026-08-23 抓到:早先按"weather 即将改成仅当前+预报、
|
|
38
|
+
# 不再存历史"的未来态把这条声明成了 KEEP_FOREVER 之外/None,
|
|
39
|
+
# 但 SHAPE/history.record_daily 从未真的改过去,导致四张声明表互相矛盾。
|
|
40
|
+
# 真要把 weather 改成不存历史时,这一行、attribution.ATTRIBUTION 里的
|
|
41
|
+
# weather 条目、history.SHAPE 里的 weather 条目,三处必须在同一批一起删,
|
|
42
|
+
# 不许只删一处。
|
|
43
|
+
"weather": 90,
|
|
44
|
+
# 60 天:采集窗口是前后 14 天,够覆盖「未来的会 → 过去的会 → 再留一个月回看」
|
|
45
|
+
"calendar_next_event": 60,
|
|
46
|
+
"reminders": 60,
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
# 改判「测量时间」之后的保质期。只列与 catalog 现值不同的;
|
|
50
|
+
# 「现在测现在传」的信号(位置/运动/专注/音频/播放)不需要改。
|
|
51
|
+
MEASURED_AT_TTL_SEC: dict[str, float] = {
|
|
52
|
+
"health_body": 90 * _DAY, # 三个月内称过就还算数
|
|
53
|
+
"health_metabolic": 30 * _DAY, # 一个月内测过就还算数
|
|
54
|
+
"health_cycle": 60 * _DAY, # 两个月内有记录就还算数
|
|
55
|
+
"health_vitals": 7 * _DAY, # 一周内测过就还算数
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ⚠️ 「永久保存」不等于「不可删除」(Codex 评审修订 K)。
|
|
60
|
+
# 以下生命周期动作必须由消费方实现,本表只是保留期,不是删除策略的全部:
|
|
61
|
+
# · 账号删除时清空
|
|
62
|
+
# · 用户主动清空
|
|
63
|
+
# · 健康权限关闭后:禁用读取与 wake(不是继续用存量)
|
|
64
|
+
# · 第三方撤权
|
|
65
|
+
# · 来源侧删除 / 纠正如何传播到我们的汇总
|
|
66
|
+
# 本模块不实现这些 —— 它零 I/O。放在这里是为了让读到保留期的人
|
|
67
|
+
# 不会把「永久」误解成「不可删」。
|
|
68
|
+
LIFECYCLE_NOTE = "retention != undeletable; see design doc 修订 K"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def stores_history(signal: str) -> bool:
|
|
72
|
+
"""这个信号进不进历史表。"""
|
|
73
|
+
return signal in RETENTION_DAYS
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def retention_days(signal: str) -> int | None:
|
|
77
|
+
"""保留天数;``KEEP_FOREVER``(None)表示永久。
|
|
78
|
+
|
|
79
|
+
不进历史表的信号调用这个是调用方的错,直接抛 —— 静默返回 None 会被
|
|
80
|
+
误当成「永久」,那是最坏的一种默认值。
|
|
81
|
+
"""
|
|
82
|
+
if signal not in RETENTION_DAYS:
|
|
83
|
+
raise KeyError(f"signal {signal!r} 不进历史表,先用 stores_history() 判断")
|
|
84
|
+
return RETENTION_DAYS[signal]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""事件规则 —— 什么条件算一件事。
|
|
2
|
+
|
|
3
|
+
types EventDefinition / RuleState / Lifecycle
|
|
4
|
+
evaluators 九种内置规则 + 自定义 evaluator 的接口
|
|
5
|
+
engine 求值 + 生命周期(scope / fire / rearm)
|
|
6
|
+
|
|
7
|
+
**不做通用表达式 DSL**,也**不解析 YAML**(这个包零依赖)。宿主想用 YAML
|
|
8
|
+
写规则完全可以 —— 自己 safe_load 成 dict 再传进来。
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from .engine import evaluate, scope_key
|
|
13
|
+
from .evaluators import BUILTIN, RuleEvaluator
|
|
14
|
+
from .types import EventDefinition, Lifecycle, RuleResult, RuleState
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"EventDefinition", "Lifecycle", "RuleState", "RuleResult",
|
|
18
|
+
"RuleEvaluator", "BUILTIN", "evaluate", "scope_key",
|
|
19
|
+
]
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""求值引擎 —— 把 evaluator 的"命中了吗"变成"该不该产生事件"。
|
|
2
|
+
|
|
3
|
+
生命周期统一在这里处理,不让九个 evaluator 各写一遍:
|
|
4
|
+
|
|
5
|
+
scope 这条规则在什么范围内计一次(每天 / 永远)
|
|
6
|
+
fire 一个范围内触发几次(一次 / 每次)
|
|
7
|
+
rearm 什么时候重新武装(次日 / 冷却后 / 永不)
|
|
8
|
+
|
|
9
|
+
**scope 换了就是一条新状态。** "今天已经触发过"不会影响明天 —— 这就是
|
|
10
|
+
``rearm=next_scope`` 的实现方式:不需要定时任务去重置,换个 scope_key 就行。
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from datetime import date, datetime
|
|
15
|
+
from typing import Any, Callable, Mapping
|
|
16
|
+
|
|
17
|
+
from .evaluators import BUILTIN
|
|
18
|
+
from .types import EventDefinition, RuleResult, RuleState
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def scope_key(definition: EventDefinition, *, local_date: date) -> str:
|
|
22
|
+
"""这条规则此刻算在哪个范围里。"""
|
|
23
|
+
if definition.lifecycle.scope == "local_day":
|
|
24
|
+
return local_date.isoformat()
|
|
25
|
+
return "forever"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _cooled_down(state: RuleState, *, now: datetime, cooldown: float) -> bool:
|
|
29
|
+
if cooldown <= 0 or not state.last_fired_at:
|
|
30
|
+
return True
|
|
31
|
+
try:
|
|
32
|
+
last = datetime.fromisoformat(state.last_fired_at)
|
|
33
|
+
except ValueError:
|
|
34
|
+
return True
|
|
35
|
+
return (now - last).total_seconds() >= cooldown
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def evaluate(
|
|
39
|
+
definition: EventDefinition,
|
|
40
|
+
state: RuleState,
|
|
41
|
+
current: Any,
|
|
42
|
+
*,
|
|
43
|
+
now: datetime,
|
|
44
|
+
context: Mapping[str, Any] | None = None,
|
|
45
|
+
extra_evaluators: Mapping[str, Callable[..., RuleResult]] | None = None,
|
|
46
|
+
) -> RuleResult:
|
|
47
|
+
"""求值一条规则,返回是否触发以及**下一版状态**。
|
|
48
|
+
|
|
49
|
+
调用方必须把返回的 state 写回去 —— 哪怕没触发。``previous_value`` 每次
|
|
50
|
+
都要更新,否则 ``threshold_crossing`` 永远拿不到正确的前值。
|
|
51
|
+
"""
|
|
52
|
+
ctx = dict(context or {})
|
|
53
|
+
if not definition.enabled:
|
|
54
|
+
return RuleResult(False, state, reason="规则已停用")
|
|
55
|
+
|
|
56
|
+
evaluators = dict(BUILTIN)
|
|
57
|
+
if extra_evaluators:
|
|
58
|
+
evaluators.update(extra_evaluators)
|
|
59
|
+
fn = evaluators.get(definition.condition_type)
|
|
60
|
+
if fn is None:
|
|
61
|
+
return RuleResult(False, state,
|
|
62
|
+
reason=f"没有 {definition.condition_type!r} 这种规则")
|
|
63
|
+
|
|
64
|
+
outcome = fn(definition, state, current, ctx)
|
|
65
|
+
lifecycle = definition.lifecycle
|
|
66
|
+
|
|
67
|
+
# 无论触发与否,前值都要推进 —— 这是 crossing / changed / delta 的前提。
|
|
68
|
+
#
|
|
69
|
+
# 但如果 evaluator 自己动了 previous_value(自定义 evaluator 往往要维护
|
|
70
|
+
# 自己的派生状态),就尊重它的版本 —— 以前无条件覆盖,导致任何需要自己
|
|
71
|
+
# 管状态的 evaluator 都没法工作。
|
|
72
|
+
evaluator_moved_previous = outcome.state.previous_value != state.previous_value
|
|
73
|
+
if evaluator_moved_previous:
|
|
74
|
+
next_previous = outcome.state.previous_value
|
|
75
|
+
elif definition.condition_type == "streak":
|
|
76
|
+
next_previous = outcome.current # streak 的"前值"是连续长度,不是观测值
|
|
77
|
+
else:
|
|
78
|
+
next_previous = current
|
|
79
|
+
advanced = RuleState(
|
|
80
|
+
previous_value=next_previous,
|
|
81
|
+
fired_in_scope=state.fired_in_scope,
|
|
82
|
+
last_fired_at=state.last_fired_at,
|
|
83
|
+
seen_keys=outcome.state.seen_keys,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
if not outcome.fired:
|
|
87
|
+
return RuleResult(False, advanced, outcome.previous, outcome.current,
|
|
88
|
+
outcome.reason)
|
|
89
|
+
|
|
90
|
+
# 冷却和 fire 模式是两件事。以前只在 fire=once 时检查冷却,
|
|
91
|
+
# 于是 fire=every + rearm=cooldown 完全没有冷却 —— 配了个不生效的东西。
|
|
92
|
+
if (lifecycle.rearm == "cooldown"
|
|
93
|
+
and not _cooled_down(state, now=now,
|
|
94
|
+
cooldown=lifecycle.cooldown_seconds)):
|
|
95
|
+
return RuleResult(False, advanced, outcome.previous, outcome.current,
|
|
96
|
+
"还在冷却中")
|
|
97
|
+
if (lifecycle.fire == "once" and state.fired_in_scope
|
|
98
|
+
and lifecycle.rearm != "cooldown"):
|
|
99
|
+
return RuleResult(False, advanced, outcome.previous, outcome.current,
|
|
100
|
+
"这个范围内已经触发过了")
|
|
101
|
+
|
|
102
|
+
fired_state = RuleState(
|
|
103
|
+
previous_value=advanced.previous_value,
|
|
104
|
+
fired_in_scope=True,
|
|
105
|
+
last_fired_at=now.isoformat(),
|
|
106
|
+
seen_keys=advanced.seen_keys,
|
|
107
|
+
)
|
|
108
|
+
return RuleResult(True, fired_state, outcome.previous, outcome.current,
|
|
109
|
+
outcome.reason)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
__all__ = ["evaluate", "scope_key"]
|