aggregate_api 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. aggregate_api/__init__.py +41 -0
  2. aggregate_api/__main__.py +154 -0
  3. aggregate_api/app.py +206 -0
  4. aggregate_api/audit.py +395 -0
  5. aggregate_api/bounds.py +331 -0
  6. aggregate_api/cache.py +319 -0
  7. aggregate_api/capability.py +823 -0
  8. aggregate_api/completion.py +219 -0
  9. aggregate_api/config.py +363 -0
  10. aggregate_api/cors.py +61 -0
  11. aggregate_api/examples.py +620 -0
  12. aggregate_api/layer_pricing.py +840 -0
  13. aggregate_api/library.py +94 -0
  14. aggregate_api/library_notes.py +96 -0
  15. aggregate_api/models.py +1407 -0
  16. aggregate_api/net.py +281 -0
  17. aggregate_api/pnl.py +101 -0
  18. aggregate_api/pricing.py +778 -0
  19. aggregate_api/resources.py +257 -0
  20. aggregate_api/routes/__init__.py +8 -0
  21. aggregate_api/routes/decl.py +327 -0
  22. aggregate_api/routes/examples.py +82 -0
  23. aggregate_api/routes/meta.py +282 -0
  24. aggregate_api/routes/objects.py +4119 -0
  25. aggregate_api/routes/status.py +466 -0
  26. aggregate_api/serializers.py +565 -0
  27. aggregate_api/sessions.py +353 -0
  28. aggregate_api/static/aggregate-api-logo-512.png +0 -0
  29. aggregate_api/static/aggregate-api-logo.png +0 -0
  30. aggregate_api/static/aggregate-api-trim.png +0 -0
  31. aggregate_api/static/android-chrome-192x192.png +0 -0
  32. aggregate_api/static/android-chrome-512x512.png +0 -0
  33. aggregate_api/static/apple-touch-icon.png +0 -0
  34. aggregate_api/static/assets/bootstrap-icons-BeopsB42.woff +0 -0
  35. aggregate_api/static/assets/bootstrap-icons-mSm7cUeB.woff2 +0 -0
  36. aggregate_api/static/assets/bootstrap-ohb1VZ53.js +5 -0
  37. aggregate_api/static/assets/codemirror-h62DHGGa.js +14 -0
  38. aggregate_api/static/assets/csv-grid.worker-DKzHGXac.js +4 -0
  39. aggregate_api/static/assets/echarts-B7o9sc00.js +40 -0
  40. aggregate_api/static/assets/echarts-gl-DG1Uf6wE.js +4282 -0
  41. aggregate_api/static/assets/lite-CUlcD8p4.css +1 -0
  42. aggregate_api/static/assets/lite-Dd2TnT4M.js +1 -0
  43. aggregate_api/static/assets/main-Bxhxa55v.css +9 -0
  44. aggregate_api/static/assets/main-CmoEiPit.js +9 -0
  45. aggregate_api/static/assets/tables-BHCF7qIF.js +8 -0
  46. aggregate_api/static/assets/tables-CxvajLr7.css +1 -0
  47. aggregate_api/static/favicon-16x16.png +0 -0
  48. aggregate_api/static/favicon-32x32.png +0 -0
  49. aggregate_api/static/favicon.ico +0 -0
  50. aggregate_api/static/index.html +912 -0
  51. aggregate_api/static/lite.html +83 -0
  52. aggregate_api/static/logo.png +0 -0
  53. aggregate_api/static/site.webmanifest +14 -0
  54. aggregate_api/static/sw.js +78 -0
  55. aggregate_api/status.py +536 -0
  56. aggregate_api/status_page.html +546 -0
  57. aggregate_api/tables.py +316 -0
  58. aggregate_api-1.0.0.dist-info/METADATA +187 -0
  59. aggregate_api-1.0.0.dist-info/RECORD +63 -0
  60. aggregate_api-1.0.0.dist-info/WHEEL +5 -0
  61. aggregate_api-1.0.0.dist-info/entry_points.txt +2 -0
  62. aggregate_api-1.0.0.dist-info/licenses/LICENSE +28 -0
  63. aggregate_api-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,331 @@
1
+ """Pricing bounds: the range of prices consistent with one calibration.
2
+
3
+ Three questions, three classes in ``aggregate.bounds``, and this module is the
4
+ thin layer that asks them.
5
+
6
+ Ordinary pricing picks a distortion and reports the number it gives. That
7
+ number is only as firm as the choice of distortion, and the choice is a
8
+ judgment. Bounds asks the other question: hold the calibration fixed, let the
9
+ distortion range over everything consistent with it, and report how wide the
10
+ answer can be. A narrow range means the calibration decided the price; a wide
11
+ one means the distortion did.
12
+
13
+ * :class:`aggregate.bounds.Bounds` is the picture. Every distortion pricing the
14
+ risk to the target premium is a point in the (s, g(s)) plane, and the envelope
15
+ is the band they sweep.
16
+ * :class:`aggregate.bounds.PricingBounds` carries the calibration across to a
17
+ *second* risk: given that some distortion prices X to P, what can it say about
18
+ Y? That is the question behind quoting a new line off an existing book.
19
+ * :class:`aggregate.bounds.AllocationBounds` turns it inward, on the units of
20
+ one portfolio: the range of natural allocation consistent with the total
21
+ premium.
22
+
23
+ Notes
24
+ -----
25
+ **The cost worry recorded in the plan was unfounded**, and this is the
26
+ measurement that settled it. Constructing a ``Bounds`` is 0.01 s, ``cloud_df``
27
+ is 0.12 s, and the fifty-resample envelope figure is 0.5 to 0.8 s, on the
28
+ fixtures the tests use. The resamples are overplotted columns drawn from a frame
29
+ that is already computed, so asking for fifty rather than none costs the drawing
30
+ and nothing else.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from typing import Any
36
+
37
+ from aggregate import Aggregate, Portfolio
38
+ from aggregate import charts as agg_charts
39
+ from aggregate.bounds import AllocationBounds, Bounds, PricingBounds
40
+
41
+ #: How many bracket columns the envelope overplots. Fifty is enough to read the
42
+ #: band as a band rather than a boundary, and cheap for the reason in the module
43
+ #: notes.
44
+ DEFAULT_RESAMPLES = 50
45
+
46
+ #: Gone at a60, with the figure it grouped: ``ENVELOPE_PANELS`` split the five
47
+ #: named distortions two-and-three because the matplotlib compositor drew three
48
+ #: panels, and a52 built that list here to work around ``plot_envelope``
49
+ #: silently ignoring the ``'space'`` shorthand the api had been passing. Both
50
+ #: problems left with the figure. The emitter puts all five on one band, which
51
+ #: is the comparison the panel is for, and the api no longer has an opinion
52
+ #: about panel contents at all.
53
+
54
+
55
+ def _document(df, formats: str = "price") -> dict | None:
56
+ """One frame as a table document, or None. Best effort, as in ``pricing``."""
57
+ from .tables import frame_document_dict
58
+
59
+ try:
60
+ return frame_document_dict(df, formats=formats)
61
+ except Exception: # noqa: BLE001 -- a static table is never worth a 500
62
+ return None
63
+
64
+
65
+ def _calibrate_for_envelope(obj: Any, premium: float, assets: float | None) -> bool:
66
+ """Calibrate the named distortions onto ``obj``, for the envelope's panel 2.
67
+
68
+ Calibrated to **the premium the request already gave**, which is the whole
69
+ point of the exhibit: panel 1 is every distortion consistent with that
70
+ premium, and panel 2 names the ones the calibration produces, so they have
71
+ to be the same premium or the two panels answer different questions.
72
+
73
+ Returns ``True`` when the object came away carrying a calibration, which is
74
+ what decides whether the emitter has a second panel to draw. It does not
75
+ build or group anything itself: which distortions go on which panel is the
76
+ emitter's business now, and through a59 this function split them across two
77
+ panels because the matplotlib compositor drew three.
78
+
79
+ ``calibrate_distortions`` takes a cost-of-capital rather than a premium, so
80
+ the premium is resolved through :func:`aggregate_api.pricing._coc_for_premium`,
81
+ which completes the pentagon and reads the cost of capital off it.
82
+
83
+ Through 1.0.0a100 that identity was written out here::
84
+
85
+ L = E[min(X, a)] the limited expected loss
86
+ M = premium - L the margin
87
+ Q = a - premium the capital
88
+ coc = M / Q
89
+
90
+ which was this repo deciding what a price means, and it predated
91
+ ``price_pentagon`` being reachable from the api. The pentagon is one library
92
+ call and the identity stays upstream where it belongs.
93
+
94
+ **That hand form was also wrong, in a small way.** ``prob_loss_assets``
95
+ snaps the asset level to the loss grid and reports ``L`` there, while ``Q``
96
+ was computed from the caller's raw request, so the margin and the capital
97
+ came off two different asset levels and the cost of capital they implied
98
+ belonged to no consistent pentagon. On a 50 claim lognormal book at
99
+ ``log2=13``, a request for 15625.068 snaps to 15624.0 and the two readings
100
+ are 0.13334437 against 0.13335957, about 1.1e-4 relative; it scales with the
101
+ distance from a grid point. Panel 2 is calibrated on one asset level now,
102
+ the one the band is drawn at.
103
+
104
+ Returns ``False`` rather than raising, in three cases, and each leaves the
105
+ document honestly one-panelled instead of falsely two:
106
+
107
+ * **no asset cap.** Capital is unbounded, so there is no cost of capital to
108
+ calibrate to. The envelope in panel 1 is still meaningful.
109
+ * **degenerate margin or capital.** ``Bounds`` already refuses a premium
110
+ below the mean or above the cap; this catches the boundary where premium
111
+ equals the cap and capital is zero, which now reaches this function as the
112
+ library's own refusal out of the pentagon rather than as a comparison here.
113
+ * **the calibration itself declines**, which it does on a book where a mass
114
+ distortion cannot be fitted.
115
+
116
+ All three are rarer than they were. Since a100 the app carries a real
117
+ calibration into this pane rather than opening on ``mean * 1.25``, and an
118
+ implied ``(P, a)`` has ``M > 0`` and ``Q > 0`` by construction, so panel 2
119
+ draws where it used to vanish. They stay because a hand typed premium can
120
+ still hit them.
121
+
122
+ Notes
123
+ -----
124
+ This **mutates** the cached object: ``calibrate_distortions`` writes
125
+ ``distortions``, ``distortion_df`` and ``calibration_df`` onto it. That is
126
+ the library's contract for the method and already how the pricing runners
127
+ use it, so the object is no more shared-mutable than before; it is worth
128
+ knowing that a Bounds request leaves a calibration behind.
129
+ """
130
+ import math
131
+
132
+ from .pricing import _coc_for_premium
133
+
134
+ if assets is None or not math.isfinite(float(assets)):
135
+ return False
136
+ assets = float(assets)
137
+ try:
138
+ coc = _coc_for_premium(obj, {"a": assets}, float(premium), None)
139
+ except Exception: # noqa: BLE001 -- an object that cannot answer gets one panel
140
+ return False
141
+ # A premium at or above the cap is negative capital and a premium below the
142
+ # limited expected loss is a negative margin. The library reports either as
143
+ # a non-positive cost of capital, so one test covers both boundaries.
144
+ if not (coc is not None and coc > 0):
145
+ return False
146
+ try:
147
+ obj.calibrate_distortions(coc, a=assets)
148
+ except Exception: # noqa: BLE001 -- reported by the document having one panel
149
+ return False
150
+ return bool(getattr(obj, "distortions", None))
151
+
152
+
153
+ def _require_risk(obj: Any) -> Any:
154
+ """Resolve to the risk the bounds classes take, or reject.
155
+
156
+ The library's own accepted set, not a kind list of ours: ``Bounds`` takes a
157
+ ``Portfolio``, an ``Aggregate``, a Series or a DataFrame, and the two the
158
+ api can hold are the first two.
159
+
160
+ A P&L resolves to its wrapped engine here, mirroring the library's own
161
+ unwrap (aggregate 1.0.0a375, [Bounds-PnL-Engine]). Resolving at the door
162
+ rather than leaning on the library's is deliberate: the envelope's second
163
+ panel calibrates onto and reads off the object handed to ``Bounds``
164
+ (``bounds._obj.distortions``), and a P&L carries no
165
+ ``calibrate_distortions`` of its own, so the engine has to be the object
166
+ the whole route works with. A kernel P&L wraps no engine and is rejected
167
+ like any other wrong kind.
168
+ """
169
+ engine = getattr(obj, "engine", None)
170
+ if not isinstance(obj, (Aggregate, Portfolio)) and \
171
+ isinstance(engine, (Aggregate, Portfolio)):
172
+ obj = engine
173
+ if not isinstance(obj, (Aggregate, Portfolio)):
174
+ raise ValueError("pricing bounds apply to an Aggregate or a Portfolio")
175
+ return obj
176
+
177
+
178
+ def run_envelope(
179
+ obj: Any,
180
+ *,
181
+ premium: float,
182
+ assets: float | None = None,
183
+ n_resamples: int = DEFAULT_RESAMPLES,
184
+ ) -> tuple[bytes, str]:
185
+ """The envelope as a chart document: canonical bytes and their hash.
186
+
187
+ Panel one is the band of admissible prices with the bracketing cloud inside
188
+ it. Panel two puts the named distortions, calibrated to this request's own
189
+ premium, on the same band, so you can see which part of the feasible space
190
+ each one occupies; it is absent rather than empty when the calibration does
191
+ not come off, and :func:`_calibrate_for_envelope` gives the three cases.
192
+
193
+ Notes
194
+ -----
195
+ **Nothing is drawn here any more.** Through a59 this rendered a matplotlib
196
+ figure and shipped SVG or PNG bytes, and it was the last thing in the api
197
+ importing matplotlib. The emitter ``charts.chart_envelope`` now publishes
198
+ the same picture as a document, so the api serves semantics and the browser
199
+ realizes them, which is the same split every other chart already keeps.
200
+
201
+ Two consequences worth stating. The reader can zoom and read values off the
202
+ band rather than squinting at a fixed raster. And the figure is no longer
203
+ two pictures maintained apart: the matplotlib compositor and the browser
204
+ now draw the same document, so they cannot disagree about what the envelope
205
+ is.
206
+
207
+ The document arrives with **two panels where the figure had three**. That is
208
+ upstream's decision and the right one: the five calibrated distortions used
209
+ to be split across the last two panels, which was an accident of the order
210
+ they were added rather than a reading anyone wants, since the question is
211
+ how the five compare and five curves on one band answer it.
212
+
213
+ Parameters
214
+ ----------
215
+ obj : Aggregate or Portfolio
216
+ The risk.
217
+ premium : float
218
+ The target premium the distortions are held to. Must sit above the
219
+ expected loss and at or below the asset cap, which the library enforces.
220
+ assets : float, optional
221
+ Asset cap; the class then bounds prices of ``min(X, a)``. Unbounded when
222
+ omitted.
223
+ n_resamples : int
224
+ Bracketing curves drawn inside the band.
225
+
226
+ Returns
227
+ -------
228
+ (bytes, str)
229
+ Canonical document JSON and its 12-hex content hash, the second so a
230
+ caller can set an ETag without parsing the body back.
231
+
232
+ Raises
233
+ ------
234
+ ValueError
235
+ Wrong kind of object, or a premium the library will not accept.
236
+ """
237
+ obj = _require_risk(obj)
238
+
239
+ kwargs = {"premium": float(premium)}
240
+ if assets is not None:
241
+ kwargs["a"] = float(assets)
242
+ bounds = Bounds(obj, **kwargs)
243
+
244
+ # Before the document, not after: the emitter reads the calibration off the
245
+ # priced object, so panel two exists only if this has already run.
246
+ _calibrate_for_envelope(obj, premium, assets)
247
+
248
+ doc = agg_charts.build_chart_doc(bounds, "envelope",
249
+ n_resamples=int(n_resamples))
250
+ return agg_charts.canonical_json(doc), doc.hash
251
+
252
+
253
+ def run_allocation(obj: Any, *, premium: float, assets: float | None = None,
254
+ ir: bool = False) -> dict:
255
+ """Per-unit natural-allocation ranges consistent with a total premium.
256
+
257
+ Portfolio only, and not by our choice: the calculation reads the
258
+ ``exeqa_*`` columns, which are what a portfolio's density frame carries and
259
+ a single aggregate has no analogue of.
260
+
261
+ Returns one row per unit with ``lower``, ``upper`` and ``width``. The width
262
+ is the reading: it is how much of each unit's price is decided by the choice
263
+ of distortion rather than by the total premium.
264
+
265
+ Parameters
266
+ ----------
267
+ obj : Portfolio
268
+ premium : float
269
+ Total premium for the book.
270
+ assets : float, optional
271
+ Asset cap.
272
+ ir : bool
273
+ Also return the table document for the static view.
274
+
275
+ Returns
276
+ -------
277
+ dict
278
+ Matches :class:`aggregate_api.models.BoundsResponse`.
279
+ """
280
+ from .serializers import frame_to_payload, reset_index_safe
281
+
282
+ if not isinstance(obj, Portfolio):
283
+ raise ValueError("allocation bounds apply to a Portfolio")
284
+ engine = AllocationBounds(obj, **({"a": float(assets)} if assets else {}))
285
+ frame = engine.bounds(float(premium))
286
+ return {
287
+ "premium": float(premium),
288
+ "table": frame_to_payload(reset_index_safe(frame)),
289
+ "ir": {"table": _document(frame)} if ir else None,
290
+ }
291
+
292
+
293
+ def run_pricing_bounds(obj: Any, *, premium: float, targets: dict,
294
+ assets: float | None = None, ir: bool = False) -> dict:
295
+ """Price ranges for other risks, given this one priced to ``premium``.
296
+
297
+ The cross-pricing question. Some distortion prices this object to the
298
+ target; every such distortion also prices anything else, and this reports
299
+ how wide that second price can be.
300
+
301
+ Parameters
302
+ ----------
303
+ obj : Aggregate or Portfolio
304
+ The reference risk carrying the pricing constraint.
305
+ premium : float
306
+ What the reference is priced to.
307
+ targets : dict of str to object
308
+ The risks whose ranges are wanted, by display name.
309
+ assets : float, optional
310
+ Asset cap, applied to both sides.
311
+ ir : bool
312
+ Also return the table document for the static view.
313
+
314
+ Returns
315
+ -------
316
+ dict
317
+ Matches :class:`aggregate_api.models.BoundsResponse`.
318
+ """
319
+ from .serializers import frame_to_payload, reset_index_safe
320
+
321
+ obj = _require_risk(obj)
322
+ if not targets:
323
+ raise ValueError("name a risk to price against this one")
324
+ engine = PricingBounds(obj, targets,
325
+ **({"a": float(assets)} if assets else {}))
326
+ frame = engine.bounds(float(premium))
327
+ return {
328
+ "premium": float(premium),
329
+ "table": frame_to_payload(reset_index_safe(frame)),
330
+ "ir": {"table": _document(frame)} if ir else None,
331
+ }
aggregate_api/cache.py ADDED
@@ -0,0 +1,319 @@
1
+ """In-memory LRU cache for built ``Aggregate`` / ``Portfolio`` objects.
2
+
3
+ The api's "Option X" cache design: a single, bounded, LRU dict
4
+ keyed by *content hash* of the DecL program. Same DecL +
5
+ ``log2`` + ``bs`` → same id → same cached object. Building is
6
+ idempotent for the cache lifetime of one server process.
7
+
8
+ Why bother
9
+ ----------
10
+
11
+ Building a moderately sized portfolio takes seconds; FFTs over
12
+ 2**18 points across many lines aren't free. The cache lets the
13
+ SPA's "per-button-fetch UX" (info, summary, stats_df, plot,
14
+ kappa, pricing) all run as O(1) lookups against the prebuilt
15
+ object, with the heavy lift paid only once per (decl, log2, bs).
16
+
17
+ Why not lru_cache
18
+ -----------------
19
+
20
+ ``functools.lru_cache`` is per-function and doesn't expose the
21
+ inspection / list / delete operations the api needs
22
+ (``GET /v1/objects``, ``DELETE /v1/objects/{id}``). An
23
+ ``OrderedDict`` does, and the move-to-end / popitem(last=False)
24
+ pair gives plain LRU semantics in ~10 lines.
25
+
26
+ Thread-safety
27
+ -------------
28
+
29
+ A single ``threading.Lock`` wraps every mutation. The cache is
30
+ small (≤50 entries by default), so coarse-grained locking is
31
+ cheaper than any per-entry alternative.
32
+
33
+ Why not weakref
34
+ ---------------
35
+
36
+ We want explicit bounded retention, not "alive until nobody
37
+ holds a reference" -- the SPA holds the id, not the object,
38
+ so weakref would have no live referents and evict immediately.
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ import hashlib
44
+ import re
45
+ import threading
46
+ from collections import OrderedDict
47
+ from dataclasses import dataclass, field
48
+ from datetime import datetime
49
+ from typing import Any
50
+
51
+
52
+ # DecL comments run from ``#`` to end-of-line. We strip them before
53
+ # hashing so trivially commented-out / annotated variants of the same
54
+ # program produce the same object id.
55
+ _COMMENT = re.compile(r"#[^\n]*")
56
+
57
+
58
+ def canonicalize_decl(decl: str) -> str:
59
+ """Strip comments and trailing whitespace for content hashing.
60
+
61
+ Note that interior whitespace is *preserved* -- two programs
62
+ that differ only in indentation hash differently. A stronger
63
+ normalization (whitespace-insensitive via Lark tree round-trip)
64
+ is flagged in the plan as a future enhancement.
65
+
66
+ Parameters
67
+ ----------
68
+ decl : str
69
+ Raw DecL source.
70
+
71
+ Returns
72
+ -------
73
+ str
74
+ Canonicalized form suitable for hashing.
75
+ """
76
+ no_comments = _COMMENT.sub("", decl)
77
+ return no_comments.rstrip()
78
+
79
+
80
+ def object_id(decl: str, log2: int, bs: float) -> str:
81
+ """Compute the cache id for a (decl, log2, bs) triple.
82
+
83
+ 16-hex-char prefix of SHA-256. Collision probability is
84
+ negligible at our scale (target ≤ 10k builds per session).
85
+
86
+ Parameters
87
+ ----------
88
+ decl : str
89
+ Already-canonicalized DecL.
90
+ log2, bs : int, float
91
+ Build knobs that affect the resulting object.
92
+ """
93
+ # ``bs!r`` because bs is a float; repr() pins exact bit pattern
94
+ # so e.g. 0.1 and 0.1000000000001 hash differently (they would
95
+ # build differently too).
96
+ payload = f"{decl}|{log2}|{bs!r}".encode("utf-8")
97
+ return hashlib.sha256(payload).hexdigest()[:16]
98
+
99
+
100
+ def qualified_object_id(session_id: str, decl: str, log2: int, bs: float) -> str:
101
+ """The cache id for a program private to one session.
102
+
103
+ Same hash, one more field. A program that resolved a name its own session
104
+ declared does not mean the same thing to anyone else, so it cannot share a
105
+ slot with the identical text typed by a different session: Alice's
106
+ ``port Book agg.Line`` and Bob's are the same eight words over different
107
+ inners, and one of them would be served the other's answer.
108
+
109
+ Parameters
110
+ ----------
111
+ session_id : str
112
+ The caller's session, from ``aggregate_api.sessions``.
113
+ decl : str
114
+ Already-canonicalized DecL.
115
+ log2, bs : int, float
116
+ Build knobs, as in :func:`object_id`.
117
+
118
+ Returns
119
+ -------
120
+ str
121
+
122
+ Notes
123
+ -----
124
+ Which of the two builders a request uses is decided by what its parse
125
+ resolved, not by anything in the text: see ``_cache_key`` in
126
+ ``routes/objects.py``. A self-contained program, or one leaning only on
127
+ library entries, keeps the shared key and is built once for everybody, which
128
+ is the case a room on one hero example is made of.
129
+ """
130
+ payload = f"{session_id}|{decl}|{log2}|{bs!r}".encode("utf-8")
131
+ return hashlib.sha256(payload).hexdigest()[:16]
132
+
133
+
134
+ @dataclass
135
+ class CacheEntry:
136
+ """One slot in the LRU.
137
+
138
+ ``obj`` is the live ``Aggregate`` or ``Portfolio`` -- the SPA
139
+ never reaches it directly, but the api's per-button endpoints
140
+ pull data off it on demand.
141
+
142
+ The other fields are metadata returned by
143
+ ``GET /v1/objects/{id}`` and used by the build endpoint to
144
+ fill out the response without consulting the underlying object.
145
+
146
+ Thread-safety of ``obj``
147
+ ------------------------
148
+
149
+ ``lock`` serializes *reads of the built object*, which sounds unnecessary
150
+ and is not. An ``Aggregate`` and a ``Portfolio`` materialize several of
151
+ their frames lazily and cache them on the instance, so a "read" is a write
152
+ the first time. Two requests that touch the same object concurrently can
153
+ therefore see a half-built frame.
154
+
155
+ That is not hypothetical. Fetching ``unit_density_df`` and ``tail_df`` for
156
+ one Portfolio at the same moment (which the Overview exhibit does, in a
157
+ single ``Promise.all``) raised
158
+ ``KeyError: "['F', 'S'] not in index"`` from inside
159
+ ``Portfolio.unit_density_df`` roughly half the time on a cold object, and
160
+ never once the frames were warm. FastAPI runs sync handlers in a thread
161
+ pool, so those two requests really are two threads on one object.
162
+
163
+ The lock is per entry rather than global so unrelated objects still serve
164
+ in parallel, and contention is confined to the first access of each frame:
165
+ afterwards every read is a cache hit and the critical section is trivial.
166
+ """
167
+
168
+ obj: Any
169
+ decl: str
170
+ log2: int
171
+ bs: float
172
+ kind: str
173
+ name: str
174
+ created_at: datetime
175
+ # What the library said while building this object, at WARNING and above.
176
+ # Stored here rather than returned once and forgotten, because a warning
177
+ # ("this splice is coarse", "the grid clips the tail") describes the
178
+ # OBJECT, not the request that happened to build it. Kept on the entry, a
179
+ # cache hit reports the same warnings as the miss that made it, so
180
+ # rebuilding a program does not silently lose the reason to worry about it.
181
+ notes: list[str] = field(default_factory=list)
182
+ # Guards reads *of the object*, not of this dataclass. See below.
183
+ lock: threading.Lock = field(default_factory=threading.Lock, repr=False,
184
+ compare=False)
185
+
186
+
187
+ class ObjectCache:
188
+ """Bounded LRU keyed by content hash.
189
+
190
+ Methods are coarsely thread-safe via a single lock. Reads
191
+ move the entry to MRU; writes evict the LRU when full. The
192
+ ``__contains__`` check does *not* move-to-MRU (peek, not
193
+ promote) -- the standard idiom for "is this in cache" before
194
+ a separate get/put decision.
195
+ """
196
+
197
+ def __init__(self, max_entries: int = 50) -> None:
198
+ self._max = max_entries
199
+ self._store: OrderedDict[str, CacheEntry] = OrderedDict()
200
+ self._lock = threading.Lock()
201
+ # Instrumentation for GET /v1/status. Incremented under the same lock
202
+ # that guards the store, so a reading of them is consistent with the
203
+ # entry list read in the same call and no second lock is introduced.
204
+ # They count this instance, and ``_get_cache`` replaces the instance
205
+ # when ``cache_max`` changes, so the page reports them against the
206
+ # process start time and that is honest for every deployment which does
207
+ # not edit its config while running.
208
+ self.hits = 0
209
+ self.misses = 0
210
+ self.puts = 0
211
+ self.evictions = 0
212
+ self.deletes = 0
213
+
214
+ def get(self, oid: str) -> CacheEntry | None:
215
+ """Return the cached entry for ``oid`` (moving it to MRU) or None."""
216
+ with self._lock:
217
+ entry = self._store.get(oid)
218
+ if entry is None:
219
+ self.misses += 1
220
+ return None
221
+ self.hits += 1
222
+ # ``move_to_end`` is the OrderedDict primitive that makes
223
+ # LRU work; entries closest to the front are evicted first.
224
+ self._store.move_to_end(oid)
225
+ return entry
226
+
227
+ def put(self, oid: str, entry: CacheEntry) -> None:
228
+ """Insert or refresh ``entry`` under ``oid``, evicting LRU if full."""
229
+ with self._lock:
230
+ self.puts += 1
231
+ if oid in self._store:
232
+ self._store.move_to_end(oid)
233
+ self._store[oid] = entry
234
+ return
235
+ self._store[oid] = entry
236
+ while len(self._store) > self._max:
237
+ # ``last=False`` pops the *least* recently used (front).
238
+ self._store.popitem(last=False)
239
+ self.evictions += 1
240
+
241
+ def delete(self, oid: str) -> bool:
242
+ """Remove ``oid`` from the cache; return True if it was present."""
243
+ with self._lock:
244
+ dropped = self._store.pop(oid, None) is not None
245
+ if dropped:
246
+ self.deletes += 1
247
+ return dropped
248
+
249
+ def list(self) -> list[CacheEntry]:
250
+ """Return a snapshot of cache contents, MRU last.
251
+
252
+ Returned list is a copy of the internal values -- safe to
253
+ iterate without holding the lock.
254
+ """
255
+ with self._lock:
256
+ return list(self._store.values())
257
+
258
+ def items(self) -> list[tuple[str, CacheEntry]]:
259
+ """Snapshot of ``(id, entry)`` pairs, **LRU first**.
260
+
261
+ Notes
262
+ -----
263
+ LRU first because the caller that wants ids as well as entries is
264
+ ``GET /v1/status``, whose cache table answers "what will I lose next",
265
+ and the front of the ``OrderedDict`` is exactly that. :meth:`list`
266
+ keeps its MRU-last ordering, which is what its callers already read.
267
+
268
+ Exists so callers stop reaching into ``_store`` directly, which two of
269
+ them in ``routes/objects.py`` do, unlocked. The pairs are a copy, so
270
+ iterating them is safe without the lock; the entries themselves are the
271
+ live objects, as :meth:`list` also returns.
272
+ """
273
+ with self._lock:
274
+ return list(self._store.items())
275
+
276
+ def clear(self) -> None:
277
+ """Drop every entry. Used by tests."""
278
+ with self._lock:
279
+ self._store.clear()
280
+
281
+ def __contains__(self, oid: str) -> bool:
282
+ with self._lock:
283
+ return oid in self._store
284
+
285
+ def __len__(self) -> int:
286
+ with self._lock:
287
+ return len(self._store)
288
+
289
+ def stats(self) -> dict:
290
+ """Counters and occupancy, read in one critical section.
291
+
292
+ Returns
293
+ -------
294
+ dict
295
+ ``entries``, ``max``, ``hits``, ``misses``, ``hit_rate``, ``puts``,
296
+ ``evictions`` and ``deletes``. ``hit_rate`` is ``None`` rather than
297
+ zero before the first lookup, because "no requests yet" and "every
298
+ request missed" are different facts and a status page that renders
299
+ them the same way is misreporting a cold start as a failure.
300
+
301
+ Notes
302
+ -----
303
+ One call rather than eight attribute reads so the numbers a caller
304
+ prints together were true together. An eviction landing between two
305
+ reads would otherwise show a hit count that does not reconcile with the
306
+ entry list.
307
+ """
308
+ with self._lock:
309
+ looks = self.hits + self.misses
310
+ return {
311
+ "entries": len(self._store),
312
+ "max": self._max,
313
+ "hits": self.hits,
314
+ "misses": self.misses,
315
+ "hit_rate": round(self.hits / looks, 4) if looks else None,
316
+ "puts": self.puts,
317
+ "evictions": self.evictions,
318
+ "deletes": self.deletes,
319
+ }