jstdata 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,357 @@
1
+ """Barebones terminal bar charts for discover preview and rank boards."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import Counter, defaultdict
6
+ from datetime import datetime, timedelta
7
+ from statistics import mean
8
+ from typing import AbstractSet, Iterable, Optional, Sequence
9
+
10
+ from ..models import Entity, TimeSeries
11
+
12
+ FREQ_TIEBREAK = ("Annual", "Quarterly", "Monthly", "Daily", "Intraday")
13
+ _EIGHTHS = " ▁▂▃▄▅▆▇█"
14
+ BAR_HEIGHT = 3
15
+ LABEL_WIDTH = 12
16
+ VALUE_WIDTH = 7
17
+
18
+
19
+ def pick_frequency(series_list: Sequence[TimeSeries]) -> Optional[str]:
20
+ """Most common frequency; ties prefer coarser frequencies."""
21
+ counts = Counter(
22
+ ts.series.frequency for ts in series_list if ts.series.frequency
23
+ )
24
+ if not counts:
25
+ return None
26
+ top = max(counts.values())
27
+ tied = [freq for freq, n in counts.items() if n == top]
28
+ rank = {name: i for i, name in enumerate(FREQ_TIEBREAK)}
29
+ tied.sort(key=lambda freq: rank.get(freq, 99))
30
+ return tied[0]
31
+
32
+
33
+ def pick_series_per_entity(
34
+ series_list: Sequence[TimeSeries],
35
+ entity_ids: Sequence[str],
36
+ frequency: str,
37
+ ) -> dict[str, Optional[TimeSeries]]:
38
+ """One series per entity at ``frequency`` (most in-window obs, then id)."""
39
+ by_entity: dict[str, list[TimeSeries]] = defaultdict(list)
40
+ wanted = set(entity_ids)
41
+ for ts in series_list:
42
+ if ts.series.frequency != frequency:
43
+ continue
44
+ matched = [ent.id for ent in ts.series.entities if ent.id in wanted]
45
+ if not matched and len(entity_ids) == 1 and not ts.series.entities:
46
+ matched = [entity_ids[0]]
47
+ for eid in matched:
48
+ by_entity[eid].append(ts)
49
+
50
+ picked: dict[str, Optional[TimeSeries]] = {}
51
+ for eid in entity_ids:
52
+ candidates = by_entity.get(eid, [])
53
+ if not candidates:
54
+ picked[eid] = None
55
+ continue
56
+ candidates.sort(key=lambda ts: (-len(ts.observations), ts.series.id))
57
+ picked[eid] = candidates[0]
58
+ return picked
59
+
60
+
61
+ def _as_naive(ts: datetime) -> datetime:
62
+ if ts.tzinfo is not None:
63
+ return ts.replace(tzinfo=None)
64
+ return ts
65
+
66
+
67
+ def downsample(
68
+ observations: Sequence,
69
+ n_bins: int,
70
+ ) -> list[tuple[datetime, Optional[float]]]:
71
+ """Bin observations into at most ``n_bins`` (mean per bin; gaps are None)."""
72
+ if n_bins <= 0 or not observations:
73
+ return []
74
+ obs = sorted(observations, key=lambda o: _as_naive(o.observation_timestamp))
75
+ if len(obs) <= n_bins:
76
+ return [(_as_naive(o.observation_timestamp), o.value) for o in obs]
77
+ start = _as_naive(obs[0].observation_timestamp)
78
+ end = _as_naive(obs[-1].observation_timestamp)
79
+ span = (end - start).total_seconds() or 1.0
80
+ buckets: list[list[float]] = [[] for _ in range(n_bins)]
81
+ for o in obs:
82
+ t = (_as_naive(o.observation_timestamp) - start).total_seconds() / span
83
+ idx = min(n_bins - 1, max(0, int(t * n_bins)))
84
+ buckets[idx].append(o.value)
85
+ out: list[tuple[datetime, Optional[float]]] = []
86
+ for i, vals in enumerate(buckets):
87
+ stamp = start + timedelta(seconds=span * (i + 0.5) / n_bins)
88
+ out.append((stamp, mean(vals) if vals else None))
89
+ return out
90
+
91
+
92
+ def _column_stack(frac: Optional[float], height: int) -> list[str]:
93
+ """Top-to-bottom characters for one time column."""
94
+ if frac is None or frac <= 0:
95
+ return [" "] * height
96
+ total = height * 8
97
+ filled = max(1, min(total, int(round(frac * total))))
98
+ cells: list[str] = []
99
+ remaining = filled
100
+ for _ in range(height):
101
+ if remaining >= 8:
102
+ cells.append("█")
103
+ remaining -= 8
104
+ elif remaining > 0:
105
+ cells.append(_EIGHTHS[remaining])
106
+ remaining = 0
107
+ else:
108
+ cells.append(" ")
109
+ return list(reversed(cells))
110
+
111
+
112
+ def _scale(values: Iterable[Optional[float]]) -> tuple[float, float]:
113
+ nums = [v for v in values if v is not None]
114
+ if not nums:
115
+ return 0.0, 1.0
116
+ lo, hi = min(nums), max(nums)
117
+ if lo >= 0:
118
+ lo = 0.0
119
+ if hi == lo:
120
+ hi = lo + 1.0
121
+ return lo, hi
122
+
123
+
124
+ def _label(text: str, width: int = LABEL_WIDTH) -> str:
125
+ text = text or ""
126
+ if len(text) <= width:
127
+ return text.ljust(width)
128
+ return text[: width - 1] + "…"
129
+
130
+
131
+ def _trim(n: float, decimals: int = 2) -> str:
132
+ text = f"{n:.{decimals}f}"
133
+ if "." in text:
134
+ text = text.rstrip("0").rstrip(".")
135
+ return text or "0"
136
+
137
+
138
+ def format_compact(value: float) -> str:
139
+ """Short magnitude for the preview gutter (SI suffixes, no unit)."""
140
+ if value != value: # NaN
141
+ return "—"
142
+ if value == 0:
143
+ return "0"
144
+ sign = "-" if value < 0 else ""
145
+ v = abs(value)
146
+ for thresh, suffix in ((1e12, "T"), (1e9, "B"), (1e6, "M"), (1e3, "k")):
147
+ if v >= thresh:
148
+ return sign + _trim(v / thresh) + suffix
149
+ if v >= 100:
150
+ return sign + _trim(v, 0)
151
+ if v >= 10:
152
+ return sign + _trim(v, 1)
153
+ if v >= 1:
154
+ return sign + _trim(v, 2)
155
+ return sign + _trim(v, 3)
156
+
157
+
158
+ def pick_entity(
159
+ entities: Sequence[Entity],
160
+ taxonomy_ids: Optional[AbstractSet[str]] = None,
161
+ ) -> Optional[Entity]:
162
+ """Prefer a taxonomy member when known; otherwise the first entity."""
163
+ if not entities:
164
+ return None
165
+ if taxonomy_ids:
166
+ for ent in entities:
167
+ if ent.id in taxonomy_ids:
168
+ return ent
169
+ return entities[0]
170
+
171
+
172
+ def last_observation_value(ts: TimeSeries) -> Optional[float]:
173
+ """Chronologically last observation value on a series (rank key)."""
174
+ if not ts.observations:
175
+ return None
176
+ latest = max(
177
+ ts.observations,
178
+ key=lambda o: _as_naive(o.observation_timestamp),
179
+ )
180
+ return latest.value
181
+
182
+
183
+ def last_observation_timestamp(ts: TimeSeries) -> Optional[datetime]:
184
+ if not ts.observations:
185
+ return None
186
+ latest = max(
187
+ ts.observations,
188
+ key=lambda o: _as_naive(o.observation_timestamp),
189
+ )
190
+ return _as_naive(latest.observation_timestamp)
191
+
192
+
193
+ def shared_pane_max(values: Iterable[Optional[float]]) -> float:
194
+ """Positive ceiling for shared-scale bars (0 → pane max)."""
195
+ nums = [v for v in values if v is not None]
196
+ if not nums:
197
+ return 1.0
198
+ hi = max(nums)
199
+ if hi <= 0:
200
+ return 1.0
201
+ return hi
202
+
203
+
204
+ def horizontal_bar(frac: float, width: int) -> str:
205
+ """One-line unicode bar; ``frac`` is 0..1 against a shared max."""
206
+ if width <= 0:
207
+ return ""
208
+ if frac is None or frac <= 0:
209
+ return " " * width
210
+ total = width * 8
211
+ filled = max(1, min(total, int(round(min(1.0, frac) * total))))
212
+ full, rem = divmod(filled, 8)
213
+ cells = ["█"] * full
214
+ if rem and len(cells) < width:
215
+ cells.append(_EIGHTHS[rem])
216
+ while len(cells) < width:
217
+ cells.append(" ")
218
+ return "".join(cells[:width])
219
+
220
+
221
+ def format_rank_line(
222
+ rank: int,
223
+ name: str,
224
+ value: float,
225
+ pane_max: float,
226
+ bar_width: int,
227
+ name_width: int = 18,
228
+ ) -> str:
229
+ """One leaderboard row: rank, name, compact value, shared-scale bar."""
230
+ frac = (value / pane_max) if pane_max > 0 else 0.0
231
+ if value < 0:
232
+ frac = 0.0
233
+ return (
234
+ f"{rank:>3} {_label(name, name_width)} "
235
+ f"{format_compact(value).rjust(VALUE_WIDTH)} "
236
+ f"{horizontal_bar(frac, bar_width)}"
237
+ )
238
+
239
+
240
+ def _axis_line(
241
+ stamps: Sequence[datetime],
242
+ bar_width: int,
243
+ gutter: int = LABEL_WIDTH + 1,
244
+ ) -> str:
245
+ if not stamps or bar_width <= 0:
246
+ return ""
247
+ years = [
248
+ stamps[0].strftime("%Y"),
249
+ stamps[len(stamps) // 2].strftime("%Y"),
250
+ stamps[-1].strftime("%Y"),
251
+ ]
252
+ uniq: list[str] = []
253
+ for year in years:
254
+ if year not in uniq:
255
+ uniq.append(year)
256
+ if bar_width < 4:
257
+ return " " * gutter + "–".join(uniq)
258
+ labels = [
259
+ (0, uniq[0]),
260
+ (max(0, bar_width // 2 - 2), uniq[len(uniq) // 2]),
261
+ (max(0, bar_width - 4), uniq[-1]),
262
+ ]
263
+ line = [" "] * bar_width
264
+ used: set[int] = set()
265
+ for pos, lab in labels:
266
+ if len(lab) > bar_width:
267
+ continue
268
+ pos = min(max(0, pos), bar_width - len(lab))
269
+ if any(pos + i in used for i in range(len(lab))):
270
+ continue
271
+ for i, ch in enumerate(lab):
272
+ line[pos + i] = ch
273
+ used.add(pos + i)
274
+ return " " * gutter + "".join(line)
275
+
276
+
277
+ def render_preview(
278
+ metric_name: str,
279
+ frequency: str,
280
+ units: str,
281
+ entity_labels: dict[str, str],
282
+ picked: dict[str, Optional[TimeSeries]],
283
+ bar_width: int = 48,
284
+ bar_height: int = BAR_HEIGHT,
285
+ ) -> str:
286
+ """Per-series unicode bars with that series' max in a right gutter.
287
+
288
+ Each small-multiple is scaled to its own range so a country with a
289
+ smaller magnitude still shows shape. The compact max is the y-axis:
290
+ the tallest bar in a row is that number.
291
+ """
292
+ plot_width = max(8, bar_width - VALUE_WIDTH - 2)
293
+ binned: dict[str, list[tuple[datetime, Optional[float]]]] = {}
294
+ series_max: dict[str, float] = {}
295
+ all_stamps: list[datetime] = []
296
+ for eid, ts in picked.items():
297
+ if ts is None or not ts.observations:
298
+ binned[eid] = []
299
+ continue
300
+ points = downsample(ts.observations, plot_width)
301
+ binned[eid] = points
302
+ raw = [o.value for o in ts.observations if o.value is not None]
303
+ if raw:
304
+ series_max[eid] = max(raw)
305
+ all_stamps.extend(t for t, v in points if v is not None)
306
+
307
+ header_bits = [metric_name or "metric"]
308
+ if frequency:
309
+ header_bits.append(frequency)
310
+ if units:
311
+ header_bits.append(units)
312
+ if all_stamps:
313
+ header_bits.append(
314
+ f"{min(all_stamps).year}–{max(all_stamps).year}"
315
+ )
316
+ lines = [" · ".join(header_bits), ""]
317
+
318
+ for eid, ts in picked.items():
319
+ name = _label(entity_labels.get(eid, eid))
320
+ points = binned.get(eid) or []
321
+ if not points:
322
+ lines.append(f"{name} (no series)")
323
+ lines.append("")
324
+ continue
325
+ vmin, vmax = _scale(v for _, v in points)
326
+ span = vmax - vmin or 1.0
327
+ columns = []
328
+ for _, value in points:
329
+ if value is None:
330
+ frac = None
331
+ else:
332
+ frac = (value - vmin) / span
333
+ columns.append(_column_stack(frac, bar_height))
334
+ peak = series_max.get(eid)
335
+ max_bit = (
336
+ format_compact(peak).rjust(VALUE_WIDTH) if peak is not None else ""
337
+ )
338
+ mid = bar_height // 2
339
+ for row in range(bar_height):
340
+ prefix = name if row == mid else " " * LABEL_WIDTH
341
+ value_col = max_bit if row == mid and max_bit else " " * VALUE_WIDTH
342
+ bars = "".join(col[row] for col in columns)
343
+ lines.append(f"{prefix} {value_col} {bars}")
344
+ lines.append("")
345
+
346
+ if all_stamps:
347
+ longest = max(binned.values(), key=len) if binned else []
348
+ stamps = [t for t, _ in longest] or sorted(all_stamps)
349
+ lines.append(
350
+ _axis_line(
351
+ stamps,
352
+ min(plot_width, len(stamps)),
353
+ gutter=LABEL_WIDTH + 1 + VALUE_WIDTH + 1,
354
+ )
355
+ )
356
+
357
+ return "\n".join(lines).rstrip() + "\n"