usage-cli 0.29.32__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. adapters/__init__.py +5 -0
  2. adapters/agy.py +68 -0
  3. adapters/claude.py +215 -0
  4. adapters/codex.py +209 -0
  5. adapters/rate_limits.py +76 -0
  6. adapters/registry.py +17 -0
  7. adapters/types.py +139 -0
  8. agy_disk_cache.py +135 -0
  9. agy_loader.py +416 -0
  10. agy_quota_probe.py +748 -0
  11. agy_window_keeper.py +185 -0
  12. analyzer/__init__.py +5 -0
  13. analyzer/aggregator.py +139 -0
  14. analyzer/blocks.py +80 -0
  15. analyzer/diagnoser.py +638 -0
  16. analyzer/insights.py +277 -0
  17. analyzer/persona_loader.py +199 -0
  18. analyzer/reporter.py +989 -0
  19. analyzer/subscription.py +108 -0
  20. burn_rate.py +75 -0
  21. cache_quarantine.py +50 -0
  22. codex_disk_cache.py +227 -0
  23. codex_events.py +136 -0
  24. codex_fork_replay.py +111 -0
  25. codex_loader.py +1426 -0
  26. codex_paths.py +20 -0
  27. critter_frames.py +26 -0
  28. discussion_bridge.py +1196 -0
  29. discussion_cli.py +844 -0
  30. discussion_session.py +622 -0
  31. discussion_usage.py +13 -0
  32. discussion_window.py +955 -0
  33. disk_cache_common.py +132 -0
  34. disk_cache_lifecycle.py +39 -0
  35. doctor.py +452 -0
  36. fsevents_watch.py +207 -0
  37. history_disk_cache.py +110 -0
  38. history_loader.py +416 -0
  39. i18n.py +88 -0
  40. jsonl_limits.py +17 -0
  41. jsonl_utils.py +40 -0
  42. login_item.py +154 -0
  43. main.py +387 -0
  44. menubar.py +1201 -0
  45. menubar_actions.py +204 -0
  46. menubar_agy.py +193 -0
  47. menubar_chrome.py +156 -0
  48. menubar_menu.py +169 -0
  49. menubar_notify.py +102 -0
  50. menubar_popover.py +233 -0
  51. menubar_prefs.py +118 -0
  52. menubar_refresh.py +285 -0
  53. menubar_state.py +1200 -0
  54. menubar_title.py +157 -0
  55. menubar_update.py +123 -0
  56. panel_window.py +78 -0
  57. panel_window_state.py +159 -0
  58. panels/__init__.py +186 -0
  59. panels/base.py +83 -0
  60. panels/dynamic_height.py +140 -0
  61. panels/payload.py +178 -0
  62. panels/web_panel.py +513 -0
  63. panels/window_drag.py +56 -0
  64. prefs.py +44 -0
  65. pricing.py +452 -0
  66. project_resolver.py +112 -0
  67. service_status.py +383 -0
  68. session_hooks.py +1154 -0
  69. setup_app.py +171 -0
  70. setup_hook.py +1011 -0
  71. statusline_settings.py +160 -0
  72. talent_market_bridge.py +243 -0
  73. time_utils.py +24 -0
  74. tui.py +288 -0
  75. tui_sprite.py +206 -0
  76. ui/__init__.py +5 -0
  77. ui/html_report.py +923 -0
  78. ui/report_scripts.py +251 -0
  79. ui/report_styles.py +370 -0
  80. ui/tables.py +888 -0
  81. update_checker.py +156 -0
  82. update_gate.py +66 -0
  83. update_release_notes.py +49 -0
  84. usage_cli-0.29.32.data/data/share/usage/i18n.json +2427 -0
  85. usage_cli-0.29.32.dist-info/METADATA +223 -0
  86. usage_cli-0.29.32.dist-info/RECORD +109 -0
  87. usage_cli-0.29.32.dist-info/WHEEL +5 -0
  88. usage_cli-0.29.32.dist-info/entry_points.txt +3 -0
  89. usage_cli-0.29.32.dist-info/licenses/LICENSE +663 -0
  90. usage_cli-0.29.32.dist-info/top_level.txt +80 -0
  91. usage_cli.py +827 -0
  92. usage_client.py +487 -0
  93. usage_diagnosis_snapshot.py +143 -0
  94. usage_dir_sweeper.py +100 -0
  95. usage_lang.py +79 -0
  96. usage_logging.py +75 -0
  97. usage_notifications.py +96 -0
  98. usage_rate.py +97 -0
  99. usage_session_resume.py +913 -0
  100. usage_statusline.py +810 -0
  101. usage_statusline_agy.py +397 -0
  102. usage_statusline_forwarder.py +88 -0
  103. usage_terse_mode.py +223 -0
  104. usage_terse_reminder.py +151 -0
  105. win_login_item.py +53 -0
  106. window_keeper.py +264 -0
  107. windows_watch.py +443 -0
  108. wintray.py +2014 -0
  109. wintray_menu.py +136 -0
history_loader.py ADDED
@@ -0,0 +1,416 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-only
2
+ # Copyright (C) 2026 lollapalooza <https://github.com/aqua5230>
3
+ #
4
+ # Part of "usage". Free software licensed under the GNU Affero General Public
5
+ # License v3.0 only; see the LICENSE file for full terms and the warranty disclaimer.
6
+
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+ import json
11
+ import logging
12
+ import math
13
+ import os
14
+ import time
15
+ from collections import OrderedDict
16
+ from collections.abc import Iterable
17
+ from dataclasses import dataclass
18
+ from datetime import UTC, datetime, timedelta
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ from disk_cache_lifecycle import (
23
+ flush_caches_if_due,
24
+ needs_cache_seed,
25
+ )
26
+ from disk_cache_lifecycle import (
27
+ flush_caches_on_terminate as _flush_caches_on_terminate,
28
+ )
29
+ from history_disk_cache import flush_caches, seed_caches
30
+ from jsonl_limits import read_bounded_jsonl_line
31
+ from project_resolver import project_from_encoded_path, resolve_project_name
32
+ from time_utils import parse_optional_iso8601_utc
33
+
34
+ logger = logging.getLogger(__name__)
35
+
36
+ # Must comfortably exceed a real user's total *.jsonl file count. A cap at or
37
+ # below that count means every load_entries() call evicts and re-parses files
38
+ # that were just cached last refresh (LRU thrashing) — measured 512 capped at
39
+ # 640 real project files into a permanent 3+ second full-reparse every single
40
+ # call, even with per-file incremental caching working correctly in isolation.
41
+ _FILE_CACHE_MAXSIZE = 4096
42
+
43
+
44
+ @dataclass(slots=True)
45
+ class _FileCacheEntry:
46
+ mtime: float
47
+ size: int
48
+ entries: list[UsageEntry]
49
+ confirmed_offset: int
50
+ confirmed_prefix_digest: bytes
51
+
52
+
53
+ _file_cache: OrderedDict[Path, _FileCacheEntry] = OrderedDict()
54
+
55
+ CLAUDE_PROJECTS_DIR = Path(os.path.expanduser("~/.claude/projects"))
56
+ HISTORY_CACHE_PATH = Path(os.path.expanduser("~/.usage/history_jsonl_cache.json"))
57
+ _HISTORY_JSONL_CACHE_SCHEMA = 2
58
+ _disk_cache_seeded = False
59
+ _DISK_CACHE_FLUSH_INTERVAL_S = 300.0
60
+ _disk_cache_dirty = False
61
+ _last_disk_cache_flush_at: float | None = None
62
+ _monotonic = time.monotonic
63
+
64
+
65
+ @dataclass(slots=True)
66
+ class UsageEntry:
67
+ timestamp: datetime
68
+ session_id: str
69
+ message_id: str
70
+ request_id: str
71
+ model: str
72
+ input_tokens: int
73
+ output_tokens: int
74
+ cache_creation_tokens: int
75
+ cache_read_tokens: int
76
+ cost_usd: float | None
77
+ project: str
78
+
79
+ @property
80
+ def total_tokens(self) -> int:
81
+ return (
82
+ self.input_tokens
83
+ + self.output_tokens
84
+ + self.cache_creation_tokens
85
+ + self.cache_read_tokens
86
+ )
87
+
88
+ @property
89
+ def active_tokens(self) -> int:
90
+ return self.input_tokens + self.output_tokens + self.cache_creation_tokens
91
+
92
+
93
+ def load_entries(
94
+ hours_back: int = 0,
95
+ *,
96
+ jsonl_paths: Iterable[Path] | None = None,
97
+ ) -> list[UsageEntry]:
98
+ global _disk_cache_dirty
99
+
100
+ _seed_caches_from_disk()
101
+
102
+ entries: list[UsageEntry] = []
103
+ seen: set[str] = set()
104
+ cutoff = datetime.now(UTC) - timedelta(hours=hours_back) if hours_back > 0 else None
105
+
106
+ if jsonl_paths is None and not CLAUDE_PROJECTS_DIR.is_dir():
107
+ return []
108
+
109
+ cutoff_ts = cutoff.timestamp() if cutoff else None
110
+ paths = (
111
+ tuple(CLAUDE_PROJECTS_DIR.rglob("*.jsonl"))
112
+ if jsonl_paths is None
113
+ else tuple(jsonl_paths)
114
+ )
115
+ file_cache_snapshot = {
116
+ str(path): (entry.mtime, entry.size) for path, entry in _file_cache.items()
117
+ }
118
+ for jsonl_path in paths:
119
+ if cutoff_ts is not None:
120
+ try:
121
+ if jsonl_path.stat().st_mtime < cutoff_ts:
122
+ continue
123
+ except OSError as exc:
124
+ logger.warning("failed to stat Claude project log %s: %s", jsonl_path, exc)
125
+ continue
126
+ project = _project_from_path(jsonl_path)
127
+ _load_file(jsonl_path, project, cutoff, seen, entries)
128
+
129
+ if {
130
+ str(path): (entry.mtime, entry.size) for path, entry in _file_cache.items()
131
+ } != file_cache_snapshot:
132
+ _disk_cache_dirty = True
133
+ _flush_caches_to_disk()
134
+ elif _disk_cache_dirty:
135
+ _flush_caches_to_disk()
136
+
137
+ entries.sort(key=lambda entry: entry.timestamp)
138
+ return entries
139
+
140
+
141
+ def _seed_caches_from_disk() -> None:
142
+ global _disk_cache_seeded
143
+
144
+ if not needs_cache_seed(_disk_cache_seeded):
145
+ return
146
+ _disk_cache_seeded = True
147
+ seed_caches(
148
+ HISTORY_CACHE_PATH,
149
+ _HISTORY_JSONL_CACHE_SCHEMA,
150
+ _FILE_CACHE_MAXSIZE,
151
+ _file_cache,
152
+ )
153
+
154
+
155
+ def _flush_caches_to_disk(*, force: bool = False) -> None:
156
+ global _disk_cache_dirty, _last_disk_cache_flush_at
157
+
158
+ _disk_cache_dirty, _last_disk_cache_flush_at = flush_caches_if_due(
159
+ _disk_cache_dirty,
160
+ _last_disk_cache_flush_at,
161
+ _monotonic,
162
+ _DISK_CACHE_FLUSH_INTERVAL_S,
163
+ lambda: flush_caches(
164
+ HISTORY_CACHE_PATH,
165
+ _HISTORY_JSONL_CACHE_SCHEMA,
166
+ _file_cache,
167
+ ),
168
+ force=force,
169
+ )
170
+
171
+
172
+ def flush_caches_on_terminate() -> None:
173
+ """Best-effort persistence of cache changes still waiting for the throttle."""
174
+ _flush_caches_on_terminate(lambda: _flush_caches_to_disk(force=True))
175
+
176
+
177
+ def _load_file(
178
+ path: Path,
179
+ project: str,
180
+ cutoff: datetime | None,
181
+ seen: set[str],
182
+ entries: list[UsageEntry],
183
+ ) -> None:
184
+ try:
185
+ st = path.stat()
186
+ except OSError as exc:
187
+ logger.warning("failed to stat Claude project log %s: %s", path, exc)
188
+ return
189
+
190
+ cached = _file_cache.get(path)
191
+ if cached is not None and cached.mtime == st.st_mtime and cached.size == st.st_size:
192
+ _file_cache.move_to_end(path)
193
+ for entry in cached.entries:
194
+ if cutoff is not None and entry.timestamp < cutoff:
195
+ continue
196
+ dedup_key = _dedup_key(entry)
197
+ if dedup_key in seen:
198
+ continue
199
+ seen.add(dedup_key)
200
+ entries.append(entry)
201
+ return
202
+
203
+ refreshed = _refresh_cache(path, st, project, cached)
204
+ if refreshed is None:
205
+ return
206
+
207
+ if path not in _file_cache and len(_file_cache) >= _FILE_CACHE_MAXSIZE:
208
+ _file_cache.popitem(last=False)
209
+ _file_cache[path] = refreshed
210
+
211
+ for entry in refreshed.entries:
212
+ if cutoff is not None and entry.timestamp < cutoff:
213
+ continue
214
+ dedup_key = _dedup_key(entry)
215
+ if dedup_key in seen:
216
+ continue
217
+ seen.add(dedup_key)
218
+ entries.append(entry)
219
+
220
+
221
+ def _refresh_cache(
222
+ path: Path,
223
+ st: os.stat_result,
224
+ project: str,
225
+ cached: _FileCacheEntry | None,
226
+ ) -> _FileCacheEntry | None:
227
+ prefix_hasher = (
228
+ _confirmed_prefix_hasher(path, cached)
229
+ if cached is not None and st.st_size >= cached.confirmed_offset and st.st_size > cached.size
230
+ else None
231
+ )
232
+ if prefix_hasher is not None:
233
+ assert cached is not None
234
+ incremental_entries = list(cached.entries)
235
+ try:
236
+ with path.open("rb") as file:
237
+ file.seek(cached.confirmed_offset)
238
+ confirmed_offset = _parse_complete_lines(
239
+ file,
240
+ project,
241
+ incremental_entries,
242
+ prefix_hasher,
243
+ cached.confirmed_offset,
244
+ )
245
+ except OSError as exc:
246
+ logger.warning("failed to read Claude project log %s: %s", path, exc)
247
+ return None
248
+ return _FileCacheEntry(
249
+ mtime=st.st_mtime,
250
+ size=st.st_size,
251
+ entries=incremental_entries,
252
+ confirmed_offset=confirmed_offset,
253
+ confirmed_prefix_digest=prefix_hasher.digest(),
254
+ )
255
+
256
+ parsed_entries: list[UsageEntry] = []
257
+ digest = hashlib.blake2b(digest_size=16)
258
+ try:
259
+ with path.open("rb") as file:
260
+ confirmed_offset = _parse_complete_lines(file, project, parsed_entries, digest, 0)
261
+ except OSError as exc:
262
+ logger.warning("failed to read Claude project log %s: %s", path, exc)
263
+ return None
264
+ return _FileCacheEntry(
265
+ mtime=st.st_mtime,
266
+ size=st.st_size,
267
+ entries=parsed_entries,
268
+ confirmed_offset=confirmed_offset,
269
+ confirmed_prefix_digest=digest.digest(),
270
+ )
271
+
272
+
273
+ def _confirmed_prefix_hasher(path: Path, cached: _FileCacheEntry) -> Any | None:
274
+ if cached.confirmed_offset == 0:
275
+ return hashlib.blake2b(digest_size=16)
276
+
277
+ digest = hashlib.blake2b(digest_size=16)
278
+ remaining = cached.confirmed_offset
279
+ try:
280
+ with path.open("rb") as file:
281
+ while remaining > 0:
282
+ chunk = file.read(min(remaining, 65536))
283
+ if not chunk:
284
+ return None
285
+ digest.update(chunk)
286
+ remaining -= len(chunk)
287
+ except OSError:
288
+ return None
289
+ if digest.digest() != cached.confirmed_prefix_digest:
290
+ return None
291
+ return digest
292
+
293
+
294
+ def _parse_complete_lines(
295
+ file: Any,
296
+ project: str,
297
+ parsed_entries: list[UsageEntry],
298
+ digest: Any,
299
+ confirmed_offset: int,
300
+ ) -> int:
301
+ while True:
302
+ line_start = int(file.tell())
303
+ line, too_long = read_bounded_jsonl_line(file)
304
+ if too_long:
305
+ logger.warning("skipping oversized JSONL line in Claude project log %s", project)
306
+ confirmed_offset = int(file.tell())
307
+ continue
308
+ if not line:
309
+ return confirmed_offset
310
+ parsed_entry = _parse_line(line.decode("utf-8", errors="replace"), project)
311
+ if not line.endswith(b"\n") and parsed_entry is None:
312
+ return line_start
313
+ digest.update(line)
314
+ confirmed_offset = int(file.tell())
315
+ if parsed_entry is not None:
316
+ parsed_entries.append(parsed_entry)
317
+
318
+
319
+ def _parse_line(line: str, project: str) -> UsageEntry | None:
320
+ try:
321
+ data = json.loads(line)
322
+ except (json.JSONDecodeError, RecursionError):
323
+ return None
324
+
325
+ if not isinstance(data, dict) or data.get("type") != "assistant":
326
+ return None
327
+
328
+ message = data.get("message")
329
+ if not isinstance(message, dict):
330
+ return None
331
+
332
+ usage = message.get("usage")
333
+ if not isinstance(usage, dict):
334
+ return None
335
+
336
+ timestamp = _parse_timestamp(data.get("timestamp"))
337
+ if timestamp is None:
338
+ return None
339
+
340
+ input_tokens = _as_int(usage.get("input_tokens"))
341
+ output_tokens = _as_int(usage.get("output_tokens"))
342
+ cache_creation_tokens = _as_int(usage.get("cache_creation_input_tokens"))
343
+ cache_read_tokens = _as_int(usage.get("cache_read_input_tokens"))
344
+ if input_tokens + output_tokens + cache_creation_tokens + cache_read_tokens == 0:
345
+ return None
346
+
347
+ cwd = data.get("cwd")
348
+ if isinstance(cwd, str) and cwd:
349
+ project = _project_from_cwd(cwd)
350
+
351
+ return UsageEntry(
352
+ timestamp=timestamp,
353
+ session_id=_as_str(data.get("sessionId")),
354
+ message_id=_as_str(message.get("id")),
355
+ request_id=_as_str(data.get("requestId")),
356
+ model=_as_str(message.get("model")) or "unknown",
357
+ input_tokens=input_tokens,
358
+ output_tokens=output_tokens,
359
+ cache_creation_tokens=cache_creation_tokens,
360
+ cache_read_tokens=cache_read_tokens,
361
+ cost_usd=_as_optional_float(data.get("costUSD")),
362
+ project=project,
363
+ )
364
+
365
+
366
+ def _parse_timestamp(value: Any) -> datetime | None:
367
+ return parse_optional_iso8601_utc(value)
368
+
369
+
370
+ def _project_from_path(jsonl_path: Path) -> str:
371
+ return project_from_encoded_path(jsonl_path, CLAUDE_PROJECTS_DIR)
372
+
373
+
374
+ def _project_from_cwd(cwd: str) -> str:
375
+ return resolve_project_name(cwd)
376
+
377
+
378
+ def _dedup_key(entry: UsageEntry) -> str:
379
+ if entry.message_id or entry.request_id:
380
+ return f"message:{entry.message_id}:{entry.request_id}"
381
+ return (
382
+ f"entry:{entry.session_id}:{entry.timestamp.isoformat()}:{entry.model}:"
383
+ f"{entry.input_tokens}:{entry.output_tokens}:"
384
+ f"{entry.cache_creation_tokens}:{entry.cache_read_tokens}"
385
+ )
386
+
387
+
388
+ def _as_int(value: Any) -> int:
389
+ if isinstance(value, bool):
390
+ return 0
391
+ if isinstance(value, int):
392
+ return max(0, int(value))
393
+ if isinstance(value, str):
394
+ normalized = value.strip()
395
+ if normalized.isascii() and (
396
+ normalized.isdigit()
397
+ or (normalized.startswith("+") and normalized[1:].isdigit())
398
+ ):
399
+ return int(normalized)
400
+ return 0
401
+
402
+
403
+ def _as_str(value: Any) -> str:
404
+ return value if isinstance(value, str) else ""
405
+
406
+
407
+ def _as_optional_float(value: Any) -> float | None:
408
+ if isinstance(value, bool):
409
+ return None
410
+ try:
411
+ number = float(value)
412
+ except (TypeError, ValueError):
413
+ return None
414
+ if not math.isfinite(number):
415
+ return None
416
+ return number
i18n.py ADDED
@@ -0,0 +1,88 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-only
2
+ # Copyright (C) 2026 lollapalooza <https://github.com/aqua5230>
3
+ #
4
+ # Part of "usage". Free software licensed under the GNU Affero General Public
5
+ # License v3.0 only; see the LICENSE file for full terms and the warranty disclaimer.
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ import sys
12
+ import sysconfig
13
+ from functools import lru_cache
14
+ from pathlib import Path
15
+
16
+ from usage_lang import detect_lang
17
+
18
+
19
+ def packaged_resource_path(filename: str, source_mode_path: Path) -> Path:
20
+ """Resolve a data file across source-mode and py2app-bundle layouts.
21
+
22
+ py2app declares data files via setup_app.py ``OPTIONS["resources"]`` and
23
+ copies them to ``Contents/Resources/`` — adjacent to ``lib/python313.zip``,
24
+ not inside it. py2app injects the ``RESOURCEPATH`` env var at launch
25
+ pointing at that directory; we prefer it when present.
26
+
27
+ Why this exists: in py2app builds this module is compiled into
28
+ ``lib/python313.zip``, so ``Path(__file__).with_name("i18n.json")``
29
+ resolves to ``lib/python313.zip/i18n.json`` — an invalid path through
30
+ the zipfile that raises ``NotADirectoryError`` at first read. In source
31
+ mode (and tests) ``RESOURCEPATH`` is unset and the source-adjacent
32
+ fallback path is correct. Wheels installed by pip or uvx place data files
33
+ under the interpreter's sysconfig data directory, so that location is
34
+ checked before the source fallback.
35
+
36
+ The callers pass the source-mode path explicitly (as the literal
37
+ ``Path(__file__).with_name("...")``) so that
38
+ ``tests/test_packaged_resources.py`` can still statically detect every
39
+ declared resource and enforce that ``setup_app.py`` lists it.
40
+ """
41
+ resource_root = os.environ.get("RESOURCEPATH")
42
+ if resource_root:
43
+ bundled = Path(resource_root) / filename
44
+ if bundled.exists():
45
+ return bundled
46
+ frozen_root = getattr(sys, "_MEIPASS", None)
47
+ if frozen_root:
48
+ bundled = Path(frozen_root) / filename
49
+ if bundled.exists():
50
+ return bundled
51
+ wheel_data = Path(sysconfig.get_path("data")) / "share" / "usage" / filename
52
+ if wheel_data.exists():
53
+ return wheel_data
54
+ return source_mode_path
55
+
56
+
57
+ I18N_PATH = packaged_resource_path("i18n.json", Path(__file__).with_name("i18n.json"))
58
+
59
+
60
+ @lru_cache(maxsize=1)
61
+ def _load_i18n_bundle() -> dict[str, dict[str, str]]:
62
+ data = json.loads(I18N_PATH.read_text(encoding="utf-8"))
63
+ return {
64
+ str(lang): {str(key): str(value) for key, value in values.items()}
65
+ for lang, values in data.items()
66
+ }
67
+
68
+
69
+ def _t(language: str, key: str, **kwargs: object) -> str:
70
+ bundle = _load_i18n_bundle()
71
+ table = bundle.get(language) or bundle["en"]
72
+ template = table.get(key) or bundle["en"].get(key) or key
73
+ try:
74
+ return template.format(**kwargs)
75
+ except (KeyError, IndexError, ValueError, TypeError):
76
+ # A malformed placeholder in one locale's string must not crash the UI;
77
+ # fall back to the English template, then to the raw key.
78
+ en_template = bundle["en"].get(key)
79
+ if en_template is not None and en_template != template:
80
+ try:
81
+ return en_template.format(**kwargs)
82
+ except (KeyError, IndexError, ValueError, TypeError):
83
+ pass
84
+ return key
85
+
86
+
87
+ def t(key: str, **kwargs: object) -> str:
88
+ return _t(detect_lang(), key, **kwargs)
jsonl_limits.py ADDED
@@ -0,0 +1,17 @@
1
+ """Shared JSONL line-size limit and bounded binary reader."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import BinaryIO
6
+
7
+ MAX_JSONL_LINE_BYTES = 64 * 1024 * 1024
8
+
9
+
10
+ def read_bounded_jsonl_line(file: BinaryIO) -> tuple[bytes, bool]:
11
+ """Read one line, draining and flagging it when it exceeds the shared limit."""
12
+ line = file.readline(MAX_JSONL_LINE_BYTES + 1)
13
+ if len(line) <= MAX_JSONL_LINE_BYTES:
14
+ return line, False
15
+ while line and not line.endswith(b"\n"):
16
+ line = file.readline(65536)
17
+ return b"", True
jsonl_utils.py ADDED
@@ -0,0 +1,40 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import logging
5
+ from collections.abc import Iterator
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from jsonl_limits import read_bounded_jsonl_line
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ def iter_jsonl_dicts(
15
+ path: Path,
16
+ *,
17
+ encoding: str = "utf-8",
18
+ errors: str | None = None,
19
+ ) -> Iterator[dict[str, Any]]:
20
+ with path.open("rb") as file:
21
+ while True:
22
+ raw_bytes, too_long = read_bounded_jsonl_line(file)
23
+ if too_long:
24
+ logger.warning("skipping oversized JSONL line in %s", path)
25
+ continue
26
+ if not raw_bytes:
27
+ break
28
+ line = (
29
+ raw_bytes.decode(encoding)
30
+ if errors is None
31
+ else raw_bytes.decode(encoding, errors=errors)
32
+ ).strip()
33
+ if not line:
34
+ continue
35
+ try:
36
+ data = json.loads(line)
37
+ except (json.JSONDecodeError, RecursionError):
38
+ continue
39
+ if isinstance(data, dict):
40
+ yield data