briskapi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- briskapi/LICENSE-pybrisk.txt +21 -0
- briskapi/__init__.py +67 -0
- briskapi/__main__.py +3 -0
- briskapi/_archive.py +62 -0
- briskapi/_live.py +376 -0
- briskapi/_market.py +136 -0
- briskapi/_recording.py +192 -0
- briskapi/archive.json +5 -0
- briskapi/cli.py +337 -0
- briskapi/decoder/assets.json +13 -0
- briskapi/decoder/decoder.cjs +282 -0
- briskapi/decoder/sbi.cjs +144 -0
- briskapi/decoder/web.cjs +44 -0
- briskapi/references/historical_mock.json +4140 -0
- briskapi/sbi.py +266 -0
- briskapi/schema.py +351 -0
- briskapi/timing.py +72 -0
- briskapi-0.2.0.dist-info/METADATA +201 -0
- briskapi-0.2.0.dist-info/RECORD +23 -0
- briskapi-0.2.0.dist-info/WHEEL +5 -0
- briskapi-0.2.0.dist-info/entry_points.txt +2 -0
- briskapi-0.2.0.dist-info/licenses/LICENSE +21 -0
- briskapi-0.2.0.dist-info/top_level.txt +1 -0
briskapi/_recording.py
ADDED
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Decoded recordings on disk and the friendly quote/master views built from them."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import datetime as dt
|
|
5
|
+
import gzip
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
import re
|
|
10
|
+
from typing import Iterator
|
|
11
|
+
|
|
12
|
+
from briskapi.schema import canonical_lines, validate_stream
|
|
13
|
+
|
|
14
|
+
JST = dt.timezone(dt.timedelta(hours=9), 'JST')
|
|
15
|
+
PRICES = ('last_price10', 'open_price10', 'bid_price10', 'ask_price10', 'indicative_price10',
|
|
16
|
+
'closing_indicative_price10', 'indicative_open_price10', 'auction_reference_price10')
|
|
17
|
+
MASTER_PRICES = ('base_price10', 'limit_up10', 'limit_down10')
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BriskError(Exception):
|
|
21
|
+
"""Base class for API errors."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class NotFoundError(BriskError, KeyError):
|
|
25
|
+
"""A security code or recording is not available."""
|
|
26
|
+
|
|
27
|
+
def __str__(self):
|
|
28
|
+
return str(self.args[0]) if self.args else ''
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Table(list):
|
|
32
|
+
"""Rows as dicts; `to_pandas()` returns a DataFrame when pandas is installed."""
|
|
33
|
+
|
|
34
|
+
def to_pandas(self):
|
|
35
|
+
try:
|
|
36
|
+
import pandas as pd
|
|
37
|
+
except ImportError as e: # pragma: no cover - depends on the environment
|
|
38
|
+
raise ImportError("Install pandas (pip install 'briskapi[pandas]') to use to_pandas()") from e
|
|
39
|
+
return pd.DataFrame(list(self))
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class _Lines:
|
|
43
|
+
def __init__(self, lines):
|
|
44
|
+
self.lines = iter(lines)
|
|
45
|
+
|
|
46
|
+
def readline(self, size=-1):
|
|
47
|
+
return next(self.lines, b'')
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def timestamp(date: str, micros: int) -> dt.datetime:
|
|
51
|
+
"""JST datetime for microseconds since midnight on a YYYYMMDD trading date."""
|
|
52
|
+
day = dt.datetime.strptime(date, '%Y%m%d').replace(tzinfo=JST)
|
|
53
|
+
return day + dt.timedelta(microseconds=micros)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def micros(date: str, at) -> int | None:
|
|
57
|
+
"""Accept None, source microseconds, 'HH:MM[:SS[.ffffff]]', time or aware/naive JST datetime."""
|
|
58
|
+
if at is None or (type(at) is int and at >= 0):
|
|
59
|
+
return at
|
|
60
|
+
if isinstance(at, str):
|
|
61
|
+
match = re.fullmatch(r'(\d{1,2}):(\d{2})(?::(\d{2})(?:\.(\d{1,6}))?)?', at)
|
|
62
|
+
if not match:
|
|
63
|
+
raise ValueError(f'Use HH:MM[:SS[.ffffff]] for times, not {at!r}')
|
|
64
|
+
h, m, s, frac = match.groups()
|
|
65
|
+
at = dt.time(int(h), int(m), int(s or 0), int((frac or '0').ljust(6, '0')))
|
|
66
|
+
if isinstance(at, dt.datetime):
|
|
67
|
+
at = at.astimezone(JST) if at.tzinfo else at.replace(tzinfo=JST)
|
|
68
|
+
if at.strftime('%Y%m%d') != date:
|
|
69
|
+
raise ValueError(f'{at.isoformat()} is not on trading date {date}')
|
|
70
|
+
at = at.timetz()
|
|
71
|
+
if isinstance(at, dt.time):
|
|
72
|
+
return ((at.hour * 60 + at.minute) * 60 + at.second) * 1_000_000 + at.microsecond
|
|
73
|
+
raise TypeError(f'Unsupported time: {at!r}')
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def quote_view(quote: dict, date: str) -> dict:
|
|
77
|
+
"""Yen prices (None for the vendor's zero sentinel), JST times and the raw quantities/enums."""
|
|
78
|
+
view = {'code': quote['code'], 'issue_id': quote['issue_id'], 'time': timestamp(date, quote['source_time_us'])}
|
|
79
|
+
for key, value in quote.items():
|
|
80
|
+
if key in PRICES:
|
|
81
|
+
view[key[:-2]] = value / 10 if value else None
|
|
82
|
+
elif key == 'special_quote_time_us':
|
|
83
|
+
view['special_quote_time'] = timestamp(date, value) if value else None
|
|
84
|
+
elif key not in view and key != 'source_time_us':
|
|
85
|
+
view[key] = value
|
|
86
|
+
return view
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def master_view(entry: dict) -> dict:
|
|
90
|
+
"""Master entry with base price and daily limits in yen."""
|
|
91
|
+
return {key[:-2] if key in MASTER_PRICES else key: value / 10 if key in MASTER_PRICES else value
|
|
92
|
+
for key, value in entry.items()}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class Recording:
|
|
96
|
+
"""A decoded BRiSK stream: events.jsonl, events.jsonl.gz or a directory holding one.
|
|
97
|
+
|
|
98
|
+
Reading is streaming; whole-market queries parse the file once (about ten
|
|
99
|
+
seconds for the complete 420 MB demo replay) and the final state is cached.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
contribution = None # Publication status when record() contributed this recording.
|
|
103
|
+
|
|
104
|
+
def __init__(self, path):
|
|
105
|
+
path = Path(path).expanduser()
|
|
106
|
+
self.directory = path if path.is_dir() else path.parent
|
|
107
|
+
if path.is_dir():
|
|
108
|
+
path = next((path / n for n in ('events.jsonl', 'events.jsonl.gz') if (path / n).exists()), None)
|
|
109
|
+
if path is None:
|
|
110
|
+
raise NotFoundError(f'No events.jsonl(.gz) in {self.directory}')
|
|
111
|
+
if not path.exists():
|
|
112
|
+
raise NotFoundError(f'No recording at {path}')
|
|
113
|
+
self.path = path
|
|
114
|
+
manifest = self.directory / 'manifest.json'
|
|
115
|
+
self.manifest = json.loads(manifest.read_text()) if manifest.exists() else None
|
|
116
|
+
self._bootstrap = None
|
|
117
|
+
self._final = None
|
|
118
|
+
|
|
119
|
+
def __repr__(self):
|
|
120
|
+
return f'Recording({str(self.path)!r})'
|
|
121
|
+
|
|
122
|
+
def open(self):
|
|
123
|
+
return gzip.open(self.path, 'rb') if self.path.suffix == '.gz' else self.path.open('rb')
|
|
124
|
+
|
|
125
|
+
def lines(self) -> Iterator[bytes]:
|
|
126
|
+
with self.open() as f:
|
|
127
|
+
yield from f
|
|
128
|
+
|
|
129
|
+
def batches(self) -> Iterator[dict]:
|
|
130
|
+
"""Every decoded batch in order, exactly as recorded."""
|
|
131
|
+
for line in self.lines():
|
|
132
|
+
yield json.loads(line)
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def bootstrap(self) -> dict:
|
|
136
|
+
if self._bootstrap is None:
|
|
137
|
+
with self.open() as f:
|
|
138
|
+
self._bootstrap = json.loads(f.readline())
|
|
139
|
+
return self._bootstrap
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def trading_date(self) -> str:
|
|
143
|
+
return self.bootstrap['trading_date']
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def source(self) -> str:
|
|
147
|
+
return self.bootstrap['source']
|
|
148
|
+
|
|
149
|
+
@property
|
|
150
|
+
def codes(self) -> list[str]:
|
|
151
|
+
return [m['code'] for m in self.bootstrap['master']]
|
|
152
|
+
|
|
153
|
+
def issue(self, code: str) -> dict:
|
|
154
|
+
for entry in self.bootstrap['master']:
|
|
155
|
+
if entry['code'] == str(code):
|
|
156
|
+
return entry
|
|
157
|
+
raise NotFoundError(f'{code} is not in this recording')
|
|
158
|
+
|
|
159
|
+
def validate(self) -> dict:
|
|
160
|
+
"""Run the archive's full checks (canonical form, timing bounds, reference replay)."""
|
|
161
|
+
with self.open() as f:
|
|
162
|
+
return validate_stream(_Lines(canonical_lines(f)))
|
|
163
|
+
|
|
164
|
+
def state(self, at=None) -> dict[int, dict]:
|
|
165
|
+
"""Raw latest quote per issue ID at a source time (default: end of recording)."""
|
|
166
|
+
limit = micros(self.trading_date, at)
|
|
167
|
+
if limit is None and self._final is not None:
|
|
168
|
+
return dict(self._final)
|
|
169
|
+
quotes = {}
|
|
170
|
+
for batch in self.batches():
|
|
171
|
+
if limit is not None and batch['source_time_us'] > limit:
|
|
172
|
+
break
|
|
173
|
+
quotes.update((q['issue_id'], q) for q in batch.get('quotes', ()))
|
|
174
|
+
if limit is None:
|
|
175
|
+
self._final = dict(quotes)
|
|
176
|
+
return quotes
|
|
177
|
+
|
|
178
|
+
def updates(self, code: str, start=None, end=None) -> Iterator[dict]:
|
|
179
|
+
"""Raw quote updates for one security, skipping unrelated batches without parsing them."""
|
|
180
|
+
issue = self.issue(code)['issue_id']
|
|
181
|
+
needle = json.dumps(str(code)).encode() # Matches any JSON spacing; a false hit only costs a parse.
|
|
182
|
+
low, high = micros(self.trading_date, start), micros(self.trading_date, end)
|
|
183
|
+
for line in self.lines():
|
|
184
|
+
if needle not in line:
|
|
185
|
+
continue
|
|
186
|
+
for quote in json.loads(line).get('quotes', ()):
|
|
187
|
+
if quote['issue_id'] != issue:
|
|
188
|
+
continue
|
|
189
|
+
if high is not None and quote['source_time_us'] > high:
|
|
190
|
+
return
|
|
191
|
+
if low is None or quote['source_time_us'] >= low:
|
|
192
|
+
yield quote
|
briskapi/archive.json
ADDED
briskapi/cli.py
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Consume BRiSK auction data live, record and automatically contribute it, and use the shared archive."""
|
|
3
|
+
import argparse
|
|
4
|
+
import datetime as dt
|
|
5
|
+
import gzip
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
import re
|
|
10
|
+
import secrets
|
|
11
|
+
import shutil
|
|
12
|
+
import subprocess
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
import time
|
|
16
|
+
import urllib.request
|
|
17
|
+
import uuid
|
|
18
|
+
import warnings
|
|
19
|
+
|
|
20
|
+
import boto3
|
|
21
|
+
from botocore import UNSIGNED
|
|
22
|
+
from botocore.config import Config
|
|
23
|
+
from briskapi.schema import (SCHEMA, MAX_TIMING_BYTES, canonical_lines, compress, digest, inspect_package, validate_manifest,
|
|
24
|
+
validate_stream, validate_timing, require)
|
|
25
|
+
|
|
26
|
+
PACKAGE = Path(__file__).resolve().parent
|
|
27
|
+
# BRiSK's own WASM decoder runs under Node; the package ships only this host and SHA-256 pins.
|
|
28
|
+
DECODER = PACKAGE / 'decoder' / 'decoder.cjs'
|
|
29
|
+
LICENSES = ('CC0-1.0', 'CC-BY-4.0')
|
|
30
|
+
# Bump with any PRIVACY.md change to what is collected; saved choices then lapse.
|
|
31
|
+
POLICY_VERSION = 2
|
|
32
|
+
NOTICE = '''\
|
|
33
|
+
Sessions are contributed to the shared public BRiSK archive automatically.
|
|
34
|
+
After each clean, complete demo replay the contribution contains:
|
|
35
|
+
- the decoded market data you recorded (checked against the reference replay);
|
|
36
|
+
- local timing measurements: decode durations, replay lateness, asset download
|
|
37
|
+
time and your computer's receipt clock, which shows when you recorded;
|
|
38
|
+
- your public alias and data license, in the published manifest.
|
|
39
|
+
SBI BRiSK sessions contribute only a timing summary: decode time, data age and
|
|
40
|
+
frame spacing percentiles, stalls, frame count, date and start/end minute. No
|
|
41
|
+
prices, quantities or codes.
|
|
42
|
+
Contributions are public and permanent. Your IP address is used only for
|
|
43
|
+
upload rate limiting. No account, file, hostname or system details are sent.
|
|
44
|
+
Accepting declares that you may redistribute these recordings under that license.
|
|
45
|
+
Policy: https://github.com/honvl/BRiSKapi/blob/main/PRIVACY.md
|
|
46
|
+
Opt out at any time: `brisk consent --revoke` or BRISK_CONTRIBUTE=0.
|
|
47
|
+
'''
|
|
48
|
+
|
|
49
|
+
def settings(path=None):
|
|
50
|
+
return json.loads((path or PACKAGE / 'archive.json').read_text())
|
|
51
|
+
|
|
52
|
+
def consent_path():
|
|
53
|
+
return Path(os.environ.get('XDG_CONFIG_HOME') or Path.home() / '.config') / 'brisk' / 'contribution.json'
|
|
54
|
+
|
|
55
|
+
def load_consent():
|
|
56
|
+
"""The saved contribution choice under the current policy, or None if undecided."""
|
|
57
|
+
try:
|
|
58
|
+
choice = json.loads(consent_path().read_text())
|
|
59
|
+
except (OSError, ValueError):
|
|
60
|
+
return None
|
|
61
|
+
return choice if isinstance(choice, dict) and choice.get('policy_version') == POLICY_VERSION else None
|
|
62
|
+
|
|
63
|
+
def save_consent(enabled, contributor=None, license=None):
|
|
64
|
+
choice = dict(policy_version=POLICY_VERSION, enabled=bool(enabled),
|
|
65
|
+
decided_at=dt.datetime.now(dt.timezone.utc).isoformat(timespec='seconds'))
|
|
66
|
+
if enabled:
|
|
67
|
+
choice.update(contributor=contributor or f'anon-{secrets.token_hex(4)}', license=license or 'CC0-1.0')
|
|
68
|
+
require(re.fullmatch(r'[A-Za-z0-9_.-]{1,64}', choice['contributor']), 'Alias: 1-64 letters, digits, _ . -')
|
|
69
|
+
require(choice['license'] in LICENSES, 'Unsupported data license')
|
|
70
|
+
path = consent_path()
|
|
71
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
72
|
+
path.write_text(json.dumps(choice, indent=2) + '\n')
|
|
73
|
+
return choice
|
|
74
|
+
|
|
75
|
+
def interactive():
|
|
76
|
+
return sys.stdin.isatty() and sys.stderr.isatty()
|
|
77
|
+
|
|
78
|
+
def ask_consent():
|
|
79
|
+
"""One-time first-run question; Enter accepts. Prompts go to stderr, never stdout."""
|
|
80
|
+
alias = f'anon-{secrets.token_hex(4)}'
|
|
81
|
+
sys.stderr.write(NOTICE + f"Contribute automatically as '{alias}' under CC0-1.0? [Y/n] ")
|
|
82
|
+
sys.stderr.flush()
|
|
83
|
+
accepted = sys.stdin.readline().strip().lower() in {'', 'y', 'yes'}
|
|
84
|
+
return save_consent(accepted, alias)
|
|
85
|
+
|
|
86
|
+
def declaration(args):
|
|
87
|
+
"""Alias and license for this run, or None to keep the recording local."""
|
|
88
|
+
if args.contributor or args.license or args.redistribution_permitted:
|
|
89
|
+
require(args.contributor and args.license and args.redistribution_permitted,
|
|
90
|
+
'Use --contributor, --license and --redistribution-permitted together')
|
|
91
|
+
return dict(contributor=args.contributor, license=args.license)
|
|
92
|
+
choice = load_consent()
|
|
93
|
+
if choice is None and args.upload is not False and os.environ.get('BRISK_CONTRIBUTE') != '0' and interactive():
|
|
94
|
+
choice = ask_consent()
|
|
95
|
+
return {k: choice[k] for k in ('contributor', 'license')} if choice and choice['enabled'] else None
|
|
96
|
+
|
|
97
|
+
def package(events, output, contributor, license):
|
|
98
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
target = output / 'events.jsonl.gz'
|
|
100
|
+
require(not target.exists(), 'Package output already exists')
|
|
101
|
+
try:
|
|
102
|
+
with events.open('rb') as source:
|
|
103
|
+
compress(canonical_lines(source), target)
|
|
104
|
+
with gzip.open(target, 'rb') as stream:
|
|
105
|
+
summary = validate_stream(stream)
|
|
106
|
+
except BaseException:
|
|
107
|
+
target.unlink(missing_ok=True)
|
|
108
|
+
raise
|
|
109
|
+
m = validate_manifest(dict(schema=SCHEMA, sha256=digest(target), bytes=target.stat().st_size, summary=summary,
|
|
110
|
+
contributor=contributor, license=license, redistribution_permitted=True))
|
|
111
|
+
(output / 'manifest.json').write_text(json.dumps(m, indent=2) + '\n')
|
|
112
|
+
return m
|
|
113
|
+
|
|
114
|
+
def request_json(url, data=None):
|
|
115
|
+
request = urllib.request.Request(url, data=None if data is None else json.dumps(data).encode(),
|
|
116
|
+
headers={'Content-Type': 'application/json'})
|
|
117
|
+
with urllib.request.urlopen(request, timeout=60) as response:
|
|
118
|
+
return json.load(response)
|
|
119
|
+
|
|
120
|
+
def contribute(directory, api_url, timeout=660, verify=True, out=None):
|
|
121
|
+
m = json.loads((directory / 'manifest.json').read_text())
|
|
122
|
+
path = directory / 'events.jsonl.gz'
|
|
123
|
+
if verify:
|
|
124
|
+
inspect_package(path, m)
|
|
125
|
+
else:
|
|
126
|
+
require(path.stat().st_size == m['bytes'] and digest(path) == m['sha256'], 'Size/hash mismatch')
|
|
127
|
+
ticket = request_json(api_url, m)
|
|
128
|
+
# S3 browser POST policies bind key, encryption, content type and exact size.
|
|
129
|
+
# The archive's 64 MiB cap bounds the multipart body below 65 MiB.
|
|
130
|
+
boundary = uuid.uuid4().hex
|
|
131
|
+
parts = []
|
|
132
|
+
for key, value in ticket['upload']['fields'].items():
|
|
133
|
+
parts.append(f'--{boundary}\r\nContent-Disposition: form-data; name="{key}"\r\n\r\n{value}\r\n'.encode())
|
|
134
|
+
parts.append(f'--{boundary}\r\nContent-Disposition: form-data; name="file"; filename="events.jsonl.gz"\r\nContent-Type: application/gzip\r\n\r\n'.encode())
|
|
135
|
+
parts.extend([path.read_bytes(), f'\r\n--{boundary}--\r\n'.encode()])
|
|
136
|
+
request = urllib.request.Request(ticket['upload']['url'], data=b''.join(parts),
|
|
137
|
+
headers={'Content-Type': f'multipart/form-data; boundary={boundary}'}, method='POST')
|
|
138
|
+
with urllib.request.urlopen(request, timeout=120) as response:
|
|
139
|
+
response.read()
|
|
140
|
+
status_url = api_url.rstrip('/') + '/?ticket=' + ticket['ticket']
|
|
141
|
+
print(json.dumps({'ticket': ticket['ticket'], 'status_url': status_url}), file=out, flush=True)
|
|
142
|
+
deadline = time.monotonic() + timeout
|
|
143
|
+
while time.monotonic() < deadline:
|
|
144
|
+
status = request_json(status_url)
|
|
145
|
+
if status['status'] == 'published':
|
|
146
|
+
return status
|
|
147
|
+
require(status['status'] != 'rejected', 'Automatic validation rejected recording: ' + status.get('reason', 'unknown'))
|
|
148
|
+
time.sleep(3)
|
|
149
|
+
raise TimeoutError(f'Publication still pending; check {status_url}')
|
|
150
|
+
|
|
151
|
+
def client(config):
|
|
152
|
+
return boto3.client('s3', region_name=config['region'], config=Config(signature_version=UNSIGNED))
|
|
153
|
+
|
|
154
|
+
def contribute_timing(report, api_url):
|
|
155
|
+
"""Publish a timing-only report (no market data); returns the service's response."""
|
|
156
|
+
return request_json(api_url, {'timing': report})
|
|
157
|
+
|
|
158
|
+
def timing_reports(s3, bucket, date=None):
|
|
159
|
+
"""Published timing reports, newest dates last."""
|
|
160
|
+
prefix = 'timing/'
|
|
161
|
+
if date:
|
|
162
|
+
require(len(date) == 8 and date.isdigit(), 'Date must be YYYYMMDD')
|
|
163
|
+
prefix += date + '/'
|
|
164
|
+
for page in s3.get_paginator('list_objects_v2').paginate(Bucket=bucket, Prefix=prefix):
|
|
165
|
+
for item in page.get('Contents', []):
|
|
166
|
+
body = s3.get_object(Bucket=bucket, Key=item['Key'])['Body']
|
|
167
|
+
with body:
|
|
168
|
+
yield item['Key'], validate_timing(json.loads(body.read(MAX_TIMING_BYTES + 1)))
|
|
169
|
+
|
|
170
|
+
def manifests(s3, bucket, date=None):
|
|
171
|
+
prefix = 'archive/'
|
|
172
|
+
if date:
|
|
173
|
+
require(len(date) == 8 and date.isdigit(), 'Date must be YYYYMMDD')
|
|
174
|
+
prefix += date + '/'
|
|
175
|
+
for page in s3.get_paginator('list_objects_v2').paginate(Bucket=bucket, Prefix=prefix):
|
|
176
|
+
for item in page.get('Contents', []):
|
|
177
|
+
if item['Key'].endswith('/manifest.json'):
|
|
178
|
+
body = s3.get_object(Bucket=bucket, Key=item['Key'])['Body']
|
|
179
|
+
with body:
|
|
180
|
+
m = json.loads(body.read(65537))
|
|
181
|
+
validate_manifest(m)
|
|
182
|
+
yield item['Key'].rsplit('/', 1)[0], m
|
|
183
|
+
|
|
184
|
+
def pull(s3, bucket, prefix, output):
|
|
185
|
+
require(re.fullmatch(r'archive/\d{8}/[0-9a-f]{64}', prefix), 'Invalid archive prefix')
|
|
186
|
+
require(not output.exists(), 'Download output already exists')
|
|
187
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
188
|
+
with tempfile.TemporaryDirectory(dir=output.parent) as tmp:
|
|
189
|
+
directory = Path(tmp)
|
|
190
|
+
for name in ('manifest.json', 'events.jsonl.gz'):
|
|
191
|
+
obj = s3.get_object(Bucket=bucket, Key=f'{prefix}/{name}')
|
|
192
|
+
limit = 65536 if name == 'manifest.json' else 64 * 1024**2
|
|
193
|
+
require(obj['ContentLength'] <= limit, 'Remote object exceeds limit')
|
|
194
|
+
with obj['Body'] as body, (directory / name).open('wb') as target:
|
|
195
|
+
size = 0
|
|
196
|
+
while chunk := body.read(1024 * 1024):
|
|
197
|
+
size += len(chunk)
|
|
198
|
+
require(size <= limit, 'Remote object exceeds limit')
|
|
199
|
+
target.write(chunk)
|
|
200
|
+
m = json.loads((directory / 'manifest.json').read_text())
|
|
201
|
+
inspect_package(directory / 'events.jsonl.gz', m)
|
|
202
|
+
require(prefix == f"archive/{m['summary']['trading_date']}/{m['sha256']}", 'Archive identity mismatch')
|
|
203
|
+
with gzip.open(directory / 'events.jsonl.gz', 'rb') as source, (directory / 'events.jsonl').open('wb') as target:
|
|
204
|
+
shutil.copyfileobj(source, target)
|
|
205
|
+
shutil.move(str(directory), str(output))
|
|
206
|
+
return m
|
|
207
|
+
|
|
208
|
+
def record_events(events, web=False, cache=None, codes=None, limit_frames=None, speed=1, binary=None, node='node'):
|
|
209
|
+
"""Replay the pinned demo and save its decoded batches.
|
|
210
|
+
|
|
211
|
+
Needs only Node. A Rust collector binary (`brisk_quote_ingest`) is optional; it
|
|
212
|
+
records the same batches and adds its own state validation and latency display.
|
|
213
|
+
"""
|
|
214
|
+
options = ['--web'] if web else ['--cache', str(cache)]
|
|
215
|
+
options += ['--speed', str(speed)]
|
|
216
|
+
if codes:
|
|
217
|
+
options += ['--codes', codes if isinstance(codes, str) else ','.join(codes)]
|
|
218
|
+
if limit_frames:
|
|
219
|
+
options += ['--limit-frames', str(limit_frames)]
|
|
220
|
+
if binary:
|
|
221
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
222
|
+
subprocess.run([str(binary), '--decoder', str(DECODER), '--events', str(events),
|
|
223
|
+
'--latest', str(Path(tmp) / 'latest.json'), *options], check=True)
|
|
224
|
+
return
|
|
225
|
+
with open(events, 'wb') as out:
|
|
226
|
+
subprocess.run([node, str(DECODER), *options], check=True, stdout=out)
|
|
227
|
+
|
|
228
|
+
def live(args):
|
|
229
|
+
"""Print each quote update as one JSON line; a complete session is contributed per consent."""
|
|
230
|
+
from briskapi import connect, sbi # The API package builds on this module.
|
|
231
|
+
codes = args.codes.split(',') if args.codes else None
|
|
232
|
+
if load_consent() is None and os.environ.get('BRISK_CONTRIBUTE') != '0' and interactive():
|
|
233
|
+
ask_consent()
|
|
234
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
235
|
+
warnings.simplefilter('always')
|
|
236
|
+
if args.sbi: # Market data stays local; only a timing summary is contributed.
|
|
237
|
+
sbi.login()
|
|
238
|
+
feed = sbi.connect(codes=codes)
|
|
239
|
+
else:
|
|
240
|
+
feed = connect(web=args.web, cache=args.cache, codes=codes, speed=args.speed, limit_frames=args.limit_frames)
|
|
241
|
+
for warning in caught:
|
|
242
|
+
print(warning.message, file=sys.stderr)
|
|
243
|
+
try:
|
|
244
|
+
for quote in feed.quotes(raw=args.raw):
|
|
245
|
+
print(json.dumps(quote, ensure_ascii=False, default=lambda value: value.isoformat()), flush=True)
|
|
246
|
+
feed.wait()
|
|
247
|
+
finally:
|
|
248
|
+
feed.close()
|
|
249
|
+
if feed.contribution:
|
|
250
|
+
print(json.dumps({'contribution': feed.contribution}), file=sys.stderr)
|
|
251
|
+
|
|
252
|
+
def main(argv=None):
|
|
253
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
254
|
+
parser.add_argument('--config', type=Path, default=PACKAGE / 'archive.json')
|
|
255
|
+
sub = parser.add_subparsers(dest='command', required=True)
|
|
256
|
+
for name in ('record', 'package'):
|
|
257
|
+
p = sub.add_parser(name, help='Record a demo replay' if name == 'record' else 'Package a recording')
|
|
258
|
+
p.add_argument('--output', type=Path, required=True)
|
|
259
|
+
p.add_argument('--contributor', help='Public alias (defaults to the saved consent choice)')
|
|
260
|
+
p.add_argument('--license', choices=LICENSES)
|
|
261
|
+
p.add_argument('--redistribution-permitted', action='store_true',
|
|
262
|
+
help='Declare permission to redistribute this recording under the selected data license')
|
|
263
|
+
p.add_argument('--upload', action=argparse.BooleanOptionalAction, default=None,
|
|
264
|
+
help='Contribute after packaging (default: yes once contribution consent is saved)')
|
|
265
|
+
if name == 'record':
|
|
266
|
+
group = p.add_mutually_exclusive_group(required=True)
|
|
267
|
+
group.add_argument('--web', action='store_true')
|
|
268
|
+
group.add_argument('--cache', type=Path)
|
|
269
|
+
p.add_argument('--codes')
|
|
270
|
+
p.add_argument('--limit-frames', type=int)
|
|
271
|
+
p.add_argument('--speed', type=float, default=1)
|
|
272
|
+
p.add_argument('--binary', type=Path, help='Optional Rust collector (brisk_quote_ingest) for state validation and latency display')
|
|
273
|
+
else:
|
|
274
|
+
p.add_argument('--events', type=Path, required=True)
|
|
275
|
+
p = sub.add_parser('live', help='Stream live quote updates as JSON lines')
|
|
276
|
+
group = p.add_mutually_exclusive_group(required=True)
|
|
277
|
+
group.add_argument('--web', action='store_true')
|
|
278
|
+
group.add_argument('--cache', type=Path)
|
|
279
|
+
p.add_argument('--codes', help='Comma-separated security codes (default: whole market)')
|
|
280
|
+
p.add_argument('--speed', type=float, default=1)
|
|
281
|
+
p.add_argument('--limit-frames', type=int)
|
|
282
|
+
p.add_argument('--raw', action='store_true', help='Vendor fields (price10, microseconds) instead of yen/ISO times')
|
|
283
|
+
group.add_argument('--sbi', action='store_true',
|
|
284
|
+
help='Live SBI BRiSK (experimental); cookies from BRISK_SBI_COOKIES or saved with sbi.login(remember=True)')
|
|
285
|
+
p = sub.add_parser('upload', help='Contribute a prepared package'); p.add_argument('directory', type=Path)
|
|
286
|
+
p = sub.add_parser('list', help='List published recordings'); p.add_argument('--date'); p.add_argument('--source', choices=['historical_mock','synthetic_test'])
|
|
287
|
+
p = sub.add_parser('pull', help='Download and verify a recording'); p.add_argument('prefix'); p.add_argument('--output', type=Path, required=True)
|
|
288
|
+
p = sub.add_parser('consent', help='Show or change automatic contribution')
|
|
289
|
+
group = p.add_mutually_exclusive_group()
|
|
290
|
+
group.add_argument('--accept', action='store_true')
|
|
291
|
+
group.add_argument('--revoke', action='store_true')
|
|
292
|
+
p.add_argument('--contributor'); p.add_argument('--license', choices=LICENSES)
|
|
293
|
+
args = parser.parse_args(argv)
|
|
294
|
+
config = settings(args.config)
|
|
295
|
+
if args.command == 'consent':
|
|
296
|
+
if args.accept:
|
|
297
|
+
sys.stderr.write(NOTICE)
|
|
298
|
+
choice = save_consent(True, args.contributor, args.license)
|
|
299
|
+
else:
|
|
300
|
+
choice = save_consent(False) if args.revoke else load_consent() or {'policy_version': POLICY_VERSION, 'enabled': None}
|
|
301
|
+
print(json.dumps(choice))
|
|
302
|
+
elif args.command in {'record', 'package'}:
|
|
303
|
+
partial = args.command == 'record' and args.limit_frames is not None
|
|
304
|
+
require(not (partial and args.upload), 'Only complete replays can be contributed; omit --limit-frames')
|
|
305
|
+
found = None if partial else declaration(args)
|
|
306
|
+
require(found or not args.upload, f'Uploading needs `{parser.prog} consent --accept` or --contributor/--license/--redistribution-permitted')
|
|
307
|
+
if args.command == 'record':
|
|
308
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
309
|
+
events = Path(tmp) / 'events.jsonl'
|
|
310
|
+
record_events(events, args.web, args.cache, args.codes, args.limit_frames, args.speed, args.binary)
|
|
311
|
+
if found is None:
|
|
312
|
+
args.output.mkdir(parents=True, exist_ok=True)
|
|
313
|
+
require(not (args.output / 'events.jsonl').exists(), 'Recording output already exists')
|
|
314
|
+
shutil.move(str(events), str(args.output / 'events.jsonl'))
|
|
315
|
+
why = 'partial replays stay local' if partial else f'contribution is off (`{parser.prog} consent`)'
|
|
316
|
+
print(f'Saved {args.output / "events.jsonl"}; not contributed: {why}.', file=sys.stderr)
|
|
317
|
+
return
|
|
318
|
+
m = package(events, args.output, **found)
|
|
319
|
+
else:
|
|
320
|
+
require(found, f'Packaging needs `{parser.prog} consent --accept` or --contributor/--license/--redistribution-permitted')
|
|
321
|
+
m = package(args.events, args.output, **found)
|
|
322
|
+
print(json.dumps(m))
|
|
323
|
+
if args.upload is not False and os.environ.get('BRISK_CONTRIBUTE') != '0':
|
|
324
|
+
print(json.dumps(contribute(args.output, config['api_url'], verify=False)))
|
|
325
|
+
elif args.command == 'live':
|
|
326
|
+
live(args)
|
|
327
|
+
elif args.command == 'upload':
|
|
328
|
+
print(json.dumps(contribute(args.directory, config['api_url'])))
|
|
329
|
+
elif args.command == 'list':
|
|
330
|
+
for prefix, m in manifests(client(config), config['bucket'], args.date):
|
|
331
|
+
if args.source is None or m['summary']['source'] == args.source:
|
|
332
|
+
print(json.dumps({'prefix': prefix, **m}))
|
|
333
|
+
elif args.command == 'pull':
|
|
334
|
+
print(json.dumps(pull(client(config), config['bucket'], args.prefix, args.output)))
|
|
335
|
+
|
|
336
|
+
if __name__ == '__main__':
|
|
337
|
+
main()
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
{
|
|
2
|
+
"source": "https://next-demo.brisk.jp/",
|
|
3
|
+
"trading_date": "2021-09-27",
|
|
4
|
+
"protocol_version": 16000,
|
|
5
|
+
"ohlc_length": 60,
|
|
6
|
+
"assets": {
|
|
7
|
+
"fita.js": {"url": "https://next-demo.brisk.jp/assets/wasm/fita.3bebb8c6e713be6b68d6e8e7dd07945cb8b9f9ec.js", "sha256": "e970de67768c22236a24ab77c60b0106ff591f4f3d3618ee33dfe1abeb6ad698"},
|
|
8
|
+
"fita.wasm": {"url": "https://next-demo.brisk.jp/assets/wasm/fita.3bebb8c6e713be6b68d6e8e7dd07945cb8b9f9ec.wasm", "sha256": "1f901b9ccfbb457365fb5836b4187353f479cc7961d40b35a080f26d8359a1cb"},
|
|
9
|
+
"master.dat": {"url": "https://next-demo.brisk.jp/assets/754a650d27a7596b9987eac66eb92a76/master.dat", "sha256": "51e8f45f01e8b8f9e72476b61964332e90340b6940e84e3fe3b9f3fecc92506f"},
|
|
10
|
+
"snapshot.dat": {"url": "https://next-demo.brisk.jp/assets/754a650d27a7596b9987eac66eb92a76/snapshot.dat", "sha256": "6c82db8e1f8fb4a511db75b54cb6899d5539c3c7c4db2a3b5c9c66e28b5bfde9"},
|
|
11
|
+
"ws.dat": {"url": "https://next-demo.brisk.jp/assets/754a650d27a7596b9987eac66eb92a76/ws.dat", "sha256": "6a17eefa46608dab50c771703ff465f94fcab590695b7ff305c15eea32c0af2b"}
|
|
12
|
+
}
|
|
13
|
+
}
|