jstdata 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
jstdata/session.py ADDED
@@ -0,0 +1,251 @@
1
+ """Portable representation of analytical query intent.
2
+
3
+ A Session mirrors the parameters of ``JSTDataClient.query``. It contains
4
+ no observations — only what is needed to (re)execute a query.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import csv
10
+ import json
11
+ from dataclasses import asdict, dataclass, field
12
+ from datetime import datetime
13
+ from pathlib import Path
14
+ from typing import TYPE_CHECKING, Any, List, Mapping, Optional, Union
15
+
16
+ if TYPE_CHECKING:
17
+ import pandas as pd
18
+
19
+ from .client import JSTDataClient
20
+ from .models import TimeSeries
21
+
22
+ PathLike = Union[str, Path]
23
+
24
+
25
+ @dataclass
26
+ class Session:
27
+ """Analytical intent for a Jefferson Street query."""
28
+
29
+ metric: list[str] = field(default_factory=list)
30
+ entity: list[str] = field(default_factory=list)
31
+ series: list[str] = field(default_factory=list)
32
+ frequency: Optional[str] = None
33
+ taxonomy: Optional[str] = None
34
+ head: Optional[int] = None
35
+ tail: Optional[int] = None
36
+ as_of: Optional[str] = None
37
+ sort_by: Optional[str] = None
38
+ start_date: Optional[str] = None
39
+ end_date: Optional[str] = None
40
+ start_time: Optional[int] = None
41
+ end_time: Optional[int] = None
42
+ order_by: Optional[str] = None
43
+
44
+ def is_empty(self) -> bool:
45
+ """True when the session has no resource selectors."""
46
+ return not (self.metric or self.entity or self.series)
47
+
48
+ def resource_ids(self) -> list[str]:
49
+ """All metric, entity, and series IDs in the session."""
50
+ return [*self.metric, *self.entity, *self.series]
51
+
52
+ def to_query_kwargs(self) -> dict[str, Any]:
53
+ """kwargs suitable for ``JSTDataClient.query`` / ``query_df``.
54
+
55
+ Date/time fields are ignored: ``/query`` is head/tail/as_of only.
56
+ """
57
+ kwargs: dict[str, Any] = {}
58
+ if self.metric:
59
+ kwargs["metric"] = list(self.metric)
60
+ if self.entity:
61
+ kwargs["entity"] = list(self.entity)
62
+ if self.series:
63
+ kwargs["series"] = list(self.series)
64
+ if self.frequency is not None:
65
+ kwargs["frequency"] = self.frequency
66
+ if self.taxonomy is not None:
67
+ kwargs["taxonomy"] = self.taxonomy
68
+ if self.head is not None:
69
+ kwargs["head"] = self.head
70
+ if self.tail is not None:
71
+ kwargs["tail"] = self.tail
72
+ if self.as_of is not None:
73
+ kwargs["as_of"] = self.as_of
74
+ if self.sort_by is not None:
75
+ kwargs["sort_by"] = self.sort_by
76
+ if self.order_by is not None:
77
+ kwargs["order_by"] = self.order_by
78
+ return kwargs
79
+
80
+ def to_dict(self) -> dict[str, Any]:
81
+ """JSON-serializable dict; omits null optional fields."""
82
+ data = asdict(self)
83
+ return {k: v for k, v in data.items() if v is not None}
84
+
85
+ @classmethod
86
+ def from_dict(cls, data: Mapping[str, Any]) -> Session:
87
+ """Build a session from a dict.
88
+
89
+ Accepts both query-param names (``metric``/``entity``) and the
90
+ older plural keys (``metrics``/``entities``) used by early saves.
91
+ Date windows from older sessions are preserved but not sent to
92
+ ``/query``.
93
+ """
94
+ metric = data.get("metric")
95
+ if metric is None:
96
+ metric = data.get("metrics", [])
97
+ entity = data.get("entity")
98
+ if entity is None:
99
+ entity = data.get("entities", [])
100
+ series = data.get("series", [])
101
+
102
+ return cls(
103
+ metric=list(metric or []),
104
+ entity=list(entity or []),
105
+ series=list(series or []),
106
+ frequency=data.get("frequency"),
107
+ taxonomy=data.get("taxonomy"),
108
+ head=data.get("head"),
109
+ tail=data.get("tail"),
110
+ as_of=data.get("as_of"),
111
+ sort_by=data.get("sort_by"),
112
+ start_date=data.get("start_date"),
113
+ end_date=data.get("end_date"),
114
+ start_time=data.get("start_time"),
115
+ end_time=data.get("end_time"),
116
+ order_by=data.get("order_by"),
117
+ )
118
+
119
+ def save(self, path: PathLike) -> None:
120
+ """Write this session to a JSON file."""
121
+ with open(path, "w", encoding="utf-8") as f:
122
+ json.dump(self.to_dict(), f, indent=2)
123
+
124
+ @classmethod
125
+ def load(cls, path: PathLike) -> Session:
126
+ """Load a session from a JSON file."""
127
+ with open(path, "r", encoding="utf-8") as f:
128
+ return cls.from_dict(json.load(f))
129
+
130
+ def add_metric(self, resource_id: str) -> bool:
131
+ """Append a metric ID if not already present. Returns True if added."""
132
+ if resource_id in self.metric:
133
+ return False
134
+ self.metric.append(resource_id)
135
+ return True
136
+
137
+ def add_entity(self, resource_id: str) -> bool:
138
+ """Append an entity ID if not already present. Returns True if added."""
139
+ if resource_id in self.entity:
140
+ return False
141
+ self.entity.append(resource_id)
142
+ return True
143
+
144
+ def add_series(self, resource_id: str) -> bool:
145
+ """Append a series ID if not already present. Returns True if added."""
146
+ if resource_id in self.series:
147
+ return False
148
+ self.series.append(resource_id)
149
+ return True
150
+
151
+ def remove_id(self, resource_id: str) -> bool:
152
+ """Remove an ID from metric/entity/series lists. Returns True if removed."""
153
+ removed = False
154
+ for collection in (self.metric, self.entity, self.series):
155
+ while resource_id in collection:
156
+ collection.remove(resource_id)
157
+ removed = True
158
+ return removed
159
+
160
+ # --- Session operations ---
161
+
162
+ def execute(
163
+ self, client: JSTDataClient, **overrides: Any
164
+ ) -> List[TimeSeries]:
165
+ """Run this session against ``client.query``.
166
+
167
+ ``overrides`` are merged on top of ``to_query_kwargs()`` (e.g.
168
+ ``order_by="desc"`` for a display preference).
169
+ """
170
+ kwargs = self.to_query_kwargs()
171
+ kwargs.update(overrides)
172
+ return client.query(**kwargs)
173
+
174
+ def execute_df(self, client: JSTDataClient, **overrides: Any) -> pd.DataFrame:
175
+ """Run this session and return a flattened DataFrame."""
176
+ kwargs = self.to_query_kwargs()
177
+ kwargs.update(overrides)
178
+ return client.query_df(**kwargs)
179
+
180
+ def to_cli(self, program: str = "jst") -> str:
181
+ """Render a reproducible CLI invocation for this session."""
182
+ parts = [program, "query"]
183
+ for m in self.metric:
184
+ parts.append(f"--metric {m}")
185
+ for e in self.entity:
186
+ parts.append(f"--entity {e}")
187
+ for s in self.series:
188
+ parts.append(f"--series {s}")
189
+ if self.frequency:
190
+ parts.append(f"--frequency {self.frequency}")
191
+ if self.taxonomy:
192
+ parts.append(f"--taxonomy {self.taxonomy}")
193
+ if self.head is not None:
194
+ parts.append(f"--head {self.head}")
195
+ if self.tail is not None:
196
+ parts.append(f"--tail {self.tail}")
197
+ if self.as_of:
198
+ parts.append(f"--as-of {self.as_of}")
199
+ if self.sort_by:
200
+ parts.append(f"--sort-by {self.sort_by}")
201
+ return " ".join(parts)
202
+
203
+ def to_python(self) -> str:
204
+ """Render a reproducible Python snippet for this session."""
205
+ kwargs = self.to_query_kwargs()
206
+ if not kwargs:
207
+ args = ""
208
+ else:
209
+ args = ",\n".join(f" {key}={value!r}" for key, value in kwargs.items())
210
+ args = f"\n{args}\n"
211
+ return (
212
+ "from jstdata import JSTDataClient\n"
213
+ "\n"
214
+ "client = JSTDataClient()\n"
215
+ f"df = client.query_df({args})\n"
216
+ "print(df)"
217
+ )
218
+
219
+ def to_csv(
220
+ self,
221
+ client: JSTDataClient,
222
+ path: Optional[PathLike] = None,
223
+ **overrides: Any,
224
+ ) -> Path:
225
+ """Execute this session and write observations to a CSV file.
226
+
227
+ Returns the path written. When ``path`` is omitted, a timestamped
228
+ ``jst_export_*.csv`` is created in the current directory.
229
+ """
230
+ results = self.execute(client, **overrides)
231
+ out = Path(
232
+ path
233
+ if path is not None
234
+ else f"jst_export_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv"
235
+ )
236
+
237
+ with open(out, "w", newline="", encoding="utf-8") as f:
238
+ writer = csv.writer(f)
239
+ writer.writerow(["DATE", "LABEL", "VALUE", "UNITS", "SOURCE"])
240
+ for ts in results:
241
+ for obs in ts.observations:
242
+ writer.writerow(
243
+ [
244
+ obs.observation_timestamp.strftime("%Y-%m-%d"),
245
+ ts.series.label,
246
+ obs.value,
247
+ ts.series.units,
248
+ ts.series.source,
249
+ ]
250
+ )
251
+ return out
jstdata/utils.py ADDED
@@ -0,0 +1,103 @@
1
+ import json
2
+ from io import StringIO
3
+ from dataclasses import is_dataclass, asdict
4
+
5
+ import click
6
+ import pandas as pd
7
+ from tabulate import tabulate
8
+
9
+
10
+ def common_params(f):
11
+ """
12
+ Decorator to apply common CLI parameters: limit, offset, and format.
13
+ """
14
+ f = click.option(
15
+ "--limit",
16
+ default=100,
17
+ help="Maximum number of records to return (default: 100)",
18
+ )(f)
19
+ f = click.option(
20
+ "--offset", default=0, help="Number of records to skip (default: 0)"
21
+ )(f)
22
+ f = click.option(
23
+ "--format",
24
+ default="pretty",
25
+ help="Output format. Valid formats are: json, csv, pretty.",
26
+ )(f)
27
+ return f
28
+
29
+ def common_search_params(f):
30
+ """
31
+ Decorator to apply common CLI parameters: limit, offset, and format.
32
+ Includes different defaults for limit
33
+ """
34
+ f = click.option(
35
+ "--limit",
36
+ default=5,
37
+ help="Maximum number of records to return (default: 100)",
38
+ )(f)
39
+ f = click.option(
40
+ "--offset", default=0, help="Number of records to skip (default: 0)"
41
+ )(f)
42
+ f = click.option(
43
+ "--format",
44
+ default="pretty",
45
+ help="Output format. Valid formats are: json, csv, pretty.",
46
+ )(f)
47
+ return f
48
+
49
+
50
+
51
+ def df_to_csv_string(df):
52
+ csv_buffer = StringIO()
53
+ df.to_csv(csv_buffer, index=False)
54
+ return csv_buffer.getvalue()
55
+
56
+
57
+ def format_and_print(response_data, format):
58
+ # Convert dataclasses to dicts if necessary
59
+ if isinstance(response_data, list):
60
+ data = [asdict(r) if is_dataclass(r) else r for r in response_data]
61
+ elif is_dataclass(response_data):
62
+ data = asdict(response_data)
63
+ else:
64
+ data = response_data
65
+
66
+ if format == "json":
67
+ # Handle datetime serialization in JSON
68
+ def default_serializer(obj):
69
+ if hasattr(obj, "isoformat"):
70
+ return obj.isoformat()
71
+ raise TypeError(f"Object of type {type(obj)} is not JSON serializable")
72
+
73
+ click.echo(json.dumps(data, indent=2, default=default_serializer))
74
+ elif format == "csv":
75
+ if isinstance(data, dict):
76
+ data = [data]
77
+ click.echo(df_to_csv_string(pd.DataFrame(data)))
78
+ elif format == "pretty":
79
+ if not data:
80
+ click.echo("No records found.")
81
+ return
82
+
83
+ if isinstance(data, dict):
84
+ # Single record: vertical table
85
+ table_data = [(k, v) for k, v in data.items()]
86
+ click.echo(
87
+ tabulate(table_data, headers=["key", "value"], tablefmt="pretty")
88
+ )
89
+ return
90
+
91
+ # List of records: horizontal table
92
+ # Flatten nested structures (like 'entities' in Series) for better display
93
+ flattened_data = []
94
+ for item in data:
95
+ flat_item = {}
96
+ for k, v in item.items():
97
+ if isinstance(v, list):
98
+ flat_item[k] = ", ".join([str(i.get("id", i)) if isinstance(i, dict) else str(i) for i in v])
99
+ else:
100
+ flat_item[k] = v
101
+ flattened_data.append(flat_item)
102
+
103
+ click.echo(tabulate(flattened_data, headers="keys", tablefmt="pretty"))
@@ -0,0 +1,88 @@
1
+ """Interactive steps and saved workflows.
2
+
3
+ ::
4
+
5
+ jst steps
6
+ jst step console
7
+ jst run console : discover --mode union
8
+ jst workflow create --id gdp -- console : rank
9
+ jst workflow run gdp
10
+ """
11
+
12
+ from .base import (
13
+ PipelineError,
14
+ ResolvedStep,
15
+ StepArgument,
16
+ StepBinding,
17
+ StepSpec,
18
+ default_session_path,
19
+ format_step_help,
20
+ get_step,
21
+ list_steps,
22
+ parse_step_kwargs,
23
+ resolve_pipeline,
24
+ run_pipeline,
25
+ run_resolved_pipeline,
26
+ split_pipeline,
27
+ )
28
+ from .store import (
29
+ BUNDLED_WORKFLOW_IDS,
30
+ SavedStep,
31
+ SavedWorkflow,
32
+ TUTORIAL_WORKFLOW_ID,
33
+ WorkflowStoreError,
34
+ create_workflow_from_tokens,
35
+ delete_workflow,
36
+ format_pipeline,
37
+ is_bundled_workflow,
38
+ list_bundled_workflows,
39
+ list_saved_workflows,
40
+ load_bundled_workflow,
41
+ load_workflow,
42
+ resolve_saved_workflow,
43
+ save_workflow,
44
+ validate_workflow_id,
45
+ workflow_exists,
46
+ workflow_to_tokens,
47
+ )
48
+ from .tutorial import TutorialCoach, run_tutorial
49
+ from . import console as _console # noqa: F401 — registers console
50
+ from . import discover as _discover # noqa: F401 — registers discover
51
+ from . import rank as _rank # noqa: F401 — registers rank
52
+
53
+ __all__ = [
54
+ "BUNDLED_WORKFLOW_IDS",
55
+ "PipelineError",
56
+ "ResolvedStep",
57
+ "SavedStep",
58
+ "SavedWorkflow",
59
+ "StepArgument",
60
+ "StepBinding",
61
+ "StepSpec",
62
+ "TUTORIAL_WORKFLOW_ID",
63
+ "TutorialCoach",
64
+ "WorkflowStoreError",
65
+ "create_workflow_from_tokens",
66
+ "default_session_path",
67
+ "delete_workflow",
68
+ "format_pipeline",
69
+ "format_step_help",
70
+ "get_step",
71
+ "is_bundled_workflow",
72
+ "list_bundled_workflows",
73
+ "list_saved_workflows",
74
+ "list_steps",
75
+ "load_bundled_workflow",
76
+ "load_workflow",
77
+ "parse_step_kwargs",
78
+ "resolve_pipeline",
79
+ "resolve_saved_workflow",
80
+ "run_pipeline",
81
+ "run_resolved_pipeline",
82
+ "run_tutorial",
83
+ "save_workflow",
84
+ "split_pipeline",
85
+ "validate_workflow_id",
86
+ "workflow_exists",
87
+ "workflow_to_tokens",
88
+ ]