ohmydata 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ohmydata-0.2.2 → ohmydata-0.2.3}/CHANGELOG.md +14 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/PKG-INFO +14 -5
- {ohmydata-0.2.2 → ohmydata-0.2.3}/README.md +12 -3
- {ohmydata-0.2.2 → ohmydata-0.2.3}/pyproject.toml +2 -2
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/__init__.py +1 -1
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/__init__.py +2 -0
- ohmydata-0.2.3/src/ohmydata/providers/sec/_statement_parser.py +375 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/cli.py +6 -16
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/edgartools_adapter.py +107 -157
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/financials.py +35 -2
- ohmydata-0.2.3/src/ohmydata/providers/sec/financials_dataset.py +284 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/__init__.py +2 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/_fundamentals_parsers.py +27 -45
- ohmydata-0.2.3/src/ohmydata/providers/yfinance/_fundamentals_periods.py +121 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/fundamentals.py +146 -36
- {ohmydata-0.2.2 → ohmydata-0.2.3}/uv.lock +2 -2
- ohmydata-0.2.2/src/ohmydata/providers/sec/financials_dataset.py +0 -203
- {ohmydata-0.2.2 → ohmydata-0.2.3}/.gitignore +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/adapters/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/adapters/polars.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/cli.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/_vintage_lock.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/availability.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/facts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/policy.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/provenance.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/rate_limit.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/snapshot.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/specs.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/vintage.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/artifacts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/batch.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/core_dataset.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/edgar.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/http.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/nport.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/qualification.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/qualification_dataset.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/client.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/capture.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/serialization.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_adjusted_bars.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_index_mapping.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_pcf_history.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/index_weight_vintage.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/lookthrough_bundle.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/shared.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/weighted_dividend_yield.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_artifacts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_canonical.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_gates.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_plane.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_producer.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/client.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/quality.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/cli.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/drift_audit.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/universes.py +0 -0
|
@@ -1,5 +1,19 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.2.3 — 2026-09-09
|
|
4
|
+
|
|
5
|
+
- Repair SEC financial-period extraction and filing eligibility in the OMD adapter
|
|
6
|
+
against the existing edgartools 5.56.0 baseline; pin that tested optional dependency.
|
|
7
|
+
- Introduce financial dataset/identity v2 for period and dimension metadata and
|
|
8
|
+
explicit coverage, with lossless decimal strings in Parquet and atomic immutable
|
|
9
|
+
partition publication. Existing v1 datasets require rebuilding in a new root.
|
|
10
|
+
- Bind Yahoo financial values to actual report columns, retain nulls and per-metric
|
|
11
|
+
source periods, and match prior-year columns by a documented calendar policy.
|
|
12
|
+
Remove the silent EBIT fallback for operating income.
|
|
13
|
+
- Use actual quotes for FY1 calibration, preserve quote and forecast provenance,
|
|
14
|
+
and expose unknown accounting comparability. Raw fallback remains available.
|
|
15
|
+
- See [migration and changed semantics](docs/financial-period-integrity.md).
|
|
16
|
+
|
|
3
17
|
## 0.2.2 — 2026-09-06
|
|
4
18
|
|
|
5
19
|
- Enhanced `yfinance` fundamentals pipeline with institutional Forward P/E calibration and GAAP distortion detection:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: ohmydata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Financial and alternative market data ingestion SDK.
|
|
5
5
|
Author: OMD contributors
|
|
6
6
|
Requires-Python: <3.13,>=3.11
|
|
@@ -11,7 +11,7 @@ Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'polars'
|
|
|
11
11
|
Provides-Extra: sec-cli
|
|
12
12
|
Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-cli'
|
|
13
13
|
Provides-Extra: sec-financials
|
|
14
|
-
Requires-Dist: edgartools
|
|
14
|
+
Requires-Dist: edgartools==5.56.0; extra == 'sec-financials'
|
|
15
15
|
Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-financials'
|
|
16
16
|
Provides-Extra: tushare
|
|
17
17
|
Requires-Dist: pandas<3.0,>=2.0; extra == 'tushare'
|
|
@@ -506,6 +506,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
|
|
|
506
506
|
beside standardized XBRL categories (`standard_concept`) for cross-company
|
|
507
507
|
quantitative comparisons.
|
|
508
508
|
|
|
509
|
+
The SEC extra supports `edgartools==5.56.0`. See the
|
|
510
|
+
[financial period contract and v2 migration](docs/financial-period-integrity.md)
|
|
511
|
+
before rebuilding existing financial datasets.
|
|
512
|
+
|
|
509
513
|
### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
|
|
510
514
|
|
|
511
515
|
Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
|
|
@@ -535,7 +539,7 @@ bars_req = YFinanceDailyBarsRequest(
|
|
|
535
539
|
bars_result = client.fetch_daily_bars(bars_req)
|
|
536
540
|
df = bars_result.dataframe
|
|
537
541
|
|
|
538
|
-
# 2. Fetch fundamentals with
|
|
542
|
+
# 2. Fetch fundamentals with FY1 Forward P/E and source metadata
|
|
539
543
|
fund_req = YFinanceFundamentalsRequest(
|
|
540
544
|
symbols=("NVDA", "GEV"),
|
|
541
545
|
include_financials=True,
|
|
@@ -550,11 +554,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
|
|
|
550
554
|
print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
|
|
551
555
|
|
|
552
556
|
gev = fund_result.records["GEV"]
|
|
553
|
-
#
|
|
557
|
+
# Legacy flag measures EPS-source divergence; accounting basis remains unknown.
|
|
554
558
|
if gev.estimates.has_gaap_distortion:
|
|
555
|
-
print(f"GEV
|
|
559
|
+
print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
|
|
556
560
|
```
|
|
557
561
|
|
|
562
|
+
Financial values bind to actual statement columns, with per-metric dates and
|
|
563
|
+
coverage. FY1 calibration requires an actual quote and compatible currencies;
|
|
564
|
+
otherwise raw values remain available. See the
|
|
565
|
+
[period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
|
|
566
|
+
|
|
558
567
|
#### Zero-Drift Audit CLI (`omd audit-drift`)
|
|
559
568
|
|
|
560
569
|
Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
|
|
@@ -480,6 +480,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
|
|
|
480
480
|
beside standardized XBRL categories (`standard_concept`) for cross-company
|
|
481
481
|
quantitative comparisons.
|
|
482
482
|
|
|
483
|
+
The SEC extra supports `edgartools==5.56.0`. See the
|
|
484
|
+
[financial period contract and v2 migration](docs/financial-period-integrity.md)
|
|
485
|
+
before rebuilding existing financial datasets.
|
|
486
|
+
|
|
483
487
|
### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
|
|
484
488
|
|
|
485
489
|
Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
|
|
@@ -509,7 +513,7 @@ bars_req = YFinanceDailyBarsRequest(
|
|
|
509
513
|
bars_result = client.fetch_daily_bars(bars_req)
|
|
510
514
|
df = bars_result.dataframe
|
|
511
515
|
|
|
512
|
-
# 2. Fetch fundamentals with
|
|
516
|
+
# 2. Fetch fundamentals with FY1 Forward P/E and source metadata
|
|
513
517
|
fund_req = YFinanceFundamentalsRequest(
|
|
514
518
|
symbols=("NVDA", "GEV"),
|
|
515
519
|
include_financials=True,
|
|
@@ -524,11 +528,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
|
|
|
524
528
|
print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
|
|
525
529
|
|
|
526
530
|
gev = fund_result.records["GEV"]
|
|
527
|
-
#
|
|
531
|
+
# Legacy flag measures EPS-source divergence; accounting basis remains unknown.
|
|
528
532
|
if gev.estimates.has_gaap_distortion:
|
|
529
|
-
print(f"GEV
|
|
533
|
+
print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
|
|
530
534
|
```
|
|
531
535
|
|
|
536
|
+
Financial values bind to actual statement columns, with per-metric dates and
|
|
537
|
+
coverage. FY1 calibration requires an actual quote and compatible currencies;
|
|
538
|
+
otherwise raw values remain available. See the
|
|
539
|
+
[period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
|
|
540
|
+
|
|
532
541
|
#### Zero-Drift Audit CLI (`omd audit-drift`)
|
|
533
542
|
|
|
534
543
|
Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "ohmydata"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.3"
|
|
8
8
|
description = "Financial and alternative market data ingestion SDK."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11,<3.13"
|
|
@@ -25,7 +25,7 @@ polars = [
|
|
|
25
25
|
vintage-plane = ["pandas>=2.0,<3.0", "pyarrow>=23.0.1,<24.0"]
|
|
26
26
|
sec-cli = ["pyarrow>=23.0.1,<24.0"]
|
|
27
27
|
sec-financials = [
|
|
28
|
-
"edgartools
|
|
28
|
+
"edgartools==5.56.0",
|
|
29
29
|
"pyarrow>=23.0.1,<24.0",
|
|
30
30
|
]
|
|
31
31
|
|
|
@@ -36,6 +36,7 @@ from .edgar import (
|
|
|
36
36
|
)
|
|
37
37
|
from .edgartools_adapter import (
|
|
38
38
|
SecFinancialsClient,
|
|
39
|
+
SecStatementParseError,
|
|
39
40
|
ensure_edgar_available,
|
|
40
41
|
parse_statement_rows,
|
|
41
42
|
validate_user_agent,
|
|
@@ -107,6 +108,7 @@ __all__ = [
|
|
|
107
108
|
"SecPayloadReceipt",
|
|
108
109
|
"SecReplaySession",
|
|
109
110
|
"SecScheduledFundSelector",
|
|
111
|
+
"SecStatementParseError",
|
|
110
112
|
"SecStatementRow",
|
|
111
113
|
"SecTransportEvidence",
|
|
112
114
|
"SecUnavailableResult",
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
"""Edgar 5.56 statement boundary: native facts first, display frames as compatibility input."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from dataclasses import replace
|
|
9
|
+
from datetime import date
|
|
10
|
+
from decimal import Decimal, InvalidOperation
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from .financials import SecStatementRow, StatementType
|
|
14
|
+
|
|
15
|
+
_DISPLAY_PERIOD = re.compile(r"^(\d{4}-\d{2}-\d{2})(?:\s+\((FY|Q[1-4]|YTD)\))?$")
|
|
16
|
+
_STRUCTURED_PERIOD = re.compile(
|
|
17
|
+
r"^(?:(instant)_(\d{4}-\d{2}-\d{2})|(duration)_(\d{4}-\d{2}-\d{2})_(\d{4}-\d{2}-\d{2}))$"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class SecStatementParseError(ValueError):
|
|
22
|
+
"""A supplied financial statement could not be interpreted without losing identity."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _missing(value: Any) -> bool:
|
|
26
|
+
return (
|
|
27
|
+
value is None
|
|
28
|
+
or type(value).__name__ in {"NAType", "NaTType"}
|
|
29
|
+
or str(value).strip().lower() in {"", "nan", "none", "null", "<na>", "nat"}
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _flag(value: Any) -> bool:
|
|
34
|
+
if _missing(value):
|
|
35
|
+
return False
|
|
36
|
+
if type(value).__name__ in {"bool", "bool_"}:
|
|
37
|
+
return bool(value)
|
|
38
|
+
raise SecStatementParseError("invalid statement boolean metadata")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _text(value: Any) -> str | None:
|
|
42
|
+
return None if _missing(value) else str(value).strip()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _number(value: Any) -> Decimal | None:
|
|
46
|
+
if isinstance(value, Decimal) and not value.is_finite():
|
|
47
|
+
raise SecStatementParseError("non-finite statement fact")
|
|
48
|
+
if type(value).__name__ in {"bool", "bool_"} or _missing(value):
|
|
49
|
+
return None
|
|
50
|
+
try:
|
|
51
|
+
result = Decimal(str(value).strip())
|
|
52
|
+
except (InvalidOperation, TypeError, ValueError) as exc:
|
|
53
|
+
raise SecStatementParseError("non-numeric statement fact") from exc
|
|
54
|
+
if not result.is_finite():
|
|
55
|
+
raise SecStatementParseError("non-finite statement fact")
|
|
56
|
+
return result
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _period(key: str) -> tuple[str, date | None, date]:
|
|
60
|
+
match = _STRUCTURED_PERIOD.fullmatch(key)
|
|
61
|
+
if not match:
|
|
62
|
+
raise SecStatementParseError("unrecognized structured period key")
|
|
63
|
+
try:
|
|
64
|
+
if match.group(1):
|
|
65
|
+
return "instant", None, date.fromisoformat(match.group(2))
|
|
66
|
+
start, end = date.fromisoformat(match.group(4)), date.fromisoformat(match.group(5))
|
|
67
|
+
if start > end:
|
|
68
|
+
raise ValueError("reversed period")
|
|
69
|
+
return "duration", start, end
|
|
70
|
+
except ValueError as exc:
|
|
71
|
+
raise SecStatementParseError("invalid structured period dates") from exc
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _dimensions(item: Mapping[str, Any]) -> dict[str, str]:
|
|
75
|
+
metadata = item.get("dimension_metadata")
|
|
76
|
+
if metadata is None:
|
|
77
|
+
return {}
|
|
78
|
+
if not isinstance(metadata, list):
|
|
79
|
+
raise SecStatementParseError("invalid dimension metadata")
|
|
80
|
+
result = {}
|
|
81
|
+
for dim in metadata:
|
|
82
|
+
if not isinstance(dim, dict) or not dim.get("dimension") or not dim.get("member"):
|
|
83
|
+
raise SecStatementParseError("dimension identity unavailable")
|
|
84
|
+
axis, member = str(dim["dimension"]), str(dim["member"])
|
|
85
|
+
if axis in result and result[axis] != member:
|
|
86
|
+
raise SecStatementParseError("conflicting dimension identity")
|
|
87
|
+
result[axis] = member
|
|
88
|
+
return result
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _dim_key(dimensions: Mapping[str, Any]) -> str:
|
|
92
|
+
return json.dumps(dict(dimensions), sort_keys=True, separators=(",", ":"))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _native_index(xbrl: Any, concepts: set[str]) -> dict[tuple[str, str, str], list[Any]]:
|
|
96
|
+
"""One pass over facts, avoiding a full source scan for each row or period."""
|
|
97
|
+
result: dict[tuple[str, str, str], list[Any]] = {}
|
|
98
|
+
seen: set[int] = set()
|
|
99
|
+
for fact in xbrl.facts.values():
|
|
100
|
+
if id(fact) in seen:
|
|
101
|
+
continue
|
|
102
|
+
seen.add(id(fact))
|
|
103
|
+
concept = fact.element_id.replace(":", "_")
|
|
104
|
+
if concept not in concepts:
|
|
105
|
+
continue
|
|
106
|
+
context = xbrl.contexts.get(fact.context_ref)
|
|
107
|
+
period = xbrl.context_period_map.get(fact.context_ref)
|
|
108
|
+
if context is None or period is None:
|
|
109
|
+
raise SecStatementParseError("native fact context unavailable")
|
|
110
|
+
key = concept, period, _dim_key(context.dimensions)
|
|
111
|
+
result.setdefault(key, []).append(fact)
|
|
112
|
+
return result
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _structured_rows(
|
|
116
|
+
data: list[Any],
|
|
117
|
+
kind: StatementType,
|
|
118
|
+
dimensions: bool,
|
|
119
|
+
xbrl: Any = None,
|
|
120
|
+
native_index: dict[tuple[str, str, str], list[Any]] | None = None,
|
|
121
|
+
) -> list[SecStatementRow]:
|
|
122
|
+
if any(not isinstance(item, dict) for item in data):
|
|
123
|
+
raise SecStatementParseError("invalid structured statement rows")
|
|
124
|
+
concepts = {str(item.get("concept", "")).replace(":", "_") for item in data}
|
|
125
|
+
index = (
|
|
126
|
+
native_index
|
|
127
|
+
if native_index is not None
|
|
128
|
+
else _native_index(xbrl, concepts)
|
|
129
|
+
if xbrl is not None
|
|
130
|
+
else None
|
|
131
|
+
)
|
|
132
|
+
rows = []
|
|
133
|
+
seen: dict[tuple[str, str, str, str | None], SecStatementRow] = {}
|
|
134
|
+
for item in data:
|
|
135
|
+
if _flag(item.get("is_abstract", item.get("abstract"))):
|
|
136
|
+
continue
|
|
137
|
+
dims = _dimensions(item)
|
|
138
|
+
dimensional = _flag(item.get("is_dimension", item.get("dimension"))) or bool(dims)
|
|
139
|
+
if dimensional and not dimensions:
|
|
140
|
+
continue
|
|
141
|
+
if dimensional and not dims:
|
|
142
|
+
raise SecStatementParseError("dimension identity unavailable")
|
|
143
|
+
concept = _text(item.get("concept"))
|
|
144
|
+
if not concept:
|
|
145
|
+
raise SecStatementParseError("statement concept unavailable")
|
|
146
|
+
values = item.get("values", {})
|
|
147
|
+
if not isinstance(values, dict):
|
|
148
|
+
raise SecStatementParseError("invalid structured values")
|
|
149
|
+
for key, raw in values.items():
|
|
150
|
+
period_type, start, end = _period(str(key))
|
|
151
|
+
declared_type = (item.get("period_types") or {}).get(key)
|
|
152
|
+
if declared_type is not None and declared_type != period_type:
|
|
153
|
+
raise SecStatementParseError("conflicting period type")
|
|
154
|
+
if index is None and _number(raw) is None:
|
|
155
|
+
continue
|
|
156
|
+
unit_ref = (item.get("units") or {}).get(key)
|
|
157
|
+
native_decimals = (item.get("decimals") or {}).get(key)
|
|
158
|
+
context_ref = None
|
|
159
|
+
context_refs: list[str | None] = [None]
|
|
160
|
+
if index is not None:
|
|
161
|
+
facts = index.get((concept.replace(":", "_"), key, _dim_key(dims)), [])
|
|
162
|
+
if not facts:
|
|
163
|
+
raise SecStatementParseError("statement value has no native fact context")
|
|
164
|
+
identities = {(f.value, f.unit_ref, str(f.decimals)) for f in facts}
|
|
165
|
+
if len(identities) != 1:
|
|
166
|
+
raise SecStatementParseError(
|
|
167
|
+
"conflicting native facts for period and dimensions"
|
|
168
|
+
)
|
|
169
|
+
fact = min(facts, key=lambda f: f.context_ref)
|
|
170
|
+
raw, unit_ref, native_decimals = fact.value, fact.unit_ref, fact.decimals
|
|
171
|
+
context_ref = fact.context_ref
|
|
172
|
+
context_refs = sorted({f.context_ref for f in facts})
|
|
173
|
+
value = _number(raw)
|
|
174
|
+
if value is None:
|
|
175
|
+
continue
|
|
176
|
+
unit = _text(unit_ref) or _text(item.get("unit"))
|
|
177
|
+
if xbrl is not None and unit_ref:
|
|
178
|
+
definition = xbrl.units.get(unit_ref)
|
|
179
|
+
if definition is None:
|
|
180
|
+
unit = None
|
|
181
|
+
elif isinstance(definition, dict):
|
|
182
|
+
unit = _text(definition.get("measure")) or json.dumps(
|
|
183
|
+
definition, sort_keys=True
|
|
184
|
+
)
|
|
185
|
+
else:
|
|
186
|
+
raise SecStatementParseError("invalid native unit definition")
|
|
187
|
+
decimals = None
|
|
188
|
+
if native_decimals is not None and str(native_decimals) != "INF":
|
|
189
|
+
try:
|
|
190
|
+
decimals = int(native_decimals)
|
|
191
|
+
except (TypeError, ValueError) as exc:
|
|
192
|
+
raise SecStatementParseError("invalid fact precision") from exc
|
|
193
|
+
row = SecStatementRow(
|
|
194
|
+
statement_type=kind,
|
|
195
|
+
standard_concept=_text(item.get("standard_concept")) or concept,
|
|
196
|
+
concept=concept,
|
|
197
|
+
label=_text(item.get("label")) or "",
|
|
198
|
+
value=value,
|
|
199
|
+
value_native=str(raw),
|
|
200
|
+
unit=unit,
|
|
201
|
+
unit_ref=_text(unit_ref),
|
|
202
|
+
decimals=decimals,
|
|
203
|
+
decimals_native=_text(native_decimals),
|
|
204
|
+
period_start=start,
|
|
205
|
+
period_end=end,
|
|
206
|
+
period_type=period_type,
|
|
207
|
+
period_key=str(key),
|
|
208
|
+
context_ref=context_ref,
|
|
209
|
+
dimension=_dim_key(dims) if dims else None,
|
|
210
|
+
is_point_in_time=period_type == "instant",
|
|
211
|
+
period_source="xbrl-context" if index is not None else "structured-period-key",
|
|
212
|
+
)
|
|
213
|
+
for reference in context_refs:
|
|
214
|
+
contextual_row = replace(row, context_ref=reference)
|
|
215
|
+
identity = concept, str(key), _dim_key(dims), reference
|
|
216
|
+
if identity in seen:
|
|
217
|
+
previous = seen[identity]
|
|
218
|
+
if (previous.value, previous.unit) != (row.value, row.unit):
|
|
219
|
+
raise SecStatementParseError("conflicting duplicate statement facts")
|
|
220
|
+
continue
|
|
221
|
+
seen[identity] = contextual_row
|
|
222
|
+
rows.append(contextual_row)
|
|
223
|
+
return rows
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _native_statement_rows(
|
|
227
|
+
statement: Any, kind: StatementType, dimensions: bool
|
|
228
|
+
) -> list[SecStatementRow]:
|
|
229
|
+
"""Use presentation membership plus instance facts, bypassing display deduplication."""
|
|
230
|
+
xbrl = statement.xbrl
|
|
231
|
+
_, role, _ = xbrl.find_statement(statement.canonical_type or statement.role_or_type)
|
|
232
|
+
tree = xbrl.presentation_trees.get(role)
|
|
233
|
+
if tree is None:
|
|
234
|
+
raise SecStatementParseError("statement presentation tree unavailable")
|
|
235
|
+
nodes = {
|
|
236
|
+
node.element_id.replace(":", "_"): node
|
|
237
|
+
for node in tree.all_nodes.values()
|
|
238
|
+
if not node.is_abstract
|
|
239
|
+
}
|
|
240
|
+
index = _native_index(xbrl, set(nodes))
|
|
241
|
+
data = []
|
|
242
|
+
for (concept, key, dimension_key), facts in sorted(index.items()):
|
|
243
|
+
dims = json.loads(dimension_key)
|
|
244
|
+
if dims and not dimensions:
|
|
245
|
+
continue
|
|
246
|
+
node = nodes[concept]
|
|
247
|
+
data.append(
|
|
248
|
+
{
|
|
249
|
+
"concept": node.element_id,
|
|
250
|
+
"label": node.display_label,
|
|
251
|
+
"is_dimension": bool(dims),
|
|
252
|
+
"dimension_metadata": [
|
|
253
|
+
{"dimension": axis, "member": member} for axis, member in dims.items()
|
|
254
|
+
],
|
|
255
|
+
"values": {key: facts[0].value},
|
|
256
|
+
}
|
|
257
|
+
)
|
|
258
|
+
return _structured_rows(data, kind, dimensions, xbrl, index)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _display_rows(statement: Any, kind: StatementType, dimensions: bool) -> list[SecStatementRow]:
|
|
262
|
+
try:
|
|
263
|
+
frame = statement.to_dataframe(
|
|
264
|
+
standard=False,
|
|
265
|
+
include_unit=True,
|
|
266
|
+
include_point_in_time=True,
|
|
267
|
+
include_standardization=True,
|
|
268
|
+
presentation=False,
|
|
269
|
+
)
|
|
270
|
+
except Exception as exc:
|
|
271
|
+
raise SecStatementParseError("failed to parse statement dataframe") from exc
|
|
272
|
+
if frame is None:
|
|
273
|
+
raise SecStatementParseError("statement dataframe unavailable")
|
|
274
|
+
if frame.empty:
|
|
275
|
+
return []
|
|
276
|
+
metadata_columns = {
|
|
277
|
+
"concept",
|
|
278
|
+
"label",
|
|
279
|
+
"standard_concept",
|
|
280
|
+
"unit",
|
|
281
|
+
"abstract",
|
|
282
|
+
"dimension",
|
|
283
|
+
"point_in_time",
|
|
284
|
+
"dimension_axis",
|
|
285
|
+
"dimension_member",
|
|
286
|
+
"dimension_member_label",
|
|
287
|
+
"dimension_label",
|
|
288
|
+
"level",
|
|
289
|
+
"balance",
|
|
290
|
+
"weight",
|
|
291
|
+
"preferred_sign",
|
|
292
|
+
"is_breakdown",
|
|
293
|
+
"parent_concept",
|
|
294
|
+
"parent_abstract_concept",
|
|
295
|
+
"original_label",
|
|
296
|
+
"is_standardized",
|
|
297
|
+
}
|
|
298
|
+
for col in frame.columns:
|
|
299
|
+
if col not in metadata_columns and not _DISPLAY_PERIOD.fullmatch(str(col)):
|
|
300
|
+
raise SecStatementParseError("unrecognized financial period column")
|
|
301
|
+
period_cols = [(col, _DISPLAY_PERIOD.fullmatch(str(col))) for col in frame.columns]
|
|
302
|
+
period_cols = [(col, match) for col, match in period_cols if match is not None]
|
|
303
|
+
if not period_cols:
|
|
304
|
+
raise SecStatementParseError("no recognized financial period columns")
|
|
305
|
+
if frame.columns.has_duplicates:
|
|
306
|
+
raise SecStatementParseError("duplicate statement columns")
|
|
307
|
+
rows = []
|
|
308
|
+
for _, item in frame.iterrows():
|
|
309
|
+
if _flag(item.get("abstract")):
|
|
310
|
+
continue
|
|
311
|
+
dimensional = _flag(item.get("dimension"))
|
|
312
|
+
if dimensional and not dimensions:
|
|
313
|
+
continue
|
|
314
|
+
dim = None
|
|
315
|
+
if dimensional:
|
|
316
|
+
axis, member = _text(item.get("dimension_axis")), _text(item.get("dimension_member"))
|
|
317
|
+
if not axis or not member:
|
|
318
|
+
raise SecStatementParseError("dimension identity unavailable in display input")
|
|
319
|
+
dim = _dim_key({axis: member})
|
|
320
|
+
concept = _text(item.get("concept"))
|
|
321
|
+
if not concept:
|
|
322
|
+
raise SecStatementParseError("statement concept unavailable")
|
|
323
|
+
for col, match in period_cols:
|
|
324
|
+
raw = item[col]
|
|
325
|
+
value = _number(raw)
|
|
326
|
+
if value is None:
|
|
327
|
+
continue
|
|
328
|
+
try:
|
|
329
|
+
end = date.fromisoformat(match.group(1))
|
|
330
|
+
except ValueError as exc:
|
|
331
|
+
raise SecStatementParseError("invalid display period date") from exc
|
|
332
|
+
suffix = match.group(2)
|
|
333
|
+
instant = _flag(item.get("point_in_time")) or kind == "balance_sheet"
|
|
334
|
+
if instant and suffix:
|
|
335
|
+
raise SecStatementParseError("conflicting display period type")
|
|
336
|
+
rows.append(
|
|
337
|
+
SecStatementRow(
|
|
338
|
+
statement_type=kind,
|
|
339
|
+
standard_concept=_text(item.get("standard_concept")) or concept,
|
|
340
|
+
concept=concept,
|
|
341
|
+
label=_text(item.get("label")) or "",
|
|
342
|
+
value=value,
|
|
343
|
+
value_native=str(raw),
|
|
344
|
+
unit=_text(item.get("unit")),
|
|
345
|
+
period_end=end,
|
|
346
|
+
period_type="instant" if instant else "duration",
|
|
347
|
+
period_key=str(col),
|
|
348
|
+
dimension=dim,
|
|
349
|
+
is_point_in_time=instant,
|
|
350
|
+
period_source="display-column-unknown-start",
|
|
351
|
+
)
|
|
352
|
+
)
|
|
353
|
+
return rows
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def parse_statement_rows(
|
|
357
|
+
statement: Any, statement_type: StatementType, *, include_dimensions: bool = False
|
|
358
|
+
) -> list[SecStatementRow]:
|
|
359
|
+
"""Preserve native fact identity; never infer duration starts from display suffixes."""
|
|
360
|
+
if statement is None:
|
|
361
|
+
return []
|
|
362
|
+
from edgar.xbrl.statements import Statement
|
|
363
|
+
|
|
364
|
+
try:
|
|
365
|
+
if isinstance(statement, Statement):
|
|
366
|
+
return _native_statement_rows(statement, statement_type, include_dimensions)
|
|
367
|
+
getter = getattr(statement, "get_raw_data", None)
|
|
368
|
+
raw = getter() if callable(getter) else None
|
|
369
|
+
if isinstance(raw, list):
|
|
370
|
+
return _structured_rows(raw, statement_type, include_dimensions)
|
|
371
|
+
return _display_rows(statement, statement_type, include_dimensions)
|
|
372
|
+
except SecStatementParseError:
|
|
373
|
+
raise
|
|
374
|
+
except Exception as exc:
|
|
375
|
+
raise SecStatementParseError("invalid financial statement structure") from exc
|
|
@@ -468,7 +468,6 @@ def run_qualify(args: Any) -> int:
|
|
|
468
468
|
|
|
469
469
|
|
|
470
470
|
def run_financials(args: Any) -> int:
|
|
471
|
-
import hashlib
|
|
472
471
|
import importlib
|
|
473
472
|
|
|
474
473
|
if getattr(args, "config", None):
|
|
@@ -502,7 +501,9 @@ def run_financials(args: Any) -> int:
|
|
|
502
501
|
m_file = p_dir / "manifest.json"
|
|
503
502
|
if not m_file.is_file():
|
|
504
503
|
raise ValueError(f"manifest.json missing for symbol: {sym_str}")
|
|
505
|
-
|
|
504
|
+
from .financials_dataset import validate_financials_partition
|
|
505
|
+
|
|
506
|
+
m_data = validate_financials_partition(p_dir)
|
|
506
507
|
payload.update(m_data)
|
|
507
508
|
|
|
508
509
|
if getattr(args, "rows", False):
|
|
@@ -513,24 +514,13 @@ def run_financials(args: Any) -> int:
|
|
|
513
514
|
payload["rows"] = tbl.to_pylist()[:100]
|
|
514
515
|
|
|
515
516
|
elif cmd == "validate":
|
|
517
|
+
from .financials_dataset import validate_financials_partition
|
|
518
|
+
|
|
516
519
|
root_path = Path(args.root)
|
|
517
520
|
partitions = list(root_path.glob("symbol=*"))
|
|
518
521
|
verified_count = 0
|
|
519
522
|
for p in partitions:
|
|
520
|
-
|
|
521
|
-
if not m_file.is_file():
|
|
522
|
-
raise ValueError(f"manifest missing in {p}")
|
|
523
|
-
m: dict[str, Any] = json.loads(m_file.read_text(encoding="utf-8"))
|
|
524
|
-
files = m.get("files", {})
|
|
525
|
-
for fname, meta in files.items():
|
|
526
|
-
fpath = p / fname
|
|
527
|
-
if not fpath.is_file():
|
|
528
|
-
raise ValueError(f"file missing: {fpath}")
|
|
529
|
-
h = hashlib.sha256(fpath.read_bytes()).hexdigest()
|
|
530
|
-
if h != meta.get("sha256"):
|
|
531
|
-
raise ValueError(
|
|
532
|
-
f"hash mismatch for {fpath}: expected {meta.get('sha256')}, got {h}"
|
|
533
|
-
)
|
|
523
|
+
validate_financials_partition(p)
|
|
534
524
|
verified_count += 1
|
|
535
525
|
payload["status"] = "completed"
|
|
536
526
|
payload["partitions_verified"] = verified_count
|