ohmydata 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ohmydata-0.2.2 → ohmydata-0.2.4}/CHANGELOG.md +29 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/PKG-INFO +14 -5
- {ohmydata-0.2.2 → ohmydata-0.2.4}/README.md +12 -3
- {ohmydata-0.2.2 → ohmydata-0.2.4}/pyproject.toml +2 -2
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/__init__.py +1 -1
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/__init__.py +4 -0
- ohmydata-0.2.4/src/ohmydata/providers/sec/_statement_parser.py +487 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/cli.py +6 -16
- ohmydata-0.2.4/src/ohmydata/providers/sec/edgartools_adapter.py +293 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/financials.py +35 -2
- ohmydata-0.2.4/src/ohmydata/providers/sec/financials_dataset.py +284 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/__init__.py +2 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/_fundamentals_parsers.py +27 -45
- ohmydata-0.2.4/src/ohmydata/providers/yfinance/_fundamentals_periods.py +121 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/fundamentals.py +146 -36
- {ohmydata-0.2.2 → ohmydata-0.2.4}/uv.lock +2 -2
- ohmydata-0.2.2/src/ohmydata/providers/sec/edgartools_adapter.py +0 -324
- ohmydata-0.2.2/src/ohmydata/providers/sec/financials_dataset.py +0 -203
- {ohmydata-0.2.2 → ohmydata-0.2.4}/.gitignore +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/adapters/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/adapters/polars.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/cli.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/_vintage_lock.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/availability.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/facts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/policy.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/provenance.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/rate_limit.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/snapshot.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/specs.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/vintage.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/artifacts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/batch.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/core_dataset.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/edgar.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/http.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/nport.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/qualification.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/qualification_dataset.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/client.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/capture.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/serialization.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_adjusted_bars.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_index_mapping.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_pcf_history.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/index_weight_vintage.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/lookthrough_bundle.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/shared.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/weighted_dividend_yield.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_artifacts.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_canonical.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_gates.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_plane.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_producer.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/client.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/endpoints.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/errors.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/quality.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/__init__.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/cli.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/drift_audit.py +0 -0
- {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/universes.py +0 -0
|
@@ -1,5 +1,34 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.2.4 — 2026-09-09
|
|
4
|
+
|
|
5
|
+
- Fix the SEC native-fact adapter for edgartools 5.56.0: `XBRL.facts` is a
|
|
6
|
+
`FactsView`, while original `Fact` objects belong to `XBRL.parser.facts`.
|
|
7
|
+
Regression fixtures now use actual XBRL/FactsView/Statement objects.
|
|
8
|
+
- A selected filing with zero rows and parsing failures now raises
|
|
9
|
+
`SecFinancialsParseError`; its `vintage` retains coverage/accession and its
|
|
10
|
+
cause retains the first failure. Missing/empty filings without parser failures
|
|
11
|
+
and explicitly flagged partial coverage retain their existing behavior.
|
|
12
|
+
See [the financials contract](docs/financial-period-integrity.md).
|
|
13
|
+
- Native SEC duplicate facts now use exact Decimal precision intervals and retain
|
|
14
|
+
the highest precision original fact per context, while rejecting true value or
|
|
15
|
+
unit conflicts, mixed known/unknown precision, and arithmetic inputs beyond
|
|
16
|
+
the bounded 10,000-digit guard.
|
|
17
|
+
|
|
18
|
+
## 0.2.3 — 2026-09-09
|
|
19
|
+
|
|
20
|
+
- Repair SEC financial-period extraction and filing eligibility in the OMD adapter
|
|
21
|
+
against the existing edgartools 5.56.0 baseline; pin that tested optional dependency.
|
|
22
|
+
- Introduce financial dataset/identity v2 for period and dimension metadata and
|
|
23
|
+
explicit coverage, with lossless decimal strings in Parquet and atomic immutable
|
|
24
|
+
partition publication. Existing v1 datasets require rebuilding in a new root.
|
|
25
|
+
- Bind Yahoo financial values to actual report columns, retain nulls and per-metric
|
|
26
|
+
source periods, and match prior-year columns by a documented calendar policy.
|
|
27
|
+
Remove the silent EBIT fallback for operating income.
|
|
28
|
+
- Use actual quotes for FY1 calibration, preserve quote and forecast provenance,
|
|
29
|
+
and expose unknown accounting comparability. Raw fallback remains available.
|
|
30
|
+
- See [migration and changed semantics](docs/financial-period-integrity.md).
|
|
31
|
+
|
|
3
32
|
## 0.2.2 — 2026-09-06
|
|
4
33
|
|
|
5
34
|
- Enhanced `yfinance` fundamentals pipeline with institutional Forward P/E calibration and GAAP distortion detection:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: ohmydata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: Financial and alternative market data ingestion SDK.
|
|
5
5
|
Author: OMD contributors
|
|
6
6
|
Requires-Python: <3.13,>=3.11
|
|
@@ -11,7 +11,7 @@ Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'polars'
|
|
|
11
11
|
Provides-Extra: sec-cli
|
|
12
12
|
Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-cli'
|
|
13
13
|
Provides-Extra: sec-financials
|
|
14
|
-
Requires-Dist: edgartools
|
|
14
|
+
Requires-Dist: edgartools==5.56.0; extra == 'sec-financials'
|
|
15
15
|
Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-financials'
|
|
16
16
|
Provides-Extra: tushare
|
|
17
17
|
Requires-Dist: pandas<3.0,>=2.0; extra == 'tushare'
|
|
@@ -506,6 +506,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
|
|
|
506
506
|
beside standardized XBRL categories (`standard_concept`) for cross-company
|
|
507
507
|
quantitative comparisons.
|
|
508
508
|
|
|
509
|
+
The SEC extra supports `edgartools==5.56.0`. See the
|
|
510
|
+
[financial period contract and v2 migration](docs/financial-period-integrity.md)
|
|
511
|
+
before rebuilding existing financial datasets.
|
|
512
|
+
|
|
509
513
|
### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
|
|
510
514
|
|
|
511
515
|
Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
|
|
@@ -535,7 +539,7 @@ bars_req = YFinanceDailyBarsRequest(
|
|
|
535
539
|
bars_result = client.fetch_daily_bars(bars_req)
|
|
536
540
|
df = bars_result.dataframe
|
|
537
541
|
|
|
538
|
-
# 2. Fetch fundamentals with
|
|
542
|
+
# 2. Fetch fundamentals with FY1 Forward P/E and source metadata
|
|
539
543
|
fund_req = YFinanceFundamentalsRequest(
|
|
540
544
|
symbols=("NVDA", "GEV"),
|
|
541
545
|
include_financials=True,
|
|
@@ -550,11 +554,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
|
|
|
550
554
|
print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
|
|
551
555
|
|
|
552
556
|
gev = fund_result.records["GEV"]
|
|
553
|
-
#
|
|
557
|
+
# Legacy flag measures EPS-source divergence; accounting basis remains unknown.
|
|
554
558
|
if gev.estimates.has_gaap_distortion:
|
|
555
|
-
print(f"GEV
|
|
559
|
+
print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
|
|
556
560
|
```
|
|
557
561
|
|
|
562
|
+
Financial values bind to actual statement columns, with per-metric dates and
|
|
563
|
+
coverage. FY1 calibration requires an actual quote and compatible currencies;
|
|
564
|
+
otherwise raw values remain available. See the
|
|
565
|
+
[period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
|
|
566
|
+
|
|
558
567
|
#### Zero-Drift Audit CLI (`omd audit-drift`)
|
|
559
568
|
|
|
560
569
|
Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
|
|
@@ -480,6 +480,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
|
|
|
480
480
|
beside standardized XBRL categories (`standard_concept`) for cross-company
|
|
481
481
|
quantitative comparisons.
|
|
482
482
|
|
|
483
|
+
The SEC extra supports `edgartools==5.56.0`. See the
|
|
484
|
+
[financial period contract and v2 migration](docs/financial-period-integrity.md)
|
|
485
|
+
before rebuilding existing financial datasets.
|
|
486
|
+
|
|
483
487
|
### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
|
|
484
488
|
|
|
485
489
|
Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
|
|
@@ -509,7 +513,7 @@ bars_req = YFinanceDailyBarsRequest(
|
|
|
509
513
|
bars_result = client.fetch_daily_bars(bars_req)
|
|
510
514
|
df = bars_result.dataframe
|
|
511
515
|
|
|
512
|
-
# 2. Fetch fundamentals with
|
|
516
|
+
# 2. Fetch fundamentals with FY1 Forward P/E and source metadata
|
|
513
517
|
fund_req = YFinanceFundamentalsRequest(
|
|
514
518
|
symbols=("NVDA", "GEV"),
|
|
515
519
|
include_financials=True,
|
|
@@ -524,11 +528,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
|
|
|
524
528
|
print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
|
|
525
529
|
|
|
526
530
|
gev = fund_result.records["GEV"]
|
|
527
|
-
#
|
|
531
|
+
# Legacy flag measures EPS-source divergence; accounting basis remains unknown.
|
|
528
532
|
if gev.estimates.has_gaap_distortion:
|
|
529
|
-
print(f"GEV
|
|
533
|
+
print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
|
|
530
534
|
```
|
|
531
535
|
|
|
536
|
+
Financial values bind to actual statement columns, with per-metric dates and
|
|
537
|
+
coverage. FY1 calibration requires an actual quote and compatible currencies;
|
|
538
|
+
otherwise raw values remain available. See the
|
|
539
|
+
[period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
|
|
540
|
+
|
|
532
541
|
#### Zero-Drift Audit CLI (`omd audit-drift`)
|
|
533
542
|
|
|
534
543
|
Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "ohmydata"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.4"
|
|
8
8
|
description = "Financial and alternative market data ingestion SDK."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11,<3.13"
|
|
@@ -25,7 +25,7 @@ polars = [
|
|
|
25
25
|
vintage-plane = ["pandas>=2.0,<3.0", "pyarrow>=23.0.1,<24.0"]
|
|
26
26
|
sec-cli = ["pyarrow>=23.0.1,<24.0"]
|
|
27
27
|
sec-financials = [
|
|
28
|
-
"edgartools
|
|
28
|
+
"edgartools==5.56.0",
|
|
29
29
|
"pyarrow>=23.0.1,<24.0",
|
|
30
30
|
]
|
|
31
31
|
|
|
@@ -36,6 +36,8 @@ from .edgar import (
|
|
|
36
36
|
)
|
|
37
37
|
from .edgartools_adapter import (
|
|
38
38
|
SecFinancialsClient,
|
|
39
|
+
SecFinancialsParseError,
|
|
40
|
+
SecStatementParseError,
|
|
39
41
|
ensure_edgar_available,
|
|
40
42
|
parse_statement_rows,
|
|
41
43
|
validate_user_agent,
|
|
@@ -89,6 +91,7 @@ __all__ = [
|
|
|
89
91
|
"SecEmptyPolicy",
|
|
90
92
|
"SecEquityEtfUniverse",
|
|
91
93
|
"SecFinancialsClient",
|
|
94
|
+
"SecFinancialsParseError",
|
|
92
95
|
"SecFinancialsRequest",
|
|
93
96
|
"SecFundHoldingVintage",
|
|
94
97
|
"SecFundSelector",
|
|
@@ -107,6 +110,7 @@ __all__ = [
|
|
|
107
110
|
"SecPayloadReceipt",
|
|
108
111
|
"SecReplaySession",
|
|
109
112
|
"SecScheduledFundSelector",
|
|
113
|
+
"SecStatementParseError",
|
|
110
114
|
"SecStatementRow",
|
|
111
115
|
"SecTransportEvidence",
|
|
112
116
|
"SecUnavailableResult",
|
|
@@ -0,0 +1,487 @@
|
|
|
1
|
+
"""Edgar 5.56 statement boundary: native facts first, display frames as compatibility input."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from datetime import date
|
|
9
|
+
from decimal import Decimal, InvalidOperation
|
|
10
|
+
from fractions import Fraction
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from .financials import SecStatementRow, StatementType
|
|
14
|
+
|
|
15
|
+
_DISPLAY_PERIOD = re.compile(r"^(\d{4}-\d{2}-\d{2})(?:\s+\((FY|Q[1-4]|YTD)\))?$")
|
|
16
|
+
_STRUCTURED_PERIOD = re.compile(
|
|
17
|
+
r"^(?:(instant)_(\d{4}-\d{2}-\d{2})|(duration)_(\d{4}-\d{2}-\d{2})_(\d{4}-\d{2}-\d{2}))$"
|
|
18
|
+
)
|
|
19
|
+
_MAX_FACT_ARITHMETIC_DIGITS = 10_000
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class SecStatementParseError(ValueError):
|
|
23
|
+
"""A supplied financial statement could not be interpreted without losing identity."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _missing(value: Any) -> bool:
|
|
27
|
+
return (
|
|
28
|
+
value is None
|
|
29
|
+
or type(value).__name__ in {"NAType", "NaTType"}
|
|
30
|
+
or str(value).strip().lower() in {"", "nan", "none", "null", "<na>", "nat"}
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _flag(value: Any) -> bool:
|
|
35
|
+
if _missing(value):
|
|
36
|
+
return False
|
|
37
|
+
if type(value).__name__ in {"bool", "bool_"}:
|
|
38
|
+
return bool(value)
|
|
39
|
+
raise SecStatementParseError("invalid statement boolean metadata")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _text(value: Any) -> str | None:
|
|
43
|
+
return None if _missing(value) else str(value).strip()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _number(value: Any) -> Decimal | None:
|
|
47
|
+
if isinstance(value, Decimal) and not value.is_finite():
|
|
48
|
+
raise SecStatementParseError("non-finite statement fact")
|
|
49
|
+
if type(value).__name__ in {"bool", "bool_"} or _missing(value):
|
|
50
|
+
return None
|
|
51
|
+
try:
|
|
52
|
+
result = Decimal(str(value).strip())
|
|
53
|
+
except (InvalidOperation, TypeError, ValueError) as exc:
|
|
54
|
+
raise SecStatementParseError("non-numeric statement fact") from exc
|
|
55
|
+
if not result.is_finite():
|
|
56
|
+
raise SecStatementParseError("non-finite statement fact")
|
|
57
|
+
return result
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _period(key: str) -> tuple[str, date | None, date]:
|
|
61
|
+
match = _STRUCTURED_PERIOD.fullmatch(key)
|
|
62
|
+
if not match:
|
|
63
|
+
raise SecStatementParseError("unrecognized structured period key")
|
|
64
|
+
try:
|
|
65
|
+
if match.group(1):
|
|
66
|
+
return "instant", None, date.fromisoformat(match.group(2))
|
|
67
|
+
start, end = date.fromisoformat(match.group(4)), date.fromisoformat(match.group(5))
|
|
68
|
+
if start > end:
|
|
69
|
+
raise ValueError("reversed period")
|
|
70
|
+
return "duration", start, end
|
|
71
|
+
except ValueError as exc:
|
|
72
|
+
raise SecStatementParseError("invalid structured period dates") from exc
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _dimensions(item: Mapping[str, Any]) -> dict[str, str]:
|
|
76
|
+
metadata = item.get("dimension_metadata")
|
|
77
|
+
if metadata is None:
|
|
78
|
+
return {}
|
|
79
|
+
if not isinstance(metadata, list):
|
|
80
|
+
raise SecStatementParseError("invalid dimension metadata")
|
|
81
|
+
result = {}
|
|
82
|
+
for dim in metadata:
|
|
83
|
+
if not isinstance(dim, dict) or not dim.get("dimension") or not dim.get("member"):
|
|
84
|
+
raise SecStatementParseError("dimension identity unavailable")
|
|
85
|
+
axis, member = str(dim["dimension"]), str(dim["member"])
|
|
86
|
+
if axis in result and result[axis] != member:
|
|
87
|
+
raise SecStatementParseError("conflicting dimension identity")
|
|
88
|
+
result[axis] = member
|
|
89
|
+
return result
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _dim_key(dimensions: Mapping[str, Any]) -> str:
|
|
93
|
+
return json.dumps(dict(dimensions), sort_keys=True, separators=(",", ":"))
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _fact_precision(fact: Any) -> int | str | None:
|
|
97
|
+
"""Return finite precision, None for missing, and ``INF`` for infinity."""
|
|
98
|
+
decimals = fact.decimals
|
|
99
|
+
if decimals is None or type(decimals).__name__ in {"NAType", "NaTType"}:
|
|
100
|
+
return None
|
|
101
|
+
if isinstance(decimals, str) and decimals.strip() == "INF":
|
|
102
|
+
return "INF"
|
|
103
|
+
if isinstance(decimals, bool):
|
|
104
|
+
raise SecStatementParseError("invalid fact precision")
|
|
105
|
+
if isinstance(decimals, int):
|
|
106
|
+
return decimals
|
|
107
|
+
if isinstance(decimals, str) and re.fullmatch(r"[+-]?\d+", decimals.strip()):
|
|
108
|
+
return int(decimals.strip())
|
|
109
|
+
raise SecStatementParseError("invalid fact precision")
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _fact_interval(value: Decimal, precision: int | str | None) -> tuple[Fraction, Fraction]:
|
|
113
|
+
value_tuple = value.as_tuple()
|
|
114
|
+
if not isinstance(value_tuple.exponent, int):
|
|
115
|
+
raise SecStatementParseError("non-finite statement fact")
|
|
116
|
+
if (
|
|
117
|
+
abs(value_tuple.exponent) > _MAX_FACT_ARITHMETIC_DIGITS
|
|
118
|
+
or len(value_tuple.digits) > _MAX_FACT_ARITHMETIC_DIGITS
|
|
119
|
+
):
|
|
120
|
+
raise SecStatementParseError("statement fact exceeds arithmetic bounds")
|
|
121
|
+
if isinstance(precision, int) and abs(precision) > _MAX_FACT_ARITHMETIC_DIGITS:
|
|
122
|
+
raise SecStatementParseError("statement fact exceeds arithmetic bounds")
|
|
123
|
+
exact = Fraction(value)
|
|
124
|
+
if precision is None or precision == "INF":
|
|
125
|
+
return exact, exact
|
|
126
|
+
if not isinstance(precision, int):
|
|
127
|
+
raise SecStatementParseError("invalid fact precision")
|
|
128
|
+
half_unit = Fraction(10 ** (-precision), 2) if precision < 0 else Fraction(1, 2 * 10**precision)
|
|
129
|
+
return exact - half_unit, exact + half_unit
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _select_native_facts(facts: list[Any]) -> list[Any]:
|
|
133
|
+
"""Validate one duplicate group and select its best fact per context."""
|
|
134
|
+
parsed: list[tuple[Any, Decimal, int | str | None]] = []
|
|
135
|
+
unit_refs = set()
|
|
136
|
+
intervals = []
|
|
137
|
+
missing_precision = False
|
|
138
|
+
values_by_precision: dict[int | str, set[Decimal]] = {}
|
|
139
|
+
best_by_context: dict[str, tuple[tuple[int, int], tuple[str, str], Any]] = {}
|
|
140
|
+
|
|
141
|
+
def rank(precision: int | str | None) -> tuple[int, int]:
|
|
142
|
+
if precision is None:
|
|
143
|
+
return (0, 0)
|
|
144
|
+
if precision == "INF":
|
|
145
|
+
return (2, 0)
|
|
146
|
+
if not isinstance(precision, int):
|
|
147
|
+
raise SecStatementParseError("invalid fact precision")
|
|
148
|
+
return (1, precision)
|
|
149
|
+
|
|
150
|
+
for fact in facts:
|
|
151
|
+
value = _number(fact.value)
|
|
152
|
+
precision = _fact_precision(fact)
|
|
153
|
+
missing_precision |= precision is None
|
|
154
|
+
unit_refs.add(fact.unit_ref)
|
|
155
|
+
if value is None:
|
|
156
|
+
continue
|
|
157
|
+
parsed.append((fact, value, precision))
|
|
158
|
+
intervals.append(_fact_interval(value, precision))
|
|
159
|
+
if precision is not None:
|
|
160
|
+
values_by_precision.setdefault(precision, set()).add(value)
|
|
161
|
+
context = fact.context_ref
|
|
162
|
+
tie_break = (str(fact.value), str(fact.decimals))
|
|
163
|
+
candidate = (rank(precision), tie_break, fact)
|
|
164
|
+
current = best_by_context.get(context)
|
|
165
|
+
if (
|
|
166
|
+
current is None
|
|
167
|
+
or candidate[0] > current[0]
|
|
168
|
+
or (candidate[0] == current[0] and candidate[1] < current[1])
|
|
169
|
+
):
|
|
170
|
+
best_by_context[context] = candidate
|
|
171
|
+
if len(unit_refs) != 1:
|
|
172
|
+
raise SecStatementParseError("conflicting native facts for period and dimensions")
|
|
173
|
+
if not parsed:
|
|
174
|
+
return []
|
|
175
|
+
if len(parsed) != len(facts):
|
|
176
|
+
raise SecStatementParseError("non-numeric statement fact")
|
|
177
|
+
# Every interval must share one common point; pairwise or adjacent checks
|
|
178
|
+
# can incorrectly accept a chain of individually overlapping intervals.
|
|
179
|
+
lower = max(interval[0] for interval in intervals)
|
|
180
|
+
upper = min(interval[1] for interval in intervals)
|
|
181
|
+
if missing_precision and any(precision is not None for _, _, precision in parsed):
|
|
182
|
+
raise SecStatementParseError("conflicting native facts for period and dimensions")
|
|
183
|
+
if missing_precision and len({value for _, value, _ in parsed}) != 1:
|
|
184
|
+
raise SecStatementParseError("conflicting native facts for period and dimensions")
|
|
185
|
+
for values in values_by_precision.values():
|
|
186
|
+
if len(values) != 1:
|
|
187
|
+
raise SecStatementParseError("conflicting native facts for period and dimensions")
|
|
188
|
+
if lower > upper:
|
|
189
|
+
raise SecStatementParseError("conflicting native facts for period and dimensions")
|
|
190
|
+
selected: list[Any] = []
|
|
191
|
+
for context_ref in sorted(best_by_context):
|
|
192
|
+
selected.append(best_by_context[context_ref][2])
|
|
193
|
+
return selected
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _native_index(xbrl: Any, concepts: set[str]) -> dict[tuple[str, str, str], list[Any]]:
|
|
197
|
+
"""One pass over facts, avoiding a full source scan for each row or period."""
|
|
198
|
+
# Edgar 5.56 XBRL.facts is FactsView, an enriched query interface whose
|
|
199
|
+
# get_facts() may rewrite concept identifiers. The raw Fact objects are
|
|
200
|
+
# owned by XBRLParser.facts (also exposed upstream as XBRL._facts).
|
|
201
|
+
# Keep this version-specific boundary explicit: never use display values
|
|
202
|
+
# or the enriched view as a fallback for unavailable native facts.
|
|
203
|
+
try:
|
|
204
|
+
native_facts = xbrl.parser.facts
|
|
205
|
+
except AttributeError as exc:
|
|
206
|
+
raise SecStatementParseError("native parser fact mapping unavailable") from exc
|
|
207
|
+
if not isinstance(native_facts, Mapping):
|
|
208
|
+
raise SecStatementParseError("native parser facts must be a mapping")
|
|
209
|
+
result: dict[tuple[str, str, str], list[Any]] = {}
|
|
210
|
+
seen: set[int] = set()
|
|
211
|
+
for fact in native_facts.values():
|
|
212
|
+
if id(fact) in seen:
|
|
213
|
+
continue
|
|
214
|
+
seen.add(id(fact))
|
|
215
|
+
concept = fact.element_id.replace(":", "_")
|
|
216
|
+
if concept not in concepts:
|
|
217
|
+
continue
|
|
218
|
+
context = xbrl.contexts.get(fact.context_ref)
|
|
219
|
+
period = xbrl.context_period_map.get(fact.context_ref)
|
|
220
|
+
if context is None or period is None:
|
|
221
|
+
raise SecStatementParseError("native fact context unavailable")
|
|
222
|
+
key = concept, period, _dim_key(context.dimensions)
|
|
223
|
+
result.setdefault(key, []).append(fact)
|
|
224
|
+
return result
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _structured_rows(
|
|
228
|
+
data: list[Any],
|
|
229
|
+
kind: StatementType,
|
|
230
|
+
dimensions: bool,
|
|
231
|
+
xbrl: Any = None,
|
|
232
|
+
native_index: dict[tuple[str, str, str], list[Any]] | None = None,
|
|
233
|
+
) -> list[SecStatementRow]:
|
|
234
|
+
if any(not isinstance(item, dict) for item in data):
|
|
235
|
+
raise SecStatementParseError("invalid structured statement rows")
|
|
236
|
+
concepts = {str(item.get("concept", "")).replace(":", "_") for item in data}
|
|
237
|
+
index = (
|
|
238
|
+
native_index
|
|
239
|
+
if native_index is not None
|
|
240
|
+
else _native_index(xbrl, concepts)
|
|
241
|
+
if xbrl is not None
|
|
242
|
+
else None
|
|
243
|
+
)
|
|
244
|
+
rows = []
|
|
245
|
+
seen: dict[tuple[str, str, str, str | None], SecStatementRow] = {}
|
|
246
|
+
for item in data:
|
|
247
|
+
if _flag(item.get("is_abstract", item.get("abstract"))):
|
|
248
|
+
continue
|
|
249
|
+
dims = _dimensions(item)
|
|
250
|
+
dimensional = _flag(item.get("is_dimension", item.get("dimension"))) or bool(dims)
|
|
251
|
+
if dimensional and not dimensions:
|
|
252
|
+
continue
|
|
253
|
+
if dimensional and not dims:
|
|
254
|
+
raise SecStatementParseError("dimension identity unavailable")
|
|
255
|
+
concept = _text(item.get("concept"))
|
|
256
|
+
if not concept:
|
|
257
|
+
raise SecStatementParseError("statement concept unavailable")
|
|
258
|
+
values = item.get("values", {})
|
|
259
|
+
if not isinstance(values, dict):
|
|
260
|
+
raise SecStatementParseError("invalid structured values")
|
|
261
|
+
for key, raw in values.items():
|
|
262
|
+
period_type, start, end = _period(str(key))
|
|
263
|
+
declared_type = (item.get("period_types") or {}).get(key)
|
|
264
|
+
if declared_type is not None and declared_type != period_type:
|
|
265
|
+
raise SecStatementParseError("conflicting period type")
|
|
266
|
+
if index is None and _number(raw) is None:
|
|
267
|
+
continue
|
|
268
|
+
unit_ref = (item.get("units") or {}).get(key)
|
|
269
|
+
native_decimals = (item.get("decimals") or {}).get(key)
|
|
270
|
+
context_ref = None
|
|
271
|
+
if index is not None:
|
|
272
|
+
facts = index.get((concept.replace(":", "_"), key, _dim_key(dims)), [])
|
|
273
|
+
if not facts:
|
|
274
|
+
raise SecStatementParseError("statement value has no native fact context")
|
|
275
|
+
selected_facts = _select_native_facts(facts)
|
|
276
|
+
else:
|
|
277
|
+
selected_facts = [None]
|
|
278
|
+
for selected_fact in selected_facts:
|
|
279
|
+
selected_raw = raw
|
|
280
|
+
selected_unit_ref = unit_ref
|
|
281
|
+
selected_decimals = native_decimals
|
|
282
|
+
if selected_fact is not None:
|
|
283
|
+
selected_raw = selected_fact.value
|
|
284
|
+
selected_unit_ref = selected_fact.unit_ref
|
|
285
|
+
selected_decimals = selected_fact.decimals
|
|
286
|
+
context_ref = selected_fact.context_ref
|
|
287
|
+
value = _number(selected_raw)
|
|
288
|
+
if value is None:
|
|
289
|
+
continue
|
|
290
|
+
unit = _text(selected_unit_ref) or _text(item.get("unit"))
|
|
291
|
+
if xbrl is not None and selected_unit_ref:
|
|
292
|
+
definition = xbrl.units.get(selected_unit_ref)
|
|
293
|
+
if definition is None:
|
|
294
|
+
unit = None
|
|
295
|
+
elif isinstance(definition, dict):
|
|
296
|
+
unit = _text(definition.get("measure")) or json.dumps(
|
|
297
|
+
definition, sort_keys=True
|
|
298
|
+
)
|
|
299
|
+
else:
|
|
300
|
+
raise SecStatementParseError("invalid native unit definition")
|
|
301
|
+
decimals = None
|
|
302
|
+
if selected_decimals is not None and str(selected_decimals).strip() != "INF":
|
|
303
|
+
try:
|
|
304
|
+
decimals = int(str(selected_decimals).strip())
|
|
305
|
+
except (TypeError, ValueError) as exc:
|
|
306
|
+
raise SecStatementParseError("invalid fact precision") from exc
|
|
307
|
+
row = SecStatementRow(
|
|
308
|
+
statement_type=kind,
|
|
309
|
+
standard_concept=_text(item.get("standard_concept")) or concept,
|
|
310
|
+
concept=concept,
|
|
311
|
+
label=_text(item.get("label")) or "",
|
|
312
|
+
value=value,
|
|
313
|
+
value_native=str(selected_raw),
|
|
314
|
+
unit=unit,
|
|
315
|
+
unit_ref=_text(selected_unit_ref),
|
|
316
|
+
decimals=decimals,
|
|
317
|
+
decimals_native=_text(selected_decimals),
|
|
318
|
+
period_start=start,
|
|
319
|
+
period_end=end,
|
|
320
|
+
period_type=period_type,
|
|
321
|
+
period_key=str(key),
|
|
322
|
+
context_ref=context_ref,
|
|
323
|
+
dimension=_dim_key(dims) if dims else None,
|
|
324
|
+
is_point_in_time=period_type == "instant",
|
|
325
|
+
period_source="xbrl-context" if index is not None else "structured-period-key",
|
|
326
|
+
)
|
|
327
|
+
identity = concept, str(key), _dim_key(dims), context_ref
|
|
328
|
+
if identity in seen:
|
|
329
|
+
previous = seen[identity]
|
|
330
|
+
if (previous.value, previous.unit) != (row.value, row.unit):
|
|
331
|
+
raise SecStatementParseError("conflicting duplicate statement facts")
|
|
332
|
+
continue
|
|
333
|
+
seen[identity] = row
|
|
334
|
+
rows.append(row)
|
|
335
|
+
return rows
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _native_statement_rows(
|
|
339
|
+
statement: Any, kind: StatementType, dimensions: bool
|
|
340
|
+
) -> list[SecStatementRow]:
|
|
341
|
+
"""Use presentation membership plus instance facts, bypassing display deduplication."""
|
|
342
|
+
xbrl = statement.xbrl
|
|
343
|
+
_, role, _ = xbrl.find_statement(statement.canonical_type or statement.role_or_type)
|
|
344
|
+
tree = xbrl.presentation_trees.get(role)
|
|
345
|
+
if tree is None:
|
|
346
|
+
raise SecStatementParseError("statement presentation tree unavailable")
|
|
347
|
+
nodes = {
|
|
348
|
+
node.element_id.replace(":", "_"): node
|
|
349
|
+
for node in tree.all_nodes.values()
|
|
350
|
+
if not node.is_abstract
|
|
351
|
+
}
|
|
352
|
+
index = _native_index(xbrl, set(nodes))
|
|
353
|
+
data = []
|
|
354
|
+
for (concept, key, dimension_key), facts in sorted(index.items()):
|
|
355
|
+
dims = json.loads(dimension_key)
|
|
356
|
+
if dims and not dimensions:
|
|
357
|
+
continue
|
|
358
|
+
node = nodes[concept]
|
|
359
|
+
data.append(
|
|
360
|
+
{
|
|
361
|
+
"concept": node.element_id,
|
|
362
|
+
"label": node.display_label,
|
|
363
|
+
"is_dimension": bool(dims),
|
|
364
|
+
"dimension_metadata": [
|
|
365
|
+
{"dimension": axis, "member": member} for axis, member in dims.items()
|
|
366
|
+
],
|
|
367
|
+
"values": {key: facts[0].value},
|
|
368
|
+
}
|
|
369
|
+
)
|
|
370
|
+
return _structured_rows(data, kind, dimensions, xbrl, index)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _display_rows(statement: Any, kind: StatementType, dimensions: bool) -> list[SecStatementRow]:
|
|
374
|
+
try:
|
|
375
|
+
frame = statement.to_dataframe(
|
|
376
|
+
standard=False,
|
|
377
|
+
include_unit=True,
|
|
378
|
+
include_point_in_time=True,
|
|
379
|
+
include_standardization=True,
|
|
380
|
+
presentation=False,
|
|
381
|
+
)
|
|
382
|
+
except Exception as exc:
|
|
383
|
+
raise SecStatementParseError("failed to parse statement dataframe") from exc
|
|
384
|
+
if frame is None:
|
|
385
|
+
raise SecStatementParseError("statement dataframe unavailable")
|
|
386
|
+
if frame.empty:
|
|
387
|
+
return []
|
|
388
|
+
metadata_columns = {
|
|
389
|
+
"concept",
|
|
390
|
+
"label",
|
|
391
|
+
"standard_concept",
|
|
392
|
+
"unit",
|
|
393
|
+
"abstract",
|
|
394
|
+
"dimension",
|
|
395
|
+
"point_in_time",
|
|
396
|
+
"dimension_axis",
|
|
397
|
+
"dimension_member",
|
|
398
|
+
"dimension_member_label",
|
|
399
|
+
"dimension_label",
|
|
400
|
+
"level",
|
|
401
|
+
"balance",
|
|
402
|
+
"weight",
|
|
403
|
+
"preferred_sign",
|
|
404
|
+
"is_breakdown",
|
|
405
|
+
"parent_concept",
|
|
406
|
+
"parent_abstract_concept",
|
|
407
|
+
"original_label",
|
|
408
|
+
"is_standardized",
|
|
409
|
+
}
|
|
410
|
+
for col in frame.columns:
|
|
411
|
+
if col not in metadata_columns and not _DISPLAY_PERIOD.fullmatch(str(col)):
|
|
412
|
+
raise SecStatementParseError("unrecognized financial period column")
|
|
413
|
+
period_cols = [(col, _DISPLAY_PERIOD.fullmatch(str(col))) for col in frame.columns]
|
|
414
|
+
period_cols = [(col, match) for col, match in period_cols if match is not None]
|
|
415
|
+
if not period_cols:
|
|
416
|
+
raise SecStatementParseError("no recognized financial period columns")
|
|
417
|
+
if frame.columns.has_duplicates:
|
|
418
|
+
raise SecStatementParseError("duplicate statement columns")
|
|
419
|
+
rows = []
|
|
420
|
+
for _, item in frame.iterrows():
|
|
421
|
+
if _flag(item.get("abstract")):
|
|
422
|
+
continue
|
|
423
|
+
dimensional = _flag(item.get("dimension"))
|
|
424
|
+
if dimensional and not dimensions:
|
|
425
|
+
continue
|
|
426
|
+
dim = None
|
|
427
|
+
if dimensional:
|
|
428
|
+
axis, member = _text(item.get("dimension_axis")), _text(item.get("dimension_member"))
|
|
429
|
+
if not axis or not member:
|
|
430
|
+
raise SecStatementParseError("dimension identity unavailable in display input")
|
|
431
|
+
dim = _dim_key({axis: member})
|
|
432
|
+
concept = _text(item.get("concept"))
|
|
433
|
+
if not concept:
|
|
434
|
+
raise SecStatementParseError("statement concept unavailable")
|
|
435
|
+
for col, match in period_cols:
|
|
436
|
+
raw = item[col]
|
|
437
|
+
value = _number(raw)
|
|
438
|
+
if value is None:
|
|
439
|
+
continue
|
|
440
|
+
try:
|
|
441
|
+
end = date.fromisoformat(match.group(1))
|
|
442
|
+
except ValueError as exc:
|
|
443
|
+
raise SecStatementParseError("invalid display period date") from exc
|
|
444
|
+
suffix = match.group(2)
|
|
445
|
+
instant = _flag(item.get("point_in_time")) or kind == "balance_sheet"
|
|
446
|
+
if instant and suffix:
|
|
447
|
+
raise SecStatementParseError("conflicting display period type")
|
|
448
|
+
rows.append(
|
|
449
|
+
SecStatementRow(
|
|
450
|
+
statement_type=kind,
|
|
451
|
+
standard_concept=_text(item.get("standard_concept")) or concept,
|
|
452
|
+
concept=concept,
|
|
453
|
+
label=_text(item.get("label")) or "",
|
|
454
|
+
value=value,
|
|
455
|
+
value_native=str(raw),
|
|
456
|
+
unit=_text(item.get("unit")),
|
|
457
|
+
period_end=end,
|
|
458
|
+
period_type="instant" if instant else "duration",
|
|
459
|
+
period_key=str(col),
|
|
460
|
+
dimension=dim,
|
|
461
|
+
is_point_in_time=instant,
|
|
462
|
+
period_source="display-column-unknown-start",
|
|
463
|
+
)
|
|
464
|
+
)
|
|
465
|
+
return rows
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def parse_statement_rows(
|
|
469
|
+
statement: Any, statement_type: StatementType, *, include_dimensions: bool = False
|
|
470
|
+
) -> list[SecStatementRow]:
|
|
471
|
+
"""Preserve native fact identity; never infer duration starts from display suffixes."""
|
|
472
|
+
if statement is None:
|
|
473
|
+
return []
|
|
474
|
+
from edgar.xbrl.statements import Statement
|
|
475
|
+
|
|
476
|
+
try:
|
|
477
|
+
if isinstance(statement, Statement):
|
|
478
|
+
return _native_statement_rows(statement, statement_type, include_dimensions)
|
|
479
|
+
getter = getattr(statement, "get_raw_data", None)
|
|
480
|
+
raw = getter() if callable(getter) else None
|
|
481
|
+
if isinstance(raw, list):
|
|
482
|
+
return _structured_rows(raw, statement_type, include_dimensions)
|
|
483
|
+
return _display_rows(statement, statement_type, include_dimensions)
|
|
484
|
+
except SecStatementParseError:
|
|
485
|
+
raise
|
|
486
|
+
except Exception as exc:
|
|
487
|
+
raise SecStatementParseError("invalid financial statement structure") from exc
|