ohmydata 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {ohmydata-0.2.2 → ohmydata-0.2.4}/CHANGELOG.md +29 -0
  2. {ohmydata-0.2.2 → ohmydata-0.2.4}/PKG-INFO +14 -5
  3. {ohmydata-0.2.2 → ohmydata-0.2.4}/README.md +12 -3
  4. {ohmydata-0.2.2 → ohmydata-0.2.4}/pyproject.toml +2 -2
  5. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/__init__.py +1 -1
  6. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/__init__.py +4 -0
  7. ohmydata-0.2.4/src/ohmydata/providers/sec/_statement_parser.py +487 -0
  8. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/cli.py +6 -16
  9. ohmydata-0.2.4/src/ohmydata/providers/sec/edgartools_adapter.py +293 -0
  10. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/financials.py +35 -2
  11. ohmydata-0.2.4/src/ohmydata/providers/sec/financials_dataset.py +284 -0
  12. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/__init__.py +2 -0
  13. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/_fundamentals_parsers.py +27 -45
  14. ohmydata-0.2.4/src/ohmydata/providers/yfinance/_fundamentals_periods.py +121 -0
  15. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/fundamentals.py +146 -36
  16. {ohmydata-0.2.2 → ohmydata-0.2.4}/uv.lock +2 -2
  17. ohmydata-0.2.2/src/ohmydata/providers/sec/edgartools_adapter.py +0 -324
  18. ohmydata-0.2.2/src/ohmydata/providers/sec/financials_dataset.py +0 -203
  19. {ohmydata-0.2.2 → ohmydata-0.2.4}/.gitignore +0 -0
  20. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/adapters/__init__.py +0 -0
  21. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/adapters/polars.py +0 -0
  22. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/cli.py +0 -0
  23. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/__init__.py +0 -0
  24. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/_vintage_lock.py +0 -0
  25. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/availability.py +0 -0
  26. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/errors.py +0 -0
  27. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/facts.py +0 -0
  28. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/policy.py +0 -0
  29. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/provenance.py +0 -0
  30. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/rate_limit.py +0 -0
  31. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/snapshot.py +0 -0
  32. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/specs.py +0 -0
  33. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/core/vintage.py +0 -0
  34. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/__init__.py +0 -0
  35. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/artifacts.py +0 -0
  36. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/batch.py +0 -0
  37. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/core_dataset.py +0 -0
  38. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/edgar.py +0 -0
  39. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/endpoints.py +0 -0
  40. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/errors.py +0 -0
  41. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/http.py +0 -0
  42. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/nport.py +0 -0
  43. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/qualification.py +0 -0
  44. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/sec/qualification_dataset.py +0 -0
  45. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/__init__.py +0 -0
  46. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/client.py +0 -0
  47. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/endpoints.py +0 -0
  48. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/errors.py +0 -0
  49. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/__init__.py +0 -0
  50. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/capture.py +0 -0
  51. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/observations/serialization.py +0 -0
  52. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/__init__.py +0 -0
  53. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_adjusted_bars.py +0 -0
  54. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_index_mapping.py +0 -0
  55. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/etf_pcf_history.py +0 -0
  56. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/index_weight_vintage.py +0 -0
  57. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/lookthrough_bundle.py +0 -0
  58. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/shared.py +0 -0
  59. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/recipes/weighted_dividend_yield.py +0 -0
  60. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_artifacts.py +0 -0
  61. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_canonical.py +0 -0
  62. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_gates.py +0 -0
  63. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_plane.py +0 -0
  64. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/tushare/vintage_producer.py +0 -0
  65. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/client.py +0 -0
  66. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/endpoints.py +0 -0
  67. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/errors.py +0 -0
  68. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/providers/yfinance/quality.py +0 -0
  69. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/__init__.py +0 -0
  70. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/cli.py +0 -0
  71. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/drift_audit.py +0 -0
  72. {ohmydata-0.2.2 → ohmydata-0.2.4}/src/ohmydata/tools/universes.py +0 -0
@@ -1,5 +1,34 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.2.4 — 2026-09-09
4
+
5
+ - Fix the SEC native-fact adapter for edgartools 5.56.0: `XBRL.facts` is a
6
+ `FactsView`, while original `Fact` objects belong to `XBRL.parser.facts`.
7
+ Regression fixtures now use actual XBRL/FactsView/Statement objects.
8
+ - A selected filing with zero rows and parsing failures now raises
9
+ `SecFinancialsParseError`; its `vintage` retains coverage/accession and its
10
+ cause retains the first failure. Missing/empty filings without parser failures
11
+ and explicitly flagged partial coverage retain their existing behavior.
12
+ See [the financials contract](docs/financial-period-integrity.md).
13
+ - Native SEC duplicate facts now use exact Decimal precision intervals and retain
14
+ the highest precision original fact per context, while rejecting true value or
15
+ unit conflicts, mixed known/unknown precision, and arithmetic inputs beyond
16
+ the bounded 10,000-digit guard.
17
+
18
+ ## 0.2.3 — 2026-09-09
19
+
20
+ - Repair SEC financial-period extraction and filing eligibility in the OMD adapter
21
+ against the existing edgartools 5.56.0 baseline; pin that tested optional dependency.
22
+ - Introduce financial dataset/identity v2 for period and dimension metadata and
23
+ explicit coverage, with lossless decimal strings in Parquet and atomic immutable
24
+ partition publication. Existing v1 datasets require rebuilding in a new root.
25
+ - Bind Yahoo financial values to actual report columns, retain nulls and per-metric
26
+ source periods, and match prior-year columns by a documented calendar policy.
27
+ Remove the silent EBIT fallback for operating income.
28
+ - Use actual quotes for FY1 calibration, preserve quote and forecast provenance,
29
+ and expose unknown accounting comparability. Raw fallback remains available.
30
+ - See [migration and changed semantics](docs/financial-period-integrity.md).
31
+
3
32
  ## 0.2.2 — 2026-09-06
4
33
 
5
34
  - Enhanced `yfinance` fundamentals pipeline with institutional Forward P/E calibration and GAAP distortion detection:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ohmydata
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: Financial and alternative market data ingestion SDK.
5
5
  Author: OMD contributors
6
6
  Requires-Python: <3.13,>=3.11
@@ -11,7 +11,7 @@ Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'polars'
11
11
  Provides-Extra: sec-cli
12
12
  Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-cli'
13
13
  Provides-Extra: sec-financials
14
- Requires-Dist: edgartools>=5.0.0; extra == 'sec-financials'
14
+ Requires-Dist: edgartools==5.56.0; extra == 'sec-financials'
15
15
  Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-financials'
16
16
  Provides-Extra: tushare
17
17
  Requires-Dist: pandas<3.0,>=2.0; extra == 'tushare'
@@ -506,6 +506,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
506
506
  beside standardized XBRL categories (`standard_concept`) for cross-company
507
507
  quantitative comparisons.
508
508
 
509
+ The SEC extra supports `edgartools==5.56.0`. See the
510
+ [financial period contract and v2 migration](docs/financial-period-integrity.md)
511
+ before rebuilding existing financial datasets.
512
+
509
513
  ### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
510
514
 
511
515
  Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
@@ -535,7 +539,7 @@ bars_req = YFinanceDailyBarsRequest(
535
539
  bars_result = client.fetch_daily_bars(bars_req)
536
540
  df = bars_result.dataframe
537
541
 
538
- # 2. Fetch fundamentals with institutional FY1 Forward P/E calibration
542
+ # 2. Fetch fundamentals with FY1 Forward P/E and source metadata
539
543
  fund_req = YFinanceFundamentalsRequest(
540
544
  symbols=("NVDA", "GEV"),
541
545
  include_financials=True,
@@ -550,11 +554,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
550
554
  print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
551
555
 
552
556
  gev = fund_result.records["GEV"]
553
- # GAAP vs Non-GAAP accounting distortion detection (flags one-off windfalls > 25%)
557
+ # Legacy flag measures EPS-source divergence; accounting basis remains unknown.
554
558
  if gev.estimates.has_gaap_distortion:
555
- print(f"GEV GAAP distortion flagged! Gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
559
+ print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
556
560
  ```
557
561
 
562
+ Financial values bind to actual statement columns, with per-metric dates and
563
+ coverage. FY1 calibration requires an actual quote and compatible currencies;
564
+ otherwise raw values remain available. See the
565
+ [period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
566
+
558
567
  #### Zero-Drift Audit CLI (`omd audit-drift`)
559
568
 
560
569
  Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
@@ -480,6 +480,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
480
480
  beside standardized XBRL categories (`standard_concept`) for cross-company
481
481
  quantitative comparisons.
482
482
 
483
+ The SEC extra supports `edgartools==5.56.0`. See the
484
+ [financial period contract and v2 migration](docs/financial-period-integrity.md)
485
+ before rebuilding existing financial datasets.
486
+
483
487
  ### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
484
488
 
485
489
  Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
@@ -509,7 +513,7 @@ bars_req = YFinanceDailyBarsRequest(
509
513
  bars_result = client.fetch_daily_bars(bars_req)
510
514
  df = bars_result.dataframe
511
515
 
512
- # 2. Fetch fundamentals with institutional FY1 Forward P/E calibration
516
+ # 2. Fetch fundamentals with FY1 Forward P/E and source metadata
513
517
  fund_req = YFinanceFundamentalsRequest(
514
518
  symbols=("NVDA", "GEV"),
515
519
  include_financials=True,
@@ -524,11 +528,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
524
528
  print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
525
529
 
526
530
  gev = fund_result.records["GEV"]
527
- # GAAP vs Non-GAAP accounting distortion detection (flags one-off windfalls > 25%)
531
+ # Legacy flag measures EPS-source divergence; accounting basis remains unknown.
528
532
  if gev.estimates.has_gaap_distortion:
529
- print(f"GEV GAAP distortion flagged! Gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
533
+ print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
530
534
  ```
531
535
 
536
+ Financial values bind to actual statement columns, with per-metric dates and
537
+ coverage. FY1 calibration requires an actual quote and compatible currencies;
538
+ otherwise raw values remain available. See the
539
+ [period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
540
+
532
541
  #### Zero-Drift Audit CLI (`omd audit-drift`)
533
542
 
534
543
  Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "ohmydata"
7
- version = "0.2.2"
7
+ version = "0.2.4"
8
8
  description = "Financial and alternative market data ingestion SDK."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11,<3.13"
@@ -25,7 +25,7 @@ polars = [
25
25
  vintage-plane = ["pandas>=2.0,<3.0", "pyarrow>=23.0.1,<24.0"]
26
26
  sec-cli = ["pyarrow>=23.0.1,<24.0"]
27
27
  sec-financials = [
28
- "edgartools>=5.0.0",
28
+ "edgartools==5.56.0",
29
29
  "pyarrow>=23.0.1,<24.0",
30
30
  ]
31
31
 
@@ -1,5 +1,5 @@
1
1
  """Provisional offline-only Oh My Data package scaffold."""
2
2
 
3
- __version__ = "0.2.2"
3
+ __version__ = "0.2.4"
4
4
 
5
5
  __all__ = ["__version__"]
@@ -36,6 +36,8 @@ from .edgar import (
36
36
  )
37
37
  from .edgartools_adapter import (
38
38
  SecFinancialsClient,
39
+ SecFinancialsParseError,
40
+ SecStatementParseError,
39
41
  ensure_edgar_available,
40
42
  parse_statement_rows,
41
43
  validate_user_agent,
@@ -89,6 +91,7 @@ __all__ = [
89
91
  "SecEmptyPolicy",
90
92
  "SecEquityEtfUniverse",
91
93
  "SecFinancialsClient",
94
+ "SecFinancialsParseError",
92
95
  "SecFinancialsRequest",
93
96
  "SecFundHoldingVintage",
94
97
  "SecFundSelector",
@@ -107,6 +110,7 @@ __all__ = [
107
110
  "SecPayloadReceipt",
108
111
  "SecReplaySession",
109
112
  "SecScheduledFundSelector",
113
+ "SecStatementParseError",
110
114
  "SecStatementRow",
111
115
  "SecTransportEvidence",
112
116
  "SecUnavailableResult",
@@ -0,0 +1,487 @@
1
+ """Edgar 5.56 statement boundary: native facts first, display frames as compatibility input."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from collections.abc import Mapping
8
+ from datetime import date
9
+ from decimal import Decimal, InvalidOperation
10
+ from fractions import Fraction
11
+ from typing import Any
12
+
13
+ from .financials import SecStatementRow, StatementType
14
+
15
+ _DISPLAY_PERIOD = re.compile(r"^(\d{4}-\d{2}-\d{2})(?:\s+\((FY|Q[1-4]|YTD)\))?$")
16
+ _STRUCTURED_PERIOD = re.compile(
17
+ r"^(?:(instant)_(\d{4}-\d{2}-\d{2})|(duration)_(\d{4}-\d{2}-\d{2})_(\d{4}-\d{2}-\d{2}))$"
18
+ )
19
+ _MAX_FACT_ARITHMETIC_DIGITS = 10_000
20
+
21
+
22
+ class SecStatementParseError(ValueError):
23
+ """A supplied financial statement could not be interpreted without losing identity."""
24
+
25
+
26
+ def _missing(value: Any) -> bool:
27
+ return (
28
+ value is None
29
+ or type(value).__name__ in {"NAType", "NaTType"}
30
+ or str(value).strip().lower() in {"", "nan", "none", "null", "<na>", "nat"}
31
+ )
32
+
33
+
34
+ def _flag(value: Any) -> bool:
35
+ if _missing(value):
36
+ return False
37
+ if type(value).__name__ in {"bool", "bool_"}:
38
+ return bool(value)
39
+ raise SecStatementParseError("invalid statement boolean metadata")
40
+
41
+
42
+ def _text(value: Any) -> str | None:
43
+ return None if _missing(value) else str(value).strip()
44
+
45
+
46
+ def _number(value: Any) -> Decimal | None:
47
+ if isinstance(value, Decimal) and not value.is_finite():
48
+ raise SecStatementParseError("non-finite statement fact")
49
+ if type(value).__name__ in {"bool", "bool_"} or _missing(value):
50
+ return None
51
+ try:
52
+ result = Decimal(str(value).strip())
53
+ except (InvalidOperation, TypeError, ValueError) as exc:
54
+ raise SecStatementParseError("non-numeric statement fact") from exc
55
+ if not result.is_finite():
56
+ raise SecStatementParseError("non-finite statement fact")
57
+ return result
58
+
59
+
60
+ def _period(key: str) -> tuple[str, date | None, date]:
61
+ match = _STRUCTURED_PERIOD.fullmatch(key)
62
+ if not match:
63
+ raise SecStatementParseError("unrecognized structured period key")
64
+ try:
65
+ if match.group(1):
66
+ return "instant", None, date.fromisoformat(match.group(2))
67
+ start, end = date.fromisoformat(match.group(4)), date.fromisoformat(match.group(5))
68
+ if start > end:
69
+ raise ValueError("reversed period")
70
+ return "duration", start, end
71
+ except ValueError as exc:
72
+ raise SecStatementParseError("invalid structured period dates") from exc
73
+
74
+
75
+ def _dimensions(item: Mapping[str, Any]) -> dict[str, str]:
76
+ metadata = item.get("dimension_metadata")
77
+ if metadata is None:
78
+ return {}
79
+ if not isinstance(metadata, list):
80
+ raise SecStatementParseError("invalid dimension metadata")
81
+ result = {}
82
+ for dim in metadata:
83
+ if not isinstance(dim, dict) or not dim.get("dimension") or not dim.get("member"):
84
+ raise SecStatementParseError("dimension identity unavailable")
85
+ axis, member = str(dim["dimension"]), str(dim["member"])
86
+ if axis in result and result[axis] != member:
87
+ raise SecStatementParseError("conflicting dimension identity")
88
+ result[axis] = member
89
+ return result
90
+
91
+
92
+ def _dim_key(dimensions: Mapping[str, Any]) -> str:
93
+ return json.dumps(dict(dimensions), sort_keys=True, separators=(",", ":"))
94
+
95
+
96
+ def _fact_precision(fact: Any) -> int | str | None:
97
+ """Return finite precision, None for missing, and ``INF`` for infinity."""
98
+ decimals = fact.decimals
99
+ if decimals is None or type(decimals).__name__ in {"NAType", "NaTType"}:
100
+ return None
101
+ if isinstance(decimals, str) and decimals.strip() == "INF":
102
+ return "INF"
103
+ if isinstance(decimals, bool):
104
+ raise SecStatementParseError("invalid fact precision")
105
+ if isinstance(decimals, int):
106
+ return decimals
107
+ if isinstance(decimals, str) and re.fullmatch(r"[+-]?\d+", decimals.strip()):
108
+ return int(decimals.strip())
109
+ raise SecStatementParseError("invalid fact precision")
110
+
111
+
112
+ def _fact_interval(value: Decimal, precision: int | str | None) -> tuple[Fraction, Fraction]:
113
+ value_tuple = value.as_tuple()
114
+ if not isinstance(value_tuple.exponent, int):
115
+ raise SecStatementParseError("non-finite statement fact")
116
+ if (
117
+ abs(value_tuple.exponent) > _MAX_FACT_ARITHMETIC_DIGITS
118
+ or len(value_tuple.digits) > _MAX_FACT_ARITHMETIC_DIGITS
119
+ ):
120
+ raise SecStatementParseError("statement fact exceeds arithmetic bounds")
121
+ if isinstance(precision, int) and abs(precision) > _MAX_FACT_ARITHMETIC_DIGITS:
122
+ raise SecStatementParseError("statement fact exceeds arithmetic bounds")
123
+ exact = Fraction(value)
124
+ if precision is None or precision == "INF":
125
+ return exact, exact
126
+ if not isinstance(precision, int):
127
+ raise SecStatementParseError("invalid fact precision")
128
+ half_unit = Fraction(10 ** (-precision), 2) if precision < 0 else Fraction(1, 2 * 10**precision)
129
+ return exact - half_unit, exact + half_unit
130
+
131
+
132
+ def _select_native_facts(facts: list[Any]) -> list[Any]:
133
+ """Validate one duplicate group and select its best fact per context."""
134
+ parsed: list[tuple[Any, Decimal, int | str | None]] = []
135
+ unit_refs = set()
136
+ intervals = []
137
+ missing_precision = False
138
+ values_by_precision: dict[int | str, set[Decimal]] = {}
139
+ best_by_context: dict[str, tuple[tuple[int, int], tuple[str, str], Any]] = {}
140
+
141
+ def rank(precision: int | str | None) -> tuple[int, int]:
142
+ if precision is None:
143
+ return (0, 0)
144
+ if precision == "INF":
145
+ return (2, 0)
146
+ if not isinstance(precision, int):
147
+ raise SecStatementParseError("invalid fact precision")
148
+ return (1, precision)
149
+
150
+ for fact in facts:
151
+ value = _number(fact.value)
152
+ precision = _fact_precision(fact)
153
+ missing_precision |= precision is None
154
+ unit_refs.add(fact.unit_ref)
155
+ if value is None:
156
+ continue
157
+ parsed.append((fact, value, precision))
158
+ intervals.append(_fact_interval(value, precision))
159
+ if precision is not None:
160
+ values_by_precision.setdefault(precision, set()).add(value)
161
+ context = fact.context_ref
162
+ tie_break = (str(fact.value), str(fact.decimals))
163
+ candidate = (rank(precision), tie_break, fact)
164
+ current = best_by_context.get(context)
165
+ if (
166
+ current is None
167
+ or candidate[0] > current[0]
168
+ or (candidate[0] == current[0] and candidate[1] < current[1])
169
+ ):
170
+ best_by_context[context] = candidate
171
+ if len(unit_refs) != 1:
172
+ raise SecStatementParseError("conflicting native facts for period and dimensions")
173
+ if not parsed:
174
+ return []
175
+ if len(parsed) != len(facts):
176
+ raise SecStatementParseError("non-numeric statement fact")
177
+ # Every interval must share one common point; pairwise or adjacent checks
178
+ # can incorrectly accept a chain of individually overlapping intervals.
179
+ lower = max(interval[0] for interval in intervals)
180
+ upper = min(interval[1] for interval in intervals)
181
+ if missing_precision and any(precision is not None for _, _, precision in parsed):
182
+ raise SecStatementParseError("conflicting native facts for period and dimensions")
183
+ if missing_precision and len({value for _, value, _ in parsed}) != 1:
184
+ raise SecStatementParseError("conflicting native facts for period and dimensions")
185
+ for values in values_by_precision.values():
186
+ if len(values) != 1:
187
+ raise SecStatementParseError("conflicting native facts for period and dimensions")
188
+ if lower > upper:
189
+ raise SecStatementParseError("conflicting native facts for period and dimensions")
190
+ selected: list[Any] = []
191
+ for context_ref in sorted(best_by_context):
192
+ selected.append(best_by_context[context_ref][2])
193
+ return selected
194
+
195
+
196
+ def _native_index(xbrl: Any, concepts: set[str]) -> dict[tuple[str, str, str], list[Any]]:
197
+ """One pass over facts, avoiding a full source scan for each row or period."""
198
+ # Edgar 5.56 XBRL.facts is FactsView, an enriched query interface whose
199
+ # get_facts() may rewrite concept identifiers. The raw Fact objects are
200
+ # owned by XBRLParser.facts (also exposed upstream as XBRL._facts).
201
+ # Keep this version-specific boundary explicit: never use display values
202
+ # or the enriched view as a fallback for unavailable native facts.
203
+ try:
204
+ native_facts = xbrl.parser.facts
205
+ except AttributeError as exc:
206
+ raise SecStatementParseError("native parser fact mapping unavailable") from exc
207
+ if not isinstance(native_facts, Mapping):
208
+ raise SecStatementParseError("native parser facts must be a mapping")
209
+ result: dict[tuple[str, str, str], list[Any]] = {}
210
+ seen: set[int] = set()
211
+ for fact in native_facts.values():
212
+ if id(fact) in seen:
213
+ continue
214
+ seen.add(id(fact))
215
+ concept = fact.element_id.replace(":", "_")
216
+ if concept not in concepts:
217
+ continue
218
+ context = xbrl.contexts.get(fact.context_ref)
219
+ period = xbrl.context_period_map.get(fact.context_ref)
220
+ if context is None or period is None:
221
+ raise SecStatementParseError("native fact context unavailable")
222
+ key = concept, period, _dim_key(context.dimensions)
223
+ result.setdefault(key, []).append(fact)
224
+ return result
225
+
226
+
227
+ def _structured_rows(
228
+ data: list[Any],
229
+ kind: StatementType,
230
+ dimensions: bool,
231
+ xbrl: Any = None,
232
+ native_index: dict[tuple[str, str, str], list[Any]] | None = None,
233
+ ) -> list[SecStatementRow]:
234
+ if any(not isinstance(item, dict) for item in data):
235
+ raise SecStatementParseError("invalid structured statement rows")
236
+ concepts = {str(item.get("concept", "")).replace(":", "_") for item in data}
237
+ index = (
238
+ native_index
239
+ if native_index is not None
240
+ else _native_index(xbrl, concepts)
241
+ if xbrl is not None
242
+ else None
243
+ )
244
+ rows = []
245
+ seen: dict[tuple[str, str, str, str | None], SecStatementRow] = {}
246
+ for item in data:
247
+ if _flag(item.get("is_abstract", item.get("abstract"))):
248
+ continue
249
+ dims = _dimensions(item)
250
+ dimensional = _flag(item.get("is_dimension", item.get("dimension"))) or bool(dims)
251
+ if dimensional and not dimensions:
252
+ continue
253
+ if dimensional and not dims:
254
+ raise SecStatementParseError("dimension identity unavailable")
255
+ concept = _text(item.get("concept"))
256
+ if not concept:
257
+ raise SecStatementParseError("statement concept unavailable")
258
+ values = item.get("values", {})
259
+ if not isinstance(values, dict):
260
+ raise SecStatementParseError("invalid structured values")
261
+ for key, raw in values.items():
262
+ period_type, start, end = _period(str(key))
263
+ declared_type = (item.get("period_types") or {}).get(key)
264
+ if declared_type is not None and declared_type != period_type:
265
+ raise SecStatementParseError("conflicting period type")
266
+ if index is None and _number(raw) is None:
267
+ continue
268
+ unit_ref = (item.get("units") or {}).get(key)
269
+ native_decimals = (item.get("decimals") or {}).get(key)
270
+ context_ref = None
271
+ if index is not None:
272
+ facts = index.get((concept.replace(":", "_"), key, _dim_key(dims)), [])
273
+ if not facts:
274
+ raise SecStatementParseError("statement value has no native fact context")
275
+ selected_facts = _select_native_facts(facts)
276
+ else:
277
+ selected_facts = [None]
278
+ for selected_fact in selected_facts:
279
+ selected_raw = raw
280
+ selected_unit_ref = unit_ref
281
+ selected_decimals = native_decimals
282
+ if selected_fact is not None:
283
+ selected_raw = selected_fact.value
284
+ selected_unit_ref = selected_fact.unit_ref
285
+ selected_decimals = selected_fact.decimals
286
+ context_ref = selected_fact.context_ref
287
+ value = _number(selected_raw)
288
+ if value is None:
289
+ continue
290
+ unit = _text(selected_unit_ref) or _text(item.get("unit"))
291
+ if xbrl is not None and selected_unit_ref:
292
+ definition = xbrl.units.get(selected_unit_ref)
293
+ if definition is None:
294
+ unit = None
295
+ elif isinstance(definition, dict):
296
+ unit = _text(definition.get("measure")) or json.dumps(
297
+ definition, sort_keys=True
298
+ )
299
+ else:
300
+ raise SecStatementParseError("invalid native unit definition")
301
+ decimals = None
302
+ if selected_decimals is not None and str(selected_decimals).strip() != "INF":
303
+ try:
304
+ decimals = int(str(selected_decimals).strip())
305
+ except (TypeError, ValueError) as exc:
306
+ raise SecStatementParseError("invalid fact precision") from exc
307
+ row = SecStatementRow(
308
+ statement_type=kind,
309
+ standard_concept=_text(item.get("standard_concept")) or concept,
310
+ concept=concept,
311
+ label=_text(item.get("label")) or "",
312
+ value=value,
313
+ value_native=str(selected_raw),
314
+ unit=unit,
315
+ unit_ref=_text(selected_unit_ref),
316
+ decimals=decimals,
317
+ decimals_native=_text(selected_decimals),
318
+ period_start=start,
319
+ period_end=end,
320
+ period_type=period_type,
321
+ period_key=str(key),
322
+ context_ref=context_ref,
323
+ dimension=_dim_key(dims) if dims else None,
324
+ is_point_in_time=period_type == "instant",
325
+ period_source="xbrl-context" if index is not None else "structured-period-key",
326
+ )
327
+ identity = concept, str(key), _dim_key(dims), context_ref
328
+ if identity in seen:
329
+ previous = seen[identity]
330
+ if (previous.value, previous.unit) != (row.value, row.unit):
331
+ raise SecStatementParseError("conflicting duplicate statement facts")
332
+ continue
333
+ seen[identity] = row
334
+ rows.append(row)
335
+ return rows
336
+
337
+
338
+ def _native_statement_rows(
339
+ statement: Any, kind: StatementType, dimensions: bool
340
+ ) -> list[SecStatementRow]:
341
+ """Use presentation membership plus instance facts, bypassing display deduplication."""
342
+ xbrl = statement.xbrl
343
+ _, role, _ = xbrl.find_statement(statement.canonical_type or statement.role_or_type)
344
+ tree = xbrl.presentation_trees.get(role)
345
+ if tree is None:
346
+ raise SecStatementParseError("statement presentation tree unavailable")
347
+ nodes = {
348
+ node.element_id.replace(":", "_"): node
349
+ for node in tree.all_nodes.values()
350
+ if not node.is_abstract
351
+ }
352
+ index = _native_index(xbrl, set(nodes))
353
+ data = []
354
+ for (concept, key, dimension_key), facts in sorted(index.items()):
355
+ dims = json.loads(dimension_key)
356
+ if dims and not dimensions:
357
+ continue
358
+ node = nodes[concept]
359
+ data.append(
360
+ {
361
+ "concept": node.element_id,
362
+ "label": node.display_label,
363
+ "is_dimension": bool(dims),
364
+ "dimension_metadata": [
365
+ {"dimension": axis, "member": member} for axis, member in dims.items()
366
+ ],
367
+ "values": {key: facts[0].value},
368
+ }
369
+ )
370
+ return _structured_rows(data, kind, dimensions, xbrl, index)
371
+
372
+
373
+ def _display_rows(statement: Any, kind: StatementType, dimensions: bool) -> list[SecStatementRow]:
374
+ try:
375
+ frame = statement.to_dataframe(
376
+ standard=False,
377
+ include_unit=True,
378
+ include_point_in_time=True,
379
+ include_standardization=True,
380
+ presentation=False,
381
+ )
382
+ except Exception as exc:
383
+ raise SecStatementParseError("failed to parse statement dataframe") from exc
384
+ if frame is None:
385
+ raise SecStatementParseError("statement dataframe unavailable")
386
+ if frame.empty:
387
+ return []
388
+ metadata_columns = {
389
+ "concept",
390
+ "label",
391
+ "standard_concept",
392
+ "unit",
393
+ "abstract",
394
+ "dimension",
395
+ "point_in_time",
396
+ "dimension_axis",
397
+ "dimension_member",
398
+ "dimension_member_label",
399
+ "dimension_label",
400
+ "level",
401
+ "balance",
402
+ "weight",
403
+ "preferred_sign",
404
+ "is_breakdown",
405
+ "parent_concept",
406
+ "parent_abstract_concept",
407
+ "original_label",
408
+ "is_standardized",
409
+ }
410
+ for col in frame.columns:
411
+ if col not in metadata_columns and not _DISPLAY_PERIOD.fullmatch(str(col)):
412
+ raise SecStatementParseError("unrecognized financial period column")
413
+ period_cols = [(col, _DISPLAY_PERIOD.fullmatch(str(col))) for col in frame.columns]
414
+ period_cols = [(col, match) for col, match in period_cols if match is not None]
415
+ if not period_cols:
416
+ raise SecStatementParseError("no recognized financial period columns")
417
+ if frame.columns.has_duplicates:
418
+ raise SecStatementParseError("duplicate statement columns")
419
+ rows = []
420
+ for _, item in frame.iterrows():
421
+ if _flag(item.get("abstract")):
422
+ continue
423
+ dimensional = _flag(item.get("dimension"))
424
+ if dimensional and not dimensions:
425
+ continue
426
+ dim = None
427
+ if dimensional:
428
+ axis, member = _text(item.get("dimension_axis")), _text(item.get("dimension_member"))
429
+ if not axis or not member:
430
+ raise SecStatementParseError("dimension identity unavailable in display input")
431
+ dim = _dim_key({axis: member})
432
+ concept = _text(item.get("concept"))
433
+ if not concept:
434
+ raise SecStatementParseError("statement concept unavailable")
435
+ for col, match in period_cols:
436
+ raw = item[col]
437
+ value = _number(raw)
438
+ if value is None:
439
+ continue
440
+ try:
441
+ end = date.fromisoformat(match.group(1))
442
+ except ValueError as exc:
443
+ raise SecStatementParseError("invalid display period date") from exc
444
+ suffix = match.group(2)
445
+ instant = _flag(item.get("point_in_time")) or kind == "balance_sheet"
446
+ if instant and suffix:
447
+ raise SecStatementParseError("conflicting display period type")
448
+ rows.append(
449
+ SecStatementRow(
450
+ statement_type=kind,
451
+ standard_concept=_text(item.get("standard_concept")) or concept,
452
+ concept=concept,
453
+ label=_text(item.get("label")) or "",
454
+ value=value,
455
+ value_native=str(raw),
456
+ unit=_text(item.get("unit")),
457
+ period_end=end,
458
+ period_type="instant" if instant else "duration",
459
+ period_key=str(col),
460
+ dimension=dim,
461
+ is_point_in_time=instant,
462
+ period_source="display-column-unknown-start",
463
+ )
464
+ )
465
+ return rows
466
+
467
+
468
+ def parse_statement_rows(
469
+ statement: Any, statement_type: StatementType, *, include_dimensions: bool = False
470
+ ) -> list[SecStatementRow]:
471
+ """Preserve native fact identity; never infer duration starts from display suffixes."""
472
+ if statement is None:
473
+ return []
474
+ from edgar.xbrl.statements import Statement
475
+
476
+ try:
477
+ if isinstance(statement, Statement):
478
+ return _native_statement_rows(statement, statement_type, include_dimensions)
479
+ getter = getattr(statement, "get_raw_data", None)
480
+ raw = getter() if callable(getter) else None
481
+ if isinstance(raw, list):
482
+ return _structured_rows(raw, statement_type, include_dimensions)
483
+ return _display_rows(statement, statement_type, include_dimensions)
484
+ except SecStatementParseError:
485
+ raise
486
+ except Exception as exc:
487
+ raise SecStatementParseError("invalid financial statement structure") from exc