ohmydata 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {ohmydata-0.2.2 → ohmydata-0.2.3}/CHANGELOG.md +14 -0
  2. {ohmydata-0.2.2 → ohmydata-0.2.3}/PKG-INFO +14 -5
  3. {ohmydata-0.2.2 → ohmydata-0.2.3}/README.md +12 -3
  4. {ohmydata-0.2.2 → ohmydata-0.2.3}/pyproject.toml +2 -2
  5. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/__init__.py +1 -1
  6. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/__init__.py +2 -0
  7. ohmydata-0.2.3/src/ohmydata/providers/sec/_statement_parser.py +375 -0
  8. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/cli.py +6 -16
  9. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/edgartools_adapter.py +107 -157
  10. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/financials.py +35 -2
  11. ohmydata-0.2.3/src/ohmydata/providers/sec/financials_dataset.py +284 -0
  12. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/__init__.py +2 -0
  13. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/_fundamentals_parsers.py +27 -45
  14. ohmydata-0.2.3/src/ohmydata/providers/yfinance/_fundamentals_periods.py +121 -0
  15. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/fundamentals.py +146 -36
  16. {ohmydata-0.2.2 → ohmydata-0.2.3}/uv.lock +2 -2
  17. ohmydata-0.2.2/src/ohmydata/providers/sec/financials_dataset.py +0 -203
  18. {ohmydata-0.2.2 → ohmydata-0.2.3}/.gitignore +0 -0
  19. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/adapters/__init__.py +0 -0
  20. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/adapters/polars.py +0 -0
  21. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/cli.py +0 -0
  22. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/__init__.py +0 -0
  23. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/_vintage_lock.py +0 -0
  24. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/availability.py +0 -0
  25. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/errors.py +0 -0
  26. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/facts.py +0 -0
  27. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/policy.py +0 -0
  28. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/provenance.py +0 -0
  29. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/rate_limit.py +0 -0
  30. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/snapshot.py +0 -0
  31. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/specs.py +0 -0
  32. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/core/vintage.py +0 -0
  33. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/__init__.py +0 -0
  34. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/artifacts.py +0 -0
  35. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/batch.py +0 -0
  36. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/core_dataset.py +0 -0
  37. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/edgar.py +0 -0
  38. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/endpoints.py +0 -0
  39. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/errors.py +0 -0
  40. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/http.py +0 -0
  41. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/nport.py +0 -0
  42. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/qualification.py +0 -0
  43. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/sec/qualification_dataset.py +0 -0
  44. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/__init__.py +0 -0
  45. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/client.py +0 -0
  46. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/endpoints.py +0 -0
  47. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/errors.py +0 -0
  48. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/__init__.py +0 -0
  49. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/capture.py +0 -0
  50. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/observations/serialization.py +0 -0
  51. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/__init__.py +0 -0
  52. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_adjusted_bars.py +0 -0
  53. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_index_mapping.py +0 -0
  54. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/etf_pcf_history.py +0 -0
  55. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/index_weight_vintage.py +0 -0
  56. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/lookthrough_bundle.py +0 -0
  57. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/shared.py +0 -0
  58. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/recipes/weighted_dividend_yield.py +0 -0
  59. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_artifacts.py +0 -0
  60. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_canonical.py +0 -0
  61. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_gates.py +0 -0
  62. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_plane.py +0 -0
  63. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/tushare/vintage_producer.py +0 -0
  64. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/client.py +0 -0
  65. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/endpoints.py +0 -0
  66. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/errors.py +0 -0
  67. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/providers/yfinance/quality.py +0 -0
  68. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/__init__.py +0 -0
  69. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/cli.py +0 -0
  70. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/drift_audit.py +0 -0
  71. {ohmydata-0.2.2 → ohmydata-0.2.3}/src/ohmydata/tools/universes.py +0 -0
@@ -1,5 +1,19 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.2.3 — 2026-09-09
4
+
5
+ - Repair SEC financial-period extraction and filing eligibility in the OMD adapter
6
+ against the existing edgartools 5.56.0 baseline; pin that tested optional dependency.
7
+ - Introduce financial dataset/identity v2 for period and dimension metadata and
8
+ explicit coverage, with lossless decimal strings in Parquet and atomic immutable
9
+ partition publication. Existing v1 datasets require rebuilding in a new root.
10
+ - Bind Yahoo financial values to actual report columns, retain nulls and per-metric
11
+ source periods, and match prior-year columns by a documented calendar policy.
12
+ Remove the silent EBIT fallback for operating income.
13
+ - Use actual quotes for FY1 calibration, preserve quote and forecast provenance,
14
+ and expose unknown accounting comparability. Raw fallback remains available.
15
+ - See [migration and changed semantics](docs/financial-period-integrity.md).
16
+
3
17
  ## 0.2.2 — 2026-09-06
4
18
 
5
19
  - Enhanced `yfinance` fundamentals pipeline with institutional Forward P/E calibration and GAAP distortion detection:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ohmydata
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Financial and alternative market data ingestion SDK.
5
5
  Author: OMD contributors
6
6
  Requires-Python: <3.13,>=3.11
@@ -11,7 +11,7 @@ Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'polars'
11
11
  Provides-Extra: sec-cli
12
12
  Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-cli'
13
13
  Provides-Extra: sec-financials
14
- Requires-Dist: edgartools>=5.0.0; extra == 'sec-financials'
14
+ Requires-Dist: edgartools==5.56.0; extra == 'sec-financials'
15
15
  Requires-Dist: pyarrow<24.0,>=23.0.1; extra == 'sec-financials'
16
16
  Provides-Extra: tushare
17
17
  Requires-Dist: pandas<3.0,>=2.0; extra == 'tushare'
@@ -506,6 +506,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
506
506
  beside standardized XBRL categories (`standard_concept`) for cross-company
507
507
  quantitative comparisons.
508
508
 
509
+ The SEC extra supports `edgartools==5.56.0`. See the
510
+ [financial period contract and v2 migration](docs/financial-period-integrity.md)
511
+ before rebuilding existing financial datasets.
512
+
509
513
  ### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
510
514
 
511
515
  Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
@@ -535,7 +539,7 @@ bars_req = YFinanceDailyBarsRequest(
535
539
  bars_result = client.fetch_daily_bars(bars_req)
536
540
  df = bars_result.dataframe
537
541
 
538
- # 2. Fetch fundamentals with institutional FY1 Forward P/E calibration
542
+ # 2. Fetch fundamentals with FY1 Forward P/E and source metadata
539
543
  fund_req = YFinanceFundamentalsRequest(
540
544
  symbols=("NVDA", "GEV"),
541
545
  include_financials=True,
@@ -550,11 +554,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
550
554
  print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
551
555
 
552
556
  gev = fund_result.records["GEV"]
553
- # GAAP vs Non-GAAP accounting distortion detection (flags one-off windfalls > 25%)
557
+ # Legacy flag measures EPS-source divergence; accounting basis remains unknown.
554
558
  if gev.estimates.has_gaap_distortion:
555
- print(f"GEV GAAP distortion flagged! Gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
559
+ print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
556
560
  ```
557
561
 
562
+ Financial values bind to actual statement columns, with per-metric dates and
563
+ coverage. FY1 calibration requires an actual quote and compatible currencies;
564
+ otherwise raw values remain available. See the
565
+ [period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
566
+
558
567
  #### Zero-Drift Audit CLI (`omd audit-drift`)
559
568
 
560
569
  Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
@@ -480,6 +480,10 @@ preserve native line item labels and concepts (`concept`, `label`, `value_native
480
480
  beside standardized XBRL categories (`standard_concept`) for cross-company
481
481
  quantitative comparisons.
482
482
 
483
+ The SEC extra supports `edgartools==5.56.0`. See the
484
+ [financial period contract and v2 migration](docs/financial-period-integrity.md)
485
+ before rebuilding existing financial datasets.
486
+
483
487
  ### yfinance (US & Global Market Data, Fundamentals, and Zero-Drift Audit)
484
488
 
485
489
  Install `ohmydata[yfinance]` to access normalized market data, valuation ratios,
@@ -509,7 +513,7 @@ bars_req = YFinanceDailyBarsRequest(
509
513
  bars_result = client.fetch_daily_bars(bars_req)
510
514
  df = bars_result.dataframe
511
515
 
512
- # 2. Fetch fundamentals with institutional FY1 Forward P/E calibration
516
+ # 2. Fetch fundamentals with FY1 Forward P/E and source metadata
513
517
  fund_req = YFinanceFundamentalsRequest(
514
518
  symbols=("NVDA", "GEV"),
515
519
  include_financials=True,
@@ -524,11 +528,16 @@ print("NVDA Calibrated FPE:", nvda.valuation.forward_pe, nvda.valuation.forward_
524
528
  print("NVDA Raw Yahoo FPE:", nvda.valuation.raw_forward_pe)
525
529
 
526
530
  gev = fund_result.records["GEV"]
527
- # GAAP vs Non-GAAP accounting distortion detection (flags one-off windfalls > 25%)
531
+ # Legacy flag measures EPS-source divergence; accounting basis remains unknown.
528
532
  if gev.estimates.has_gaap_distortion:
529
- print(f"GEV GAAP distortion flagged! Gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
533
+ print(f"GEV EPS-source gap: {gev.estimates.gaap_diff_pct * 100:.1f}%")
530
534
  ```
531
535
 
536
+ Financial values bind to actual statement columns, with per-metric dates and
537
+ coverage. FY1 calibration requires an actual quote and compatible currencies;
538
+ otherwise raw values remain available. See the
539
+ [period selection, valuation provenance and migration guide](docs/financial-period-integrity.md).
540
+
532
541
  #### Zero-Drift Audit CLI (`omd audit-drift`)
533
542
 
534
543
  Audit 10+ years of historical data against the 13-ETF `r10a0` benchmark universe before
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "ohmydata"
7
- version = "0.2.2"
7
+ version = "0.2.3"
8
8
  description = "Financial and alternative market data ingestion SDK."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11,<3.13"
@@ -25,7 +25,7 @@ polars = [
25
25
  vintage-plane = ["pandas>=2.0,<3.0", "pyarrow>=23.0.1,<24.0"]
26
26
  sec-cli = ["pyarrow>=23.0.1,<24.0"]
27
27
  sec-financials = [
28
- "edgartools>=5.0.0",
28
+ "edgartools==5.56.0",
29
29
  "pyarrow>=23.0.1,<24.0",
30
30
  ]
31
31
 
@@ -1,5 +1,5 @@
1
1
  """Provisional offline-only Oh My Data package scaffold."""
2
2
 
3
- __version__ = "0.2.2"
3
+ __version__ = "0.2.3"
4
4
 
5
5
  __all__ = ["__version__"]
@@ -36,6 +36,7 @@ from .edgar import (
36
36
  )
37
37
  from .edgartools_adapter import (
38
38
  SecFinancialsClient,
39
+ SecStatementParseError,
39
40
  ensure_edgar_available,
40
41
  parse_statement_rows,
41
42
  validate_user_agent,
@@ -107,6 +108,7 @@ __all__ = [
107
108
  "SecPayloadReceipt",
108
109
  "SecReplaySession",
109
110
  "SecScheduledFundSelector",
111
+ "SecStatementParseError",
110
112
  "SecStatementRow",
111
113
  "SecTransportEvidence",
112
114
  "SecUnavailableResult",
@@ -0,0 +1,375 @@
1
+ """Edgar 5.56 statement boundary: native facts first, display frames as compatibility input."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from collections.abc import Mapping
8
+ from dataclasses import replace
9
+ from datetime import date
10
+ from decimal import Decimal, InvalidOperation
11
+ from typing import Any
12
+
13
+ from .financials import SecStatementRow, StatementType
14
+
15
+ _DISPLAY_PERIOD = re.compile(r"^(\d{4}-\d{2}-\d{2})(?:\s+\((FY|Q[1-4]|YTD)\))?$")
16
+ _STRUCTURED_PERIOD = re.compile(
17
+ r"^(?:(instant)_(\d{4}-\d{2}-\d{2})|(duration)_(\d{4}-\d{2}-\d{2})_(\d{4}-\d{2}-\d{2}))$"
18
+ )
19
+
20
+
21
+ class SecStatementParseError(ValueError):
22
+ """A supplied financial statement could not be interpreted without losing identity."""
23
+
24
+
25
+ def _missing(value: Any) -> bool:
26
+ return (
27
+ value is None
28
+ or type(value).__name__ in {"NAType", "NaTType"}
29
+ or str(value).strip().lower() in {"", "nan", "none", "null", "<na>", "nat"}
30
+ )
31
+
32
+
33
+ def _flag(value: Any) -> bool:
34
+ if _missing(value):
35
+ return False
36
+ if type(value).__name__ in {"bool", "bool_"}:
37
+ return bool(value)
38
+ raise SecStatementParseError("invalid statement boolean metadata")
39
+
40
+
41
+ def _text(value: Any) -> str | None:
42
+ return None if _missing(value) else str(value).strip()
43
+
44
+
45
+ def _number(value: Any) -> Decimal | None:
46
+ if isinstance(value, Decimal) and not value.is_finite():
47
+ raise SecStatementParseError("non-finite statement fact")
48
+ if type(value).__name__ in {"bool", "bool_"} or _missing(value):
49
+ return None
50
+ try:
51
+ result = Decimal(str(value).strip())
52
+ except (InvalidOperation, TypeError, ValueError) as exc:
53
+ raise SecStatementParseError("non-numeric statement fact") from exc
54
+ if not result.is_finite():
55
+ raise SecStatementParseError("non-finite statement fact")
56
+ return result
57
+
58
+
59
+ def _period(key: str) -> tuple[str, date | None, date]:
60
+ match = _STRUCTURED_PERIOD.fullmatch(key)
61
+ if not match:
62
+ raise SecStatementParseError("unrecognized structured period key")
63
+ try:
64
+ if match.group(1):
65
+ return "instant", None, date.fromisoformat(match.group(2))
66
+ start, end = date.fromisoformat(match.group(4)), date.fromisoformat(match.group(5))
67
+ if start > end:
68
+ raise ValueError("reversed period")
69
+ return "duration", start, end
70
+ except ValueError as exc:
71
+ raise SecStatementParseError("invalid structured period dates") from exc
72
+
73
+
74
+ def _dimensions(item: Mapping[str, Any]) -> dict[str, str]:
75
+ metadata = item.get("dimension_metadata")
76
+ if metadata is None:
77
+ return {}
78
+ if not isinstance(metadata, list):
79
+ raise SecStatementParseError("invalid dimension metadata")
80
+ result = {}
81
+ for dim in metadata:
82
+ if not isinstance(dim, dict) or not dim.get("dimension") or not dim.get("member"):
83
+ raise SecStatementParseError("dimension identity unavailable")
84
+ axis, member = str(dim["dimension"]), str(dim["member"])
85
+ if axis in result and result[axis] != member:
86
+ raise SecStatementParseError("conflicting dimension identity")
87
+ result[axis] = member
88
+ return result
89
+
90
+
91
+ def _dim_key(dimensions: Mapping[str, Any]) -> str:
92
+ return json.dumps(dict(dimensions), sort_keys=True, separators=(",", ":"))
93
+
94
+
95
+ def _native_index(xbrl: Any, concepts: set[str]) -> dict[tuple[str, str, str], list[Any]]:
96
+ """One pass over facts, avoiding a full source scan for each row or period."""
97
+ result: dict[tuple[str, str, str], list[Any]] = {}
98
+ seen: set[int] = set()
99
+ for fact in xbrl.facts.values():
100
+ if id(fact) in seen:
101
+ continue
102
+ seen.add(id(fact))
103
+ concept = fact.element_id.replace(":", "_")
104
+ if concept not in concepts:
105
+ continue
106
+ context = xbrl.contexts.get(fact.context_ref)
107
+ period = xbrl.context_period_map.get(fact.context_ref)
108
+ if context is None or period is None:
109
+ raise SecStatementParseError("native fact context unavailable")
110
+ key = concept, period, _dim_key(context.dimensions)
111
+ result.setdefault(key, []).append(fact)
112
+ return result
113
+
114
+
115
+ def _structured_rows(
116
+ data: list[Any],
117
+ kind: StatementType,
118
+ dimensions: bool,
119
+ xbrl: Any = None,
120
+ native_index: dict[tuple[str, str, str], list[Any]] | None = None,
121
+ ) -> list[SecStatementRow]:
122
+ if any(not isinstance(item, dict) for item in data):
123
+ raise SecStatementParseError("invalid structured statement rows")
124
+ concepts = {str(item.get("concept", "")).replace(":", "_") for item in data}
125
+ index = (
126
+ native_index
127
+ if native_index is not None
128
+ else _native_index(xbrl, concepts)
129
+ if xbrl is not None
130
+ else None
131
+ )
132
+ rows = []
133
+ seen: dict[tuple[str, str, str, str | None], SecStatementRow] = {}
134
+ for item in data:
135
+ if _flag(item.get("is_abstract", item.get("abstract"))):
136
+ continue
137
+ dims = _dimensions(item)
138
+ dimensional = _flag(item.get("is_dimension", item.get("dimension"))) or bool(dims)
139
+ if dimensional and not dimensions:
140
+ continue
141
+ if dimensional and not dims:
142
+ raise SecStatementParseError("dimension identity unavailable")
143
+ concept = _text(item.get("concept"))
144
+ if not concept:
145
+ raise SecStatementParseError("statement concept unavailable")
146
+ values = item.get("values", {})
147
+ if not isinstance(values, dict):
148
+ raise SecStatementParseError("invalid structured values")
149
+ for key, raw in values.items():
150
+ period_type, start, end = _period(str(key))
151
+ declared_type = (item.get("period_types") or {}).get(key)
152
+ if declared_type is not None and declared_type != period_type:
153
+ raise SecStatementParseError("conflicting period type")
154
+ if index is None and _number(raw) is None:
155
+ continue
156
+ unit_ref = (item.get("units") or {}).get(key)
157
+ native_decimals = (item.get("decimals") or {}).get(key)
158
+ context_ref = None
159
+ context_refs: list[str | None] = [None]
160
+ if index is not None:
161
+ facts = index.get((concept.replace(":", "_"), key, _dim_key(dims)), [])
162
+ if not facts:
163
+ raise SecStatementParseError("statement value has no native fact context")
164
+ identities = {(f.value, f.unit_ref, str(f.decimals)) for f in facts}
165
+ if len(identities) != 1:
166
+ raise SecStatementParseError(
167
+ "conflicting native facts for period and dimensions"
168
+ )
169
+ fact = min(facts, key=lambda f: f.context_ref)
170
+ raw, unit_ref, native_decimals = fact.value, fact.unit_ref, fact.decimals
171
+ context_ref = fact.context_ref
172
+ context_refs = sorted({f.context_ref for f in facts})
173
+ value = _number(raw)
174
+ if value is None:
175
+ continue
176
+ unit = _text(unit_ref) or _text(item.get("unit"))
177
+ if xbrl is not None and unit_ref:
178
+ definition = xbrl.units.get(unit_ref)
179
+ if definition is None:
180
+ unit = None
181
+ elif isinstance(definition, dict):
182
+ unit = _text(definition.get("measure")) or json.dumps(
183
+ definition, sort_keys=True
184
+ )
185
+ else:
186
+ raise SecStatementParseError("invalid native unit definition")
187
+ decimals = None
188
+ if native_decimals is not None and str(native_decimals) != "INF":
189
+ try:
190
+ decimals = int(native_decimals)
191
+ except (TypeError, ValueError) as exc:
192
+ raise SecStatementParseError("invalid fact precision") from exc
193
+ row = SecStatementRow(
194
+ statement_type=kind,
195
+ standard_concept=_text(item.get("standard_concept")) or concept,
196
+ concept=concept,
197
+ label=_text(item.get("label")) or "",
198
+ value=value,
199
+ value_native=str(raw),
200
+ unit=unit,
201
+ unit_ref=_text(unit_ref),
202
+ decimals=decimals,
203
+ decimals_native=_text(native_decimals),
204
+ period_start=start,
205
+ period_end=end,
206
+ period_type=period_type,
207
+ period_key=str(key),
208
+ context_ref=context_ref,
209
+ dimension=_dim_key(dims) if dims else None,
210
+ is_point_in_time=period_type == "instant",
211
+ period_source="xbrl-context" if index is not None else "structured-period-key",
212
+ )
213
+ for reference in context_refs:
214
+ contextual_row = replace(row, context_ref=reference)
215
+ identity = concept, str(key), _dim_key(dims), reference
216
+ if identity in seen:
217
+ previous = seen[identity]
218
+ if (previous.value, previous.unit) != (row.value, row.unit):
219
+ raise SecStatementParseError("conflicting duplicate statement facts")
220
+ continue
221
+ seen[identity] = contextual_row
222
+ rows.append(contextual_row)
223
+ return rows
224
+
225
+
226
+ def _native_statement_rows(
227
+ statement: Any, kind: StatementType, dimensions: bool
228
+ ) -> list[SecStatementRow]:
229
+ """Use presentation membership plus instance facts, bypassing display deduplication."""
230
+ xbrl = statement.xbrl
231
+ _, role, _ = xbrl.find_statement(statement.canonical_type or statement.role_or_type)
232
+ tree = xbrl.presentation_trees.get(role)
233
+ if tree is None:
234
+ raise SecStatementParseError("statement presentation tree unavailable")
235
+ nodes = {
236
+ node.element_id.replace(":", "_"): node
237
+ for node in tree.all_nodes.values()
238
+ if not node.is_abstract
239
+ }
240
+ index = _native_index(xbrl, set(nodes))
241
+ data = []
242
+ for (concept, key, dimension_key), facts in sorted(index.items()):
243
+ dims = json.loads(dimension_key)
244
+ if dims and not dimensions:
245
+ continue
246
+ node = nodes[concept]
247
+ data.append(
248
+ {
249
+ "concept": node.element_id,
250
+ "label": node.display_label,
251
+ "is_dimension": bool(dims),
252
+ "dimension_metadata": [
253
+ {"dimension": axis, "member": member} for axis, member in dims.items()
254
+ ],
255
+ "values": {key: facts[0].value},
256
+ }
257
+ )
258
+ return _structured_rows(data, kind, dimensions, xbrl, index)
259
+
260
+
261
+ def _display_rows(statement: Any, kind: StatementType, dimensions: bool) -> list[SecStatementRow]:
262
+ try:
263
+ frame = statement.to_dataframe(
264
+ standard=False,
265
+ include_unit=True,
266
+ include_point_in_time=True,
267
+ include_standardization=True,
268
+ presentation=False,
269
+ )
270
+ except Exception as exc:
271
+ raise SecStatementParseError("failed to parse statement dataframe") from exc
272
+ if frame is None:
273
+ raise SecStatementParseError("statement dataframe unavailable")
274
+ if frame.empty:
275
+ return []
276
+ metadata_columns = {
277
+ "concept",
278
+ "label",
279
+ "standard_concept",
280
+ "unit",
281
+ "abstract",
282
+ "dimension",
283
+ "point_in_time",
284
+ "dimension_axis",
285
+ "dimension_member",
286
+ "dimension_member_label",
287
+ "dimension_label",
288
+ "level",
289
+ "balance",
290
+ "weight",
291
+ "preferred_sign",
292
+ "is_breakdown",
293
+ "parent_concept",
294
+ "parent_abstract_concept",
295
+ "original_label",
296
+ "is_standardized",
297
+ }
298
+ for col in frame.columns:
299
+ if col not in metadata_columns and not _DISPLAY_PERIOD.fullmatch(str(col)):
300
+ raise SecStatementParseError("unrecognized financial period column")
301
+ period_cols = [(col, _DISPLAY_PERIOD.fullmatch(str(col))) for col in frame.columns]
302
+ period_cols = [(col, match) for col, match in period_cols if match is not None]
303
+ if not period_cols:
304
+ raise SecStatementParseError("no recognized financial period columns")
305
+ if frame.columns.has_duplicates:
306
+ raise SecStatementParseError("duplicate statement columns")
307
+ rows = []
308
+ for _, item in frame.iterrows():
309
+ if _flag(item.get("abstract")):
310
+ continue
311
+ dimensional = _flag(item.get("dimension"))
312
+ if dimensional and not dimensions:
313
+ continue
314
+ dim = None
315
+ if dimensional:
316
+ axis, member = _text(item.get("dimension_axis")), _text(item.get("dimension_member"))
317
+ if not axis or not member:
318
+ raise SecStatementParseError("dimension identity unavailable in display input")
319
+ dim = _dim_key({axis: member})
320
+ concept = _text(item.get("concept"))
321
+ if not concept:
322
+ raise SecStatementParseError("statement concept unavailable")
323
+ for col, match in period_cols:
324
+ raw = item[col]
325
+ value = _number(raw)
326
+ if value is None:
327
+ continue
328
+ try:
329
+ end = date.fromisoformat(match.group(1))
330
+ except ValueError as exc:
331
+ raise SecStatementParseError("invalid display period date") from exc
332
+ suffix = match.group(2)
333
+ instant = _flag(item.get("point_in_time")) or kind == "balance_sheet"
334
+ if instant and suffix:
335
+ raise SecStatementParseError("conflicting display period type")
336
+ rows.append(
337
+ SecStatementRow(
338
+ statement_type=kind,
339
+ standard_concept=_text(item.get("standard_concept")) or concept,
340
+ concept=concept,
341
+ label=_text(item.get("label")) or "",
342
+ value=value,
343
+ value_native=str(raw),
344
+ unit=_text(item.get("unit")),
345
+ period_end=end,
346
+ period_type="instant" if instant else "duration",
347
+ period_key=str(col),
348
+ dimension=dim,
349
+ is_point_in_time=instant,
350
+ period_source="display-column-unknown-start",
351
+ )
352
+ )
353
+ return rows
354
+
355
+
356
+ def parse_statement_rows(
357
+ statement: Any, statement_type: StatementType, *, include_dimensions: bool = False
358
+ ) -> list[SecStatementRow]:
359
+ """Preserve native fact identity; never infer duration starts from display suffixes."""
360
+ if statement is None:
361
+ return []
362
+ from edgar.xbrl.statements import Statement
363
+
364
+ try:
365
+ if isinstance(statement, Statement):
366
+ return _native_statement_rows(statement, statement_type, include_dimensions)
367
+ getter = getattr(statement, "get_raw_data", None)
368
+ raw = getter() if callable(getter) else None
369
+ if isinstance(raw, list):
370
+ return _structured_rows(raw, statement_type, include_dimensions)
371
+ return _display_rows(statement, statement_type, include_dimensions)
372
+ except SecStatementParseError:
373
+ raise
374
+ except Exception as exc:
375
+ raise SecStatementParseError("invalid financial statement structure") from exc
@@ -468,7 +468,6 @@ def run_qualify(args: Any) -> int:
468
468
 
469
469
 
470
470
  def run_financials(args: Any) -> int:
471
- import hashlib
472
471
  import importlib
473
472
 
474
473
  if getattr(args, "config", None):
@@ -502,7 +501,9 @@ def run_financials(args: Any) -> int:
502
501
  m_file = p_dir / "manifest.json"
503
502
  if not m_file.is_file():
504
503
  raise ValueError(f"manifest.json missing for symbol: {sym_str}")
505
- m_data: dict[str, Any] = json.loads(m_file.read_text(encoding="utf-8"))
504
+ from .financials_dataset import validate_financials_partition
505
+
506
+ m_data = validate_financials_partition(p_dir)
506
507
  payload.update(m_data)
507
508
 
508
509
  if getattr(args, "rows", False):
@@ -513,24 +514,13 @@ def run_financials(args: Any) -> int:
513
514
  payload["rows"] = tbl.to_pylist()[:100]
514
515
 
515
516
  elif cmd == "validate":
517
+ from .financials_dataset import validate_financials_partition
518
+
516
519
  root_path = Path(args.root)
517
520
  partitions = list(root_path.glob("symbol=*"))
518
521
  verified_count = 0
519
522
  for p in partitions:
520
- m_file = p / "manifest.json"
521
- if not m_file.is_file():
522
- raise ValueError(f"manifest missing in {p}")
523
- m: dict[str, Any] = json.loads(m_file.read_text(encoding="utf-8"))
524
- files = m.get("files", {})
525
- for fname, meta in files.items():
526
- fpath = p / fname
527
- if not fpath.is_file():
528
- raise ValueError(f"file missing: {fpath}")
529
- h = hashlib.sha256(fpath.read_bytes()).hexdigest()
530
- if h != meta.get("sha256"):
531
- raise ValueError(
532
- f"hash mismatch for {fpath}: expected {meta.get('sha256')}, got {h}"
533
- )
523
+ validate_financials_partition(p)
534
524
  verified_count += 1
535
525
  payload["status"] = "completed"
536
526
  payload["partitions_verified"] = verified_count