calcfinc 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. calcfinc/__init__.py +30 -0
  2. calcfinc/adapters/__init__.py +1 -0
  3. calcfinc/adapters/ind_as_xbrl/__init__.py +29 -0
  4. calcfinc/adapters/ind_as_xbrl/canonical.py +171 -0
  5. calcfinc/adapters/ind_as_xbrl/load.py +231 -0
  6. calcfinc/adapters/ind_as_xbrl/tags.py +144 -0
  7. calcfinc/adapters/ind_as_xbrl/vocab.py +129 -0
  8. calcfinc/adapters/ind_as_xbrl/xbrl.py +128 -0
  9. calcfinc/adapters/sec_companyfacts/__init__.py +24 -0
  10. calcfinc/adapters/sec_companyfacts/fetch.py +70 -0
  11. calcfinc/adapters/sec_companyfacts/load.py +81 -0
  12. calcfinc/adapters/sec_companyfacts/parse.py +393 -0
  13. calcfinc/adapters/sec_companyfacts/tags.py +93 -0
  14. calcfinc/engine/__init__.py +11 -0
  15. calcfinc/engine/check.py +95 -0
  16. calcfinc/engine/decompose.py +80 -0
  17. calcfinc/engine/engine.py +690 -0
  18. calcfinc/engine/evaluate.py +189 -0
  19. calcfinc/engine/growth.py +28 -0
  20. calcfinc/engine/records.py +157 -0
  21. calcfinc/engine/segments.py +179 -0
  22. calcfinc/entity.py +41 -0
  23. calcfinc/fact.py +177 -0
  24. calcfinc/formula.py +180 -0
  25. calcfinc/loaders/__init__.py +6 -0
  26. calcfinc/loaders/csv.py +86 -0
  27. calcfinc/loaders/dataframe.py +93 -0
  28. calcfinc/loaders/records.py +255 -0
  29. calcfinc/num.py +96 -0
  30. calcfinc/period.py +112 -0
  31. calcfinc/py.typed +0 -0
  32. calcfinc/registry/__init__.py +6 -0
  33. calcfinc/registry/metrics.py +159 -0
  34. calcfinc/registry/ratios.py +427 -0
  35. calcfinc/store/__init__.py +5 -0
  36. calcfinc/store/base.py +91 -0
  37. calcfinc/store/schema.sql +104 -0
  38. calcfinc/store/sqlite.py +418 -0
  39. calcfinc-0.1.0.dist-info/METADATA +167 -0
  40. calcfinc-0.1.0.dist-info/RECORD +42 -0
  41. calcfinc-0.1.0.dist-info/WHEEL +4 -0
  42. calcfinc-0.1.0.dist-info/licenses/LICENSE +21 -0
calcfinc/__init__.py ADDED
@@ -0,0 +1,30 @@
1
+ """calcfinc: auditable financial metrics for any entity."""
2
+ import logging
3
+
4
+ from calcfinc.engine import CheckResult, EngineError, EngineResult, FactRef, FinancialEngine
5
+ from calcfinc.entity import Entity
6
+ from calcfinc.fact import (
7
+ Basis,
8
+ FinancialFact,
9
+ MappingConfidence,
10
+ Segment,
11
+ SegmentFact,
12
+ SharePrice,
13
+ Source,
14
+ StatementType,
15
+ )
16
+ from calcfinc.period import PeriodWindows
17
+ from calcfinc.registry import RatioSpec, register_metric, register_ratio
18
+ from calcfinc.store import SqliteRepositories
19
+
20
+ __version__ = "0.1.0"
21
+
22
+ # A library logs through the standard `logging` module and leaves the handlers to the application.
23
+ logging.getLogger(__name__).addHandler(logging.NullHandler())
24
+
25
+ __all__ = [
26
+ "Basis", "CheckResult", "Entity", "EngineError", "EngineResult", "FactRef", "FinancialEngine",
27
+ "FinancialFact", "MappingConfidence", "PeriodWindows", "RatioSpec", "Segment", "SegmentFact",
28
+ "SharePrice", "Source", "SqliteRepositories", "StatementType", "__version__", "register_metric",
29
+ "register_ratio",
30
+ ]
@@ -0,0 +1 @@
1
+ """Source adapters: each maps a source's tags and conventions to canonical facts."""
@@ -0,0 +1,29 @@
1
+ """Ind-AS (India) adapter.
2
+
3
+ Reads exchange XBRL filings (a .xbrl file, raw fact rows, or canonical records) into
4
+ calcfinc facts: April-March fiscal year, INR, exact Decimal values, review flags for records
5
+ whose own arithmetic fails. Also holds the India-only vocabulary and the `india.*` ratios
6
+ (`register()`), so the core stays free of country names.
7
+
8
+ from calcfinc.adapters import ind_as_xbrl
9
+ ind_as_xbrl.load_xbrl_file(repos, "results_consolidated.xbrl", entity="ACME")
10
+ """
11
+ from calcfinc.adapters.ind_as_xbrl.canonical import consistency_issues, map_facts, parse_number
12
+ from calcfinc.adapters.ind_as_xbrl.load import (
13
+ IndAsReport,
14
+ load_canonical,
15
+ load_canonical_file,
16
+ load_raw_facts,
17
+ load_xbrl_file,
18
+ load_xbrl_files,
19
+ read_canonical_json,
20
+ record_to_facts,
21
+ )
22
+ from calcfinc.adapters.ind_as_xbrl.vocab import RENAMES, register, shares_outstanding
23
+ from calcfinc.adapters.ind_as_xbrl.xbrl import parse_xbrl_file
24
+
25
+ __all__ = [
26
+ "IndAsReport", "RENAMES", "consistency_issues", "load_canonical", "load_canonical_file",
27
+ "load_raw_facts", "load_xbrl_file", "load_xbrl_files", "map_facts", "parse_number", "parse_xbrl_file",
28
+ "read_canonical_json", "record_to_facts", "register", "shares_outstanding",
29
+ ]
@@ -0,0 +1,171 @@
1
+ """Raw XBRL facts -> canonical records, with exact Decimal values.
2
+
3
+ A *raw fact* is a dict as produced by `xbrl.parse_xbrl_file`: line_item_tag, value (text),
4
+ context_id, period_start / period_end / instant, sign. A *canonical record* is one context's
5
+ facts under the source-side canonical names (`tags.TAG_MAP`), plus its period.
6
+
7
+ Only whole-company contexts are used (OneD, OneI, PY_D...), never segment or note breakdowns.
8
+ Duration and instant contexts are kept as separate records; the engine merges a balance sheet
9
+ into the duration record that ends on the same day.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from collections import defaultdict
14
+ from collections.abc import Iterable, Mapping
15
+ from decimal import Decimal, InvalidOperation
16
+ from typing import Any
17
+
18
+ from calcfinc.adapters.ind_as_xbrl.tags import (
19
+ BANK_BALANCE_SHEET_AUX_TAGS,
20
+ BANK_EQUITY_AUX_TAGS,
21
+ DEBT_ALT_TAGS,
22
+ HIGH_SEVERITY,
23
+ SECTOR_ALT_TAGS,
24
+ SNAPSHOT_FIELDS,
25
+ TAG_MAP,
26
+ is_primary_context,
27
+ )
28
+ from calcfinc.num import HUNDRED, add, div, mul, sub
29
+
30
+ _MISSING = {"", "-", "—", "NA", "N/A"}
31
+
32
+
33
+ def parse_number(raw: object) -> Decimal | None:
34
+ """Parse a reported number exactly. Handles Indian formatting ('59,553.00'), accounting
35
+ negatives ('(1,234.5)'), a rupee sign and trailing footnote markers. Returns None for
36
+ anything that is not a number. A float (from JSON) is taken through its shortest text,
37
+ which is the original text for any figure of 17 significant digits or fewer."""
38
+ if raw is None or isinstance(raw, bool):
39
+ return None
40
+ if isinstance(raw, Decimal):
41
+ return raw if raw.is_finite() else None
42
+ if isinstance(raw, int):
43
+ return Decimal(raw)
44
+ s = repr(raw) if isinstance(raw, float) else str(raw).strip()
45
+ if s in _MISSING:
46
+ return None
47
+ negative = s.startswith("(") and s.endswith(")")
48
+ s = s.strip("()").replace(",", "").replace("₹", "").strip()
49
+ for candidate in (s, "".join(c for c in s if c.isdigit() or c in ".-")):
50
+ try:
51
+ value = Decimal(candidate)
52
+ except InvalidOperation:
53
+ continue
54
+ if value.is_finite():
55
+ return -value if negative else value
56
+ return None
57
+
58
+
59
+ def _tag_index() -> tuple[dict[str, str], dict[str, str]]:
60
+ by_tag = {v: k for k, v in TAG_MAP.items()}
61
+ by_tag.update({v: k for k, v in BANK_EQUITY_AUX_TAGS.items()})
62
+ by_tag.update({v: k for k, v in BANK_BALANCE_SHEET_AUX_TAGS.items()})
63
+ by_tag.update(SECTOR_ALT_TAGS)
64
+ by_tag.update(DEBT_ALT_TAGS)
65
+ # Older Regulation 33 filings use the in-bse-fin: prefix with the same local concept name.
66
+ # An exact full-tag match always wins; the local name is only a fallback.
67
+ by_local: dict[str, str] = {}
68
+ for tag, name in by_tag.items():
69
+ by_local.setdefault(tag.split(":", 1)[1] if ":" in tag else tag, name)
70
+ return by_tag, by_local
71
+
72
+
73
+ def map_facts(facts: Iterable[Mapping[str, Any]]) -> list[dict[str, Any]]:
74
+ """Group raw facts into canonical records, newest first."""
75
+ by_tag, by_local = _tag_index()
76
+ fields: dict[str, dict[str, Decimal | None]] = defaultdict(dict)
77
+ periods: dict[str, dict[str, Any]] = {}
78
+
79
+ for fact in facts:
80
+ ctx = fact.get("context_id")
81
+ if not isinstance(ctx, str) or not is_primary_context(ctx):
82
+ continue
83
+ tag = fact.get("line_item_tag")
84
+ name = by_tag.get(tag or "")
85
+ if name is None and tag and ":" in tag:
86
+ name = by_local.get(tag.split(":", 1)[1])
87
+ if name is None:
88
+ continue # not a field we track
89
+ value = parse_number(fact.get("value"))
90
+ if value is not None and fact.get("sign") == "-":
91
+ value = -value
92
+ fields[ctx][name] = value
93
+ periods[ctx] = {"context_id": ctx, "period_start": fact.get("period_start"),
94
+ "period_end": fact.get("period_end"), "instant": fact.get("instant")}
95
+
96
+ records = []
97
+ for ctx, values in fields.items():
98
+ record: dict[str, Any] = {**periods[ctx], **values}
99
+ _derive_bank_lines(record)
100
+ records.append(record)
101
+ records.sort(key=lambda r: r.get("period_end") or r.get("instant") or "", reverse=True)
102
+ return records
103
+
104
+
105
+ def _derive_bank_lines(record: dict[str, Any]) -> None:
106
+ """Banks report their equity, cash and liabilities as sub-lines. Build the totals only when
107
+ the total itself is absent and every needed part is present."""
108
+ capital = record.pop("_bank_capital", None)
109
+ reserves = record.pop("_bank_reserves_and_surplus", None)
110
+ if record.get("total_equity") is None and capital is not None and reserves is not None:
111
+ record["total_equity"] = add(capital, reserves)
112
+ cash_with_central_bank = record.pop("_bank_cash_with_rbi", None)
113
+ balances_with_banks = record.pop("_bank_balances_with_banks", None)
114
+ if (record.get("cash_and_equivalents") is None and cash_with_central_bank is not None
115
+ and balances_with_banks is not None):
116
+ record["cash_and_equivalents"] = add(cash_with_central_bank, balances_with_banks)
117
+ other = record.pop("_bank_other_liabilities_and_provisions", None)
118
+ if (record.get("total_liabilities") is None and other is not None
119
+ and record.get("deposits_debt") is not None and record.get("borrowings_noncurrent") is not None):
120
+ record["total_liabilities"] = add(add(record["deposits_debt"], record["borrowings_noncurrent"]), other)
121
+
122
+
123
+ # Fields a filing reports as an exact 0 when it means "not applicable". Found in real filings:
124
+ # a consolidated bank filing carries 0 for its whole NPA block; paid-up capital, face value and
125
+ # CET1 are never genuinely 0; a coverage ratio of exactly 0 is a placeholder, not a result.
126
+ NPA_BLOCK = ("gross_npa", "net_npa", "gross_npa_ratio", "net_npa_ratio")
127
+ BANK_REGULATORY = (*NPA_BLOCK, "return_on_assets", "cet1_ratio", "additional_tier1_ratio")
128
+ ZERO_IS_MISSING = ("paid_up_equity_capital", "face_value_per_share", "cet1_ratio",
129
+ "debt_service_coverage_ratio_reported", "interest_service_coverage_ratio_reported")
130
+
131
+
132
+ def drop_placeholder_zeros(record: Mapping[str, Any]) -> tuple[dict[str, Any], list[str]]:
133
+ """The record without the zeros that mean 'not reported', and the names removed. A genuine
134
+ zero (no exceptional items, no borrowings, no minority interest) is left alone."""
135
+ out = dict(record)
136
+ dropped: list[str] = []
137
+ npa = [record.get(k) for k in NPA_BLOCK]
138
+ if all(isinstance(v, Decimal) and v == 0 for v in npa): # the whole block is zero together
139
+ dropped += [k for k in BANK_REGULATORY if isinstance(record.get(k), Decimal)]
140
+ dropped += [k for k in ZERO_IS_MISSING if isinstance(record.get(k), Decimal) and record[k] == 0
141
+ and k not in dropped]
142
+ for k in dropped:
143
+ out.pop(k, None)
144
+ return out, dropped
145
+
146
+
147
+ def consistency_issues(records: Iterable[Mapping[str, Any]],
148
+ tolerance: Decimal = Decimal(1)) -> list[dict[str, Any]]:
149
+ """Records that end on the same date but disagree on a balance-sheet / capital field, as a
150
+ filing sometimes does between its quarterly and cumulative contexts. Differences within
151
+ `tolerance` (rounding) are ignored."""
152
+ by_date: dict[str, list[Mapping[str, Any]]] = defaultdict(list)
153
+ for r in records:
154
+ key = r.get("period_end") or r.get("instant")
155
+ if key:
156
+ by_date[key].append(r)
157
+ issues: list[dict[str, Any]] = []
158
+ for date_key, group in by_date.items():
159
+ if len(group) < 2:
160
+ continue
161
+ for name in SNAPSHOT_FIELDS:
162
+ values: dict[Any, Decimal] = {r.get("context_id"): r[name] for r in group if r.get(name) is not None}
163
+ distinct = set(values.values())
164
+ if len(distinct) <= 1 or sub(max(distinct), min(distinct)) <= tolerance:
165
+ continue
166
+ nonzero = [v for v in distinct if v != 0]
167
+ smallest = min(nonzero, key=abs) if nonzero else Decimal(1)
168
+ pct = div(mul(sub(max(distinct), min(distinct)), HUNDRED), abs(smallest)).quantize(Decimal("0.01"))
169
+ issues.append({"period_end": date_key, "field": name, "values": values,
170
+ "severity": "high" if name in HIGH_SEVERITY else "low", "magnitude_pct": pct})
171
+ return issues
@@ -0,0 +1,231 @@
1
+ """Canonical records -> facts in a store.
2
+
3
+ Ind-AS reporting: an April-March fiscal year and INR by default. Values are exact Decimals; a
4
+ record whose own arithmetic does not add up (assets vs liabilities + equity, income vs
5
+ expenses...) still loads, but every fact from it carries a review reason so it can be routed
6
+ to a person.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ import json
12
+ import re
13
+ from collections.abc import Iterable, Mapping
14
+ from dataclasses import dataclass, field, replace
15
+ from datetime import UTC, date, datetime
16
+ from decimal import Decimal
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ from calcfinc.adapters.ind_as_xbrl.canonical import consistency_issues, drop_placeholder_zeros, map_facts
21
+ from calcfinc.adapters.ind_as_xbrl.tags import META_KEYS, NAME_MAP
22
+ from calcfinc.adapters.ind_as_xbrl.vocab import register
23
+ from calcfinc.adapters.ind_as_xbrl.xbrl import parse_xbrl_file
24
+ from calcfinc.engine.check import check_values
25
+ from calcfinc.entity import Entity
26
+ from calcfinc.fact import Basis, FinancialFact, MappingConfidence, Source
27
+ from calcfinc.num import div
28
+ from calcfinc.period import DEFAULT_WINDOWS, PeriodWindows, ResolvedPeriod, classify_range, fiscal_year
29
+ from calcfinc.registry import metrics
30
+
31
+ _FILENAME = re.compile(r"^(?P<symbol>.+)_(?P<basis>consolidated|standalone)_(?P<period>.+)_canonical\.json$")
32
+ _XBRL_BASIS = re.compile(r"consolidated|standalone", re.I)
33
+
34
+
35
+ @dataclass(frozen=True, slots=True)
36
+ class IndAsReport:
37
+ entity: str
38
+ records: int
39
+ facts: int
40
+ needs_review: tuple[str, ...] = () # period labels of records whose arithmetic failed
41
+ skipped: tuple[str, ...] = () # facts that could not be placed in a period
42
+ consistency: tuple[dict[str, Any], ...] = field(default=()) # contexts that disagree on a balance
43
+ conflicts: tuple[str, ...] = () # figures two contexts reported differently; one was used
44
+
45
+
46
+ _ORDER = {"One": 0, "Two": 1, "Three": 2, "Four": 3, "Five": 4, "Six": 5}
47
+ _CONTEXT_START = re.compile(r"(One|Two|Three|Four|Five|Six)")
48
+
49
+
50
+ def _rank(context_id: Any, fact: FinancialFact) -> tuple[int, str]:
51
+ """How much to trust the context a fact came from: OneD / OneI describe the current period,
52
+ TwoD... earlier ones, PY_ the prior year. A balance comes from the instant half of a merged
53
+ id ('FourD+OneI'), a flow or a derived figure from the duration half."""
54
+ parts = str(context_id or "").split("+")
55
+ from_instant = fact.is_point_in_time and fact.mapping_confidence is MappingConfidence.EXACT
56
+ part = parts[-1] if from_instant else parts[0]
57
+ m = _CONTEXT_START.match(part)
58
+ return (_ORDER[m.group(1)] if m else 9 if part.startswith("PY_") else 8), part
59
+
60
+
61
+ def _resolve_conflicts(items: list[tuple[int, str, FinancialFact]]) -> tuple[list[FinancialFact], list[str]]:
62
+ """Two contexts of one filing can claim the same figure for the same period with different
63
+ values (a cumulative context given the quarter's dates, a stale prior-year capital). The most
64
+ current context wins, the winner is flagged, and the disagreement is reported; the store
65
+ never gets to pick silently by load order."""
66
+ groups: dict[tuple[Any, ...], list[tuple[int, str, FinancialFact]]] = {}
67
+ for item in items:
68
+ f = item[2]
69
+ groups.setdefault((f.metric, f.basis, f.period_start, f.period_end), []).append(item)
70
+ kept: list[FinancialFact] = []
71
+ notes: list[str] = []
72
+ for candidates in groups.values():
73
+ if len({c[2].value for c in candidates}) == 1:
74
+ kept.append(candidates[0][2])
75
+ continue
76
+ candidates.sort(key=lambda c: c[0])
77
+ _, winner_ctx, winner = candidates[0]
78
+ others = "; ".join(f"{ctx} reports {f.value}" for _, ctx, f in candidates[1:] if f.value != winner.value)
79
+ why = f"contexts disagree on this figure ({others}); {winner_ctx} used"
80
+ kept.append(replace(winner, mapping_reason=f"{winner.mapping_reason}; {why}" if winner.mapping_reason else why))
81
+ notes.append(f"{winner.metric} {winner.period_end}: {winner_ctx} = {winner.value} used; {others}")
82
+ return kept, notes
83
+
84
+
85
+ def _date(v: Any) -> date | None:
86
+ if isinstance(v, date):
87
+ return v
88
+ return date.fromisoformat(str(v)[:10]) if v else None
89
+
90
+
91
+ def record_to_facts(record: Mapping[str, Any], *, entity_id: int, basis: Basis | str = "consolidated",
92
+ currency: str = "INR", fiscal_year_end_month: int = 3,
93
+ windows: PeriodWindows = DEFAULT_WINDOWS, source_id: int | None = None,
94
+ reported_at: date | None = None,
95
+ placeholder_zeros: bool = True) -> tuple[list[FinancialFact], list[str]]:
96
+ """Facts for one canonical record, and notes for any value that could not be placed or that
97
+ was dropped. With `placeholder_zeros` (the default), figures the filing reports as an exact 0
98
+ to mean 'not applicable' are treated as not reported (see `canonical.drop_placeholder_zeros`)."""
99
+ register()
100
+ notes: list[str] = []
101
+ if placeholder_zeros:
102
+ record, dropped = drop_placeholder_zeros(record)
103
+ if dropped:
104
+ notes.append(f"context {record.get('context_id')}: exactly 0, treated as not reported: "
105
+ + ", ".join(dropped))
106
+ start, end = _date(record.get("period_start")), _date(record.get("period_end") or record.get("instant"))
107
+ values = {NAME_MAP.get(k, k): v for k, v in record.items()
108
+ if k not in META_KEYS and not k.startswith("_") and isinstance(v, Decimal)}
109
+ if end is None or not values:
110
+ return [], notes + [f"context {record.get('context_id')}: no period or no values"]
111
+ failed = check_values(values)
112
+ review = ("source record failed arithmetic checks: " + ", ".join(k for k, ok in failed.items() if not ok)
113
+ if not all(failed.values()) else None)
114
+ fye = fiscal_year_end_month
115
+ duration = classify_range(start, end, fye, windows) if start is not None else None
116
+ instant = ResolvedPeriod(None, end, fiscal_year(end, fye), None, False)
117
+
118
+ facts: list[FinancialFact] = []
119
+ for name, value in values.items():
120
+ spec = metrics.get(name)
121
+ if spec is None:
122
+ continue
123
+ pit = spec.is_point_in_time
124
+ period = instant if pit else duration
125
+ if period is None:
126
+ notes.append(f"{name} ({record.get('context_id')}): a flow figure with no start date, not loaded")
127
+ continue
128
+ facts.append(FinancialFact(
129
+ entity_id=entity_id, metric=name, value=value, statement_type=spec.statement_type,
130
+ basis=Basis(basis), currency=currency if spec.kind in metrics.CURRENCY_KINDS else None,
131
+ period_start=None if pit else period.start, period_end=period.end,
132
+ financial_year=period.financial_year, quarter=None if pit else period.quarter,
133
+ is_annual=period.is_annual and not pit, is_point_in_time=pit, reported_at=reported_at,
134
+ source_id=source_id, mapping_confidence=MappingConfidence.EXACT, mapping_reason=review))
135
+ paid_up, face = values.get("paid_up_equity_capital"), values.get("face_value_per_share")
136
+ if paid_up is not None and face:
137
+ facts.append(FinancialFact(
138
+ entity_id=entity_id, metric="shares_outstanding", value=div(paid_up, face),
139
+ statement_type=metrics.BS, basis=Basis(basis), period_end=end,
140
+ financial_year=instant.financial_year, is_point_in_time=True, reported_at=reported_at,
141
+ source_id=source_id, mapping_confidence=MappingConfidence.DERIVED,
142
+ mapping_reason=review or "derived: paid-up equity capital / face value per share"))
143
+ return facts, notes
144
+
145
+
146
+ def load_canonical(repos: Any, records: Iterable[Mapping[str, Any]], *, entity: str,
147
+ basis: Basis | str = "consolidated", currency: str = "INR",
148
+ fiscal_year_end_month: int = 3, windows: PeriodWindows = DEFAULT_WINDOWS,
149
+ reported_at: date | None = None, source: Source | None = None,
150
+ placeholder_zeros: bool = True) -> IndAsReport:
151
+ """Load canonical records for one entity and basis. An existing entity keeps its own
152
+ currency and fiscal year end; a new one is created with the arguments given."""
153
+ register()
154
+ records = list(records)
155
+ ent = repos.entities.resolve(entity)
156
+ fye = ent.fiscal_year_end_month if ent else fiscal_year_end_month
157
+ try:
158
+ if ent is None:
159
+ ent = repos.entities.upsert(Entity(name=entity, currency=currency, fiscal_year_end_month=fye))
160
+ source_id = repos.sources.add(source).source_id if source is not None else None
161
+ items: list[tuple[int, str, FinancialFact]] = []
162
+ skipped: list[str] = []
163
+ review: list[str] = []
164
+ for record in records:
165
+ facts, notes = record_to_facts(record, entity_id=int(ent.id or 0), basis=basis, currency=currency,
166
+ fiscal_year_end_month=fye, windows=windows, source_id=source_id,
167
+ reported_at=reported_at, placeholder_zeros=placeholder_zeros)
168
+ skipped += notes
169
+ if any(f.mapping_reason and f.mapping_reason.startswith("source record failed") for f in facts):
170
+ review.append(str(record.get("period_end") or record.get("instant")))
171
+ items += [(*_rank(record.get("context_id"), f), f) for f in facts]
172
+ all_facts, conflicts = _resolve_conflicts(items)
173
+ repos.facts.add_many(all_facts)
174
+ repos.commit()
175
+ except Exception:
176
+ repos.rollback()
177
+ raise
178
+ return IndAsReport(entity=ent.name, records=len(records), facts=len(all_facts), needs_review=tuple(review),
179
+ skipped=tuple(skipped), consistency=tuple(consistency_issues(records)),
180
+ conflicts=tuple(conflicts))
181
+
182
+
183
+ def load_raw_facts(repos: Any, facts: Iterable[Mapping[str, Any]], **kwargs: Any) -> IndAsReport:
184
+ """Raw XBRL fact rows (see `xbrl.parse_xbrl_file`) -> canonical records -> store."""
185
+ return load_canonical(repos, map_facts(facts), **kwargs)
186
+
187
+
188
+ def load_xbrl_file(repos: Any, path: str | Path, *, entity: str, basis: Basis | str | None = None,
189
+ **kwargs: Any) -> IndAsReport:
190
+ """Load one .xbrl filing. The basis is read from the file name (it must say consolidated or
191
+ standalone) unless you pass it."""
192
+ p = Path(path)
193
+ if basis is None:
194
+ m = _XBRL_BASIS.search(p.stem)
195
+ if m is None:
196
+ raise ValueError(f"{p.name}: cannot tell consolidated from standalone; pass basis=")
197
+ basis = m.group(0).lower()
198
+ source = Source(kind="xbrl", document_title=p.name, uri=str(p.resolve()),
199
+ content_hash=hashlib.sha256(p.read_bytes()).hexdigest(),
200
+ retrieved_at=datetime.now(UTC))
201
+ return load_raw_facts(repos, parse_xbrl_file(p, entity), entity=entity, basis=basis, source=source, **kwargs)
202
+
203
+
204
+ def load_xbrl_files(repos: Any, paths: Iterable[str | Path], *, entity: str, **kwargs: Any) -> list[IndAsReport]:
205
+ """Load several filings, each "Revision" after the "Original" it corrects, whatever order they
206
+ are given in. The filing carries no date, so with no `reported_at` a later load overwrites an
207
+ earlier one for the same period, and a revision must therefore come last."""
208
+ ordered = sorted((Path(p) for p in paths), key=lambda p: ("revision" in p.stem.lower(), p.name))
209
+ return [load_xbrl_file(repos, p, entity=entity, **kwargs) for p in ordered]
210
+
211
+
212
+ def read_canonical_json(path: str | Path) -> list[dict[str, Any]]:
213
+ """Canonical JSON with numbers read as exact Decimals (never through float)."""
214
+ data = json.loads(Path(path).read_text(encoding="utf-8"), parse_float=Decimal, parse_int=Decimal)
215
+ return [data] if isinstance(data, dict) else list(data)
216
+
217
+
218
+ def load_canonical_file(repos: Any, path: str | Path, *, entity: str | None = None,
219
+ basis: Basis | str | None = None, **kwargs: Any) -> IndAsReport:
220
+ """Load a `<SYMBOL>_<consolidated|standalone>_<period>_canonical.json` file. The entity and
221
+ basis default to what the file name says."""
222
+ p = Path(path)
223
+ m = _FILENAME.match(p.name)
224
+ if m is None and (entity is None or basis is None):
225
+ raise ValueError(f"{p.name}: not a <SYMBOL>_<basis>_<period>_canonical.json name; "
226
+ "pass entity= and basis=")
227
+ source = Source(kind="xbrl", document_title=p.name, uri=str(p.resolve()),
228
+ content_hash=hashlib.sha256(p.read_bytes()).hexdigest(),
229
+ retrieved_at=datetime.now(UTC), period_label=m["period"] if m else None)
230
+ return load_canonical(repos, read_canonical_json(p), entity=entity or m["symbol"], # type: ignore[index]
231
+ basis=basis or m["basis"], source=source, **kwargs) # type: ignore[index]
@@ -0,0 +1,144 @@
1
+ """Ind-AS XBRL tag map: exchange taxonomy concepts -> canonical names.
2
+
3
+ Ported from the author's earlier extraction code. The names on the left are the *source-side*
4
+ canonical names that code used; `NAME_MAP` renames the few that differ from calcfinc's
5
+ vocabulary. Concepts live under the `in-capmkt:` taxonomy; older Regulation 33 filings use
6
+ `in-bse-fin:` with the same local names, so `map_facts` falls back to the local name.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import re
11
+
12
+ TAG_MAP: dict[str, str] = {
13
+ # P&L
14
+ "revenue": "in-capmkt:RevenueFromOperations",
15
+ "other_income": "in-capmkt:OtherIncome",
16
+ "total_income": "in-capmkt:Income",
17
+ "employee_expense": "in-capmkt:EmployeeBenefitExpense",
18
+ "depreciation": "in-capmkt:DepreciationDepletionAndAmortisationExpense",
19
+ "other_expenses": "in-capmkt:OtherExpenses",
20
+ "finance_costs": "in-capmkt:FinanceCosts",
21
+ "total_expenses": "in-capmkt:Expenses",
22
+ "pbt_before_exceptional": "in-capmkt:ProfitBeforeExceptionalItemsAndTax",
23
+ "exceptional_items": "in-capmkt:ExceptionalItemsBeforeTax",
24
+ "pbt": "in-capmkt:ProfitBeforeTax",
25
+ "current_tax": "in-capmkt:CurrentTax",
26
+ "deferred_tax": "in-capmkt:DeferredTax",
27
+ "tax_expense": "in-capmkt:TaxExpense",
28
+ "pat_continuing_ops": "in-capmkt:ProfitLossForPeriodFromContinuingOperations",
29
+ "net_profit": "in-capmkt:ProfitLossForPeriod",
30
+ "oci": "in-capmkt:OtherComprehensiveIncomeNetOfTaxes",
31
+ "total_comprehensive_income": "in-capmkt:ComprehensiveIncomeForThePeriod",
32
+ "net_profit_owners": "in-capmkt:ProfitOrLossAttributableToOwnersOfParent",
33
+ "net_profit_nci": "in-capmkt:ProfitOrLossAttributableToNonControllingInterests",
34
+ "eps_basic": "in-capmkt:BasicEarningsLossPerShareFromContinuingAndDiscontinuedOperations",
35
+ "eps_diluted": "in-capmkt:DilutedEarningsLossPerShareFromContinuingAndDiscontinuedOperations",
36
+ # equity / capital and the ratios SEBI makes issuers report
37
+ "paid_up_equity_capital": "in-capmkt:PaidUpValueOfEquityShareCapital",
38
+ "face_value_per_share": "in-capmkt:FaceValueOfEquityShareCapital",
39
+ "debt_equity_ratio_reported": "in-capmkt:DebtEquityRatio",
40
+ "debt_service_coverage_ratio_reported": "in-capmkt:DebtServiceCoverageRatio",
41
+ "interest_service_coverage_ratio_reported": "in-capmkt:InterestServiceCoverageRatio",
42
+ # balance sheet
43
+ "total_assets": "in-capmkt:Assets",
44
+ "total_liabilities": "in-capmkt:Liabilities",
45
+ "total_equity": "in-capmkt:Equity",
46
+ "current_assets": "in-capmkt:CurrentAssets",
47
+ "noncurrent_assets": "in-capmkt:NoncurrentAssets",
48
+ "current_liabilities": "in-capmkt:CurrentLiabilities",
49
+ "noncurrent_liabilities": "in-capmkt:NoncurrentLiabilities",
50
+ "borrowings_current": "in-capmkt:BorrowingsCurrent",
51
+ "borrowings_noncurrent": "in-capmkt:BorrowingsNoncurrent",
52
+ "cash_and_equivalents": "in-capmkt:CashAndCashEquivalents",
53
+ # cash flow
54
+ "operating_cash_flow": "in-capmkt:CashFlowsFromUsedInOperatingActivities",
55
+ "investing_cash_flow": "in-capmkt:CashFlowsFromUsedInInvestingActivities",
56
+ "financing_cash_flow": "in-capmkt:CashFlowsFromUsedInFinancingActivities",
57
+ "dividends": "in-capmkt:DividendsPaidClassifiedAsFinancingActivities",
58
+ "capex_ppe": "in-capmkt:PurchaseOfPropertyPlantAndEquipmentClassifiedAsInvestingActivities",
59
+ "capex_intangibles": "in-capmkt:PurchaseOfIntangibleAssetsClassifiedAsInvestingActivities",
60
+ # insurers (life insurers file net premium income; it marks the filer as an insurer)
61
+ "insurance_net_premium": "in-capmkt:NetPremiumIncome",
62
+ # banks
63
+ "bank_interest_earned": "in-capmkt:InterestEarned",
64
+ "bank_interest_expended": "in-capmkt:InterestExpended",
65
+ "bank_operating_profit": "in-capmkt:OperatingProfitBeforeProvisionAndContingencies",
66
+ "bank_provisions": "in-capmkt:ProvisionsOtherThanTaxAndContingencies",
67
+ "bank_employee_cost": "in-capmkt:EmployeesCost",
68
+ "bank_other_operating_expenses": "in-capmkt:OtherOperatingExpenses",
69
+ "advances": "in-capmkt:Advances",
70
+ # Bank capital adequacy and asset quality are reported at the bank entity (standalone)
71
+ # level only; consolidated filings carry zeros.
72
+ "cet1_ratio": "in-capmkt:CET1Ratio",
73
+ "additional_tier1_ratio": "in-capmkt:AdditionalTier1Ratio",
74
+ "gross_npa": "in-capmkt:GrossNonPerformingAssets",
75
+ # `NonPerformingAssets` is the NET figure (confirmed against a bank's own NPA ratios).
76
+ "net_npa": "in-capmkt:NonPerformingAssets",
77
+ "gross_npa_ratio": "in-capmkt:PercentageOfGrossNpa",
78
+ "net_npa_ratio": "in-capmkt:PercentageOfNpa",
79
+ # Filers tag this with a currency unit although it is a plain ratio; the unit is taken
80
+ # from the metric registry, not from the raw tag, so the mistake is harmless.
81
+ "return_on_assets": "in-capmkt:ReturnOnAssets",
82
+ }
83
+
84
+ BANK_EQUITY_AUX_TAGS = {
85
+ "_bank_capital": "in-capmkt:Capital",
86
+ "_bank_reserves_and_surplus": "in-capmkt:ReservesAndSurplus",
87
+ }
88
+ BANK_BALANCE_SHEET_AUX_TAGS = {
89
+ "_bank_cash_with_rbi": "in-capmkt:CashAndBalancesWithReserveBankOfIndia",
90
+ "_bank_balances_with_banks": "in-capmkt:BalancesWithBanksAndMoneyAtCallAndShortNotice",
91
+ "_bank_other_liabilities_and_provisions": "in-capmkt:OtherLiabilitiesAndProvisions",
92
+ }
93
+ SECTOR_ALT_TAGS = {
94
+ "in-capmkt:ShareholdersFunds": "total_equity",
95
+ "in-capmkt:ProfitLossForThePeriod": "net_profit",
96
+ "in-capmkt:ProfitLossAfterTaxAndExtraordinaryItems": "net_profit",
97
+ "in-capmkt:ProfitLossFromOrdinaryActivitiesBeforeTax": "pbt",
98
+ }
99
+ DEBT_ALT_TAGS = {
100
+ "in-capmkt:Borrowings": "borrowings_noncurrent",
101
+ "in-capmkt:LongTermBorrowings": "borrowings_noncurrent",
102
+ "in-capmkt:ShortTermBorrowings": "borrowings_current",
103
+ "in-capmkt:DebtSecurities": "debt_securities",
104
+ "in-capmkt:Deposits": "deposits_debt",
105
+ }
106
+
107
+ # Source-side canonical name -> calcfinc metric name (names not listed are unchanged).
108
+ NAME_MAP = {
109
+ "pat_continuing_ops": "profit_continuing_ops",
110
+ "insurance_net_premium": "insurance.net_earned_premium",
111
+ "bank_interest_earned": "bank.interest_earned",
112
+ "bank_interest_expended": "bank.interest_expended",
113
+ "bank_operating_profit": "bank.operating_profit",
114
+ "bank_provisions": "bank.provisions",
115
+ "bank_employee_cost": "bank.employee_cost",
116
+ "bank_other_operating_expenses": "bank.other_operating_expenses",
117
+ "advances": "bank.advances",
118
+ "cet1_ratio": "bank.cet1_ratio",
119
+ "additional_tier1_ratio": "bank.additional_tier1_ratio",
120
+ "gross_npa": "bank.gross_npa",
121
+ "net_npa": "bank.net_npa",
122
+ "gross_npa_ratio": "bank.gross_npa_ratio",
123
+ "net_npa_ratio": "bank.net_npa_ratio",
124
+ "return_on_assets": "india.return_on_assets_reported",
125
+ }
126
+
127
+ # The primary whole-company contexts: OneD / OneI, TwoD ..., and PY_D / PY_I for the prior
128
+ # year. Segment and note-breakdown contexts (OneReportable1D, OneExpenses2D...) are rejected.
129
+ PRIMARY_CONTEXT = re.compile(r"^(One|Two|Three|Four|Five|Six)[DI]$|^PY_[DI]$")
130
+
131
+ # Fields compared across contexts that share a date, to catch a filing that disagrees with itself.
132
+ SNAPSHOT_FIELDS = (
133
+ "total_assets", "total_liabilities", "total_equity", "current_assets", "noncurrent_assets",
134
+ "current_liabilities", "noncurrent_liabilities", "borrowings_current", "borrowings_noncurrent",
135
+ "debt_securities", "deposits_debt", "cash_and_equivalents", "paid_up_equity_capital",
136
+ "face_value_per_share",
137
+ )
138
+ HIGH_SEVERITY = frozenset({"paid_up_equity_capital", "face_value_per_share", "total_assets", "total_equity"})
139
+
140
+ META_KEYS = frozenset({"context_id", "period_start", "period_end", "instant"})
141
+
142
+
143
+ def is_primary_context(context_id: str | None) -> bool:
144
+ return bool(PRIMARY_CONTEXT.match(context_id or ""))