calcfinc 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calcfinc/__init__.py +30 -0
- calcfinc/adapters/__init__.py +1 -0
- calcfinc/adapters/ind_as_xbrl/__init__.py +29 -0
- calcfinc/adapters/ind_as_xbrl/canonical.py +171 -0
- calcfinc/adapters/ind_as_xbrl/load.py +231 -0
- calcfinc/adapters/ind_as_xbrl/tags.py +144 -0
- calcfinc/adapters/ind_as_xbrl/vocab.py +129 -0
- calcfinc/adapters/ind_as_xbrl/xbrl.py +128 -0
- calcfinc/adapters/sec_companyfacts/__init__.py +24 -0
- calcfinc/adapters/sec_companyfacts/fetch.py +70 -0
- calcfinc/adapters/sec_companyfacts/load.py +81 -0
- calcfinc/adapters/sec_companyfacts/parse.py +393 -0
- calcfinc/adapters/sec_companyfacts/tags.py +93 -0
- calcfinc/engine/__init__.py +11 -0
- calcfinc/engine/check.py +95 -0
- calcfinc/engine/decompose.py +80 -0
- calcfinc/engine/engine.py +690 -0
- calcfinc/engine/evaluate.py +189 -0
- calcfinc/engine/growth.py +28 -0
- calcfinc/engine/records.py +157 -0
- calcfinc/engine/segments.py +179 -0
- calcfinc/entity.py +41 -0
- calcfinc/fact.py +177 -0
- calcfinc/formula.py +180 -0
- calcfinc/loaders/__init__.py +6 -0
- calcfinc/loaders/csv.py +86 -0
- calcfinc/loaders/dataframe.py +93 -0
- calcfinc/loaders/records.py +255 -0
- calcfinc/num.py +96 -0
- calcfinc/period.py +112 -0
- calcfinc/py.typed +0 -0
- calcfinc/registry/__init__.py +6 -0
- calcfinc/registry/metrics.py +159 -0
- calcfinc/registry/ratios.py +427 -0
- calcfinc/store/__init__.py +5 -0
- calcfinc/store/base.py +91 -0
- calcfinc/store/schema.sql +104 -0
- calcfinc/store/sqlite.py +418 -0
- calcfinc-0.1.0.dist-info/METADATA +167 -0
- calcfinc-0.1.0.dist-info/RECORD +42 -0
- calcfinc-0.1.0.dist-info/WHEEL +4 -0
- calcfinc-0.1.0.dist-info/licenses/LICENSE +21 -0
calcfinc/__init__.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""calcfinc: auditable financial metrics for any entity."""
|
|
2
|
+
import logging
|
|
3
|
+
|
|
4
|
+
from calcfinc.engine import CheckResult, EngineError, EngineResult, FactRef, FinancialEngine
|
|
5
|
+
from calcfinc.entity import Entity
|
|
6
|
+
from calcfinc.fact import (
|
|
7
|
+
Basis,
|
|
8
|
+
FinancialFact,
|
|
9
|
+
MappingConfidence,
|
|
10
|
+
Segment,
|
|
11
|
+
SegmentFact,
|
|
12
|
+
SharePrice,
|
|
13
|
+
Source,
|
|
14
|
+
StatementType,
|
|
15
|
+
)
|
|
16
|
+
from calcfinc.period import PeriodWindows
|
|
17
|
+
from calcfinc.registry import RatioSpec, register_metric, register_ratio
|
|
18
|
+
from calcfinc.store import SqliteRepositories
|
|
19
|
+
|
|
20
|
+
__version__ = "0.1.0"
|
|
21
|
+
|
|
22
|
+
# A library logs through the standard `logging` module and leaves the handlers to the application.
|
|
23
|
+
logging.getLogger(__name__).addHandler(logging.NullHandler())
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"Basis", "CheckResult", "Entity", "EngineError", "EngineResult", "FactRef", "FinancialEngine",
|
|
27
|
+
"FinancialFact", "MappingConfidence", "PeriodWindows", "RatioSpec", "Segment", "SegmentFact",
|
|
28
|
+
"SharePrice", "Source", "SqliteRepositories", "StatementType", "__version__", "register_metric",
|
|
29
|
+
"register_ratio",
|
|
30
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Source adapters: each maps a source's tags and conventions to canonical facts."""
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Ind-AS (India) adapter.
|
|
2
|
+
|
|
3
|
+
Reads exchange XBRL filings (a .xbrl file, raw fact rows, or canonical records) into
|
|
4
|
+
calcfinc facts: April-March fiscal year, INR, exact Decimal values, review flags for records
|
|
5
|
+
whose own arithmetic fails. Also holds the India-only vocabulary and the `india.*` ratios
|
|
6
|
+
(`register()`), so the core stays free of country names.
|
|
7
|
+
|
|
8
|
+
from calcfinc.adapters import ind_as_xbrl
|
|
9
|
+
ind_as_xbrl.load_xbrl_file(repos, "results_consolidated.xbrl", entity="ACME")
|
|
10
|
+
"""
|
|
11
|
+
from calcfinc.adapters.ind_as_xbrl.canonical import consistency_issues, map_facts, parse_number
|
|
12
|
+
from calcfinc.adapters.ind_as_xbrl.load import (
|
|
13
|
+
IndAsReport,
|
|
14
|
+
load_canonical,
|
|
15
|
+
load_canonical_file,
|
|
16
|
+
load_raw_facts,
|
|
17
|
+
load_xbrl_file,
|
|
18
|
+
load_xbrl_files,
|
|
19
|
+
read_canonical_json,
|
|
20
|
+
record_to_facts,
|
|
21
|
+
)
|
|
22
|
+
from calcfinc.adapters.ind_as_xbrl.vocab import RENAMES, register, shares_outstanding
|
|
23
|
+
from calcfinc.adapters.ind_as_xbrl.xbrl import parse_xbrl_file
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"IndAsReport", "RENAMES", "consistency_issues", "load_canonical", "load_canonical_file",
|
|
27
|
+
"load_raw_facts", "load_xbrl_file", "load_xbrl_files", "map_facts", "parse_number", "parse_xbrl_file",
|
|
28
|
+
"read_canonical_json", "record_to_facts", "register", "shares_outstanding",
|
|
29
|
+
]
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""Raw XBRL facts -> canonical records, with exact Decimal values.
|
|
2
|
+
|
|
3
|
+
A *raw fact* is a dict as produced by `xbrl.parse_xbrl_file`: line_item_tag, value (text),
|
|
4
|
+
context_id, period_start / period_end / instant, sign. A *canonical record* is one context's
|
|
5
|
+
facts under the source-side canonical names (`tags.TAG_MAP`), plus its period.
|
|
6
|
+
|
|
7
|
+
Only whole-company contexts are used (OneD, OneI, PY_D...), never segment or note breakdowns.
|
|
8
|
+
Duration and instant contexts are kept as separate records; the engine merges a balance sheet
|
|
9
|
+
into the duration record that ends on the same day.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections import defaultdict
|
|
14
|
+
from collections.abc import Iterable, Mapping
|
|
15
|
+
from decimal import Decimal, InvalidOperation
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from calcfinc.adapters.ind_as_xbrl.tags import (
|
|
19
|
+
BANK_BALANCE_SHEET_AUX_TAGS,
|
|
20
|
+
BANK_EQUITY_AUX_TAGS,
|
|
21
|
+
DEBT_ALT_TAGS,
|
|
22
|
+
HIGH_SEVERITY,
|
|
23
|
+
SECTOR_ALT_TAGS,
|
|
24
|
+
SNAPSHOT_FIELDS,
|
|
25
|
+
TAG_MAP,
|
|
26
|
+
is_primary_context,
|
|
27
|
+
)
|
|
28
|
+
from calcfinc.num import HUNDRED, add, div, mul, sub
|
|
29
|
+
|
|
30
|
+
_MISSING = {"", "-", "—", "NA", "N/A"}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def parse_number(raw: object) -> Decimal | None:
|
|
34
|
+
"""Parse a reported number exactly. Handles Indian formatting ('59,553.00'), accounting
|
|
35
|
+
negatives ('(1,234.5)'), a rupee sign and trailing footnote markers. Returns None for
|
|
36
|
+
anything that is not a number. A float (from JSON) is taken through its shortest text,
|
|
37
|
+
which is the original text for any figure of 17 significant digits or fewer."""
|
|
38
|
+
if raw is None or isinstance(raw, bool):
|
|
39
|
+
return None
|
|
40
|
+
if isinstance(raw, Decimal):
|
|
41
|
+
return raw if raw.is_finite() else None
|
|
42
|
+
if isinstance(raw, int):
|
|
43
|
+
return Decimal(raw)
|
|
44
|
+
s = repr(raw) if isinstance(raw, float) else str(raw).strip()
|
|
45
|
+
if s in _MISSING:
|
|
46
|
+
return None
|
|
47
|
+
negative = s.startswith("(") and s.endswith(")")
|
|
48
|
+
s = s.strip("()").replace(",", "").replace("₹", "").strip()
|
|
49
|
+
for candidate in (s, "".join(c for c in s if c.isdigit() or c in ".-")):
|
|
50
|
+
try:
|
|
51
|
+
value = Decimal(candidate)
|
|
52
|
+
except InvalidOperation:
|
|
53
|
+
continue
|
|
54
|
+
if value.is_finite():
|
|
55
|
+
return -value if negative else value
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _tag_index() -> tuple[dict[str, str], dict[str, str]]:
|
|
60
|
+
by_tag = {v: k for k, v in TAG_MAP.items()}
|
|
61
|
+
by_tag.update({v: k for k, v in BANK_EQUITY_AUX_TAGS.items()})
|
|
62
|
+
by_tag.update({v: k for k, v in BANK_BALANCE_SHEET_AUX_TAGS.items()})
|
|
63
|
+
by_tag.update(SECTOR_ALT_TAGS)
|
|
64
|
+
by_tag.update(DEBT_ALT_TAGS)
|
|
65
|
+
# Older Regulation 33 filings use the in-bse-fin: prefix with the same local concept name.
|
|
66
|
+
# An exact full-tag match always wins; the local name is only a fallback.
|
|
67
|
+
by_local: dict[str, str] = {}
|
|
68
|
+
for tag, name in by_tag.items():
|
|
69
|
+
by_local.setdefault(tag.split(":", 1)[1] if ":" in tag else tag, name)
|
|
70
|
+
return by_tag, by_local
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def map_facts(facts: Iterable[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
|
74
|
+
"""Group raw facts into canonical records, newest first."""
|
|
75
|
+
by_tag, by_local = _tag_index()
|
|
76
|
+
fields: dict[str, dict[str, Decimal | None]] = defaultdict(dict)
|
|
77
|
+
periods: dict[str, dict[str, Any]] = {}
|
|
78
|
+
|
|
79
|
+
for fact in facts:
|
|
80
|
+
ctx = fact.get("context_id")
|
|
81
|
+
if not isinstance(ctx, str) or not is_primary_context(ctx):
|
|
82
|
+
continue
|
|
83
|
+
tag = fact.get("line_item_tag")
|
|
84
|
+
name = by_tag.get(tag or "")
|
|
85
|
+
if name is None and tag and ":" in tag:
|
|
86
|
+
name = by_local.get(tag.split(":", 1)[1])
|
|
87
|
+
if name is None:
|
|
88
|
+
continue # not a field we track
|
|
89
|
+
value = parse_number(fact.get("value"))
|
|
90
|
+
if value is not None and fact.get("sign") == "-":
|
|
91
|
+
value = -value
|
|
92
|
+
fields[ctx][name] = value
|
|
93
|
+
periods[ctx] = {"context_id": ctx, "period_start": fact.get("period_start"),
|
|
94
|
+
"period_end": fact.get("period_end"), "instant": fact.get("instant")}
|
|
95
|
+
|
|
96
|
+
records = []
|
|
97
|
+
for ctx, values in fields.items():
|
|
98
|
+
record: dict[str, Any] = {**periods[ctx], **values}
|
|
99
|
+
_derive_bank_lines(record)
|
|
100
|
+
records.append(record)
|
|
101
|
+
records.sort(key=lambda r: r.get("period_end") or r.get("instant") or "", reverse=True)
|
|
102
|
+
return records
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _derive_bank_lines(record: dict[str, Any]) -> None:
|
|
106
|
+
"""Banks report their equity, cash and liabilities as sub-lines. Build the totals only when
|
|
107
|
+
the total itself is absent and every needed part is present."""
|
|
108
|
+
capital = record.pop("_bank_capital", None)
|
|
109
|
+
reserves = record.pop("_bank_reserves_and_surplus", None)
|
|
110
|
+
if record.get("total_equity") is None and capital is not None and reserves is not None:
|
|
111
|
+
record["total_equity"] = add(capital, reserves)
|
|
112
|
+
cash_with_central_bank = record.pop("_bank_cash_with_rbi", None)
|
|
113
|
+
balances_with_banks = record.pop("_bank_balances_with_banks", None)
|
|
114
|
+
if (record.get("cash_and_equivalents") is None and cash_with_central_bank is not None
|
|
115
|
+
and balances_with_banks is not None):
|
|
116
|
+
record["cash_and_equivalents"] = add(cash_with_central_bank, balances_with_banks)
|
|
117
|
+
other = record.pop("_bank_other_liabilities_and_provisions", None)
|
|
118
|
+
if (record.get("total_liabilities") is None and other is not None
|
|
119
|
+
and record.get("deposits_debt") is not None and record.get("borrowings_noncurrent") is not None):
|
|
120
|
+
record["total_liabilities"] = add(add(record["deposits_debt"], record["borrowings_noncurrent"]), other)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# Fields a filing reports as an exact 0 when it means "not applicable". Found in real filings:
|
|
124
|
+
# a consolidated bank filing carries 0 for its whole NPA block; paid-up capital, face value and
|
|
125
|
+
# CET1 are never genuinely 0; a coverage ratio of exactly 0 is a placeholder, not a result.
|
|
126
|
+
NPA_BLOCK = ("gross_npa", "net_npa", "gross_npa_ratio", "net_npa_ratio")
|
|
127
|
+
BANK_REGULATORY = (*NPA_BLOCK, "return_on_assets", "cet1_ratio", "additional_tier1_ratio")
|
|
128
|
+
ZERO_IS_MISSING = ("paid_up_equity_capital", "face_value_per_share", "cet1_ratio",
|
|
129
|
+
"debt_service_coverage_ratio_reported", "interest_service_coverage_ratio_reported")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def drop_placeholder_zeros(record: Mapping[str, Any]) -> tuple[dict[str, Any], list[str]]:
|
|
133
|
+
"""The record without the zeros that mean 'not reported', and the names removed. A genuine
|
|
134
|
+
zero (no exceptional items, no borrowings, no minority interest) is left alone."""
|
|
135
|
+
out = dict(record)
|
|
136
|
+
dropped: list[str] = []
|
|
137
|
+
npa = [record.get(k) for k in NPA_BLOCK]
|
|
138
|
+
if all(isinstance(v, Decimal) and v == 0 for v in npa): # the whole block is zero together
|
|
139
|
+
dropped += [k for k in BANK_REGULATORY if isinstance(record.get(k), Decimal)]
|
|
140
|
+
dropped += [k for k in ZERO_IS_MISSING if isinstance(record.get(k), Decimal) and record[k] == 0
|
|
141
|
+
and k not in dropped]
|
|
142
|
+
for k in dropped:
|
|
143
|
+
out.pop(k, None)
|
|
144
|
+
return out, dropped
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def consistency_issues(records: Iterable[Mapping[str, Any]],
|
|
148
|
+
tolerance: Decimal = Decimal(1)) -> list[dict[str, Any]]:
|
|
149
|
+
"""Records that end on the same date but disagree on a balance-sheet / capital field, as a
|
|
150
|
+
filing sometimes does between its quarterly and cumulative contexts. Differences within
|
|
151
|
+
`tolerance` (rounding) are ignored."""
|
|
152
|
+
by_date: dict[str, list[Mapping[str, Any]]] = defaultdict(list)
|
|
153
|
+
for r in records:
|
|
154
|
+
key = r.get("period_end") or r.get("instant")
|
|
155
|
+
if key:
|
|
156
|
+
by_date[key].append(r)
|
|
157
|
+
issues: list[dict[str, Any]] = []
|
|
158
|
+
for date_key, group in by_date.items():
|
|
159
|
+
if len(group) < 2:
|
|
160
|
+
continue
|
|
161
|
+
for name in SNAPSHOT_FIELDS:
|
|
162
|
+
values: dict[Any, Decimal] = {r.get("context_id"): r[name] for r in group if r.get(name) is not None}
|
|
163
|
+
distinct = set(values.values())
|
|
164
|
+
if len(distinct) <= 1 or sub(max(distinct), min(distinct)) <= tolerance:
|
|
165
|
+
continue
|
|
166
|
+
nonzero = [v for v in distinct if v != 0]
|
|
167
|
+
smallest = min(nonzero, key=abs) if nonzero else Decimal(1)
|
|
168
|
+
pct = div(mul(sub(max(distinct), min(distinct)), HUNDRED), abs(smallest)).quantize(Decimal("0.01"))
|
|
169
|
+
issues.append({"period_end": date_key, "field": name, "values": values,
|
|
170
|
+
"severity": "high" if name in HIGH_SEVERITY else "low", "magnitude_pct": pct})
|
|
171
|
+
return issues
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Canonical records -> facts in a store.
|
|
2
|
+
|
|
3
|
+
Ind-AS reporting: an April-March fiscal year and INR by default. Values are exact Decimals; a
|
|
4
|
+
record whose own arithmetic does not add up (assets vs liabilities + equity, income vs
|
|
5
|
+
expenses...) still loads, but every fact from it carries a review reason so it can be routed
|
|
6
|
+
to a person.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import re
|
|
13
|
+
from collections.abc import Iterable, Mapping
|
|
14
|
+
from dataclasses import dataclass, field, replace
|
|
15
|
+
from datetime import UTC, date, datetime
|
|
16
|
+
from decimal import Decimal
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from calcfinc.adapters.ind_as_xbrl.canonical import consistency_issues, drop_placeholder_zeros, map_facts
|
|
21
|
+
from calcfinc.adapters.ind_as_xbrl.tags import META_KEYS, NAME_MAP
|
|
22
|
+
from calcfinc.adapters.ind_as_xbrl.vocab import register
|
|
23
|
+
from calcfinc.adapters.ind_as_xbrl.xbrl import parse_xbrl_file
|
|
24
|
+
from calcfinc.engine.check import check_values
|
|
25
|
+
from calcfinc.entity import Entity
|
|
26
|
+
from calcfinc.fact import Basis, FinancialFact, MappingConfidence, Source
|
|
27
|
+
from calcfinc.num import div
|
|
28
|
+
from calcfinc.period import DEFAULT_WINDOWS, PeriodWindows, ResolvedPeriod, classify_range, fiscal_year
|
|
29
|
+
from calcfinc.registry import metrics
|
|
30
|
+
|
|
31
|
+
_FILENAME = re.compile(r"^(?P<symbol>.+)_(?P<basis>consolidated|standalone)_(?P<period>.+)_canonical\.json$")
|
|
32
|
+
_XBRL_BASIS = re.compile(r"consolidated|standalone", re.I)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, slots=True)
|
|
36
|
+
class IndAsReport:
|
|
37
|
+
entity: str
|
|
38
|
+
records: int
|
|
39
|
+
facts: int
|
|
40
|
+
needs_review: tuple[str, ...] = () # period labels of records whose arithmetic failed
|
|
41
|
+
skipped: tuple[str, ...] = () # facts that could not be placed in a period
|
|
42
|
+
consistency: tuple[dict[str, Any], ...] = field(default=()) # contexts that disagree on a balance
|
|
43
|
+
conflicts: tuple[str, ...] = () # figures two contexts reported differently; one was used
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
_ORDER = {"One": 0, "Two": 1, "Three": 2, "Four": 3, "Five": 4, "Six": 5}
|
|
47
|
+
_CONTEXT_START = re.compile(r"(One|Two|Three|Four|Five|Six)")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _rank(context_id: Any, fact: FinancialFact) -> tuple[int, str]:
|
|
51
|
+
"""How much to trust the context a fact came from: OneD / OneI describe the current period,
|
|
52
|
+
TwoD... earlier ones, PY_ the prior year. A balance comes from the instant half of a merged
|
|
53
|
+
id ('FourD+OneI'), a flow or a derived figure from the duration half."""
|
|
54
|
+
parts = str(context_id or "").split("+")
|
|
55
|
+
from_instant = fact.is_point_in_time and fact.mapping_confidence is MappingConfidence.EXACT
|
|
56
|
+
part = parts[-1] if from_instant else parts[0]
|
|
57
|
+
m = _CONTEXT_START.match(part)
|
|
58
|
+
return (_ORDER[m.group(1)] if m else 9 if part.startswith("PY_") else 8), part
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _resolve_conflicts(items: list[tuple[int, str, FinancialFact]]) -> tuple[list[FinancialFact], list[str]]:
|
|
62
|
+
"""Two contexts of one filing can claim the same figure for the same period with different
|
|
63
|
+
values (a cumulative context given the quarter's dates, a stale prior-year capital). The most
|
|
64
|
+
current context wins, the winner is flagged, and the disagreement is reported; the store
|
|
65
|
+
never gets to pick silently by load order."""
|
|
66
|
+
groups: dict[tuple[Any, ...], list[tuple[int, str, FinancialFact]]] = {}
|
|
67
|
+
for item in items:
|
|
68
|
+
f = item[2]
|
|
69
|
+
groups.setdefault((f.metric, f.basis, f.period_start, f.period_end), []).append(item)
|
|
70
|
+
kept: list[FinancialFact] = []
|
|
71
|
+
notes: list[str] = []
|
|
72
|
+
for candidates in groups.values():
|
|
73
|
+
if len({c[2].value for c in candidates}) == 1:
|
|
74
|
+
kept.append(candidates[0][2])
|
|
75
|
+
continue
|
|
76
|
+
candidates.sort(key=lambda c: c[0])
|
|
77
|
+
_, winner_ctx, winner = candidates[0]
|
|
78
|
+
others = "; ".join(f"{ctx} reports {f.value}" for _, ctx, f in candidates[1:] if f.value != winner.value)
|
|
79
|
+
why = f"contexts disagree on this figure ({others}); {winner_ctx} used"
|
|
80
|
+
kept.append(replace(winner, mapping_reason=f"{winner.mapping_reason}; {why}" if winner.mapping_reason else why))
|
|
81
|
+
notes.append(f"{winner.metric} {winner.period_end}: {winner_ctx} = {winner.value} used; {others}")
|
|
82
|
+
return kept, notes
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _date(v: Any) -> date | None:
|
|
86
|
+
if isinstance(v, date):
|
|
87
|
+
return v
|
|
88
|
+
return date.fromisoformat(str(v)[:10]) if v else None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def record_to_facts(record: Mapping[str, Any], *, entity_id: int, basis: Basis | str = "consolidated",
|
|
92
|
+
currency: str = "INR", fiscal_year_end_month: int = 3,
|
|
93
|
+
windows: PeriodWindows = DEFAULT_WINDOWS, source_id: int | None = None,
|
|
94
|
+
reported_at: date | None = None,
|
|
95
|
+
placeholder_zeros: bool = True) -> tuple[list[FinancialFact], list[str]]:
|
|
96
|
+
"""Facts for one canonical record, and notes for any value that could not be placed or that
|
|
97
|
+
was dropped. With `placeholder_zeros` (the default), figures the filing reports as an exact 0
|
|
98
|
+
to mean 'not applicable' are treated as not reported (see `canonical.drop_placeholder_zeros`)."""
|
|
99
|
+
register()
|
|
100
|
+
notes: list[str] = []
|
|
101
|
+
if placeholder_zeros:
|
|
102
|
+
record, dropped = drop_placeholder_zeros(record)
|
|
103
|
+
if dropped:
|
|
104
|
+
notes.append(f"context {record.get('context_id')}: exactly 0, treated as not reported: "
|
|
105
|
+
+ ", ".join(dropped))
|
|
106
|
+
start, end = _date(record.get("period_start")), _date(record.get("period_end") or record.get("instant"))
|
|
107
|
+
values = {NAME_MAP.get(k, k): v for k, v in record.items()
|
|
108
|
+
if k not in META_KEYS and not k.startswith("_") and isinstance(v, Decimal)}
|
|
109
|
+
if end is None or not values:
|
|
110
|
+
return [], notes + [f"context {record.get('context_id')}: no period or no values"]
|
|
111
|
+
failed = check_values(values)
|
|
112
|
+
review = ("source record failed arithmetic checks: " + ", ".join(k for k, ok in failed.items() if not ok)
|
|
113
|
+
if not all(failed.values()) else None)
|
|
114
|
+
fye = fiscal_year_end_month
|
|
115
|
+
duration = classify_range(start, end, fye, windows) if start is not None else None
|
|
116
|
+
instant = ResolvedPeriod(None, end, fiscal_year(end, fye), None, False)
|
|
117
|
+
|
|
118
|
+
facts: list[FinancialFact] = []
|
|
119
|
+
for name, value in values.items():
|
|
120
|
+
spec = metrics.get(name)
|
|
121
|
+
if spec is None:
|
|
122
|
+
continue
|
|
123
|
+
pit = spec.is_point_in_time
|
|
124
|
+
period = instant if pit else duration
|
|
125
|
+
if period is None:
|
|
126
|
+
notes.append(f"{name} ({record.get('context_id')}): a flow figure with no start date, not loaded")
|
|
127
|
+
continue
|
|
128
|
+
facts.append(FinancialFact(
|
|
129
|
+
entity_id=entity_id, metric=name, value=value, statement_type=spec.statement_type,
|
|
130
|
+
basis=Basis(basis), currency=currency if spec.kind in metrics.CURRENCY_KINDS else None,
|
|
131
|
+
period_start=None if pit else period.start, period_end=period.end,
|
|
132
|
+
financial_year=period.financial_year, quarter=None if pit else period.quarter,
|
|
133
|
+
is_annual=period.is_annual and not pit, is_point_in_time=pit, reported_at=reported_at,
|
|
134
|
+
source_id=source_id, mapping_confidence=MappingConfidence.EXACT, mapping_reason=review))
|
|
135
|
+
paid_up, face = values.get("paid_up_equity_capital"), values.get("face_value_per_share")
|
|
136
|
+
if paid_up is not None and face:
|
|
137
|
+
facts.append(FinancialFact(
|
|
138
|
+
entity_id=entity_id, metric="shares_outstanding", value=div(paid_up, face),
|
|
139
|
+
statement_type=metrics.BS, basis=Basis(basis), period_end=end,
|
|
140
|
+
financial_year=instant.financial_year, is_point_in_time=True, reported_at=reported_at,
|
|
141
|
+
source_id=source_id, mapping_confidence=MappingConfidence.DERIVED,
|
|
142
|
+
mapping_reason=review or "derived: paid-up equity capital / face value per share"))
|
|
143
|
+
return facts, notes
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def load_canonical(repos: Any, records: Iterable[Mapping[str, Any]], *, entity: str,
|
|
147
|
+
basis: Basis | str = "consolidated", currency: str = "INR",
|
|
148
|
+
fiscal_year_end_month: int = 3, windows: PeriodWindows = DEFAULT_WINDOWS,
|
|
149
|
+
reported_at: date | None = None, source: Source | None = None,
|
|
150
|
+
placeholder_zeros: bool = True) -> IndAsReport:
|
|
151
|
+
"""Load canonical records for one entity and basis. An existing entity keeps its own
|
|
152
|
+
currency and fiscal year end; a new one is created with the arguments given."""
|
|
153
|
+
register()
|
|
154
|
+
records = list(records)
|
|
155
|
+
ent = repos.entities.resolve(entity)
|
|
156
|
+
fye = ent.fiscal_year_end_month if ent else fiscal_year_end_month
|
|
157
|
+
try:
|
|
158
|
+
if ent is None:
|
|
159
|
+
ent = repos.entities.upsert(Entity(name=entity, currency=currency, fiscal_year_end_month=fye))
|
|
160
|
+
source_id = repos.sources.add(source).source_id if source is not None else None
|
|
161
|
+
items: list[tuple[int, str, FinancialFact]] = []
|
|
162
|
+
skipped: list[str] = []
|
|
163
|
+
review: list[str] = []
|
|
164
|
+
for record in records:
|
|
165
|
+
facts, notes = record_to_facts(record, entity_id=int(ent.id or 0), basis=basis, currency=currency,
|
|
166
|
+
fiscal_year_end_month=fye, windows=windows, source_id=source_id,
|
|
167
|
+
reported_at=reported_at, placeholder_zeros=placeholder_zeros)
|
|
168
|
+
skipped += notes
|
|
169
|
+
if any(f.mapping_reason and f.mapping_reason.startswith("source record failed") for f in facts):
|
|
170
|
+
review.append(str(record.get("period_end") or record.get("instant")))
|
|
171
|
+
items += [(*_rank(record.get("context_id"), f), f) for f in facts]
|
|
172
|
+
all_facts, conflicts = _resolve_conflicts(items)
|
|
173
|
+
repos.facts.add_many(all_facts)
|
|
174
|
+
repos.commit()
|
|
175
|
+
except Exception:
|
|
176
|
+
repos.rollback()
|
|
177
|
+
raise
|
|
178
|
+
return IndAsReport(entity=ent.name, records=len(records), facts=len(all_facts), needs_review=tuple(review),
|
|
179
|
+
skipped=tuple(skipped), consistency=tuple(consistency_issues(records)),
|
|
180
|
+
conflicts=tuple(conflicts))
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def load_raw_facts(repos: Any, facts: Iterable[Mapping[str, Any]], **kwargs: Any) -> IndAsReport:
|
|
184
|
+
"""Raw XBRL fact rows (see `xbrl.parse_xbrl_file`) -> canonical records -> store."""
|
|
185
|
+
return load_canonical(repos, map_facts(facts), **kwargs)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def load_xbrl_file(repos: Any, path: str | Path, *, entity: str, basis: Basis | str | None = None,
|
|
189
|
+
**kwargs: Any) -> IndAsReport:
|
|
190
|
+
"""Load one .xbrl filing. The basis is read from the file name (it must say consolidated or
|
|
191
|
+
standalone) unless you pass it."""
|
|
192
|
+
p = Path(path)
|
|
193
|
+
if basis is None:
|
|
194
|
+
m = _XBRL_BASIS.search(p.stem)
|
|
195
|
+
if m is None:
|
|
196
|
+
raise ValueError(f"{p.name}: cannot tell consolidated from standalone; pass basis=")
|
|
197
|
+
basis = m.group(0).lower()
|
|
198
|
+
source = Source(kind="xbrl", document_title=p.name, uri=str(p.resolve()),
|
|
199
|
+
content_hash=hashlib.sha256(p.read_bytes()).hexdigest(),
|
|
200
|
+
retrieved_at=datetime.now(UTC))
|
|
201
|
+
return load_raw_facts(repos, parse_xbrl_file(p, entity), entity=entity, basis=basis, source=source, **kwargs)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def load_xbrl_files(repos: Any, paths: Iterable[str | Path], *, entity: str, **kwargs: Any) -> list[IndAsReport]:
|
|
205
|
+
"""Load several filings, each "Revision" after the "Original" it corrects, whatever order they
|
|
206
|
+
are given in. The filing carries no date, so with no `reported_at` a later load overwrites an
|
|
207
|
+
earlier one for the same period, and a revision must therefore come last."""
|
|
208
|
+
ordered = sorted((Path(p) for p in paths), key=lambda p: ("revision" in p.stem.lower(), p.name))
|
|
209
|
+
return [load_xbrl_file(repos, p, entity=entity, **kwargs) for p in ordered]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def read_canonical_json(path: str | Path) -> list[dict[str, Any]]:
|
|
213
|
+
"""Canonical JSON with numbers read as exact Decimals (never through float)."""
|
|
214
|
+
data = json.loads(Path(path).read_text(encoding="utf-8"), parse_float=Decimal, parse_int=Decimal)
|
|
215
|
+
return [data] if isinstance(data, dict) else list(data)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def load_canonical_file(repos: Any, path: str | Path, *, entity: str | None = None,
|
|
219
|
+
basis: Basis | str | None = None, **kwargs: Any) -> IndAsReport:
|
|
220
|
+
"""Load a `<SYMBOL>_<consolidated|standalone>_<period>_canonical.json` file. The entity and
|
|
221
|
+
basis default to what the file name says."""
|
|
222
|
+
p = Path(path)
|
|
223
|
+
m = _FILENAME.match(p.name)
|
|
224
|
+
if m is None and (entity is None or basis is None):
|
|
225
|
+
raise ValueError(f"{p.name}: not a <SYMBOL>_<basis>_<period>_canonical.json name; "
|
|
226
|
+
"pass entity= and basis=")
|
|
227
|
+
source = Source(kind="xbrl", document_title=p.name, uri=str(p.resolve()),
|
|
228
|
+
content_hash=hashlib.sha256(p.read_bytes()).hexdigest(),
|
|
229
|
+
retrieved_at=datetime.now(UTC), period_label=m["period"] if m else None)
|
|
230
|
+
return load_canonical(repos, read_canonical_json(p), entity=entity or m["symbol"], # type: ignore[index]
|
|
231
|
+
basis=basis or m["basis"], source=source, **kwargs) # type: ignore[index]
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Ind-AS XBRL tag map: exchange taxonomy concepts -> canonical names.
|
|
2
|
+
|
|
3
|
+
Ported from the author's earlier extraction code. The names on the left are the *source-side*
|
|
4
|
+
canonical names that code used; `NAME_MAP` renames the few that differ from calcfinc's
|
|
5
|
+
vocabulary. Concepts live under the `in-capmkt:` taxonomy; older Regulation 33 filings use
|
|
6
|
+
`in-bse-fin:` with the same local names, so `map_facts` falls back to the local name.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
TAG_MAP: dict[str, str] = {
|
|
13
|
+
# P&L
|
|
14
|
+
"revenue": "in-capmkt:RevenueFromOperations",
|
|
15
|
+
"other_income": "in-capmkt:OtherIncome",
|
|
16
|
+
"total_income": "in-capmkt:Income",
|
|
17
|
+
"employee_expense": "in-capmkt:EmployeeBenefitExpense",
|
|
18
|
+
"depreciation": "in-capmkt:DepreciationDepletionAndAmortisationExpense",
|
|
19
|
+
"other_expenses": "in-capmkt:OtherExpenses",
|
|
20
|
+
"finance_costs": "in-capmkt:FinanceCosts",
|
|
21
|
+
"total_expenses": "in-capmkt:Expenses",
|
|
22
|
+
"pbt_before_exceptional": "in-capmkt:ProfitBeforeExceptionalItemsAndTax",
|
|
23
|
+
"exceptional_items": "in-capmkt:ExceptionalItemsBeforeTax",
|
|
24
|
+
"pbt": "in-capmkt:ProfitBeforeTax",
|
|
25
|
+
"current_tax": "in-capmkt:CurrentTax",
|
|
26
|
+
"deferred_tax": "in-capmkt:DeferredTax",
|
|
27
|
+
"tax_expense": "in-capmkt:TaxExpense",
|
|
28
|
+
"pat_continuing_ops": "in-capmkt:ProfitLossForPeriodFromContinuingOperations",
|
|
29
|
+
"net_profit": "in-capmkt:ProfitLossForPeriod",
|
|
30
|
+
"oci": "in-capmkt:OtherComprehensiveIncomeNetOfTaxes",
|
|
31
|
+
"total_comprehensive_income": "in-capmkt:ComprehensiveIncomeForThePeriod",
|
|
32
|
+
"net_profit_owners": "in-capmkt:ProfitOrLossAttributableToOwnersOfParent",
|
|
33
|
+
"net_profit_nci": "in-capmkt:ProfitOrLossAttributableToNonControllingInterests",
|
|
34
|
+
"eps_basic": "in-capmkt:BasicEarningsLossPerShareFromContinuingAndDiscontinuedOperations",
|
|
35
|
+
"eps_diluted": "in-capmkt:DilutedEarningsLossPerShareFromContinuingAndDiscontinuedOperations",
|
|
36
|
+
# equity / capital and the ratios SEBI makes issuers report
|
|
37
|
+
"paid_up_equity_capital": "in-capmkt:PaidUpValueOfEquityShareCapital",
|
|
38
|
+
"face_value_per_share": "in-capmkt:FaceValueOfEquityShareCapital",
|
|
39
|
+
"debt_equity_ratio_reported": "in-capmkt:DebtEquityRatio",
|
|
40
|
+
"debt_service_coverage_ratio_reported": "in-capmkt:DebtServiceCoverageRatio",
|
|
41
|
+
"interest_service_coverage_ratio_reported": "in-capmkt:InterestServiceCoverageRatio",
|
|
42
|
+
# balance sheet
|
|
43
|
+
"total_assets": "in-capmkt:Assets",
|
|
44
|
+
"total_liabilities": "in-capmkt:Liabilities",
|
|
45
|
+
"total_equity": "in-capmkt:Equity",
|
|
46
|
+
"current_assets": "in-capmkt:CurrentAssets",
|
|
47
|
+
"noncurrent_assets": "in-capmkt:NoncurrentAssets",
|
|
48
|
+
"current_liabilities": "in-capmkt:CurrentLiabilities",
|
|
49
|
+
"noncurrent_liabilities": "in-capmkt:NoncurrentLiabilities",
|
|
50
|
+
"borrowings_current": "in-capmkt:BorrowingsCurrent",
|
|
51
|
+
"borrowings_noncurrent": "in-capmkt:BorrowingsNoncurrent",
|
|
52
|
+
"cash_and_equivalents": "in-capmkt:CashAndCashEquivalents",
|
|
53
|
+
# cash flow
|
|
54
|
+
"operating_cash_flow": "in-capmkt:CashFlowsFromUsedInOperatingActivities",
|
|
55
|
+
"investing_cash_flow": "in-capmkt:CashFlowsFromUsedInInvestingActivities",
|
|
56
|
+
"financing_cash_flow": "in-capmkt:CashFlowsFromUsedInFinancingActivities",
|
|
57
|
+
"dividends": "in-capmkt:DividendsPaidClassifiedAsFinancingActivities",
|
|
58
|
+
"capex_ppe": "in-capmkt:PurchaseOfPropertyPlantAndEquipmentClassifiedAsInvestingActivities",
|
|
59
|
+
"capex_intangibles": "in-capmkt:PurchaseOfIntangibleAssetsClassifiedAsInvestingActivities",
|
|
60
|
+
# insurers (life insurers file net premium income; it marks the filer as an insurer)
|
|
61
|
+
"insurance_net_premium": "in-capmkt:NetPremiumIncome",
|
|
62
|
+
# banks
|
|
63
|
+
"bank_interest_earned": "in-capmkt:InterestEarned",
|
|
64
|
+
"bank_interest_expended": "in-capmkt:InterestExpended",
|
|
65
|
+
"bank_operating_profit": "in-capmkt:OperatingProfitBeforeProvisionAndContingencies",
|
|
66
|
+
"bank_provisions": "in-capmkt:ProvisionsOtherThanTaxAndContingencies",
|
|
67
|
+
"bank_employee_cost": "in-capmkt:EmployeesCost",
|
|
68
|
+
"bank_other_operating_expenses": "in-capmkt:OtherOperatingExpenses",
|
|
69
|
+
"advances": "in-capmkt:Advances",
|
|
70
|
+
# Bank capital adequacy and asset quality are reported at the bank entity (standalone)
|
|
71
|
+
# level only; consolidated filings carry zeros.
|
|
72
|
+
"cet1_ratio": "in-capmkt:CET1Ratio",
|
|
73
|
+
"additional_tier1_ratio": "in-capmkt:AdditionalTier1Ratio",
|
|
74
|
+
"gross_npa": "in-capmkt:GrossNonPerformingAssets",
|
|
75
|
+
# `NonPerformingAssets` is the NET figure (confirmed against a bank's own NPA ratios).
|
|
76
|
+
"net_npa": "in-capmkt:NonPerformingAssets",
|
|
77
|
+
"gross_npa_ratio": "in-capmkt:PercentageOfGrossNpa",
|
|
78
|
+
"net_npa_ratio": "in-capmkt:PercentageOfNpa",
|
|
79
|
+
# Filers tag this with a currency unit although it is a plain ratio; the unit is taken
|
|
80
|
+
# from the metric registry, not from the raw tag, so the mistake is harmless.
|
|
81
|
+
"return_on_assets": "in-capmkt:ReturnOnAssets",
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
BANK_EQUITY_AUX_TAGS = {
|
|
85
|
+
"_bank_capital": "in-capmkt:Capital",
|
|
86
|
+
"_bank_reserves_and_surplus": "in-capmkt:ReservesAndSurplus",
|
|
87
|
+
}
|
|
88
|
+
BANK_BALANCE_SHEET_AUX_TAGS = {
|
|
89
|
+
"_bank_cash_with_rbi": "in-capmkt:CashAndBalancesWithReserveBankOfIndia",
|
|
90
|
+
"_bank_balances_with_banks": "in-capmkt:BalancesWithBanksAndMoneyAtCallAndShortNotice",
|
|
91
|
+
"_bank_other_liabilities_and_provisions": "in-capmkt:OtherLiabilitiesAndProvisions",
|
|
92
|
+
}
|
|
93
|
+
SECTOR_ALT_TAGS = {
|
|
94
|
+
"in-capmkt:ShareholdersFunds": "total_equity",
|
|
95
|
+
"in-capmkt:ProfitLossForThePeriod": "net_profit",
|
|
96
|
+
"in-capmkt:ProfitLossAfterTaxAndExtraordinaryItems": "net_profit",
|
|
97
|
+
"in-capmkt:ProfitLossFromOrdinaryActivitiesBeforeTax": "pbt",
|
|
98
|
+
}
|
|
99
|
+
DEBT_ALT_TAGS = {
|
|
100
|
+
"in-capmkt:Borrowings": "borrowings_noncurrent",
|
|
101
|
+
"in-capmkt:LongTermBorrowings": "borrowings_noncurrent",
|
|
102
|
+
"in-capmkt:ShortTermBorrowings": "borrowings_current",
|
|
103
|
+
"in-capmkt:DebtSecurities": "debt_securities",
|
|
104
|
+
"in-capmkt:Deposits": "deposits_debt",
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
# Source-side canonical name -> calcfinc metric name (names not listed are unchanged).
|
|
108
|
+
NAME_MAP = {
|
|
109
|
+
"pat_continuing_ops": "profit_continuing_ops",
|
|
110
|
+
"insurance_net_premium": "insurance.net_earned_premium",
|
|
111
|
+
"bank_interest_earned": "bank.interest_earned",
|
|
112
|
+
"bank_interest_expended": "bank.interest_expended",
|
|
113
|
+
"bank_operating_profit": "bank.operating_profit",
|
|
114
|
+
"bank_provisions": "bank.provisions",
|
|
115
|
+
"bank_employee_cost": "bank.employee_cost",
|
|
116
|
+
"bank_other_operating_expenses": "bank.other_operating_expenses",
|
|
117
|
+
"advances": "bank.advances",
|
|
118
|
+
"cet1_ratio": "bank.cet1_ratio",
|
|
119
|
+
"additional_tier1_ratio": "bank.additional_tier1_ratio",
|
|
120
|
+
"gross_npa": "bank.gross_npa",
|
|
121
|
+
"net_npa": "bank.net_npa",
|
|
122
|
+
"gross_npa_ratio": "bank.gross_npa_ratio",
|
|
123
|
+
"net_npa_ratio": "bank.net_npa_ratio",
|
|
124
|
+
"return_on_assets": "india.return_on_assets_reported",
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
# The primary whole-company contexts: OneD / OneI, TwoD ..., and PY_D / PY_I for the prior
|
|
128
|
+
# year. Segment and note-breakdown contexts (OneReportable1D, OneExpenses2D...) are rejected.
|
|
129
|
+
PRIMARY_CONTEXT = re.compile(r"^(One|Two|Three|Four|Five|Six)[DI]$|^PY_[DI]$")
|
|
130
|
+
|
|
131
|
+
# Fields compared across contexts that share a date, to catch a filing that disagrees with itself.
|
|
132
|
+
SNAPSHOT_FIELDS = (
|
|
133
|
+
"total_assets", "total_liabilities", "total_equity", "current_assets", "noncurrent_assets",
|
|
134
|
+
"current_liabilities", "noncurrent_liabilities", "borrowings_current", "borrowings_noncurrent",
|
|
135
|
+
"debt_securities", "deposits_debt", "cash_and_equivalents", "paid_up_equity_capital",
|
|
136
|
+
"face_value_per_share",
|
|
137
|
+
)
|
|
138
|
+
HIGH_SEVERITY = frozenset({"paid_up_equity_capital", "face_value_per_share", "total_assets", "total_equity"})
|
|
139
|
+
|
|
140
|
+
META_KEYS = frozenset({"context_id", "period_start", "period_end", "instant"})
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_primary_context(context_id: str | None) -> bool:
|
|
144
|
+
return bool(PRIMARY_CONTEXT.match(context_id or ""))
|