fintfm 0.5.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. fintfm/__init__.py +33 -0
  2. fintfm/config.py +329 -0
  3. fintfm/configs/default.yaml +87 -0
  4. fintfm/evaluation/__init__.py +24 -0
  5. fintfm/evaluation/bench.py +259 -0
  6. fintfm/evaluation/boosting.py +109 -0
  7. fintfm/evaluation/datasets.py +376 -0
  8. fintfm/evaluation/fetch.py +115 -0
  9. fintfm/evaluation/metrics.py +311 -0
  10. fintfm/experiments/__init__.py +5 -0
  11. fintfm/experiments/capability.py +1074 -0
  12. fintfm/experiments/context_sweep.py +245 -0
  13. fintfm/experiments/openml_breadth.py +227 -0
  14. fintfm/experiments/prior_ablation.py +621 -0
  15. fintfm/experiments/prior_score.py +302 -0
  16. fintfm/experiments/retrieval_grouping.py +279 -0
  17. fintfm/experiments/term_structure.py +279 -0
  18. fintfm/experiments/v4_out_of_time.py +362 -0
  19. fintfm/experiments/v4_protocol.py +600 -0
  20. fintfm/inference/__init__.py +6 -0
  21. fintfm/inference/binning.py +248 -0
  22. fintfm/inference/categorical.py +259 -0
  23. fintfm/inference/classifier.py +661 -0
  24. fintfm/inference/preprocess.py +114 -0
  25. fintfm/inference/regressor.py +190 -0
  26. fintfm/inference/retrieval.py +247 -0
  27. fintfm/modeling/__init__.py +6 -0
  28. fintfm/modeling/hazard.py +222 -0
  29. fintfm/modeling/model.py +671 -0
  30. fintfm/modeling/train.py +536 -0
  31. fintfm/prior/__init__.py +16 -0
  32. fintfm/prior/base.py +139 -0
  33. fintfm/prior/crossed.py +180 -0
  34. fintfm/prior/financial.py +634 -0
  35. fintfm/prior/mixture.py +242 -0
  36. fintfm/prior/scm.py +353 -0
  37. fintfm/prior/tree.py +169 -0
  38. fintfm/prior/trivial.py +79 -0
  39. fintfm/py.typed +0 -0
  40. fintfm-0.5.5.dist-info/METADATA +407 -0
  41. fintfm-0.5.5.dist-info/RECORD +44 -0
  42. fintfm-0.5.5.dist-info/WHEEL +4 -0
  43. fintfm-0.5.5.dist-info/entry_points.txt +13 -0
  44. fintfm-0.5.5.dist-info/licenses/LICENSE +202 -0
fintfm/__init__.py ADDED
@@ -0,0 +1,33 @@
1
+ """FinTFM: a prior-fitted tabular foundation model for corporate credit risk.
2
+
3
+ The model is pretrained once on synthetic tasks drawn from a *financial prior* (a generative
4
+ story for company tables and their default labels) mixed with a generic structural-causal
5
+ prior. At inference the labelled table is passed as context and unlabelled rows as queries;
6
+ no gradient steps happen on customer data.
7
+
8
+ Package layout follows the pipeline:
9
+
10
+ prior/ synthetic task generation — the only source of pretraining data
11
+ modeling/ architecture and the pretraining loop
12
+ inference/ in-context prediction, context construction, calibration correction
13
+ evaluation/ real datasets, calibration-aware metrics, benchmark harnesses
14
+ experiments/ designed experiments with pre-stated exit conditions
15
+ config.py the layered configuration loader; defaults in configs/default.yaml
16
+
17
+ See ``docs/design/ARCHITECTURE.md`` for how the model works and ``docs/roadmap/STRATEGY.md`` for why.
18
+ """
19
+
20
+ from fintfm.evaluation.metrics import CreditMetrics, evaluate_binary
21
+ from fintfm.inference.classifier import ContextStrategy, FinancialTFMClassifier
22
+ from fintfm.inference.regressor import FinancialTFMRegressor
23
+ from fintfm.modeling.model import FinancialTFM, ModelConfig
24
+
25
+ __all__ = [
26
+ "ContextStrategy",
27
+ "CreditMetrics",
28
+ "FinancialTFM",
29
+ "FinancialTFMClassifier",
30
+ "FinancialTFMRegressor",
31
+ "ModelConfig",
32
+ "evaluate_binary",
33
+ ]
fintfm/config.py ADDED
@@ -0,0 +1,329 @@
1
+ """Typed, strictly validated configuration, layered from a packaged default.
2
+
3
+ Why this is layered rather than simply loaded
4
+ ---------------------------------------------
5
+ Three layers, each overriding the one before: the **packaged default**
6
+ (``fintfm/configs/default.yaml``), then an optional **override file** given by ``--config`` or
7
+ ``FINTFM_CONFIG``, then **explicit command-line flags**. An override file is deep-merged, so
8
+ it need only carry the keys it changes — a sweep that differs from the default in one year
9
+ should be one line, not a copy of the whole file that silently freezes every other value at
10
+ the moment it was copied.
11
+
12
+ Why unknown keys are an error
13
+ -----------------------------
14
+ A misspelled key in a silently-tolerant loader is the worst kind of configuration bug: the run
15
+ completes, reports numbers, and used the default. This repository has already lost a
16
+ pretraining run to a value that was quietly not what it appeared to be
17
+ (``docs/results/FINDINGS.md`` §28), so :func:`load_config` refuses unknown keys and names the path of
18
+ the offender.
19
+
20
+ Why library defaults are duplicated rather than moved
21
+ -----------------------------------------------------
22
+ :class:`~fintfm.inference.classifier.FinancialTFMClassifier` keeps its own literal defaults.
23
+ A library must behave identically without reading a file from disk, and an estimator whose
24
+ defaults depend on an environment variable is not reproducible across machines. The
25
+ ``inference`` section therefore *mirrors* those defaults for experiments to read, and
26
+ ``tests/test_config.py`` asserts the two agree — the test is the drift guard, so changing one
27
+ without the other fails the suite rather than diverging quietly.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import os
33
+ from dataclasses import dataclass, field, fields, is_dataclass
34
+ from pathlib import Path
35
+ from typing import Any
36
+
37
+
38
+ class ConfigError(ValueError):
39
+ """A configuration file is malformed, unknown, or internally inconsistent.
40
+
41
+ A subclass of :class:`ValueError` so existing handlers still catch it, and its own type so
42
+ a caller can distinguish "the config is wrong" from "an argument is wrong" — which matters
43
+ for a CLI that should tell the user which file to edit.
44
+ """
45
+
46
+
47
+ #: Environment variable naming an override file, used when no ``--config`` is given.
48
+ CONFIG_ENV_VAR = "FINTFM_CONFIG"
49
+
50
+ #: The packaged default, always the base layer.
51
+ DEFAULT_CONFIG_PATH = Path(__file__).with_name("configs") / "default.yaml"
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class InferenceConfig:
56
+ """Mirror of :class:`FinancialTFMClassifier`'s defaults, for experiments to read."""
57
+
58
+ max_context: int = 2000
59
+ context_strategy: str = "uniform"
60
+ feature_transform: str = "rank"
61
+ n_ensemble: int = 1
62
+ correct_prior: bool = True
63
+ query_chunk: int = 2048
64
+ feature_chunk: int | None = 16
65
+ retrieval_groups: int = 64
66
+ retrieval_min_positive: int = 8
67
+ prototype_minority_ratio: float = 0.3
68
+ winsor_quantile: float = 0.01
69
+ transform_subsample: int = 20_000
70
+
71
+
72
+ @dataclass(frozen=True)
73
+ class EvaluationConfig:
74
+ """Thresholds that decide what counts as a scoreable result."""
75
+
76
+ bootstrap_resamples: int = 2000
77
+ alpha: float = 0.05
78
+ ece_bins: int = 10
79
+ min_rows_per_horizon: int = 100
80
+
81
+
82
+ @dataclass(frozen=True)
83
+ class HazardArm:
84
+ """One configuration of the hazard head, scored as its own arm."""
85
+
86
+ name: str
87
+ strategy: str
88
+ correct_prior: bool
89
+
90
+
91
+ @dataclass(frozen=True)
92
+ class V4FinBenchConfig:
93
+ """The out-of-time protocol on V4FinBench."""
94
+
95
+ max_rows: int = 120_000
96
+ train_until: int = 2016
97
+ test_from: int = 2017
98
+ max_context: int = 2000
99
+ hazard_arms: tuple[HazardArm, ...] = ()
100
+
101
+
102
+ @dataclass(frozen=True)
103
+ class ContextSweepConfig:
104
+ """The context-strategy sweep."""
105
+
106
+ strategies: tuple[str, ...] = ("balanced", "hybrid", "uniform", "retrieval")
107
+ context_sizes: tuple[int, ...] = (1000, 2000, 4000)
108
+ retrieval_groups: int = 64
109
+ seeds: tuple[int, ...] = (0,)
110
+
111
+
112
+ @dataclass(frozen=True)
113
+ class PriorConfigDefaults:
114
+ """The financial prior's default-rate envelope."""
115
+
116
+ sharpness_min: float = 0.3
117
+ sharpness_max: float = 12.0
118
+ min_expected_positives: float = 2.0
119
+ absolute_rate_floor: float = 0.001
120
+ rate_ceiling: float = 0.30
121
+ n_sectors: int = 12
122
+
123
+
124
+ @dataclass(frozen=True)
125
+ class Config:
126
+ """The whole configuration, with the path it was loaded from for the run record."""
127
+
128
+ inference: InferenceConfig = field(default_factory=InferenceConfig)
129
+ evaluation: EvaluationConfig = field(default_factory=EvaluationConfig)
130
+ v4finbench: V4FinBenchConfig = field(default_factory=V4FinBenchConfig)
131
+ context_sweep: ContextSweepConfig = field(default_factory=ContextSweepConfig)
132
+ prior: PriorConfigDefaults = field(default_factory=PriorConfigDefaults)
133
+ sources: tuple[str, ...] = ()
134
+
135
+ def provenance(self) -> str:
136
+ """The layers this configuration was built from, newest last.
137
+
138
+ Returns:
139
+ A string for a run record, so a result can be traced to the values that made it.
140
+ """
141
+ return " <- ".join(self.sources) if self.sources else "built-in defaults"
142
+
143
+
144
+ def _read_yaml(path: Path) -> dict[str, Any]:
145
+ """Parse a YAML mapping, refusing anything else.
146
+
147
+ Args:
148
+ path: File to read.
149
+
150
+ Returns:
151
+ The parsed mapping, or an empty dict for an empty file.
152
+
153
+ Raises:
154
+ FileNotFoundError: If the file is absent.
155
+ ConfigError: If the document is not a mapping.
156
+ """
157
+ import yaml
158
+
159
+ if not path.exists():
160
+ raise FileNotFoundError(f"config file not found: {path}")
161
+ loaded = yaml.safe_load(path.read_text()) or {}
162
+ if not isinstance(loaded, dict):
163
+ raise ConfigError(
164
+ f"{path}: top level must be a mapping, got {type(loaded).__name__}"
165
+ )
166
+ return loaded
167
+
168
+
169
+ def _deep_merge(base: dict[str, Any], over: dict[str, Any]) -> dict[str, Any]:
170
+ """Merge ``over`` onto ``base``, recursing into nested mappings.
171
+
172
+ Lists replace rather than concatenate: a sweep asking for two strategies means those two,
173
+ not those two appended to the default four.
174
+
175
+ Args:
176
+ base: The lower-priority mapping.
177
+ over: The higher-priority mapping.
178
+
179
+ Returns:
180
+ A new merged mapping; neither input is modified.
181
+ """
182
+ out = dict(base)
183
+ for key, value in over.items():
184
+ if isinstance(value, dict) and isinstance(out.get(key), dict):
185
+ out[key] = _deep_merge(out[key], value)
186
+ else:
187
+ out[key] = value
188
+ return out
189
+
190
+
191
+ def _build(cls: type, data: Any, path: str) -> Any:
192
+ """Instantiate a dataclass from parsed data, rejecting unknown keys.
193
+
194
+ Args:
195
+ cls: Target dataclass.
196
+ data: Parsed mapping, or a value for a non-dataclass field.
197
+ path: Dotted key path, used in error messages.
198
+
199
+ Returns:
200
+ An instance of ``cls``.
201
+
202
+ Raises:
203
+ ConfigError: If ``data`` is not a mapping where one is required, or carries a key the
204
+ dataclass does not define. The message names the full path and the valid keys,
205
+ because a configuration error should be fixable without reading this module.
206
+ """
207
+ if not is_dataclass(cls):
208
+ return data
209
+ if not isinstance(data, dict):
210
+ raise ConfigError(
211
+ f"{path or 'config'}: expected a mapping, got {type(data).__name__}"
212
+ )
213
+ valid = {f.name: f for f in fields(cls)}
214
+ unknown = sorted(set(data) - set(valid))
215
+ if unknown:
216
+ where = f"{path}." if path else ""
217
+ raise ConfigError(
218
+ f"unknown config key(s) {', '.join(where + u for u in unknown)}; "
219
+ f"valid keys here are {', '.join(sorted(valid))}"
220
+ )
221
+ kwargs: dict[str, Any] = {}
222
+ for name in valid:
223
+ if name not in data:
224
+ continue
225
+ sub = f"{path}.{name}" if path else name
226
+ value = data[name]
227
+ if name == "hazard_arms":
228
+ if not isinstance(value, list):
229
+ raise ConfigError(f"{sub}: expected a list of arms")
230
+ kwargs[name] = tuple(_build(HazardArm, v, f"{sub}[]") for v in value)
231
+ elif isinstance(value, list):
232
+ kwargs[name] = tuple(value)
233
+ else:
234
+ kwargs[name] = value
235
+ return cls(**kwargs)
236
+
237
+
238
+ def _validate(cfg: Config) -> None:
239
+ """Reject configurations that are internally inconsistent.
240
+
241
+ Checked here rather than at use time, so a bad value fails before a run starts instead of
242
+ hours in. An overlapping time split is the important one: it would be reported as
243
+ out-of-time validation while not being it.
244
+
245
+ Args:
246
+ cfg: The assembled configuration.
247
+
248
+ Raises:
249
+ ConfigError: Naming the offending key and why it is wrong.
250
+ """
251
+ v = cfg.v4finbench
252
+ if v.test_from <= v.train_until:
253
+ raise ConfigError(
254
+ f"v4finbench.test_from ({v.test_from}) must exceed train_until ({v.train_until}); "
255
+ "an overlapping split is not out-of-time validation"
256
+ )
257
+ if not 0.0 < cfg.evaluation.alpha < 1.0:
258
+ raise ConfigError(f"evaluation.alpha must lie in (0, 1), got {cfg.evaluation.alpha}")
259
+ if cfg.evaluation.bootstrap_resamples < 1:
260
+ raise ConfigError("evaluation.bootstrap_resamples must be at least 1")
261
+ if cfg.evaluation.min_rows_per_horizon < 1:
262
+ raise ConfigError("evaluation.min_rows_per_horizon must be at least 1")
263
+ if not 0.0 < cfg.inference.prototype_minority_ratio <= 1.0:
264
+ raise ConfigError(
265
+ "inference.prototype_minority_ratio must lie in (0, 1], got "
266
+ f"{cfg.inference.prototype_minority_ratio}"
267
+ )
268
+ if not 0.0 <= cfg.inference.winsor_quantile < 0.5:
269
+ raise ConfigError("inference.winsor_quantile must lie in [0, 0.5)")
270
+ if any(n < 1 for n in cfg.context_sweep.context_sizes):
271
+ raise ConfigError("context_sweep.context_sizes must all be at least 1")
272
+ if not cfg.context_sweep.seeds:
273
+ raise ConfigError("context_sweep.seeds must not be empty")
274
+ p = cfg.prior
275
+ if not 0.0 < p.absolute_rate_floor < p.rate_ceiling <= 1.0:
276
+ raise ConfigError(
277
+ f"prior rates must satisfy 0 < absolute_rate_floor ({p.absolute_rate_floor}) < "
278
+ f"rate_ceiling ({p.rate_ceiling}) <= 1"
279
+ )
280
+
281
+
282
+ #: Section name to dataclass. Declared explicitly rather than read off ``Config``'s
283
+ #: annotations, because ``from __future__ import annotations`` turns ``field.type`` into a
284
+ #: string and resolving it back would be indirection for its own sake.
285
+ _SECTIONS: dict[str, type] = {
286
+ "inference": InferenceConfig,
287
+ "evaluation": EvaluationConfig,
288
+ "v4finbench": V4FinBenchConfig,
289
+ "context_sweep": ContextSweepConfig,
290
+ "prior": PriorConfigDefaults,
291
+ }
292
+
293
+
294
+ def load_config(path: str | os.PathLike[str] | None = None) -> Config:
295
+ """Load configuration, layering an override over the packaged default.
296
+
297
+ Args:
298
+ path: Override file. Falls back to ``$FINTFM_CONFIG``, then to no override at all.
299
+
300
+ Returns:
301
+ A validated :class:`Config` recording the layers it came from.
302
+
303
+ Raises:
304
+ FileNotFoundError: If a path was given explicitly and does not exist. A *missing*
305
+ ``FINTFM_CONFIG`` is also an error rather than a silent fallback, because the
306
+ whole point of setting it is that it takes effect.
307
+ ConfigError: If a key is unknown, mistyped, or fails validation.
308
+ """
309
+ data = _read_yaml(DEFAULT_CONFIG_PATH)
310
+ sources = [str(DEFAULT_CONFIG_PATH)]
311
+ override = path if path is not None else os.environ.get(CONFIG_ENV_VAR)
312
+ if override:
313
+ over_path = Path(override)
314
+ data = _deep_merge(data, _read_yaml(over_path))
315
+ sources.append(str(over_path))
316
+ if "sources" in data:
317
+ raise ConfigError("config key 'sources' is set by the loader and cannot be provided")
318
+ unknown = sorted(set(data) - set(_SECTIONS))
319
+ if unknown:
320
+ raise ConfigError(
321
+ f"unknown top-level config section(s) {', '.join(unknown)}; valid sections are "
322
+ f"{', '.join(sorted(_SECTIONS))}"
323
+ )
324
+ cfg = Config(
325
+ **{name: _build(cls, data[name], name) for name, cls in _SECTIONS.items() if name in data},
326
+ sources=tuple(sources),
327
+ )
328
+ _validate(cfg)
329
+ return cfg
@@ -0,0 +1,87 @@
1
+ # fintfm default configuration.
2
+ #
3
+ # This file is the single declared place for the numbers that experiments and command-line
4
+ # entry points use. Override it with `--config path/to/file.yaml` on any experiment CLI, or
5
+ # by setting FINTFM_CONFIG; an override is deep-merged over these values, so a file need only
6
+ # carry what it changes. Individual CLI flags override both.
7
+ #
8
+ # WHAT DELIBERATELY DOES NOT LIVE HERE
9
+ # ------------------------------------
10
+ # Not every literal in the codebase is configuration, and treating them all as such makes a
11
+ # system less robust rather than more. These stay in code on purpose:
12
+ #
13
+ # * hazard.CENSORED (-1) and the observation-mask semantics -- part of the data contract.
14
+ # A configurable sentinel is a way to silently reinterpret every stored period label.
15
+ # * retrieval._SCALE_FLOOR, HazardHead.max_hazard, the log1p/clamp epsilons -- numerical
16
+ # guards whose values are tied to float32 behaviour, not to a modelling choice.
17
+ # * the accounting identities in prior/financial.py -- these are arithmetic, not settings.
18
+ # * dataset URLs and licence attributions -- provenance, and changing one silently would
19
+ # break the licence record in README.md.
20
+ #
21
+ # The rule: a value belongs here if a reasonable experiment would want it different. A value
22
+ # whose change would make results incomparable or the code incorrect stays in code.
23
+
24
+ # Library defaults for FinancialTFMClassifier. These MIRROR the literal defaults in
25
+ # inference/classifier.py rather than replacing them: a library should behave identically
26
+ # without reading a file from disk, so the code keeps its own defaults and
27
+ # tests/test_config.py asserts the two agree. That test is the drift guard -- if you change
28
+ # one, it fails until you change the other.
29
+ inference:
30
+ max_context: 2000
31
+ context_strategy: uniform # docs/design/DECISIONS.md D10; was "balanced" before 2026-09-09
32
+ feature_transform: rank # docs/results/FINDINGS.md §35
33
+ n_ensemble: 1 # D12's stated mitigation for the column-order invariance
34
+ correct_prior: true
35
+ query_chunk: 2048
36
+ feature_chunk: 16 # identity-preserving memory knob; docs/results/FINDINGS.md §81
37
+ retrieval_groups: 64
38
+ retrieval_min_positive: 8
39
+ prototype_minority_ratio: 0.3 # Kostrzewa et al.'s value; docs/results/FINDINGS.md §38
40
+ winsor_quantile: 0.01
41
+ transform_subsample: 20000
42
+
43
+ # Thresholds used when scoring. Changing these changes what counts as a result, so they are
44
+ # recorded in every run's output alongside the numbers they produced.
45
+ evaluation:
46
+ bootstrap_resamples: 2000
47
+ alpha: 0.05
48
+ ece_bins: 10
49
+ # a horizon with fewer observed rows than this, or only one class, is reported as NaN
50
+ # rather than scored -- see docs/results/FINDINGS.md on degenerate splits poisoning aggregates
51
+ min_rows_per_horizon: 100
52
+
53
+ # The V4FinBench out-of-time protocol. The split years are the load-bearing numbers here:
54
+ # train and test must not overlap, and time_split() refuses if they do.
55
+ v4finbench:
56
+ max_rows: 120000
57
+ train_until: 2016
58
+ test_from: 2017
59
+ max_context: 2000
60
+ # scored side by side; the uncorrected arm is permanent by decision D8, so that the
61
+ # base-rate distortion of docs/results/FINDINGS.md §28 is measured rather than assumed absent
62
+ hazard_arms:
63
+ - {name: fintfm_hazard, strategy: prototype, correct_prior: true} # best, §38
64
+ - {name: fintfm_hazard_uncorrected, strategy: balanced, correct_prior: false}
65
+ - {name: fintfm_hazard_uniform, strategy: uniform, correct_prior: true}
66
+
67
+ # The context-strategy sweep behind docs/results/FINDINGS.md §29, §32 and §33.
68
+ context_sweep:
69
+ strategies: [balanced, hybrid, uniform, retrieval]
70
+ # 2000 is the operating point: on three seeds 4000 ties with it for 1.7x the time (§33)
71
+ context_sizes: [1000, 2000, 4000]
72
+ retrieval_groups: 64
73
+ seeds: [0]
74
+
75
+ # The financial prior's default-rate envelope. The floor is what lets the prior reach the
76
+ # low-default regime the strategy targets: at two expected defaults a 512-row task reaches
77
+ # 0.195%, which covers V4FinBench's 0.19% (docs/infra/COMPUTE.md).
78
+ prior:
79
+ # The financial prior's signal-to-noise range, drawn log-uniformly per task. Spans
80
+ # near-pure-noise to near-deterministic (docs/results/FINDINGS.md §42), and §64 measured this as
81
+ # the variable that predicts what a prior teaches.
82
+ sharpness_min: 0.3
83
+ sharpness_max: 12.0
84
+ min_expected_positives: 2.0
85
+ absolute_rate_floor: 0.001
86
+ rate_ceiling: 0.30
87
+ n_sectors: 12
@@ -0,0 +1,24 @@
1
+ """Evaluation: real datasets, metrics that see calibration, and benchmark harnesses.
2
+
3
+ Datasets loaded here are for **evaluation only**. Nothing real may reach pretraining — that
4
+ invariant is what makes a benchmark number auditable (``docs/results/FINDINGS.md`` §1).
5
+ """
6
+
7
+ from fintfm.evaluation.datasets import (
8
+ CreditDataset,
9
+ SurvivalDataset,
10
+ load_polish_bankruptcy,
11
+ load_taiwan_bankruptcy,
12
+ load_v4finbench,
13
+ )
14
+ from fintfm.evaluation.metrics import CreditMetrics, evaluate_binary
15
+
16
+ __all__ = [
17
+ "CreditDataset",
18
+ "CreditMetrics",
19
+ "SurvivalDataset",
20
+ "evaluate_binary",
21
+ "load_polish_bankruptcy",
22
+ "load_taiwan_bankruptcy",
23
+ "load_v4finbench",
24
+ ]