fintfm 0.5.5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fintfm/__init__.py +33 -0
- fintfm/config.py +329 -0
- fintfm/configs/default.yaml +87 -0
- fintfm/evaluation/__init__.py +24 -0
- fintfm/evaluation/bench.py +259 -0
- fintfm/evaluation/boosting.py +109 -0
- fintfm/evaluation/datasets.py +376 -0
- fintfm/evaluation/fetch.py +115 -0
- fintfm/evaluation/metrics.py +311 -0
- fintfm/experiments/__init__.py +5 -0
- fintfm/experiments/capability.py +1074 -0
- fintfm/experiments/context_sweep.py +245 -0
- fintfm/experiments/openml_breadth.py +227 -0
- fintfm/experiments/prior_ablation.py +621 -0
- fintfm/experiments/prior_score.py +302 -0
- fintfm/experiments/retrieval_grouping.py +279 -0
- fintfm/experiments/term_structure.py +279 -0
- fintfm/experiments/v4_out_of_time.py +362 -0
- fintfm/experiments/v4_protocol.py +600 -0
- fintfm/inference/__init__.py +6 -0
- fintfm/inference/binning.py +248 -0
- fintfm/inference/categorical.py +259 -0
- fintfm/inference/classifier.py +661 -0
- fintfm/inference/preprocess.py +114 -0
- fintfm/inference/regressor.py +190 -0
- fintfm/inference/retrieval.py +247 -0
- fintfm/modeling/__init__.py +6 -0
- fintfm/modeling/hazard.py +222 -0
- fintfm/modeling/model.py +671 -0
- fintfm/modeling/train.py +536 -0
- fintfm/prior/__init__.py +16 -0
- fintfm/prior/base.py +139 -0
- fintfm/prior/crossed.py +180 -0
- fintfm/prior/financial.py +634 -0
- fintfm/prior/mixture.py +242 -0
- fintfm/prior/scm.py +353 -0
- fintfm/prior/tree.py +169 -0
- fintfm/prior/trivial.py +79 -0
- fintfm/py.typed +0 -0
- fintfm-0.5.5.dist-info/METADATA +407 -0
- fintfm-0.5.5.dist-info/RECORD +44 -0
- fintfm-0.5.5.dist-info/WHEEL +4 -0
- fintfm-0.5.5.dist-info/entry_points.txt +13 -0
- fintfm-0.5.5.dist-info/licenses/LICENSE +202 -0
fintfm/__init__.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""FinTFM: a prior-fitted tabular foundation model for corporate credit risk.
|
|
2
|
+
|
|
3
|
+
The model is pretrained once on synthetic tasks drawn from a *financial prior* (a generative
|
|
4
|
+
story for company tables and their default labels) mixed with a generic structural-causal
|
|
5
|
+
prior. At inference the labelled table is passed as context and unlabelled rows as queries;
|
|
6
|
+
no gradient steps happen on customer data.
|
|
7
|
+
|
|
8
|
+
Package layout follows the pipeline:
|
|
9
|
+
|
|
10
|
+
prior/ synthetic task generation — the only source of pretraining data
|
|
11
|
+
modeling/ architecture and the pretraining loop
|
|
12
|
+
inference/ in-context prediction, context construction, calibration correction
|
|
13
|
+
evaluation/ real datasets, calibration-aware metrics, benchmark harnesses
|
|
14
|
+
experiments/ designed experiments with pre-stated exit conditions
|
|
15
|
+
config.py the layered configuration loader; defaults in configs/default.yaml
|
|
16
|
+
|
|
17
|
+
See ``docs/design/ARCHITECTURE.md`` for how the model works and ``docs/roadmap/STRATEGY.md`` for why.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from fintfm.evaluation.metrics import CreditMetrics, evaluate_binary
|
|
21
|
+
from fintfm.inference.classifier import ContextStrategy, FinancialTFMClassifier
|
|
22
|
+
from fintfm.inference.regressor import FinancialTFMRegressor
|
|
23
|
+
from fintfm.modeling.model import FinancialTFM, ModelConfig
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ContextStrategy",
|
|
27
|
+
"CreditMetrics",
|
|
28
|
+
"FinancialTFM",
|
|
29
|
+
"FinancialTFMClassifier",
|
|
30
|
+
"FinancialTFMRegressor",
|
|
31
|
+
"ModelConfig",
|
|
32
|
+
"evaluate_binary",
|
|
33
|
+
]
|
fintfm/config.py
ADDED
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""Typed, strictly validated configuration, layered from a packaged default.
|
|
2
|
+
|
|
3
|
+
Why this is layered rather than simply loaded
|
|
4
|
+
---------------------------------------------
|
|
5
|
+
Three layers, each overriding the one before: the **packaged default**
|
|
6
|
+
(``fintfm/configs/default.yaml``), then an optional **override file** given by ``--config`` or
|
|
7
|
+
``FINTFM_CONFIG``, then **explicit command-line flags**. An override file is deep-merged, so
|
|
8
|
+
it need only carry the keys it changes — a sweep that differs from the default in one year
|
|
9
|
+
should be one line, not a copy of the whole file that silently freezes every other value at
|
|
10
|
+
the moment it was copied.
|
|
11
|
+
|
|
12
|
+
Why unknown keys are an error
|
|
13
|
+
-----------------------------
|
|
14
|
+
A misspelled key in a silently-tolerant loader is the worst kind of configuration bug: the run
|
|
15
|
+
completes, reports numbers, and used the default. This repository has already lost a
|
|
16
|
+
pretraining run to a value that was quietly not what it appeared to be
|
|
17
|
+
(``docs/results/FINDINGS.md`` §28), so :func:`load_config` refuses unknown keys and names the path of
|
|
18
|
+
the offender.
|
|
19
|
+
|
|
20
|
+
Why library defaults are duplicated rather than moved
|
|
21
|
+
-----------------------------------------------------
|
|
22
|
+
:class:`~fintfm.inference.classifier.FinancialTFMClassifier` keeps its own literal defaults.
|
|
23
|
+
A library must behave identically without reading a file from disk, and an estimator whose
|
|
24
|
+
defaults depend on an environment variable is not reproducible across machines. The
|
|
25
|
+
``inference`` section therefore *mirrors* those defaults for experiments to read, and
|
|
26
|
+
``tests/test_config.py`` asserts the two agree — the test is the drift guard, so changing one
|
|
27
|
+
without the other fails the suite rather than diverging quietly.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import os
|
|
33
|
+
from dataclasses import dataclass, field, fields, is_dataclass
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Any
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ConfigError(ValueError):
|
|
39
|
+
"""A configuration file is malformed, unknown, or internally inconsistent.
|
|
40
|
+
|
|
41
|
+
A subclass of :class:`ValueError` so existing handlers still catch it, and its own type so
|
|
42
|
+
a caller can distinguish "the config is wrong" from "an argument is wrong" — which matters
|
|
43
|
+
for a CLI that should tell the user which file to edit.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
#: Environment variable naming an override file, used when no ``--config`` is given.
|
|
48
|
+
CONFIG_ENV_VAR = "FINTFM_CONFIG"
|
|
49
|
+
|
|
50
|
+
#: The packaged default, always the base layer.
|
|
51
|
+
DEFAULT_CONFIG_PATH = Path(__file__).with_name("configs") / "default.yaml"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass(frozen=True)
|
|
55
|
+
class InferenceConfig:
|
|
56
|
+
"""Mirror of :class:`FinancialTFMClassifier`'s defaults, for experiments to read."""
|
|
57
|
+
|
|
58
|
+
max_context: int = 2000
|
|
59
|
+
context_strategy: str = "uniform"
|
|
60
|
+
feature_transform: str = "rank"
|
|
61
|
+
n_ensemble: int = 1
|
|
62
|
+
correct_prior: bool = True
|
|
63
|
+
query_chunk: int = 2048
|
|
64
|
+
feature_chunk: int | None = 16
|
|
65
|
+
retrieval_groups: int = 64
|
|
66
|
+
retrieval_min_positive: int = 8
|
|
67
|
+
prototype_minority_ratio: float = 0.3
|
|
68
|
+
winsor_quantile: float = 0.01
|
|
69
|
+
transform_subsample: int = 20_000
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class EvaluationConfig:
|
|
74
|
+
"""Thresholds that decide what counts as a scoreable result."""
|
|
75
|
+
|
|
76
|
+
bootstrap_resamples: int = 2000
|
|
77
|
+
alpha: float = 0.05
|
|
78
|
+
ece_bins: int = 10
|
|
79
|
+
min_rows_per_horizon: int = 100
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class HazardArm:
|
|
84
|
+
"""One configuration of the hazard head, scored as its own arm."""
|
|
85
|
+
|
|
86
|
+
name: str
|
|
87
|
+
strategy: str
|
|
88
|
+
correct_prior: bool
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True)
|
|
92
|
+
class V4FinBenchConfig:
|
|
93
|
+
"""The out-of-time protocol on V4FinBench."""
|
|
94
|
+
|
|
95
|
+
max_rows: int = 120_000
|
|
96
|
+
train_until: int = 2016
|
|
97
|
+
test_from: int = 2017
|
|
98
|
+
max_context: int = 2000
|
|
99
|
+
hazard_arms: tuple[HazardArm, ...] = ()
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@dataclass(frozen=True)
|
|
103
|
+
class ContextSweepConfig:
|
|
104
|
+
"""The context-strategy sweep."""
|
|
105
|
+
|
|
106
|
+
strategies: tuple[str, ...] = ("balanced", "hybrid", "uniform", "retrieval")
|
|
107
|
+
context_sizes: tuple[int, ...] = (1000, 2000, 4000)
|
|
108
|
+
retrieval_groups: int = 64
|
|
109
|
+
seeds: tuple[int, ...] = (0,)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass(frozen=True)
|
|
113
|
+
class PriorConfigDefaults:
|
|
114
|
+
"""The financial prior's default-rate envelope."""
|
|
115
|
+
|
|
116
|
+
sharpness_min: float = 0.3
|
|
117
|
+
sharpness_max: float = 12.0
|
|
118
|
+
min_expected_positives: float = 2.0
|
|
119
|
+
absolute_rate_floor: float = 0.001
|
|
120
|
+
rate_ceiling: float = 0.30
|
|
121
|
+
n_sectors: int = 12
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass(frozen=True)
|
|
125
|
+
class Config:
|
|
126
|
+
"""The whole configuration, with the path it was loaded from for the run record."""
|
|
127
|
+
|
|
128
|
+
inference: InferenceConfig = field(default_factory=InferenceConfig)
|
|
129
|
+
evaluation: EvaluationConfig = field(default_factory=EvaluationConfig)
|
|
130
|
+
v4finbench: V4FinBenchConfig = field(default_factory=V4FinBenchConfig)
|
|
131
|
+
context_sweep: ContextSweepConfig = field(default_factory=ContextSweepConfig)
|
|
132
|
+
prior: PriorConfigDefaults = field(default_factory=PriorConfigDefaults)
|
|
133
|
+
sources: tuple[str, ...] = ()
|
|
134
|
+
|
|
135
|
+
def provenance(self) -> str:
|
|
136
|
+
"""The layers this configuration was built from, newest last.
|
|
137
|
+
|
|
138
|
+
Returns:
|
|
139
|
+
A string for a run record, so a result can be traced to the values that made it.
|
|
140
|
+
"""
|
|
141
|
+
return " <- ".join(self.sources) if self.sources else "built-in defaults"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _read_yaml(path: Path) -> dict[str, Any]:
|
|
145
|
+
"""Parse a YAML mapping, refusing anything else.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
path: File to read.
|
|
149
|
+
|
|
150
|
+
Returns:
|
|
151
|
+
The parsed mapping, or an empty dict for an empty file.
|
|
152
|
+
|
|
153
|
+
Raises:
|
|
154
|
+
FileNotFoundError: If the file is absent.
|
|
155
|
+
ConfigError: If the document is not a mapping.
|
|
156
|
+
"""
|
|
157
|
+
import yaml
|
|
158
|
+
|
|
159
|
+
if not path.exists():
|
|
160
|
+
raise FileNotFoundError(f"config file not found: {path}")
|
|
161
|
+
loaded = yaml.safe_load(path.read_text()) or {}
|
|
162
|
+
if not isinstance(loaded, dict):
|
|
163
|
+
raise ConfigError(
|
|
164
|
+
f"{path}: top level must be a mapping, got {type(loaded).__name__}"
|
|
165
|
+
)
|
|
166
|
+
return loaded
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _deep_merge(base: dict[str, Any], over: dict[str, Any]) -> dict[str, Any]:
|
|
170
|
+
"""Merge ``over`` onto ``base``, recursing into nested mappings.
|
|
171
|
+
|
|
172
|
+
Lists replace rather than concatenate: a sweep asking for two strategies means those two,
|
|
173
|
+
not those two appended to the default four.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
base: The lower-priority mapping.
|
|
177
|
+
over: The higher-priority mapping.
|
|
178
|
+
|
|
179
|
+
Returns:
|
|
180
|
+
A new merged mapping; neither input is modified.
|
|
181
|
+
"""
|
|
182
|
+
out = dict(base)
|
|
183
|
+
for key, value in over.items():
|
|
184
|
+
if isinstance(value, dict) and isinstance(out.get(key), dict):
|
|
185
|
+
out[key] = _deep_merge(out[key], value)
|
|
186
|
+
else:
|
|
187
|
+
out[key] = value
|
|
188
|
+
return out
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _build(cls: type, data: Any, path: str) -> Any:
|
|
192
|
+
"""Instantiate a dataclass from parsed data, rejecting unknown keys.
|
|
193
|
+
|
|
194
|
+
Args:
|
|
195
|
+
cls: Target dataclass.
|
|
196
|
+
data: Parsed mapping, or a value for a non-dataclass field.
|
|
197
|
+
path: Dotted key path, used in error messages.
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
An instance of ``cls``.
|
|
201
|
+
|
|
202
|
+
Raises:
|
|
203
|
+
ConfigError: If ``data`` is not a mapping where one is required, or carries a key the
|
|
204
|
+
dataclass does not define. The message names the full path and the valid keys,
|
|
205
|
+
because a configuration error should be fixable without reading this module.
|
|
206
|
+
"""
|
|
207
|
+
if not is_dataclass(cls):
|
|
208
|
+
return data
|
|
209
|
+
if not isinstance(data, dict):
|
|
210
|
+
raise ConfigError(
|
|
211
|
+
f"{path or 'config'}: expected a mapping, got {type(data).__name__}"
|
|
212
|
+
)
|
|
213
|
+
valid = {f.name: f for f in fields(cls)}
|
|
214
|
+
unknown = sorted(set(data) - set(valid))
|
|
215
|
+
if unknown:
|
|
216
|
+
where = f"{path}." if path else ""
|
|
217
|
+
raise ConfigError(
|
|
218
|
+
f"unknown config key(s) {', '.join(where + u for u in unknown)}; "
|
|
219
|
+
f"valid keys here are {', '.join(sorted(valid))}"
|
|
220
|
+
)
|
|
221
|
+
kwargs: dict[str, Any] = {}
|
|
222
|
+
for name in valid:
|
|
223
|
+
if name not in data:
|
|
224
|
+
continue
|
|
225
|
+
sub = f"{path}.{name}" if path else name
|
|
226
|
+
value = data[name]
|
|
227
|
+
if name == "hazard_arms":
|
|
228
|
+
if not isinstance(value, list):
|
|
229
|
+
raise ConfigError(f"{sub}: expected a list of arms")
|
|
230
|
+
kwargs[name] = tuple(_build(HazardArm, v, f"{sub}[]") for v in value)
|
|
231
|
+
elif isinstance(value, list):
|
|
232
|
+
kwargs[name] = tuple(value)
|
|
233
|
+
else:
|
|
234
|
+
kwargs[name] = value
|
|
235
|
+
return cls(**kwargs)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _validate(cfg: Config) -> None:
|
|
239
|
+
"""Reject configurations that are internally inconsistent.
|
|
240
|
+
|
|
241
|
+
Checked here rather than at use time, so a bad value fails before a run starts instead of
|
|
242
|
+
hours in. An overlapping time split is the important one: it would be reported as
|
|
243
|
+
out-of-time validation while not being it.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
cfg: The assembled configuration.
|
|
247
|
+
|
|
248
|
+
Raises:
|
|
249
|
+
ConfigError: Naming the offending key and why it is wrong.
|
|
250
|
+
"""
|
|
251
|
+
v = cfg.v4finbench
|
|
252
|
+
if v.test_from <= v.train_until:
|
|
253
|
+
raise ConfigError(
|
|
254
|
+
f"v4finbench.test_from ({v.test_from}) must exceed train_until ({v.train_until}); "
|
|
255
|
+
"an overlapping split is not out-of-time validation"
|
|
256
|
+
)
|
|
257
|
+
if not 0.0 < cfg.evaluation.alpha < 1.0:
|
|
258
|
+
raise ConfigError(f"evaluation.alpha must lie in (0, 1), got {cfg.evaluation.alpha}")
|
|
259
|
+
if cfg.evaluation.bootstrap_resamples < 1:
|
|
260
|
+
raise ConfigError("evaluation.bootstrap_resamples must be at least 1")
|
|
261
|
+
if cfg.evaluation.min_rows_per_horizon < 1:
|
|
262
|
+
raise ConfigError("evaluation.min_rows_per_horizon must be at least 1")
|
|
263
|
+
if not 0.0 < cfg.inference.prototype_minority_ratio <= 1.0:
|
|
264
|
+
raise ConfigError(
|
|
265
|
+
"inference.prototype_minority_ratio must lie in (0, 1], got "
|
|
266
|
+
f"{cfg.inference.prototype_minority_ratio}"
|
|
267
|
+
)
|
|
268
|
+
if not 0.0 <= cfg.inference.winsor_quantile < 0.5:
|
|
269
|
+
raise ConfigError("inference.winsor_quantile must lie in [0, 0.5)")
|
|
270
|
+
if any(n < 1 for n in cfg.context_sweep.context_sizes):
|
|
271
|
+
raise ConfigError("context_sweep.context_sizes must all be at least 1")
|
|
272
|
+
if not cfg.context_sweep.seeds:
|
|
273
|
+
raise ConfigError("context_sweep.seeds must not be empty")
|
|
274
|
+
p = cfg.prior
|
|
275
|
+
if not 0.0 < p.absolute_rate_floor < p.rate_ceiling <= 1.0:
|
|
276
|
+
raise ConfigError(
|
|
277
|
+
f"prior rates must satisfy 0 < absolute_rate_floor ({p.absolute_rate_floor}) < "
|
|
278
|
+
f"rate_ceiling ({p.rate_ceiling}) <= 1"
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
#: Section name to dataclass. Declared explicitly rather than read off ``Config``'s
|
|
283
|
+
#: annotations, because ``from __future__ import annotations`` turns ``field.type`` into a
|
|
284
|
+
#: string and resolving it back would be indirection for its own sake.
|
|
285
|
+
_SECTIONS: dict[str, type] = {
|
|
286
|
+
"inference": InferenceConfig,
|
|
287
|
+
"evaluation": EvaluationConfig,
|
|
288
|
+
"v4finbench": V4FinBenchConfig,
|
|
289
|
+
"context_sweep": ContextSweepConfig,
|
|
290
|
+
"prior": PriorConfigDefaults,
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def load_config(path: str | os.PathLike[str] | None = None) -> Config:
|
|
295
|
+
"""Load configuration, layering an override over the packaged default.
|
|
296
|
+
|
|
297
|
+
Args:
|
|
298
|
+
path: Override file. Falls back to ``$FINTFM_CONFIG``, then to no override at all.
|
|
299
|
+
|
|
300
|
+
Returns:
|
|
301
|
+
A validated :class:`Config` recording the layers it came from.
|
|
302
|
+
|
|
303
|
+
Raises:
|
|
304
|
+
FileNotFoundError: If a path was given explicitly and does not exist. A *missing*
|
|
305
|
+
``FINTFM_CONFIG`` is also an error rather than a silent fallback, because the
|
|
306
|
+
whole point of setting it is that it takes effect.
|
|
307
|
+
ConfigError: If a key is unknown, mistyped, or fails validation.
|
|
308
|
+
"""
|
|
309
|
+
data = _read_yaml(DEFAULT_CONFIG_PATH)
|
|
310
|
+
sources = [str(DEFAULT_CONFIG_PATH)]
|
|
311
|
+
override = path if path is not None else os.environ.get(CONFIG_ENV_VAR)
|
|
312
|
+
if override:
|
|
313
|
+
over_path = Path(override)
|
|
314
|
+
data = _deep_merge(data, _read_yaml(over_path))
|
|
315
|
+
sources.append(str(over_path))
|
|
316
|
+
if "sources" in data:
|
|
317
|
+
raise ConfigError("config key 'sources' is set by the loader and cannot be provided")
|
|
318
|
+
unknown = sorted(set(data) - set(_SECTIONS))
|
|
319
|
+
if unknown:
|
|
320
|
+
raise ConfigError(
|
|
321
|
+
f"unknown top-level config section(s) {', '.join(unknown)}; valid sections are "
|
|
322
|
+
f"{', '.join(sorted(_SECTIONS))}"
|
|
323
|
+
)
|
|
324
|
+
cfg = Config(
|
|
325
|
+
**{name: _build(cls, data[name], name) for name, cls in _SECTIONS.items() if name in data},
|
|
326
|
+
sources=tuple(sources),
|
|
327
|
+
)
|
|
328
|
+
_validate(cfg)
|
|
329
|
+
return cfg
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# fintfm default configuration.
|
|
2
|
+
#
|
|
3
|
+
# This file is the single declared place for the numbers that experiments and command-line
|
|
4
|
+
# entry points use. Override it with `--config path/to/file.yaml` on any experiment CLI, or
|
|
5
|
+
# by setting FINTFM_CONFIG; an override is deep-merged over these values, so a file need only
|
|
6
|
+
# carry what it changes. Individual CLI flags override both.
|
|
7
|
+
#
|
|
8
|
+
# WHAT DELIBERATELY DOES NOT LIVE HERE
|
|
9
|
+
# ------------------------------------
|
|
10
|
+
# Not every literal in the codebase is configuration, and treating them all as such makes a
|
|
11
|
+
# system less robust rather than more. These stay in code on purpose:
|
|
12
|
+
#
|
|
13
|
+
# * hazard.CENSORED (-1) and the observation-mask semantics -- part of the data contract.
|
|
14
|
+
# A configurable sentinel is a way to silently reinterpret every stored period label.
|
|
15
|
+
# * retrieval._SCALE_FLOOR, HazardHead.max_hazard, the log1p/clamp epsilons -- numerical
|
|
16
|
+
# guards whose values are tied to float32 behaviour, not to a modelling choice.
|
|
17
|
+
# * the accounting identities in prior/financial.py -- these are arithmetic, not settings.
|
|
18
|
+
# * dataset URLs and licence attributions -- provenance, and changing one silently would
|
|
19
|
+
# break the licence record in README.md.
|
|
20
|
+
#
|
|
21
|
+
# The rule: a value belongs here if a reasonable experiment would want it different. A value
|
|
22
|
+
# whose change would make results incomparable or the code incorrect stays in code.
|
|
23
|
+
|
|
24
|
+
# Library defaults for FinancialTFMClassifier. These MIRROR the literal defaults in
|
|
25
|
+
# inference/classifier.py rather than replacing them: a library should behave identically
|
|
26
|
+
# without reading a file from disk, so the code keeps its own defaults and
|
|
27
|
+
# tests/test_config.py asserts the two agree. That test is the drift guard -- if you change
|
|
28
|
+
# one, it fails until you change the other.
|
|
29
|
+
inference:
|
|
30
|
+
max_context: 2000
|
|
31
|
+
context_strategy: uniform # docs/design/DECISIONS.md D10; was "balanced" before 2026-09-09
|
|
32
|
+
feature_transform: rank # docs/results/FINDINGS.md §35
|
|
33
|
+
n_ensemble: 1 # D12's stated mitigation for the column-order invariance
|
|
34
|
+
correct_prior: true
|
|
35
|
+
query_chunk: 2048
|
|
36
|
+
feature_chunk: 16 # identity-preserving memory knob; docs/results/FINDINGS.md §81
|
|
37
|
+
retrieval_groups: 64
|
|
38
|
+
retrieval_min_positive: 8
|
|
39
|
+
prototype_minority_ratio: 0.3 # Kostrzewa et al.'s value; docs/results/FINDINGS.md §38
|
|
40
|
+
winsor_quantile: 0.01
|
|
41
|
+
transform_subsample: 20000
|
|
42
|
+
|
|
43
|
+
# Thresholds used when scoring. Changing these changes what counts as a result, so they are
|
|
44
|
+
# recorded in every run's output alongside the numbers they produced.
|
|
45
|
+
evaluation:
|
|
46
|
+
bootstrap_resamples: 2000
|
|
47
|
+
alpha: 0.05
|
|
48
|
+
ece_bins: 10
|
|
49
|
+
# a horizon with fewer observed rows than this, or only one class, is reported as NaN
|
|
50
|
+
# rather than scored -- see docs/results/FINDINGS.md on degenerate splits poisoning aggregates
|
|
51
|
+
min_rows_per_horizon: 100
|
|
52
|
+
|
|
53
|
+
# The V4FinBench out-of-time protocol. The split years are the load-bearing numbers here:
|
|
54
|
+
# train and test must not overlap, and time_split() refuses if they do.
|
|
55
|
+
v4finbench:
|
|
56
|
+
max_rows: 120000
|
|
57
|
+
train_until: 2016
|
|
58
|
+
test_from: 2017
|
|
59
|
+
max_context: 2000
|
|
60
|
+
# scored side by side; the uncorrected arm is permanent by decision D8, so that the
|
|
61
|
+
# base-rate distortion of docs/results/FINDINGS.md §28 is measured rather than assumed absent
|
|
62
|
+
hazard_arms:
|
|
63
|
+
- {name: fintfm_hazard, strategy: prototype, correct_prior: true} # best, §38
|
|
64
|
+
- {name: fintfm_hazard_uncorrected, strategy: balanced, correct_prior: false}
|
|
65
|
+
- {name: fintfm_hazard_uniform, strategy: uniform, correct_prior: true}
|
|
66
|
+
|
|
67
|
+
# The context-strategy sweep behind docs/results/FINDINGS.md §29, §32 and §33.
|
|
68
|
+
context_sweep:
|
|
69
|
+
strategies: [balanced, hybrid, uniform, retrieval]
|
|
70
|
+
# 2000 is the operating point: on three seeds 4000 ties with it for 1.7x the time (§33)
|
|
71
|
+
context_sizes: [1000, 2000, 4000]
|
|
72
|
+
retrieval_groups: 64
|
|
73
|
+
seeds: [0]
|
|
74
|
+
|
|
75
|
+
# The financial prior's default-rate envelope. The floor is what lets the prior reach the
|
|
76
|
+
# low-default regime the strategy targets: at two expected defaults a 512-row task reaches
|
|
77
|
+
# 0.195%, which covers V4FinBench's 0.19% (docs/infra/COMPUTE.md).
|
|
78
|
+
prior:
|
|
79
|
+
# The financial prior's signal-to-noise range, drawn log-uniformly per task. Spans
|
|
80
|
+
# near-pure-noise to near-deterministic (docs/results/FINDINGS.md §42), and §64 measured this as
|
|
81
|
+
# the variable that predicts what a prior teaches.
|
|
82
|
+
sharpness_min: 0.3
|
|
83
|
+
sharpness_max: 12.0
|
|
84
|
+
min_expected_positives: 2.0
|
|
85
|
+
absolute_rate_floor: 0.001
|
|
86
|
+
rate_ceiling: 0.30
|
|
87
|
+
n_sectors: 12
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Evaluation: real datasets, metrics that see calibration, and benchmark harnesses.
|
|
2
|
+
|
|
3
|
+
Datasets loaded here are for **evaluation only**. Nothing real may reach pretraining — that
|
|
4
|
+
invariant is what makes a benchmark number auditable (``docs/results/FINDINGS.md`` §1).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from fintfm.evaluation.datasets import (
|
|
8
|
+
CreditDataset,
|
|
9
|
+
SurvivalDataset,
|
|
10
|
+
load_polish_bankruptcy,
|
|
11
|
+
load_taiwan_bankruptcy,
|
|
12
|
+
load_v4finbench,
|
|
13
|
+
)
|
|
14
|
+
from fintfm.evaluation.metrics import CreditMetrics, evaluate_binary
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"CreditDataset",
|
|
18
|
+
"CreditMetrics",
|
|
19
|
+
"SurvivalDataset",
|
|
20
|
+
"evaluate_binary",
|
|
21
|
+
"load_polish_bankruptcy",
|
|
22
|
+
"load_taiwan_bankruptcy",
|
|
23
|
+
"load_v4finbench",
|
|
24
|
+
]
|