bspe 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bspe/__init__.py +212 -0
- bspe/_vmm_serialization_common.py +176 -0
- bspe/_vmm_serialization_csv.py +222 -0
- bspe/analysis.py +130 -0
- bspe/batch_parsing.py +278 -0
- bspe/constants.py +9 -0
- bspe/domain.py +209 -0
- bspe/errors.py +292 -0
- bspe/filtering.py +73 -0
- bspe/information.py +37 -0
- bspe/markov_batch_serialization.py +143 -0
- bspe/markov_csv.py +52 -0
- bspe/markov_information.py +73 -0
- bspe/markov_serialization.py +239 -0
- bspe/markov_types.py +146 -0
- bspe/methods/__init__.py +23 -0
- bspe/methods/hmm.py +41 -0
- bspe/methods/markov.py +254 -0
- bspe/methods/shannon.py +132 -0
- bspe/methods/vmm.py +253 -0
- bspe/parsing.py +88 -0
- bspe/presentation.py +108 -0
- bspe/py.typed +0 -0
- bspe/records.py +68 -0
- bspe/schemas.py +73 -0
- bspe/serialization.py +189 -0
- bspe/stimulus_analysis.py +51 -0
- bspe/stimulus_generation.py +31 -0
- bspe/stimulus_matching.py +140 -0
- bspe/stimulus_metadata.py +44 -0
- bspe/stimulus_metrics.py +39 -0
- bspe/stimulus_search.py +211 -0
- bspe/stimulus_search_csv.py +211 -0
- bspe/stimulus_search_json.py +245 -0
- bspe/stimulus_search_results.py +189 -0
- bspe/stimulus_search_types.py +185 -0
- bspe/stimulus_targets.py +277 -0
- bspe/ui/__init__.py +9 -0
- bspe/ui/chart.py +229 -0
- bspe/ui/comparison.py +159 -0
- bspe/ui/form.py +38 -0
- bspe/ui/help_text.py +319 -0
- bspe/ui/hmm_session.py +94 -0
- bspe/ui/inputs.py +230 -0
- bspe/ui/markov_model_view.py +237 -0
- bspe/ui/markov_results.py +150 -0
- bspe/ui/markov_state.py +82 -0
- bspe/ui/markov_view.py +37 -0
- bspe/ui/method_controls.py +220 -0
- bspe/ui/model_inputs.py +106 -0
- bspe/ui/preset_inputs.py +101 -0
- bspe/ui/results.py +132 -0
- bspe/ui/results_view.py +241 -0
- bspe/ui/session.py +160 -0
- bspe/ui/setup.py +51 -0
- bspe/ui/shannon_results.py +151 -0
- bspe/ui/sidebar.py +105 -0
- bspe/ui/state.py +234 -0
- bspe/ui/stimulus_constraint_controls.py +186 -0
- bspe/ui/stimulus_controls.py +117 -0
- bspe/ui/stimulus_derived.py +284 -0
- bspe/ui/stimulus_metadata_controls.py +29 -0
- bspe/ui/stimulus_preference_controls.py +83 -0
- bspe/ui/stimulus_results.py +212 -0
- bspe/ui/stimulus_session.py +208 -0
- bspe/ui/stimulus_view.py +258 -0
- bspe/ui/stimulus_vmm_controls.py +84 -0
- bspe/ui/summary.py +273 -0
- bspe/ui/text.py +6 -0
- bspe/ui/tokens.py +30 -0
- bspe/ui/vmm_artifacts.py +187 -0
- bspe/ui/vmm_chart.py +139 -0
- bspe/ui/vmm_evidence.py +111 -0
- bspe/ui/vmm_results.py +270 -0
- bspe/ui/vmm_view.py +296 -0
- bspe/ui/workbench_state.py +281 -0
- bspe/ui/workspace_agreement.py +163 -0
- bspe/ui/workspace_charts.py +154 -0
- bspe/ui/workspace_components.py +102 -0
- bspe/ui/workspace_current.py +97 -0
- bspe/ui/workspace_exports.py +196 -0
- bspe/ui/workspace_method_common.py +51 -0
- bspe/ui/workspace_method_context.py +20 -0
- bspe/ui/workspace_method_renderers.py +197 -0
- bspe/ui/workspace_metrics.py +295 -0
- bspe/ui/workspace_mode.py +98 -0
- bspe/ui/workspace_renderers.py +157 -0
- bspe/ui/workspace_secondary_renderers.py +115 -0
- bspe/ui/workspace_selection.py +108 -0
- bspe/ui/workspace_session.py +86 -0
- bspe/vmm_serialization.py +266 -0
- bspe/vmm_types.py +156 -0
- bspe/workbench.py +153 -0
- bspe-0.1.0.dist-info/METADATA +570 -0
- bspe-0.1.0.dist-info/RECORD +97 -0
- bspe-0.1.0.dist-info/WHEEL +4 -0
- bspe-0.1.0.dist-info/licenses/LICENSE +21 -0
bspe/__init__.py
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Reusable core for binary hidden-Markov predictive entropy analysis."""
|
|
2
|
+
|
|
3
|
+
from bspe.analysis import analyze_sequence
|
|
4
|
+
from bspe.batch_parsing import (
|
|
5
|
+
CsvBatchColumns,
|
|
6
|
+
parse_csv_batch,
|
|
7
|
+
parse_manual_batch,
|
|
8
|
+
parse_txt_batch,
|
|
9
|
+
)
|
|
10
|
+
from bspe.domain import BinaryHMM, BinaryLabels, SequenceAnalysis
|
|
11
|
+
from bspe.markov_batch_serialization import markov_batch_summary_csv
|
|
12
|
+
from bspe.markov_serialization import (
|
|
13
|
+
markov_model_json,
|
|
14
|
+
markov_sequence_csv,
|
|
15
|
+
)
|
|
16
|
+
from bspe.markov_types import (
|
|
17
|
+
MarkovBatchAnalysis,
|
|
18
|
+
MarkovEstimation,
|
|
19
|
+
MarkovModel,
|
|
20
|
+
MarkovPredictionMode,
|
|
21
|
+
MarkovResultScope,
|
|
22
|
+
)
|
|
23
|
+
from bspe.methods.hmm import HMMBatchAnalysis, analyze_hmm
|
|
24
|
+
from bspe.methods.markov import (
|
|
25
|
+
analyze_markov,
|
|
26
|
+
analyze_markov_per_sequence,
|
|
27
|
+
fit_markov,
|
|
28
|
+
predict_markov,
|
|
29
|
+
)
|
|
30
|
+
from bspe.methods.shannon import (
|
|
31
|
+
ShannonBatchAnalysis,
|
|
32
|
+
ShannonPrefixResult,
|
|
33
|
+
analyze_shannon,
|
|
34
|
+
)
|
|
35
|
+
from bspe.methods.vmm import analyze_vmm, analyze_vmm_per_sequence, fit_vmm
|
|
36
|
+
from bspe.parsing import parse_sequence
|
|
37
|
+
from bspe.records import (
|
|
38
|
+
BinarySequence,
|
|
39
|
+
SequenceDataset,
|
|
40
|
+
SequenceId,
|
|
41
|
+
SequenceRecord,
|
|
42
|
+
)
|
|
43
|
+
from bspe.stimulus_generation import generate_candidate_records
|
|
44
|
+
from bspe.stimulus_matching import (
|
|
45
|
+
create_complement_candidates,
|
|
46
|
+
match_stimuli,
|
|
47
|
+
)
|
|
48
|
+
from bspe.stimulus_metrics import sequence_metrics
|
|
49
|
+
from bspe.stimulus_search import evaluate_candidate, search_stimuli
|
|
50
|
+
from bspe.stimulus_search_csv import (
|
|
51
|
+
stimulus_candidate_csv,
|
|
52
|
+
stimulus_experiment_csv,
|
|
53
|
+
stimulus_scientific_csv,
|
|
54
|
+
)
|
|
55
|
+
from bspe.stimulus_search_json import stimulus_generator_config_json
|
|
56
|
+
from bspe.stimulus_search_results import (
|
|
57
|
+
ConstraintViolation,
|
|
58
|
+
MatchingResult,
|
|
59
|
+
MatchTolerances,
|
|
60
|
+
MetricSummary,
|
|
61
|
+
SearchPartialReason,
|
|
62
|
+
SearchResult,
|
|
63
|
+
SearchStatus,
|
|
64
|
+
SequenceMetrics,
|
|
65
|
+
StimulusCandidate,
|
|
66
|
+
StimulusPair,
|
|
67
|
+
TargetCongruency,
|
|
68
|
+
ValidationReport,
|
|
69
|
+
)
|
|
70
|
+
from bspe.stimulus_search_types import (
|
|
71
|
+
InclusiveRange,
|
|
72
|
+
InvalidStimulusSearchConfigurationError,
|
|
73
|
+
PredictedSymbol,
|
|
74
|
+
PreferenceMetric,
|
|
75
|
+
SoftPreference,
|
|
76
|
+
StimulusConstraints,
|
|
77
|
+
StimulusId,
|
|
78
|
+
StimulusSearchConfig,
|
|
79
|
+
)
|
|
80
|
+
from bspe.stimulus_targets import assign_targets, validate_stimuli
|
|
81
|
+
from bspe.vmm_serialization import (
|
|
82
|
+
vmm_context_evidence_csv,
|
|
83
|
+
vmm_context_model_json,
|
|
84
|
+
vmm_evaluation_csv,
|
|
85
|
+
)
|
|
86
|
+
from bspe.vmm_types import (
|
|
87
|
+
AdditiveSmoothing,
|
|
88
|
+
InvalidVMMConfigurationError,
|
|
89
|
+
KTSmoothing,
|
|
90
|
+
MLESmoothing,
|
|
91
|
+
VMMAnalysis,
|
|
92
|
+
VMMConfig,
|
|
93
|
+
VMMContextCount,
|
|
94
|
+
VMMDepthAnalysis,
|
|
95
|
+
VMMDepthStatus,
|
|
96
|
+
VMMModel,
|
|
97
|
+
VMMRecordAnalysis,
|
|
98
|
+
VMMResultScope,
|
|
99
|
+
VMMSmoothing,
|
|
100
|
+
)
|
|
101
|
+
from bspe.workbench import (
|
|
102
|
+
AnalysisMethod,
|
|
103
|
+
HMMAnalysisRequest,
|
|
104
|
+
MarkovAnalysisRequest,
|
|
105
|
+
MethodComparison,
|
|
106
|
+
ShannonAnalysisRequest,
|
|
107
|
+
VMMAnalysisRequest,
|
|
108
|
+
analyze_dataset,
|
|
109
|
+
compare_methods,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
__all__ = [
|
|
113
|
+
"AdditiveSmoothing",
|
|
114
|
+
"AnalysisMethod",
|
|
115
|
+
"BinaryHMM",
|
|
116
|
+
"BinaryLabels",
|
|
117
|
+
"BinarySequence",
|
|
118
|
+
"ConstraintViolation",
|
|
119
|
+
"CsvBatchColumns",
|
|
120
|
+
"HMMAnalysisRequest",
|
|
121
|
+
"HMMBatchAnalysis",
|
|
122
|
+
"InclusiveRange",
|
|
123
|
+
"InvalidStimulusSearchConfigurationError",
|
|
124
|
+
"InvalidVMMConfigurationError",
|
|
125
|
+
"KTSmoothing",
|
|
126
|
+
"MLESmoothing",
|
|
127
|
+
"MarkovAnalysisRequest",
|
|
128
|
+
"MarkovBatchAnalysis",
|
|
129
|
+
"MarkovEstimation",
|
|
130
|
+
"MarkovModel",
|
|
131
|
+
"MarkovPredictionMode",
|
|
132
|
+
"MarkovResultScope",
|
|
133
|
+
"MatchTolerances",
|
|
134
|
+
"MatchingResult",
|
|
135
|
+
"MethodComparison",
|
|
136
|
+
"MetricSummary",
|
|
137
|
+
"PredictedSymbol",
|
|
138
|
+
"PreferenceMetric",
|
|
139
|
+
"SearchPartialReason",
|
|
140
|
+
"SearchResult",
|
|
141
|
+
"SearchStatus",
|
|
142
|
+
"SequenceAnalysis",
|
|
143
|
+
"SequenceDataset",
|
|
144
|
+
"SequenceId",
|
|
145
|
+
"SequenceMetrics",
|
|
146
|
+
"SequenceRecord",
|
|
147
|
+
"ShannonAnalysisRequest",
|
|
148
|
+
"ShannonBatchAnalysis",
|
|
149
|
+
"ShannonPrefixResult",
|
|
150
|
+
"SoftPreference",
|
|
151
|
+
"StimulusCandidate",
|
|
152
|
+
"StimulusConstraints",
|
|
153
|
+
"StimulusId",
|
|
154
|
+
"StimulusPair",
|
|
155
|
+
"StimulusSearchConfig",
|
|
156
|
+
"TargetCongruency",
|
|
157
|
+
"VMMAnalysis",
|
|
158
|
+
"VMMAnalysisRequest",
|
|
159
|
+
"VMMConfig",
|
|
160
|
+
"VMMContextCount",
|
|
161
|
+
"VMMDepthAnalysis",
|
|
162
|
+
"VMMDepthStatus",
|
|
163
|
+
"VMMModel",
|
|
164
|
+
"VMMRecordAnalysis",
|
|
165
|
+
"VMMResultScope",
|
|
166
|
+
"VMMSmoothing",
|
|
167
|
+
"ValidationReport",
|
|
168
|
+
"analyze_dataset",
|
|
169
|
+
"analyze_hmm",
|
|
170
|
+
"analyze_markov",
|
|
171
|
+
"analyze_markov_per_sequence",
|
|
172
|
+
"analyze_sequence",
|
|
173
|
+
"analyze_shannon",
|
|
174
|
+
"analyze_vmm",
|
|
175
|
+
"analyze_vmm_per_sequence",
|
|
176
|
+
"assign_targets",
|
|
177
|
+
"assign_outcomes",
|
|
178
|
+
"TargetAssignmentMode",
|
|
179
|
+
"PresentationMetadata",
|
|
180
|
+
"apply_presentation_metadata",
|
|
181
|
+
"compare_methods",
|
|
182
|
+
"create_complement_candidates",
|
|
183
|
+
"evaluate_candidate",
|
|
184
|
+
"fit_markov",
|
|
185
|
+
"fit_vmm",
|
|
186
|
+
"generate_candidate_records",
|
|
187
|
+
"markov_batch_summary_csv",
|
|
188
|
+
"markov_model_json",
|
|
189
|
+
"markov_sequence_csv",
|
|
190
|
+
"match_stimuli",
|
|
191
|
+
"parse_csv_batch",
|
|
192
|
+
"parse_manual_batch",
|
|
193
|
+
"parse_sequence",
|
|
194
|
+
"parse_txt_batch",
|
|
195
|
+
"predict_markov",
|
|
196
|
+
"search_stimuli",
|
|
197
|
+
"sequence_metrics",
|
|
198
|
+
"stimulus_candidate_csv",
|
|
199
|
+
"stimulus_experiment_csv",
|
|
200
|
+
"stimulus_generator_config_json",
|
|
201
|
+
"stimulus_scientific_csv",
|
|
202
|
+
"validate_stimuli",
|
|
203
|
+
"vmm_context_evidence_csv",
|
|
204
|
+
"vmm_context_model_json",
|
|
205
|
+
"vmm_evaluation_csv",
|
|
206
|
+
]
|
|
207
|
+
|
|
208
|
+
from bspe.stimulus_metadata import (
|
|
209
|
+
PresentationMetadata,
|
|
210
|
+
apply_presentation_metadata,
|
|
211
|
+
)
|
|
212
|
+
from bspe.stimulus_targets import TargetAssignmentMode, assign_outcomes
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Shared provenance and status values for VMM raw exports."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Final
|
|
8
|
+
|
|
9
|
+
from bspe.records import BinarySequence, SequenceDataset, SequenceRecord
|
|
10
|
+
from bspe.vmm_types import (
|
|
11
|
+
AdditiveSmoothing,
|
|
12
|
+
KTSmoothing,
|
|
13
|
+
MLESmoothing,
|
|
14
|
+
VMMAnalysis,
|
|
15
|
+
VMMDepthAnalysis,
|
|
16
|
+
VMMDepthStatus,
|
|
17
|
+
VMMRecordAnalysis,
|
|
18
|
+
VMMSmoothing,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
EXPERIMENTAL_STATUS: Final = "experimental"
|
|
22
|
+
EXPERIMENTAL_NOTICE: Final = (
|
|
23
|
+
"Experimental raw VMM artifact; retain its schema and source details when reused."
|
|
24
|
+
)
|
|
25
|
+
MLE_UNAVAILABLE_REASON: Final = (
|
|
26
|
+
"MLE unavailable: unseen context has no occurrences in the training dataset."
|
|
27
|
+
)
|
|
28
|
+
CONFIGURED_DEPTH_SELECTION: Final = "deepest_supported_suffix"
|
|
29
|
+
DATASET_ROLE: Final = "training"
|
|
30
|
+
type JsonMetric = float | str | None
|
|
31
|
+
_ESTIMATION_RULES: Final = {
|
|
32
|
+
KTSmoothing: "krichevsky_trofimov",
|
|
33
|
+
MLESmoothing: "maximum_likelihood",
|
|
34
|
+
AdditiveSmoothing: "additive_smoothing",
|
|
35
|
+
}
|
|
36
|
+
_SUPPORT_STATUSES: Final = {
|
|
37
|
+
VMMDepthStatus.ACCEPTED: ("accepted", "not_sparse"),
|
|
38
|
+
VMMDepthStatus.LOW_SUPPORT: ("low_support", "sparse"),
|
|
39
|
+
VMMDepthStatus.UNAVAILABLE: ("unavailable", "unavailable"),
|
|
40
|
+
}
|
|
41
|
+
_UNAVAILABLE_REASONS: Final = {
|
|
42
|
+
KTSmoothing: "Unseen context has no occurrences in the training dataset.",
|
|
43
|
+
MLESmoothing: MLE_UNAVAILABLE_REASON,
|
|
44
|
+
AdditiveSmoothing: "Unseen context has no occurrences in the training dataset.",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True, slots=True)
|
|
49
|
+
class ExportContext:
|
|
50
|
+
"""Inputs and stable identity shared by all three artifacts."""
|
|
51
|
+
|
|
52
|
+
analysis: VMMAnalysis
|
|
53
|
+
dataset: SequenceDataset
|
|
54
|
+
training_identifier: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True, slots=True)
|
|
58
|
+
class RecordPair:
|
|
59
|
+
"""One analyzed record paired with its ordered source stimulus."""
|
|
60
|
+
|
|
61
|
+
analysis: VMMRecordAnalysis
|
|
62
|
+
source: SequenceRecord
|
|
63
|
+
source_order: int
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def export_context(analysis: VMMAnalysis, dataset: SequenceDataset) -> ExportContext:
|
|
67
|
+
"""Build shared export provenance from immutable analysis inputs."""
|
|
68
|
+
identity = (
|
|
69
|
+
("observable_labels", dataset.labels.observables),
|
|
70
|
+
(
|
|
71
|
+
"records",
|
|
72
|
+
tuple((record.sequence_id, record.sequence) for record in dataset.records),
|
|
73
|
+
),
|
|
74
|
+
)
|
|
75
|
+
canonical = json.dumps(
|
|
76
|
+
identity,
|
|
77
|
+
ensure_ascii=False,
|
|
78
|
+
allow_nan=False,
|
|
79
|
+
separators=(",", ":"),
|
|
80
|
+
).encode("utf-8", errors="strict")
|
|
81
|
+
digest = hashlib.sha256(canonical).hexdigest()
|
|
82
|
+
return ExportContext(analysis, dataset, f"sha256:{digest}")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def record_pairs(context: ExportContext) -> tuple[RecordPair, ...]:
|
|
86
|
+
"""Pair analysis records to source records without changing source order."""
|
|
87
|
+
return tuple(
|
|
88
|
+
RecordPair(analyzed, source, source_order)
|
|
89
|
+
for source_order, (analyzed, source) in enumerate(
|
|
90
|
+
zip(context.analysis.records, context.dataset.records, strict=True),
|
|
91
|
+
start=1,
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def estimation_rule(analysis: VMMAnalysis) -> str:
|
|
97
|
+
"""Name the configured estimator without inferring from alpha alone."""
|
|
98
|
+
return _ESTIMATION_RULES[type(analysis.config.smoothing)]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def labeled_sequence(
|
|
102
|
+
sequence: BinarySequence,
|
|
103
|
+
labels: tuple[str, str],
|
|
104
|
+
) -> tuple[str, ...]:
|
|
105
|
+
"""Map internal observable indices to retained source labels."""
|
|
106
|
+
return tuple(labels[index] for index in sequence)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def csv_data(text: str) -> str:
|
|
110
|
+
"""Neutralize spreadsheet formula prefixes in untrusted text cells."""
|
|
111
|
+
return f"'{text}" if text.lstrip().startswith(("=", "+", "-", "@")) else text
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def support_status(status: VMMDepthStatus) -> tuple[str, str]:
|
|
115
|
+
"""Return separate support and sparse statuses for one examined context."""
|
|
116
|
+
return _SUPPORT_STATUSES[status]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def context_reason(row: VMMDepthAnalysis, smoothing: VMMSmoothing) -> str | None:
|
|
120
|
+
"""State why an examined context was rejected when a reason applies."""
|
|
121
|
+
reasons = {
|
|
122
|
+
VMMDepthStatus.ACCEPTED: None,
|
|
123
|
+
VMMDepthStatus.LOW_SUPPORT: (
|
|
124
|
+
"Context support is below the configured minimum."
|
|
125
|
+
),
|
|
126
|
+
VMMDepthStatus.UNAVAILABLE: _UNAVAILABLE_REASONS[type(smoothing)],
|
|
127
|
+
}
|
|
128
|
+
return reasons[row.status]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def backoff_selection(record: VMMRecordAnalysis) -> str:
|
|
132
|
+
"""Name whether the requested full suffix was selected or backed off."""
|
|
133
|
+
if record.effective_context_depth is None:
|
|
134
|
+
return "unavailable"
|
|
135
|
+
if record.effective_context_depth == len(record.sequence):
|
|
136
|
+
return "requested_depth_selected"
|
|
137
|
+
return "backed_off_to_shorter_suffix"
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def backoff_reason(
|
|
141
|
+
record: VMMRecordAnalysis,
|
|
142
|
+
smoothing: VMMSmoothing,
|
|
143
|
+
) -> str | None:
|
|
144
|
+
"""Explain the deepest requested suffix outcome from retained evidence."""
|
|
145
|
+
if record.effective_context_depth == len(record.sequence):
|
|
146
|
+
return None
|
|
147
|
+
reason = context_reason(record.depth_rows[-1], smoothing)
|
|
148
|
+
return reason or "No context meets the configured minimum support."
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def selected_row(record: VMMRecordAnalysis) -> VMMDepthAnalysis | None:
|
|
152
|
+
"""Return the retained evidence row selected for final prediction."""
|
|
153
|
+
return next(
|
|
154
|
+
(
|
|
155
|
+
row
|
|
156
|
+
for row in record.depth_rows
|
|
157
|
+
if row.depth == record.effective_context_depth
|
|
158
|
+
),
|
|
159
|
+
None,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def evaluation_status(pair: RecordPair) -> str:
|
|
164
|
+
"""Label training-record target assessment without claiming held-out data."""
|
|
165
|
+
if pair.source.actual_target_index is None:
|
|
166
|
+
return "Not supplied"
|
|
167
|
+
if pair.analysis.target_assessment is None:
|
|
168
|
+
return "In-sample evaluation unavailable"
|
|
169
|
+
return "In-sample evaluation, not held out"
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def json_metric(value: float | None) -> JsonMetric:
|
|
173
|
+
"""Represent defined infinite surprisal without non-standard JSON numbers."""
|
|
174
|
+
if value is None or math.isfinite(value):
|
|
175
|
+
return value
|
|
176
|
+
return "infinity" if value > 0.0 else "-infinity"
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""High-precision CSV rows for experimental VMM raw artifacts."""
|
|
2
|
+
|
|
3
|
+
from typing import Final
|
|
4
|
+
|
|
5
|
+
from bspe._vmm_serialization_common import (
|
|
6
|
+
CONFIGURED_DEPTH_SELECTION,
|
|
7
|
+
DATASET_ROLE,
|
|
8
|
+
EXPERIMENTAL_NOTICE,
|
|
9
|
+
EXPERIMENTAL_STATUS,
|
|
10
|
+
ExportContext,
|
|
11
|
+
RecordPair,
|
|
12
|
+
backoff_reason,
|
|
13
|
+
backoff_selection,
|
|
14
|
+
context_reason,
|
|
15
|
+
csv_data,
|
|
16
|
+
estimation_rule,
|
|
17
|
+
evaluation_status,
|
|
18
|
+
labeled_sequence,
|
|
19
|
+
record_pairs,
|
|
20
|
+
selected_row,
|
|
21
|
+
support_status,
|
|
22
|
+
)
|
|
23
|
+
from bspe.information import surprisal
|
|
24
|
+
from bspe.markov_csv import CsvCell, markov_csv_text
|
|
25
|
+
from bspe.records import BinarySequence
|
|
26
|
+
from bspe.vmm_types import VMMDepthAnalysis
|
|
27
|
+
|
|
28
|
+
COLUMNS: Final = (
|
|
29
|
+
"artifact_name",
|
|
30
|
+
"experimental_status",
|
|
31
|
+
"experimental_notice",
|
|
32
|
+
"method",
|
|
33
|
+
"dataset_role",
|
|
34
|
+
"training_dataset_identifier",
|
|
35
|
+
"evaluation_dataset_identifier",
|
|
36
|
+
"record_identifier",
|
|
37
|
+
"source_order",
|
|
38
|
+
"observable_A_label",
|
|
39
|
+
"observable_B_label",
|
|
40
|
+
"sequence_stimulus",
|
|
41
|
+
"consumed_prefix_stimulus",
|
|
42
|
+
"sequence_length",
|
|
43
|
+
"consumed_prefix_depth",
|
|
44
|
+
"displayed_context",
|
|
45
|
+
"requested_depth",
|
|
46
|
+
"actual_depth",
|
|
47
|
+
"workflow",
|
|
48
|
+
"result_scope",
|
|
49
|
+
"configured_depth_selection",
|
|
50
|
+
"estimation_rule",
|
|
51
|
+
"smoothing_alpha",
|
|
52
|
+
"suffix_backoff_selection",
|
|
53
|
+
"suffix_backoff_reason",
|
|
54
|
+
"context_occurrence_count",
|
|
55
|
+
"next_A_count",
|
|
56
|
+
"next_B_count",
|
|
57
|
+
"support_rule",
|
|
58
|
+
"support_status",
|
|
59
|
+
"sparse_status",
|
|
60
|
+
"next_A_probability",
|
|
61
|
+
"next_B_probability",
|
|
62
|
+
"predictive_entropy_bits",
|
|
63
|
+
"observed_target",
|
|
64
|
+
"target_probability",
|
|
65
|
+
"target_surprisal_bits",
|
|
66
|
+
"target_classification",
|
|
67
|
+
"evaluation_status",
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def context_evidence_csv(context: ExportContext) -> str:
|
|
72
|
+
"""Serialize every retained depth row in source and ascending-depth order."""
|
|
73
|
+
rows = tuple(
|
|
74
|
+
_evidence_row(context, pair, row)
|
|
75
|
+
for pair in record_pairs(context)
|
|
76
|
+
for row in pair.analysis.depth_rows
|
|
77
|
+
)
|
|
78
|
+
return markov_csv_text(COLUMNS, rows)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def evaluation_csv(context: ExportContext) -> str:
|
|
82
|
+
"""Serialize one final evaluation row per ordered training record."""
|
|
83
|
+
rows = tuple(_evaluation_row(context, pair) for pair in record_pairs(context))
|
|
84
|
+
return markov_csv_text(COLUMNS, rows)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _evidence_row(
|
|
88
|
+
context: ExportContext,
|
|
89
|
+
pair: RecordPair,
|
|
90
|
+
row: VMMDepthAnalysis,
|
|
91
|
+
) -> tuple[CsvCell, ...]:
|
|
92
|
+
record = pair.analysis
|
|
93
|
+
labels = context.dataset.labels.observables
|
|
94
|
+
selected = row.depth == record.effective_context_depth
|
|
95
|
+
support, sparse = support_status(row.status)
|
|
96
|
+
target = record.target_assessment if selected else None
|
|
97
|
+
target_index = pair.source.actual_target_index
|
|
98
|
+
target_probability = (
|
|
99
|
+
None
|
|
100
|
+
if target_index is None
|
|
101
|
+
else (row.probability_a, row.probability_b)[target_index]
|
|
102
|
+
)
|
|
103
|
+
if target_index is None:
|
|
104
|
+
row_evaluation_status = "Not supplied"
|
|
105
|
+
elif target_probability is None:
|
|
106
|
+
row_evaluation_status = "In-sample evaluation unavailable"
|
|
107
|
+
else:
|
|
108
|
+
row_evaluation_status = "In-sample evaluation, not held out"
|
|
109
|
+
if record.effective_context_depth is None:
|
|
110
|
+
selection = "no_context_selected"
|
|
111
|
+
elif selected:
|
|
112
|
+
selection = "selected"
|
|
113
|
+
elif row.depth > record.effective_context_depth:
|
|
114
|
+
selection = "rejected_for_backoff"
|
|
115
|
+
else:
|
|
116
|
+
selection = "not_selected_shorter_context"
|
|
117
|
+
return (
|
|
118
|
+
*_provenance(context, pair, "Context evidence export"),
|
|
119
|
+
_context_text(row.matched_suffix, labels),
|
|
120
|
+
row.depth,
|
|
121
|
+
record.effective_context_depth,
|
|
122
|
+
"variable_order_markov",
|
|
123
|
+
context.analysis.result_scope.value,
|
|
124
|
+
CONFIGURED_DEPTH_SELECTION,
|
|
125
|
+
estimation_rule(context.analysis),
|
|
126
|
+
context.analysis.config.smoothing.alpha,
|
|
127
|
+
selection,
|
|
128
|
+
context_reason(row, context.analysis.config.smoothing),
|
|
129
|
+
row.support,
|
|
130
|
+
row.count_a,
|
|
131
|
+
row.count_b,
|
|
132
|
+
f"minimum_support={context.analysis.config.minimum_support}",
|
|
133
|
+
support,
|
|
134
|
+
sparse,
|
|
135
|
+
row.probability_a,
|
|
136
|
+
row.probability_b,
|
|
137
|
+
row.predictive_entropy_bits,
|
|
138
|
+
None if target_index is None else csv_data(labels[target_index]),
|
|
139
|
+
target_probability,
|
|
140
|
+
None if target_probability is None else surprisal(target_probability),
|
|
141
|
+
None if target is None else target.classification.value,
|
|
142
|
+
row_evaluation_status,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _evaluation_row(
|
|
147
|
+
context: ExportContext,
|
|
148
|
+
pair: RecordPair,
|
|
149
|
+
) -> tuple[CsvCell, ...]:
|
|
150
|
+
record = pair.analysis
|
|
151
|
+
labels = context.dataset.labels.observables
|
|
152
|
+
selected = selected_row(record)
|
|
153
|
+
support, sparse = (
|
|
154
|
+
support_status(selected.status)
|
|
155
|
+
if selected is not None
|
|
156
|
+
else ("unavailable", "unavailable")
|
|
157
|
+
)
|
|
158
|
+
target = record.target_assessment
|
|
159
|
+
target_index = pair.source.actual_target_index
|
|
160
|
+
return (
|
|
161
|
+
*_provenance(context, pair, "Evaluation export"),
|
|
162
|
+
(
|
|
163
|
+
None
|
|
164
|
+
if record.context_used is None
|
|
165
|
+
else _context_text(record.context_used, labels)
|
|
166
|
+
),
|
|
167
|
+
len(record.sequence),
|
|
168
|
+
record.effective_context_depth,
|
|
169
|
+
"variable_order_markov",
|
|
170
|
+
context.analysis.result_scope.value,
|
|
171
|
+
CONFIGURED_DEPTH_SELECTION,
|
|
172
|
+
estimation_rule(context.analysis),
|
|
173
|
+
context.analysis.config.smoothing.alpha,
|
|
174
|
+
backoff_selection(record),
|
|
175
|
+
backoff_reason(record, context.analysis.config.smoothing),
|
|
176
|
+
None if selected is None else selected.support,
|
|
177
|
+
None if selected is None else selected.count_a,
|
|
178
|
+
None if selected is None else selected.count_b,
|
|
179
|
+
f"minimum_support={context.analysis.config.minimum_support}",
|
|
180
|
+
support,
|
|
181
|
+
sparse,
|
|
182
|
+
record.probability_a,
|
|
183
|
+
record.probability_b,
|
|
184
|
+
record.predictive_entropy_bits,
|
|
185
|
+
None if target_index is None else csv_data(labels[target_index]),
|
|
186
|
+
None if target is None else target.probability,
|
|
187
|
+
None if target is None else target.surprisal_bits,
|
|
188
|
+
None if target is None else target.classification.value,
|
|
189
|
+
evaluation_status(pair),
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _provenance(
|
|
194
|
+
context: ExportContext,
|
|
195
|
+
pair: RecordPair,
|
|
196
|
+
artifact_name: str,
|
|
197
|
+
) -> tuple[CsvCell, ...]:
|
|
198
|
+
labels = context.dataset.labels.observables
|
|
199
|
+
stimulus = csv_data(",".join(labeled_sequence(pair.analysis.sequence, labels)))
|
|
200
|
+
return (
|
|
201
|
+
artifact_name,
|
|
202
|
+
EXPERIMENTAL_STATUS,
|
|
203
|
+
EXPERIMENTAL_NOTICE,
|
|
204
|
+
"vmm",
|
|
205
|
+
DATASET_ROLE,
|
|
206
|
+
context.training_identifier,
|
|
207
|
+
None,
|
|
208
|
+
csv_data(str(pair.source.sequence_id)),
|
|
209
|
+
pair.source_order,
|
|
210
|
+
csv_data(labels[0]),
|
|
211
|
+
csv_data(labels[1]),
|
|
212
|
+
stimulus,
|
|
213
|
+
stimulus,
|
|
214
|
+
len(pair.analysis.sequence),
|
|
215
|
+
len(pair.analysis.sequence),
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _context_text(context: BinarySequence, labels: tuple[str, str]) -> str:
|
|
220
|
+
if not context:
|
|
221
|
+
return "Order 0 (no suffix)"
|
|
222
|
+
return csv_data(",".join(labeled_sequence(context, labels)))
|