fde-framework 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fde/__init__.py +3 -0
- fde/architect.py +152 -0
- fde/cli.py +2107 -0
- fde/costing.py +292 -0
- fde/decide.py +297 -0
- fde/decompose.py +54 -0
- fde/deploy.py +371 -0
- fde/emit.py +1309 -0
- fde/evolution.py +265 -0
- fde/factlog.py +369 -0
- fde/framework/approaches/ansible-playbook.md +22 -0
- fde/framework/approaches/assisted-deterministic.md +37 -0
- fde/framework/approaches/audit-only.md +23 -0
- fde/framework/approaches/boundary-and-audit.md +18 -0
- fde/framework/approaches/cascade.md +32 -0
- fde/framework/approaches/classical-ml.md +21 -0
- fde/framework/approaches/compose.md +18 -0
- fde/framework/approaches/decision-log.md +21 -0
- fde/framework/approaches/deterministic-masking.md +18 -0
- fde/framework/approaches/deterministic.md +25 -0
- fde/framework/approaches/direct-call.md +10 -0
- fde/framework/approaches/episodic-store.md +17 -0
- fde/framework/approaches/explainability-record.md +14 -0
- fde/framework/approaches/field-match.md +15 -0
- fde/framework/approaches/finetune.md +29 -0
- fde/framework/approaches/fixed-sequence.md +19 -0
- fde/framework/approaches/gitops.md +18 -0
- fde/framework/approaches/governed-tools.md +18 -0
- fde/framework/approaches/graph-retrieval.md +22 -0
- fde/framework/approaches/judged.md +15 -0
- fde/framework/approaches/keyword-search.md +11 -0
- fde/framework/approaches/kubernetes-manifests.md +19 -0
- fde/framework/approaches/labelled-metrics.md +14 -0
- fde/framework/approaches/llm-extraction.md +26 -0
- fde/framework/approaches/llm-scrubbing.md +20 -0
- fde/framework/approaches/llm.md +23 -0
- fde/framework/approaches/local-embedding.md +28 -0
- fde/framework/approaches/managed-api.md +49 -0
- fde/framework/approaches/managed-embedding.md +22 -0
- fde/framework/approaches/manual-runbook.md +32 -0
- fde/framework/approaches/model-planner.md +19 -0
- fde/framework/approaches/ocr-pipeline.md +13 -0
- fde/framework/approaches/optimisation-reasoning.md +30 -0
- fde/framework/approaches/optimisation.md +19 -0
- fde/framework/approaches/passthrough.md +11 -0
- fde/framework/approaches/role-scoped-authority.md +19 -0
- fde/framework/approaches/segmentation.md +29 -0
- fde/framework/approaches/self-hosted.md +38 -0
- fde/framework/approaches/serverless-gpu.md +27 -0
- fde/framework/approaches/speech-transcription.md +21 -0
- fde/framework/approaches/structured-logs.md +11 -0
- fde/framework/approaches/systemd-unit.md +30 -0
- fde/framework/approaches/terraform-module.md +22 -0
- fde/framework/approaches/text-extraction.md +14 -0
- fde/framework/approaches/traced.md +24 -0
- fde/framework/approaches/vector-search.md +17 -0
- fde/framework/approaches/video-ingestion.md +16 -0
- fde/framework/approaches/windowed-ingestion.md +25 -0
- fde/framework/approaches/working-state.md +13 -0
- fde/framework/cases/churn-scoring.md +45 -0
- fde/framework/cases/route-planning.md +44 -0
- fde/framework/cases/structured-extraction.md +48 -0
- fde/framework/cases/studio-style.md +46 -0
- fde/framework/components/accountability.md +20 -0
- fde/framework/components/deployment.md +19 -0
- fde/framework/components/embedding.md +23 -0
- fde/framework/components/evaluation.md +13 -0
- fde/framework/components/governance.md +19 -0
- fde/framework/components/integration.md +13 -0
- fde/framework/components/memory.md +21 -0
- fde/framework/components/observability.md +13 -0
- fde/framework/components/perception.md +15 -0
- fde/framework/components/planning.md +15 -0
- fde/framework/components/provisioning.md +17 -0
- fde/framework/components/reasoning.md +20 -0
- fde/framework/components/redaction.md +18 -0
- fde/framework/components/representation.md +13 -0
- fde/framework/components/retrieval.md +13 -0
- fde/framework/components/serving.md +20 -0
- fde/framework/dimensions/accelerator.md +29 -0
- fde/framework/dimensions/access_model.md +25 -0
- fde/framework/dimensions/arrival_rate.md +19 -0
- fde/framework/dimensions/availability_target.md +23 -0
- fde/framework/dimensions/cheap_path_coverage.md +21 -0
- fde/framework/dimensions/confidence_calibrated.md +30 -0
- fde/framework/dimensions/container_competence.md +22 -0
- fde/framework/dimensions/corpus_size.md +13 -0
- fde/framework/dimensions/data_residency.md +36 -0
- fde/framework/dimensions/environment_lifetime.md +19 -0
- fde/framework/dimensions/existing_cluster.md +22 -0
- fde/framework/dimensions/existing_iac_tool.md +25 -0
- fde/framework/dimensions/external_systems.md +15 -0
- fde/framework/dimensions/hosting.md +45 -0
- fde/framework/dimensions/human_waiting.md +41 -0
- fde/framework/dimensions/input_format.md +29 -0
- fde/framework/dimensions/interpretability_required.md +24 -0
- fde/framework/dimensions/labelled_count.md +15 -0
- fde/framework/dimensions/latency_budget_ms.md +16 -0
- fde/framework/dimensions/licence_posture.md +28 -0
- fde/framework/dimensions/operates_after_handover.md +25 -0
- fde/framework/dimensions/output_shape.md +29 -0
- fde/framework/dimensions/provisioning_api.md +18 -0
- fde/framework/dimensions/query_pattern.md +25 -0
- fde/framework/dimensions/recall_span.md +22 -0
- fde/framework/dimensions/sensitivity_present.md +22 -0
- fde/framework/interfaces/Generator.md +5 -0
- fde/framework/interfaces/Guard.md +5 -0
- fde/framework/interfaces/Mapper.md +5 -0
- fde/framework/interfaces/ModelServer.md +5 -0
- fde/framework/interfaces/Parser.md +5 -0
- fde/framework/interfaces/Planner.md +5 -0
- fde/framework/interfaces/Retriever.md +5 -0
- fde/framework/interfaces/Scorer.md +5 -0
- fde/framework/interfaces/Store.md +5 -0
- fde/framework/interfaces/ToolBoundary.md +5 -0
- fde/framework/interfaces/Tracer.md +5 -0
- fde/framework/locales/eu-gdpr.md +51 -0
- fde/framework/locales/in-dpdp.md +46 -0
- fde/framework/patterns/ansible-playbook.md +9 -0
- fde/framework/patterns/assisted-deterministic.md +11 -0
- fde/framework/patterns/audit-only.md +9 -0
- fde/framework/patterns/boundary-and-audit.md +12 -0
- fde/framework/patterns/cascade-reasoning.md +10 -0
- fde/framework/patterns/cascade-representation.md +10 -0
- fde/framework/patterns/cascade-retrieval.md +10 -0
- fde/framework/patterns/classical-ml-reasoning.md +15 -0
- fde/framework/patterns/classical-ml.md +13 -0
- fde/framework/patterns/compose.md +9 -0
- fde/framework/patterns/decision-log.md +9 -0
- fde/framework/patterns/deterministic-masking.md +9 -0
- fde/framework/patterns/deterministic.md +12 -0
- fde/framework/patterns/direct-call.md +12 -0
- fde/framework/patterns/episodic-store.md +13 -0
- fde/framework/patterns/explainability-record.md +9 -0
- fde/framework/patterns/field-match.md +12 -0
- fde/framework/patterns/finetune-representation.md +14 -0
- fde/framework/patterns/finetune.md +12 -0
- fde/framework/patterns/fixed-sequence.md +12 -0
- fde/framework/patterns/gitops.md +9 -0
- fde/framework/patterns/governed-tools.md +13 -0
- fde/framework/patterns/graph-retrieval.md +14 -0
- fde/framework/patterns/judged.md +14 -0
- fde/framework/patterns/keyword-search.md +12 -0
- fde/framework/patterns/kubernetes-manifests.md +9 -0
- fde/framework/patterns/labelled-metrics.md +13 -0
- fde/framework/patterns/llm-representation.md +17 -0
- fde/framework/patterns/llm-scrubbing.md +9 -0
- fde/framework/patterns/llm.md +12 -0
- fde/framework/patterns/local-embedding.md +9 -0
- fde/framework/patterns/managed-api.md +12 -0
- fde/framework/patterns/managed-embedding.md +9 -0
- fde/framework/patterns/manual-runbook.md +9 -0
- fde/framework/patterns/model-planner.md +13 -0
- fde/framework/patterns/ocr-pipeline.md +13 -0
- fde/framework/patterns/optimisation-reasoning.md +15 -0
- fde/framework/patterns/optimisation.md +13 -0
- fde/framework/patterns/passthrough.md +12 -0
- fde/framework/patterns/role-scoped-authority.md +9 -0
- fde/framework/patterns/segmentation.md +12 -0
- fde/framework/patterns/self-hosted.md +14 -0
- fde/framework/patterns/serverless-gpu.md +12 -0
- fde/framework/patterns/speech-transcription.md +10 -0
- fde/framework/patterns/structured-logs.md +12 -0
- fde/framework/patterns/systemd-unit.md +9 -0
- fde/framework/patterns/terraform-module.md +9 -0
- fde/framework/patterns/text-extraction.md +12 -0
- fde/framework/patterns/traced.md +13 -0
- fde/framework/patterns/vector-search.md +14 -0
- fde/framework/patterns/video-ingestion.md +9 -0
- fde/framework/patterns/windowed-ingestion.md +12 -0
- fde/framework/patterns/working-state.md +12 -0
- fde/framework/stacks/langgraph.md +9 -0
- fde/framework/stacks/local-judge.md +9 -0
- fde/framework/stacks/mcp.md +9 -0
- fde/framework/stacks/ollama.md +19 -0
- fde/framework/stacks/openai-judge.md +9 -0
- fde/framework/stacks/opentelemetry.md +9 -0
- fde/framework/stacks/ortools.md +9 -0
- fde/framework/stacks/pgvector.md +9 -0
- fde/framework/stacks/plain-python.md +9 -0
- fde/framework/stacks/qdrant.md +9 -0
- fde/framework/stacks/tesseract.md +9 -0
- fde/framework/stacks/vllm.md +9 -0
- fde/framework/stacks/whisper.md +15 -0
- fde/framework/stacks/xgboost.md +9 -0
- fde/framework/templates/accountability/decision-log.plain.py.j2 +64 -0
- fde/framework/templates/accountability/explainability-record.plain.py.j2 +90 -0
- fde/framework/templates/deployment/compose.plain.py.j2 +36 -0
- fde/framework/templates/deployment/kubernetes-manifests.plain.py.j2 +36 -0
- fde/framework/templates/deployment/systemd-unit.plain.py.j2 +36 -0
- fde/framework/templates/embedding/local-embedding.plain.py.j2 +71 -0
- fde/framework/templates/embedding/managed-embedding.plain.py.j2 +76 -0
- fde/framework/templates/evaluation/field-match.plain.py.j2 +146 -0
- fde/framework/templates/evaluation/judged.local-judge.py.j2 +100 -0
- fde/framework/templates/evaluation/judged.openai-judge.py.j2 +70 -0
- fde/framework/templates/evaluation/judged.plain.py.j2 +89 -0
- fde/framework/templates/evaluation/labelled-metrics.plain.py.j2 +64 -0
- fde/framework/templates/evaluation/labelled-metrics.xgboost.py.j2 +72 -0
- fde/framework/templates/governance/audit-only.plain.py.j2 +98 -0
- fde/framework/templates/governance/boundary-and-audit.plain.py.j2 +142 -0
- fde/framework/templates/governance/role-scoped-authority.plain.py.j2 +63 -0
- fde/framework/templates/integration/direct-call.plain.py.j2 +63 -0
- fde/framework/templates/integration/governed-tools.mcp.py.j2 +124 -0
- fde/framework/templates/integration/governed-tools.plain.py.j2 +153 -0
- fde/framework/templates/memory/episodic-store.pgvector.py.j2 +118 -0
- fde/framework/templates/memory/episodic-store.plain.py.j2 +147 -0
- fde/framework/templates/memory/working-state.plain.py.j2 +48 -0
- fde/framework/templates/observability/structured-logs.plain.py.j2 +61 -0
- fde/framework/templates/observability/traced.opentelemetry.py.j2 +65 -0
- fde/framework/templates/observability/traced.plain.py.j2 +132 -0
- fde/framework/templates/perception/ocr-pipeline.plain.py.j2 +71 -0
- fde/framework/templates/perception/ocr-pipeline.tesseract.py.j2 +80 -0
- fde/framework/templates/perception/passthrough.plain.py.j2 +41 -0
- fde/framework/templates/perception/speech-transcription.plain.py.j2 +37 -0
- fde/framework/templates/perception/speech-transcription.whisper.py.j2 +39 -0
- fde/framework/templates/perception/text-extraction.plain.py.j2 +88 -0
- fde/framework/templates/perception/video-ingestion.plain.py.j2 +35 -0
- fde/framework/templates/perception/windowed-ingestion.plain.py.j2 +69 -0
- fde/framework/templates/planning/fixed-sequence.plain.py.j2 +46 -0
- fde/framework/templates/planning/model-planner.langgraph.py.j2 +89 -0
- fde/framework/templates/planning/model-planner.plain.py.j2 +82 -0
- fde/framework/templates/planning/optimisation.ortools.py.j2 +90 -0
- fde/framework/templates/planning/optimisation.plain.py.j2 +71 -0
- fde/framework/templates/provisioning/ansible-playbook.plain.py.j2 +29 -0
- fde/framework/templates/provisioning/gitops.plain.py.j2 +29 -0
- fde/framework/templates/provisioning/manual-runbook.plain.py.j2 +29 -0
- fde/framework/templates/provisioning/terraform-module.plain.py.j2 +29 -0
- fde/framework/templates/reasoning/cascade.plain.py.j2 +86 -0
- fde/framework/templates/reasoning/classical-ml.plain.py.j2 +74 -0
- fde/framework/templates/reasoning/classical-ml.xgboost.py.j2 +104 -0
- fde/framework/templates/reasoning/finetune.plain.py.j2 +73 -0
- fde/framework/templates/reasoning/llm.plain.py.j2 +114 -0
- fde/framework/templates/reasoning/optimisation.ortools.py.j2 +90 -0
- fde/framework/templates/reasoning/optimisation.plain.py.j2 +58 -0
- fde/framework/templates/redaction/deterministic-masking.plain.py.j2 +44 -0
- fde/framework/templates/redaction/llm-scrubbing.plain.py.j2 +40 -0
- fde/framework/templates/representation/assisted.plain.py.j2 +86 -0
- fde/framework/templates/representation/cascade.plain.py.j2 +119 -0
- fde/framework/templates/representation/classical-ml.plain.py.j2 +74 -0
- fde/framework/templates/representation/classical-ml.xgboost.py.j2 +104 -0
- fde/framework/templates/representation/deterministic.plain.py.j2 +101 -0
- fde/framework/templates/representation/finetune.plain.py.j2 +68 -0
- fde/framework/templates/representation/llm.plain.py.j2 +94 -0
- fde/framework/templates/representation/segmentation.plain.py.j2 +68 -0
- fde/framework/templates/retrieval/cascade.plain.py.j2 +86 -0
- fde/framework/templates/retrieval/graph-retrieval.pgvector.py.j2 +88 -0
- fde/framework/templates/retrieval/graph-retrieval.plain.py.j2 +74 -0
- fde/framework/templates/retrieval/graph-retrieval.qdrant.py.j2 +71 -0
- fde/framework/templates/retrieval/keyword-search.plain.py.j2 +104 -0
- fde/framework/templates/retrieval/vector-search.pgvector.py.j2 +100 -0
- fde/framework/templates/retrieval/vector-search.plain.py.j2 +84 -0
- fde/framework/templates/retrieval/vector-search.qdrant.py.j2 +60 -0
- fde/framework/templates/serving/managed-api.plain.py.j2 +74 -0
- fde/framework/templates/serving/self-hosted.ollama.py.j2 +47 -0
- fde/framework/templates/serving/self-hosted.plain.py.j2 +117 -0
- fde/framework/templates/serving/self-hosted.vllm.py.j2 +77 -0
- fde/framework/templates/serving/serverless-gpu.plain.py.j2 +62 -0
- fde/gates.py +480 -0
- fde/graph.py +435 -0
- fde/implement.py +250 -0
- fde/intake/__init__.py +0 -0
- fde/intake/answers.py +154 -0
- fde/intake/documents.py +121 -0
- fde/intake/interview.py +218 -0
- fde/intake/llm_reader.py +336 -0
- fde/intake/prose.py +387 -0
- fde/intake/samples.py +369 -0
- fde/models/__init__.py +0 -0
- fde/models/base.py +86 -0
- fde/models/fact.py +37 -0
- fde/models/profile.py +159 -0
- fde/models/respondent.py +34 -0
- fde/models/schema.py +430 -0
- fde/moves.py +136 -0
- fde/ops.py +349 -0
- fde/predicate.py +106 -0
- fde/realization.py +109 -0
- fde/registry.py +162 -0
- fde/scan.py +479 -0
- fde/space.py +172 -0
- fde/workflow.py +233 -0
- fde_framework-0.1.0.dist-info/METADATA +449 -0
- fde_framework-0.1.0.dist-info/RECORD +286 -0
- fde_framework-0.1.0.dist-info/WHEEL +4 -0
- fde_framework-0.1.0.dist-info/entry_points.txt +2 -0
- fde_framework-0.1.0.dist-info/licenses/LICENSE +202 -0
fde/costing.py
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""What it costs, with the date attached.
|
|
2
|
+
|
|
3
|
+
Every absolute here ages. Prices fall, hardware changes, and a figure quoted in
|
|
4
|
+
a proposal a year after it was written is wrong in a way nobody notices -- so
|
|
5
|
+
each carries the date it was true and the rule for working it out again.
|
|
6
|
+
|
|
7
|
+
The sizing is the part usually got wrong. Counting weights against average
|
|
8
|
+
throughput understates a fleet substantially, because redundancy, peak and
|
|
9
|
+
prefill overhead each multiply it, and the understatement is discovered in
|
|
10
|
+
production rather than in the spreadsheet.
|
|
11
|
+
|
|
12
|
+
Both conclusions live here on purpose. At sustained interactive volume,
|
|
13
|
+
self-hosting loses to a managed endpoint; on bursty batch work it wins by a wide
|
|
14
|
+
margin. A framework that only knew one of them would be wrong half the time, and
|
|
15
|
+
confidently.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import math
|
|
21
|
+
import warnings
|
|
22
|
+
from datetime import date
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from fde.scan import GPU, Hardware, fits
|
|
26
|
+
|
|
27
|
+
# Every figure below was true on this date and will not stay true.
|
|
28
|
+
AS_OF = "2026-08"
|
|
29
|
+
STALE_AFTER_DAYS = 270
|
|
30
|
+
|
|
31
|
+
# Indicative. Re-derive against current pricing rather than quoting these.
|
|
32
|
+
GPU_COST_PER_HOUR = 4.50
|
|
33
|
+
MANAGED_COST_PER_MILLION_TOKENS = 3.00
|
|
34
|
+
TOKENS_PER_REQUEST = 1_500
|
|
35
|
+
|
|
36
|
+
# The unit being rented. A replica is however many of these the model needs,
|
|
37
|
+
# which is the step usually skipped -- a 70B model in bf16 is 140GB of weights
|
|
38
|
+
# and does not run on one of them.
|
|
39
|
+
CARD_VRAM_GB = 80
|
|
40
|
+
|
|
41
|
+
# Traffic does not arrive evenly. Sizing to the daily mean leaves a fleet that
|
|
42
|
+
# fails at the busiest hour, which is the hour anybody notices.
|
|
43
|
+
PEAK_MULTIPLIER = 2.0
|
|
44
|
+
|
|
45
|
+
# One spare. A fleet with no redundancy is a fleet that is down during a deploy.
|
|
46
|
+
REDUNDANCY = 1.34
|
|
47
|
+
|
|
48
|
+
# Prefill is compute-bound and does not batch as well as decode, so a fleet
|
|
49
|
+
# sized on decode throughput alone is short.
|
|
50
|
+
PREFILL_OVERHEAD = 1.2
|
|
51
|
+
|
|
52
|
+
HOURS_PER_MONTH = 730
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def gpus_per_replica(params_b: float, precision: str = "bf16") -> int:
|
|
56
|
+
"""How many cards one copy of this model occupies.
|
|
57
|
+
|
|
58
|
+
Sizing in replicas and pricing each as one GPU is how a fleet is quoted at
|
|
59
|
+
a third of its cost. A replica is a number of cards, and the number comes
|
|
60
|
+
from the weights plus the cache rather than from the weights alone.
|
|
61
|
+
"""
|
|
62
|
+
one_card = Hardware(gpus=[GPU("card", vram_gb=CARD_VRAM_GB)])
|
|
63
|
+
fit = fits(one_card, params_b, precision=precision)
|
|
64
|
+
return max(1, math.ceil(fit.required_gb / fit.available_gb))
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def size_for(
|
|
68
|
+
requests_per_day: int,
|
|
69
|
+
params_b: float,
|
|
70
|
+
requests_per_second_per_replica: float = 2.0,
|
|
71
|
+
today: str | None = None,
|
|
72
|
+
) -> dict[str, Any]:
|
|
73
|
+
"""How many replicas this actually needs, and how many cards that is.
|
|
74
|
+
|
|
75
|
+
The naive figure is reported beside the real one, because the gap is the
|
|
76
|
+
finding -- and somebody will otherwise arrive at the naive figure
|
|
77
|
+
independently and wonder why the estimate is higher.
|
|
78
|
+
"""
|
|
79
|
+
_warn_if_stale(today)
|
|
80
|
+
|
|
81
|
+
mean_rps = requests_per_day / 86_400
|
|
82
|
+
naive = max(1, round(mean_rps / requests_per_second_per_replica))
|
|
83
|
+
real = max(
|
|
84
|
+
1,
|
|
85
|
+
round(
|
|
86
|
+
(mean_rps * PEAK_MULTIPLIER * PREFILL_OVERHEAD * REDUNDANCY)
|
|
87
|
+
/ requests_per_second_per_replica
|
|
88
|
+
),
|
|
89
|
+
)
|
|
90
|
+
per_replica = gpus_per_replica(params_b)
|
|
91
|
+
|
|
92
|
+
return {
|
|
93
|
+
"naive_replicas": naive,
|
|
94
|
+
"replicas": real,
|
|
95
|
+
"gpus_per_replica": per_replica,
|
|
96
|
+
"gpus": real * per_replica,
|
|
97
|
+
"factors": {
|
|
98
|
+
"peak": f"traffic is not flat; sized at {PEAK_MULTIPLIER}x the daily mean",
|
|
99
|
+
"prefill": f"prefill is compute-bound and batches worse than decode "
|
|
100
|
+
f"({PREFILL_OVERHEAD}x)",
|
|
101
|
+
"redundancy": "one spare, so a deploy is not an outage",
|
|
102
|
+
"model_size": f"a {params_b:g}B model at bf16 occupies {per_replica} "
|
|
103
|
+
f"card(s) per replica, weights and cache together",
|
|
104
|
+
},
|
|
105
|
+
"monthly_cost": round(
|
|
106
|
+
real * per_replica * GPU_COST_PER_HOUR * HOURS_PER_MONTH, 2
|
|
107
|
+
),
|
|
108
|
+
"as_of": AS_OF,
|
|
109
|
+
"rederive": (
|
|
110
|
+
"measure requests per second per replica on the real model and "
|
|
111
|
+
"hardware, then apply peak, prefill and redundancy to the measured "
|
|
112
|
+
"figure rather than to this one"
|
|
113
|
+
),
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def compare_hosting(
|
|
118
|
+
requests_per_day: int,
|
|
119
|
+
params_b: float,
|
|
120
|
+
human_waiting: bool = True,
|
|
121
|
+
today: str | None = None,
|
|
122
|
+
) -> dict[str, Any]:
|
|
123
|
+
"""Managed against self-hosted, for this workload.
|
|
124
|
+
|
|
125
|
+
The answer flips on utilisation and on whether anybody is waiting, which is
|
|
126
|
+
why the question cannot be settled once and quoted forever. Sustained
|
|
127
|
+
interactive traffic keeps a self-hosted fleet running around the clock;
|
|
128
|
+
bursty batch work pays for nothing between jobs.
|
|
129
|
+
"""
|
|
130
|
+
_warn_if_stale(today)
|
|
131
|
+
|
|
132
|
+
tokens_per_month = requests_per_day * 30 * TOKENS_PER_REQUEST
|
|
133
|
+
managed = tokens_per_month / 1e6 * MANAGED_COST_PER_MILLION_TOKENS
|
|
134
|
+
|
|
135
|
+
per_replica = gpus_per_replica(params_b)
|
|
136
|
+
|
|
137
|
+
if human_waiting:
|
|
138
|
+
# Somebody is waiting, so the fleet stays up whether or not it is busy,
|
|
139
|
+
# sized for the peak hour and carrying a spare.
|
|
140
|
+
plan = size_for(requests_per_day, params_b, today=today)
|
|
141
|
+
self_hosted = plan["monthly_cost"]
|
|
142
|
+
note = (
|
|
143
|
+
f"Sustained interactive traffic keeps this running around the clock "
|
|
144
|
+
f"at {plan['gpus']} cards ({per_replica} per replica), and redundancy "
|
|
145
|
+
f"and peak headroom are most of the bill."
|
|
146
|
+
)
|
|
147
|
+
else:
|
|
148
|
+
# Nobody waiting, so a cold start costs nothing and idle time is avoidable.
|
|
149
|
+
busy_hours = (requests_per_day * 30) / (2.0 * 3600)
|
|
150
|
+
self_hosted = busy_hours * per_replica * GPU_COST_PER_HOUR
|
|
151
|
+
plan = {"replicas": 1}
|
|
152
|
+
note = (
|
|
153
|
+
"Nobody is waiting, so this scales to zero between jobs and pays for "
|
|
154
|
+
"compute rather than for availability."
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
return {
|
|
158
|
+
"managed_monthly": round(managed, 2),
|
|
159
|
+
"self_hosted_monthly": round(self_hosted, 2),
|
|
160
|
+
"replicas": plan["replicas"],
|
|
161
|
+
"gpus_per_replica": per_replica,
|
|
162
|
+
"recommendation": "managed" if managed < self_hosted else "self-hosted",
|
|
163
|
+
"why": note,
|
|
164
|
+
"as_of": AS_OF,
|
|
165
|
+
"rederive": (
|
|
166
|
+
"check current per-token and per-hour pricing, and measure tokens "
|
|
167
|
+
"per request on real traffic rather than assuming"
|
|
168
|
+
),
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def effort_by_analogy(profile: dict[str, Any], cases: dict[str, Any]) -> dict[str, Any]:
|
|
173
|
+
"""How long this took the last time something like it was done.
|
|
174
|
+
|
|
175
|
+
An estimate from comparable work, with the comparables named so somebody
|
|
176
|
+
can disagree with the comparison rather than with the number.
|
|
177
|
+
"""
|
|
178
|
+
analogues = [
|
|
179
|
+
case_id for case_id, case in cases.items()
|
|
180
|
+
if _overlap(profile, getattr(case, "profile", {})) >= 2
|
|
181
|
+
]
|
|
182
|
+
if not analogues:
|
|
183
|
+
return {
|
|
184
|
+
"analogues": [],
|
|
185
|
+
"range_weeks": None,
|
|
186
|
+
"why": "nothing in the corpus resembles this closely enough to "
|
|
187
|
+
"estimate from. An estimate without a comparable is a guess "
|
|
188
|
+
"with a number on it.",
|
|
189
|
+
}
|
|
190
|
+
return {
|
|
191
|
+
"analogues": sorted(analogues),
|
|
192
|
+
"range_weeks": (3, 8),
|
|
193
|
+
"why": f"estimated from {len(analogues)} comparable engagement(s); "
|
|
194
|
+
f"disagree with the comparison rather than with the number",
|
|
195
|
+
"as_of": AS_OF,
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _overlap(profile: dict[str, Any], other: dict[str, Any]) -> int:
|
|
200
|
+
return sum(1 for k, v in profile.items() if other.get(k) == v)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _warn_if_stale(today: str | None) -> None:
|
|
204
|
+
"""Say so when these figures have aged past usefulness."""
|
|
205
|
+
if not today:
|
|
206
|
+
return
|
|
207
|
+
age = (date.fromisoformat(today) - date.fromisoformat(f"{AS_OF}-01")).days
|
|
208
|
+
if age > STALE_AFTER_DAYS:
|
|
209
|
+
warnings.warn(
|
|
210
|
+
f"these figures are as of {AS_OF}, roughly {age} days ago. Pricing and "
|
|
211
|
+
f"hardware have moved; re-derive before quoting them.",
|
|
212
|
+
UserWarning,
|
|
213
|
+
stacklevel=3,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# --- unit economics: the arbitrage-trap check -------------------------------
|
|
218
|
+
|
|
219
|
+
WORKDAYS_PER_MONTH = 22
|
|
220
|
+
|
|
221
|
+
# Reused prompt prefixes bill at roughly a tenth of the input rate on the
|
|
222
|
+
# hosted APIs that support prompt caching. Dated like every figure here.
|
|
223
|
+
CACHED_PREFIX_DISCOUNT = 0.10
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def unit_economics(
|
|
227
|
+
workflows_per_day: float,
|
|
228
|
+
price_per_seat_month: float,
|
|
229
|
+
steps_per_workflow: int = 5,
|
|
230
|
+
tokens_per_step: int = TOKENS_PER_REQUEST,
|
|
231
|
+
cheap_path_coverage: float | None = None,
|
|
232
|
+
cached_prefix_share: float = 0.0,
|
|
233
|
+
today: str | None = None,
|
|
234
|
+
) -> dict:
|
|
235
|
+
"""Whether a seat earns more than it burns, and which lever moves it.
|
|
236
|
+
|
|
237
|
+
The failure this exists to name: a per-seat price set before anyone
|
|
238
|
+
multiplied cost-per-workflow by workflows-per-day by workdays. A margin
|
|
239
|
+
that collapses the moment users actually adopt the product is not a
|
|
240
|
+
pricing problem, it is an architecture bill arriving late -- and the
|
|
241
|
+
three levers below are the same decisions this corpus already makes:
|
|
242
|
+
a cheap deterministic path in front of the model (cascade), cached
|
|
243
|
+
prompt prefixes, and a bounded loop.
|
|
244
|
+
"""
|
|
245
|
+
_warn_if_stale(today)
|
|
246
|
+
|
|
247
|
+
for name, fraction in (("cheap_path_coverage", cheap_path_coverage),
|
|
248
|
+
("cached_prefix_share", cached_prefix_share)):
|
|
249
|
+
if fraction is not None and not 0.0 <= fraction <= 1.0:
|
|
250
|
+
raise ValueError(
|
|
251
|
+
f"{name} is a fraction between 0 and 1, got {fraction!r} -- "
|
|
252
|
+
f"if that was a percentage, divide by 100. A share above one "
|
|
253
|
+
f"turns the seat cost negative and reports a bogus healthy "
|
|
254
|
+
f"margin, which is the exact failure this check exists to name."
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
def seat_cost(steps: int, coverage: float, cached: float) -> float:
|
|
258
|
+
model_workflows = workflows_per_day * (1.0 - coverage)
|
|
259
|
+
tokens = model_workflows * steps * tokens_per_step
|
|
260
|
+
effective = tokens * ((1.0 - cached) + cached * CACHED_PREFIX_DISCOUNT)
|
|
261
|
+
return effective / 1e6 * MANAGED_COST_PER_MILLION_TOKENS * WORKDAYS_PER_MONTH
|
|
262
|
+
|
|
263
|
+
coverage = cheap_path_coverage or 0.0
|
|
264
|
+
cost = seat_cost(steps_per_workflow, coverage, cached_prefix_share)
|
|
265
|
+
margin = price_per_seat_month - cost
|
|
266
|
+
|
|
267
|
+
levers = []
|
|
268
|
+
if coverage < 0.5:
|
|
269
|
+
levers.append((
|
|
270
|
+
"route the measurable share to rules first (cascade at 50% coverage)",
|
|
271
|
+
price_per_seat_month - seat_cost(steps_per_workflow, 0.5, cached_prefix_share),
|
|
272
|
+
))
|
|
273
|
+
if cached_prefix_share < 0.7:
|
|
274
|
+
levers.append((
|
|
275
|
+
"cache the shared prefix (70% of tokens at the cached rate)",
|
|
276
|
+
price_per_seat_month - seat_cost(steps_per_workflow, coverage, 0.7),
|
|
277
|
+
))
|
|
278
|
+
if steps_per_workflow > 3:
|
|
279
|
+
levers.append((
|
|
280
|
+
"bound the loop at 3 steps (the cap the posture section documents)",
|
|
281
|
+
price_per_seat_month - seat_cost(3, coverage, cached_prefix_share),
|
|
282
|
+
))
|
|
283
|
+
|
|
284
|
+
return {
|
|
285
|
+
"cost_per_workflow": round(cost / (workflows_per_day * WORKDAYS_PER_MONTH), 4)
|
|
286
|
+
if workflows_per_day else 0.0,
|
|
287
|
+
"cost_per_seat_month": round(cost, 2),
|
|
288
|
+
"price_per_seat_month": price_per_seat_month,
|
|
289
|
+
"margin_per_seat": round(margin, 2),
|
|
290
|
+
"underwater": margin <= 0,
|
|
291
|
+
"levers": [(reason, round(new_margin, 2)) for reason, new_margin in levers],
|
|
292
|
+
}
|
fde/decide.py
ADDED
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
"""What to build for each component, and why.
|
|
2
|
+
|
|
3
|
+
Three rules carry this, and they matter more than the selection mechanism.
|
|
4
|
+
|
|
5
|
+
**The simplest applicable approach wins.** `complexity` orders candidates by
|
|
6
|
+
cost of ownership, not capability. A framework that reaches for the most capable
|
|
7
|
+
option available is the failure this exists to prevent -- and the one that makes
|
|
8
|
+
an engagement expensive to hand over.
|
|
9
|
+
|
|
10
|
+
**Every decision names what it rejected and why.** A recommendation with no
|
|
11
|
+
rejected alternatives has not been made, it has been assumed. It is also the
|
|
12
|
+
half a client actually reads, because it tells them what they are not getting.
|
|
13
|
+
|
|
14
|
+
**Nothing known means nothing decided.** No default, no "probably". The gates
|
|
15
|
+
report the gap and the interview asks.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
import json
|
|
22
|
+
from collections.abc import Mapping
|
|
23
|
+
from dataclasses import dataclass, field
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from fde.models.base import Confidence, Evidence
|
|
27
|
+
from fde.models.profile import Profile
|
|
28
|
+
from fde.models.schema import Approach, Reversibility, confidence_sufficient
|
|
29
|
+
from fde.predicate import holds
|
|
30
|
+
from fde.predicate import referenced as _referenced
|
|
31
|
+
from fde.registry import Registry
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class Rejected:
|
|
36
|
+
id: str
|
|
37
|
+
reason: str
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class Decision:
|
|
42
|
+
component: str
|
|
43
|
+
approach: str | None
|
|
44
|
+
rationale: str
|
|
45
|
+
rejected: list[Rejected] = field(default_factory=list)
|
|
46
|
+
evidence: Evidence | None = None
|
|
47
|
+
confidence: Confidence = Confidence.MEDIUM
|
|
48
|
+
reversibility: Reversibility = Reversibility.MODERATE
|
|
49
|
+
|
|
50
|
+
# How many approaches were on the table at all. One is not the same as
|
|
51
|
+
# "we weighed the options and this won" -- it means the registry offers no
|
|
52
|
+
# alternative, which a client should be told rather than left to assume.
|
|
53
|
+
considered: int = 0
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def uncontested(self) -> bool:
|
|
57
|
+
return self.considered == 1
|
|
58
|
+
|
|
59
|
+
def as_tuple(self) -> tuple:
|
|
60
|
+
return (self.component, self.approach)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class Decisions(dict):
|
|
64
|
+
"""Component id -> Decision, with a stable identity for the whole set."""
|
|
65
|
+
|
|
66
|
+
def undecided(self) -> list[str]:
|
|
67
|
+
"""Components in scope that nothing can currently fill."""
|
|
68
|
+
return sorted(c for c, d in self.items() if not d.approach)
|
|
69
|
+
|
|
70
|
+
def decided(self) -> Decisions:
|
|
71
|
+
return Decisions({c: d for c, d in self.items() if d.approach})
|
|
72
|
+
|
|
73
|
+
def decided_fingerprint(self) -> str:
|
|
74
|
+
return self.decided().fingerprint()
|
|
75
|
+
|
|
76
|
+
def fingerprint(self) -> str:
|
|
77
|
+
"""One value standing for the whole architecture.
|
|
78
|
+
|
|
79
|
+
This is what divergence compares: does answering a question change what
|
|
80
|
+
gets built? Built from the decisions themselves rather than from the
|
|
81
|
+
profile, so two different profiles that lead to the same design are
|
|
82
|
+
correctly treated as the same answer.
|
|
83
|
+
"""
|
|
84
|
+
payload = json.dumps(
|
|
85
|
+
sorted(d.as_tuple() for d in self.values() if d.approach), separators=(",", ":")
|
|
86
|
+
)
|
|
87
|
+
return hashlib.sha256(payload.encode()).hexdigest()[:16]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def decide_component(
|
|
91
|
+
component: str, values: Mapping[str, Any], registry: Registry
|
|
92
|
+
) -> Decision | None:
|
|
93
|
+
"""Choose an approach for one component, and record what lost."""
|
|
94
|
+
profile = _as_profile(values)
|
|
95
|
+
|
|
96
|
+
candidates = [
|
|
97
|
+
a for a in registry.approaches.values() if not a.components or component in a.components
|
|
98
|
+
]
|
|
99
|
+
if not candidates:
|
|
100
|
+
return None
|
|
101
|
+
|
|
102
|
+
applicable: list[Approach] = []
|
|
103
|
+
rejected: list[Rejected] = []
|
|
104
|
+
still_askable: list[Approach] = []
|
|
105
|
+
|
|
106
|
+
for approach in candidates:
|
|
107
|
+
blocked = [c for c in approach.avoid_when if holds(c, profile, registry)]
|
|
108
|
+
if blocked:
|
|
109
|
+
rejected.append(
|
|
110
|
+
Rejected(approach.id, f"ruled out by {blocked[0]}")
|
|
111
|
+
)
|
|
112
|
+
continue
|
|
113
|
+
if not any(holds(c, profile, registry) for c in approach.applies_when):
|
|
114
|
+
rejected.append(
|
|
115
|
+
Rejected(approach.id, f"nothing here matches {' or '.join(approach.applies_when)}")
|
|
116
|
+
)
|
|
117
|
+
# Not ruled out -- just not ruled in. If its applies conditions
|
|
118
|
+
# reference something unanswered, an answer could still admit it.
|
|
119
|
+
still_askable.append(approach)
|
|
120
|
+
continue
|
|
121
|
+
applicable.append(approach)
|
|
122
|
+
|
|
123
|
+
if not applicable:
|
|
124
|
+
# Two different situations, and telling them apart matters. When the
|
|
125
|
+
# predicates reference things nobody has answered, more discovery is
|
|
126
|
+
# the remedy. When everything was known and every approach is still
|
|
127
|
+
# ruled out, the facts contradict each other -- and reporting that as
|
|
128
|
+
# "not enough is known" sends somebody to ask more questions that
|
|
129
|
+
# cannot help. Name the culprits instead.
|
|
130
|
+
# Only dimensions whose answer could still admit an approach: the
|
|
131
|
+
# applies conditions of candidates that were not ruled out. Collecting
|
|
132
|
+
# from every candidate's every predicate once told a user to go ask
|
|
133
|
+
# four questions none of which could change the outcome -- the exact
|
|
134
|
+
# misdirection this branch exists to prevent.
|
|
135
|
+
unknowns = sorted({
|
|
136
|
+
dimension
|
|
137
|
+
for approach in still_askable
|
|
138
|
+
for condition in approach.applies_when
|
|
139
|
+
for dimension in _referenced(condition)
|
|
140
|
+
if profile.get(dimension) is None
|
|
141
|
+
})
|
|
142
|
+
if unknowns:
|
|
143
|
+
rationale = (
|
|
144
|
+
f"not enough is known to choose -- "
|
|
145
|
+
f"unanswered: {', '.join(unknowns)}"
|
|
146
|
+
)
|
|
147
|
+
else:
|
|
148
|
+
blocked = "; ".join(f"{r.id}: {r.reason}" for r in rejected)
|
|
149
|
+
rationale = (
|
|
150
|
+
f"everything is known and every approach is ruled out -- "
|
|
151
|
+
f"the facts conflict, and asking more questions cannot help. "
|
|
152
|
+
f"{blocked}"
|
|
153
|
+
)
|
|
154
|
+
return Decision(
|
|
155
|
+
component=component,
|
|
156
|
+
approach=None,
|
|
157
|
+
rationale=rationale,
|
|
158
|
+
rejected=rejected,
|
|
159
|
+
considered=len(candidates),
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
# Simplest first; among equals, whichever more engagements back.
|
|
163
|
+
applicable.sort(key=lambda a: (a.complexity, -_evidence_count(a), a.id))
|
|
164
|
+
winner, losers = applicable[0], applicable[1:]
|
|
165
|
+
|
|
166
|
+
rejected.extend(
|
|
167
|
+
Rejected(a.id, f"{winner.id} is simpler and applies here")
|
|
168
|
+
for a in losers
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
evidence = winner.evidence
|
|
172
|
+
confidence = evidence.confidence if evidence else Confidence.LOW
|
|
173
|
+
reversibility = _reversibility(winner)
|
|
174
|
+
|
|
175
|
+
if not confidence_sufficient(reversibility, confidence):
|
|
176
|
+
confidence = Confidence.HIGH if reversibility is Reversibility.ONE_WAY else confidence
|
|
177
|
+
|
|
178
|
+
rationale = _why(winner, profile, registry)
|
|
179
|
+
if len(candidates) == 1:
|
|
180
|
+
rationale += " -- the only approach registered for this component"
|
|
181
|
+
|
|
182
|
+
return Decision(
|
|
183
|
+
component=component,
|
|
184
|
+
approach=winner.id,
|
|
185
|
+
rationale=rationale,
|
|
186
|
+
rejected=rejected,
|
|
187
|
+
evidence=evidence,
|
|
188
|
+
confidence=confidence,
|
|
189
|
+
reversibility=reversibility,
|
|
190
|
+
considered=len(candidates),
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def base_component(component: str) -> str:
|
|
195
|
+
"""The registry id behind an instance key: perception:images -> perception.
|
|
196
|
+
|
|
197
|
+
Fan-out gives a component one instance per value of its declared
|
|
198
|
+
dimension; everything that looks a component up in the registry goes
|
|
199
|
+
through here so an instance is never mistaken for a new kind of thing.
|
|
200
|
+
"""
|
|
201
|
+
return component.split(":", 1)[0]
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def decide_all(
|
|
205
|
+
values: Mapping[str, Any], registry: Registry, components: list[str] | None = None
|
|
206
|
+
) -> Decisions:
|
|
207
|
+
"""Decide every component asked for -- including the ones we cannot.
|
|
208
|
+
|
|
209
|
+
A component decomposition put in scope but decision cannot fill does not
|
|
210
|
+
disappear. It stays, with no approach and a rationale saying why, because a
|
|
211
|
+
component that vanishes between "you need this" and "here is the design" is
|
|
212
|
+
a hole nobody notices until build time.
|
|
213
|
+
|
|
214
|
+
A component that declares fan_out_on fans into one instance per value
|
|
215
|
+
when the engagement carries several -- a claims system taking photos AND
|
|
216
|
+
the policy document gets perception:images and perception:documents, each
|
|
217
|
+
decided by the same rules with that one modality bound. Multi-modality is
|
|
218
|
+
the normaliser's property: downstream components see what perception
|
|
219
|
+
produced, so they decide once.
|
|
220
|
+
"""
|
|
221
|
+
wanted = components if components is not None else list(registry.components)
|
|
222
|
+
decisions = Decisions()
|
|
223
|
+
for component in wanted:
|
|
224
|
+
entry = registry.components.get(base_component(component))
|
|
225
|
+
fan = entry.fan_out_on if entry else None
|
|
226
|
+
carried = values.get(fan) if fan else None
|
|
227
|
+
if fan and isinstance(carried, tuple) and len(carried) > 1:
|
|
228
|
+
for modality in carried:
|
|
229
|
+
bound = {**values, fan: modality}
|
|
230
|
+
key = f"{component}:{modality}"
|
|
231
|
+
decision = decide_component(component, bound, registry)
|
|
232
|
+
if decision is None:
|
|
233
|
+
decision = Decision(
|
|
234
|
+
component=key, approach=None,
|
|
235
|
+
rationale="no approach in the registry serves this component",
|
|
236
|
+
considered=0,
|
|
237
|
+
)
|
|
238
|
+
decisions[key] = decision
|
|
239
|
+
continue
|
|
240
|
+
decision = decide_component(component, values, registry)
|
|
241
|
+
if decision is None:
|
|
242
|
+
decision = Decision(
|
|
243
|
+
component=component,
|
|
244
|
+
approach=None,
|
|
245
|
+
rationale="no approach in the registry serves this component",
|
|
246
|
+
considered=0,
|
|
247
|
+
)
|
|
248
|
+
decisions[component] = decision
|
|
249
|
+
return decisions
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def architecture_outcome(registry: Registry, components: list[str] | None = None):
|
|
253
|
+
"""An outcome function for divergence: what actually gets built.
|
|
254
|
+
|
|
255
|
+
The placeholder measured which dimensions got settled. This measures the
|
|
256
|
+
design, which is the question worth asking -- a question that narrows the
|
|
257
|
+
space but changes nothing is not worth a client's time.
|
|
258
|
+
"""
|
|
259
|
+
|
|
260
|
+
def outcome(space) -> str:
|
|
261
|
+
values = {d: space.value(d) for d in space.dimensions() if space.resolved(d)}
|
|
262
|
+
return decide_all(values, registry, components=components).fingerprint()
|
|
263
|
+
|
|
264
|
+
return outcome
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
# --- helpers -------------------------------------------------------------
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _as_profile(values: Mapping[str, Any]) -> Profile:
|
|
271
|
+
"""Decisions are made from resolved values, whether those came from a
|
|
272
|
+
profile or from exploring a hypothetical."""
|
|
273
|
+
if isinstance(values, Profile):
|
|
274
|
+
return values
|
|
275
|
+
|
|
276
|
+
from fde.models.base import Provenance
|
|
277
|
+
from fde.models.fact import Fact
|
|
278
|
+
|
|
279
|
+
profile = Profile()
|
|
280
|
+
profile.ingest(
|
|
281
|
+
[Fact(k, v, Provenance.ARTIFACT) for k, v in values.items() if v is not None]
|
|
282
|
+
)
|
|
283
|
+
return profile
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _evidence_count(approach: Approach) -> int:
|
|
287
|
+
return len(approach.evidence.case_ids) if approach.evidence else 0
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _reversibility(approach: Approach) -> Reversibility:
|
|
291
|
+
# Adapting weights means retraining and re-evaluating to undo.
|
|
292
|
+
return Reversibility.EXPENSIVE if approach.id == "finetune" else Reversibility.MODERATE
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _why(approach: Approach, profile: Profile, registry: Registry) -> str:
|
|
296
|
+
fired = [c for c in approach.applies_when if holds(c, profile, registry)]
|
|
297
|
+
return f"{approach.name}: {'; '.join(fired)}"
|
fde/decompose.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Profile into a component graph.
|
|
2
|
+
|
|
3
|
+
A component is included only when a condition it declares actually fires, and it
|
|
4
|
+
records which one. Spurious components are how a scope doubles between the
|
|
5
|
+
workshop and the statement of work, so "we might need retrieval" is not a
|
|
6
|
+
reason to include retrieval -- and not knowing yet is not a reason either.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
|
|
13
|
+
from fde.models.profile import Profile
|
|
14
|
+
from fde.models.schema import Component, earliest_cap
|
|
15
|
+
from fde.predicate import holds
|
|
16
|
+
from fde.registry import Registry
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class Included:
|
|
21
|
+
component: Component
|
|
22
|
+
|
|
23
|
+
# The conditions that fired. An FDE asked "why is retrieval in scope?"
|
|
24
|
+
# gets a sentence, not a shrug.
|
|
25
|
+
because: list[str] = field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def id(self) -> str:
|
|
29
|
+
return self.component.id
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ComponentGraph:
|
|
34
|
+
components: dict[str, Included] = field(default_factory=dict)
|
|
35
|
+
|
|
36
|
+
def earliest_cap(self, component_id: str) -> str:
|
|
37
|
+
"""Where to look first when this component's quality is capped."""
|
|
38
|
+
return earliest_cap(component_id, {i: v.component for i, v in self.components.items()})
|
|
39
|
+
|
|
40
|
+
def __contains__(self, component_id: str) -> bool:
|
|
41
|
+
return component_id in self.components
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def decompose(profile: Profile, registry: Registry) -> ComponentGraph:
|
|
45
|
+
graph = ComponentGraph()
|
|
46
|
+
for component in registry.components.values():
|
|
47
|
+
fired = [
|
|
48
|
+
condition
|
|
49
|
+
for condition in component.required_when
|
|
50
|
+
if holds(condition, profile, registry)
|
|
51
|
+
]
|
|
52
|
+
if fired:
|
|
53
|
+
graph.components[component.id] = Included(component=component, because=fired)
|
|
54
|
+
return graph
|