fde-framework 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (286) hide show
  1. fde/__init__.py +3 -0
  2. fde/architect.py +152 -0
  3. fde/cli.py +2107 -0
  4. fde/costing.py +292 -0
  5. fde/decide.py +297 -0
  6. fde/decompose.py +54 -0
  7. fde/deploy.py +371 -0
  8. fde/emit.py +1309 -0
  9. fde/evolution.py +265 -0
  10. fde/factlog.py +369 -0
  11. fde/framework/approaches/ansible-playbook.md +22 -0
  12. fde/framework/approaches/assisted-deterministic.md +37 -0
  13. fde/framework/approaches/audit-only.md +23 -0
  14. fde/framework/approaches/boundary-and-audit.md +18 -0
  15. fde/framework/approaches/cascade.md +32 -0
  16. fde/framework/approaches/classical-ml.md +21 -0
  17. fde/framework/approaches/compose.md +18 -0
  18. fde/framework/approaches/decision-log.md +21 -0
  19. fde/framework/approaches/deterministic-masking.md +18 -0
  20. fde/framework/approaches/deterministic.md +25 -0
  21. fde/framework/approaches/direct-call.md +10 -0
  22. fde/framework/approaches/episodic-store.md +17 -0
  23. fde/framework/approaches/explainability-record.md +14 -0
  24. fde/framework/approaches/field-match.md +15 -0
  25. fde/framework/approaches/finetune.md +29 -0
  26. fde/framework/approaches/fixed-sequence.md +19 -0
  27. fde/framework/approaches/gitops.md +18 -0
  28. fde/framework/approaches/governed-tools.md +18 -0
  29. fde/framework/approaches/graph-retrieval.md +22 -0
  30. fde/framework/approaches/judged.md +15 -0
  31. fde/framework/approaches/keyword-search.md +11 -0
  32. fde/framework/approaches/kubernetes-manifests.md +19 -0
  33. fde/framework/approaches/labelled-metrics.md +14 -0
  34. fde/framework/approaches/llm-extraction.md +26 -0
  35. fde/framework/approaches/llm-scrubbing.md +20 -0
  36. fde/framework/approaches/llm.md +23 -0
  37. fde/framework/approaches/local-embedding.md +28 -0
  38. fde/framework/approaches/managed-api.md +49 -0
  39. fde/framework/approaches/managed-embedding.md +22 -0
  40. fde/framework/approaches/manual-runbook.md +32 -0
  41. fde/framework/approaches/model-planner.md +19 -0
  42. fde/framework/approaches/ocr-pipeline.md +13 -0
  43. fde/framework/approaches/optimisation-reasoning.md +30 -0
  44. fde/framework/approaches/optimisation.md +19 -0
  45. fde/framework/approaches/passthrough.md +11 -0
  46. fde/framework/approaches/role-scoped-authority.md +19 -0
  47. fde/framework/approaches/segmentation.md +29 -0
  48. fde/framework/approaches/self-hosted.md +38 -0
  49. fde/framework/approaches/serverless-gpu.md +27 -0
  50. fde/framework/approaches/speech-transcription.md +21 -0
  51. fde/framework/approaches/structured-logs.md +11 -0
  52. fde/framework/approaches/systemd-unit.md +30 -0
  53. fde/framework/approaches/terraform-module.md +22 -0
  54. fde/framework/approaches/text-extraction.md +14 -0
  55. fde/framework/approaches/traced.md +24 -0
  56. fde/framework/approaches/vector-search.md +17 -0
  57. fde/framework/approaches/video-ingestion.md +16 -0
  58. fde/framework/approaches/windowed-ingestion.md +25 -0
  59. fde/framework/approaches/working-state.md +13 -0
  60. fde/framework/cases/churn-scoring.md +45 -0
  61. fde/framework/cases/route-planning.md +44 -0
  62. fde/framework/cases/structured-extraction.md +48 -0
  63. fde/framework/cases/studio-style.md +46 -0
  64. fde/framework/components/accountability.md +20 -0
  65. fde/framework/components/deployment.md +19 -0
  66. fde/framework/components/embedding.md +23 -0
  67. fde/framework/components/evaluation.md +13 -0
  68. fde/framework/components/governance.md +19 -0
  69. fde/framework/components/integration.md +13 -0
  70. fde/framework/components/memory.md +21 -0
  71. fde/framework/components/observability.md +13 -0
  72. fde/framework/components/perception.md +15 -0
  73. fde/framework/components/planning.md +15 -0
  74. fde/framework/components/provisioning.md +17 -0
  75. fde/framework/components/reasoning.md +20 -0
  76. fde/framework/components/redaction.md +18 -0
  77. fde/framework/components/representation.md +13 -0
  78. fde/framework/components/retrieval.md +13 -0
  79. fde/framework/components/serving.md +20 -0
  80. fde/framework/dimensions/accelerator.md +29 -0
  81. fde/framework/dimensions/access_model.md +25 -0
  82. fde/framework/dimensions/arrival_rate.md +19 -0
  83. fde/framework/dimensions/availability_target.md +23 -0
  84. fde/framework/dimensions/cheap_path_coverage.md +21 -0
  85. fde/framework/dimensions/confidence_calibrated.md +30 -0
  86. fde/framework/dimensions/container_competence.md +22 -0
  87. fde/framework/dimensions/corpus_size.md +13 -0
  88. fde/framework/dimensions/data_residency.md +36 -0
  89. fde/framework/dimensions/environment_lifetime.md +19 -0
  90. fde/framework/dimensions/existing_cluster.md +22 -0
  91. fde/framework/dimensions/existing_iac_tool.md +25 -0
  92. fde/framework/dimensions/external_systems.md +15 -0
  93. fde/framework/dimensions/hosting.md +45 -0
  94. fde/framework/dimensions/human_waiting.md +41 -0
  95. fde/framework/dimensions/input_format.md +29 -0
  96. fde/framework/dimensions/interpretability_required.md +24 -0
  97. fde/framework/dimensions/labelled_count.md +15 -0
  98. fde/framework/dimensions/latency_budget_ms.md +16 -0
  99. fde/framework/dimensions/licence_posture.md +28 -0
  100. fde/framework/dimensions/operates_after_handover.md +25 -0
  101. fde/framework/dimensions/output_shape.md +29 -0
  102. fde/framework/dimensions/provisioning_api.md +18 -0
  103. fde/framework/dimensions/query_pattern.md +25 -0
  104. fde/framework/dimensions/recall_span.md +22 -0
  105. fde/framework/dimensions/sensitivity_present.md +22 -0
  106. fde/framework/interfaces/Generator.md +5 -0
  107. fde/framework/interfaces/Guard.md +5 -0
  108. fde/framework/interfaces/Mapper.md +5 -0
  109. fde/framework/interfaces/ModelServer.md +5 -0
  110. fde/framework/interfaces/Parser.md +5 -0
  111. fde/framework/interfaces/Planner.md +5 -0
  112. fde/framework/interfaces/Retriever.md +5 -0
  113. fde/framework/interfaces/Scorer.md +5 -0
  114. fde/framework/interfaces/Store.md +5 -0
  115. fde/framework/interfaces/ToolBoundary.md +5 -0
  116. fde/framework/interfaces/Tracer.md +5 -0
  117. fde/framework/locales/eu-gdpr.md +51 -0
  118. fde/framework/locales/in-dpdp.md +46 -0
  119. fde/framework/patterns/ansible-playbook.md +9 -0
  120. fde/framework/patterns/assisted-deterministic.md +11 -0
  121. fde/framework/patterns/audit-only.md +9 -0
  122. fde/framework/patterns/boundary-and-audit.md +12 -0
  123. fde/framework/patterns/cascade-reasoning.md +10 -0
  124. fde/framework/patterns/cascade-representation.md +10 -0
  125. fde/framework/patterns/cascade-retrieval.md +10 -0
  126. fde/framework/patterns/classical-ml-reasoning.md +15 -0
  127. fde/framework/patterns/classical-ml.md +13 -0
  128. fde/framework/patterns/compose.md +9 -0
  129. fde/framework/patterns/decision-log.md +9 -0
  130. fde/framework/patterns/deterministic-masking.md +9 -0
  131. fde/framework/patterns/deterministic.md +12 -0
  132. fde/framework/patterns/direct-call.md +12 -0
  133. fde/framework/patterns/episodic-store.md +13 -0
  134. fde/framework/patterns/explainability-record.md +9 -0
  135. fde/framework/patterns/field-match.md +12 -0
  136. fde/framework/patterns/finetune-representation.md +14 -0
  137. fde/framework/patterns/finetune.md +12 -0
  138. fde/framework/patterns/fixed-sequence.md +12 -0
  139. fde/framework/patterns/gitops.md +9 -0
  140. fde/framework/patterns/governed-tools.md +13 -0
  141. fde/framework/patterns/graph-retrieval.md +14 -0
  142. fde/framework/patterns/judged.md +14 -0
  143. fde/framework/patterns/keyword-search.md +12 -0
  144. fde/framework/patterns/kubernetes-manifests.md +9 -0
  145. fde/framework/patterns/labelled-metrics.md +13 -0
  146. fde/framework/patterns/llm-representation.md +17 -0
  147. fde/framework/patterns/llm-scrubbing.md +9 -0
  148. fde/framework/patterns/llm.md +12 -0
  149. fde/framework/patterns/local-embedding.md +9 -0
  150. fde/framework/patterns/managed-api.md +12 -0
  151. fde/framework/patterns/managed-embedding.md +9 -0
  152. fde/framework/patterns/manual-runbook.md +9 -0
  153. fde/framework/patterns/model-planner.md +13 -0
  154. fde/framework/patterns/ocr-pipeline.md +13 -0
  155. fde/framework/patterns/optimisation-reasoning.md +15 -0
  156. fde/framework/patterns/optimisation.md +13 -0
  157. fde/framework/patterns/passthrough.md +12 -0
  158. fde/framework/patterns/role-scoped-authority.md +9 -0
  159. fde/framework/patterns/segmentation.md +12 -0
  160. fde/framework/patterns/self-hosted.md +14 -0
  161. fde/framework/patterns/serverless-gpu.md +12 -0
  162. fde/framework/patterns/speech-transcription.md +10 -0
  163. fde/framework/patterns/structured-logs.md +12 -0
  164. fde/framework/patterns/systemd-unit.md +9 -0
  165. fde/framework/patterns/terraform-module.md +9 -0
  166. fde/framework/patterns/text-extraction.md +12 -0
  167. fde/framework/patterns/traced.md +13 -0
  168. fde/framework/patterns/vector-search.md +14 -0
  169. fde/framework/patterns/video-ingestion.md +9 -0
  170. fde/framework/patterns/windowed-ingestion.md +12 -0
  171. fde/framework/patterns/working-state.md +12 -0
  172. fde/framework/stacks/langgraph.md +9 -0
  173. fde/framework/stacks/local-judge.md +9 -0
  174. fde/framework/stacks/mcp.md +9 -0
  175. fde/framework/stacks/ollama.md +19 -0
  176. fde/framework/stacks/openai-judge.md +9 -0
  177. fde/framework/stacks/opentelemetry.md +9 -0
  178. fde/framework/stacks/ortools.md +9 -0
  179. fde/framework/stacks/pgvector.md +9 -0
  180. fde/framework/stacks/plain-python.md +9 -0
  181. fde/framework/stacks/qdrant.md +9 -0
  182. fde/framework/stacks/tesseract.md +9 -0
  183. fde/framework/stacks/vllm.md +9 -0
  184. fde/framework/stacks/whisper.md +15 -0
  185. fde/framework/stacks/xgboost.md +9 -0
  186. fde/framework/templates/accountability/decision-log.plain.py.j2 +64 -0
  187. fde/framework/templates/accountability/explainability-record.plain.py.j2 +90 -0
  188. fde/framework/templates/deployment/compose.plain.py.j2 +36 -0
  189. fde/framework/templates/deployment/kubernetes-manifests.plain.py.j2 +36 -0
  190. fde/framework/templates/deployment/systemd-unit.plain.py.j2 +36 -0
  191. fde/framework/templates/embedding/local-embedding.plain.py.j2 +71 -0
  192. fde/framework/templates/embedding/managed-embedding.plain.py.j2 +76 -0
  193. fde/framework/templates/evaluation/field-match.plain.py.j2 +146 -0
  194. fde/framework/templates/evaluation/judged.local-judge.py.j2 +100 -0
  195. fde/framework/templates/evaluation/judged.openai-judge.py.j2 +70 -0
  196. fde/framework/templates/evaluation/judged.plain.py.j2 +89 -0
  197. fde/framework/templates/evaluation/labelled-metrics.plain.py.j2 +64 -0
  198. fde/framework/templates/evaluation/labelled-metrics.xgboost.py.j2 +72 -0
  199. fde/framework/templates/governance/audit-only.plain.py.j2 +98 -0
  200. fde/framework/templates/governance/boundary-and-audit.plain.py.j2 +142 -0
  201. fde/framework/templates/governance/role-scoped-authority.plain.py.j2 +63 -0
  202. fde/framework/templates/integration/direct-call.plain.py.j2 +63 -0
  203. fde/framework/templates/integration/governed-tools.mcp.py.j2 +124 -0
  204. fde/framework/templates/integration/governed-tools.plain.py.j2 +153 -0
  205. fde/framework/templates/memory/episodic-store.pgvector.py.j2 +118 -0
  206. fde/framework/templates/memory/episodic-store.plain.py.j2 +147 -0
  207. fde/framework/templates/memory/working-state.plain.py.j2 +48 -0
  208. fde/framework/templates/observability/structured-logs.plain.py.j2 +61 -0
  209. fde/framework/templates/observability/traced.opentelemetry.py.j2 +65 -0
  210. fde/framework/templates/observability/traced.plain.py.j2 +132 -0
  211. fde/framework/templates/perception/ocr-pipeline.plain.py.j2 +71 -0
  212. fde/framework/templates/perception/ocr-pipeline.tesseract.py.j2 +80 -0
  213. fde/framework/templates/perception/passthrough.plain.py.j2 +41 -0
  214. fde/framework/templates/perception/speech-transcription.plain.py.j2 +37 -0
  215. fde/framework/templates/perception/speech-transcription.whisper.py.j2 +39 -0
  216. fde/framework/templates/perception/text-extraction.plain.py.j2 +88 -0
  217. fde/framework/templates/perception/video-ingestion.plain.py.j2 +35 -0
  218. fde/framework/templates/perception/windowed-ingestion.plain.py.j2 +69 -0
  219. fde/framework/templates/planning/fixed-sequence.plain.py.j2 +46 -0
  220. fde/framework/templates/planning/model-planner.langgraph.py.j2 +89 -0
  221. fde/framework/templates/planning/model-planner.plain.py.j2 +82 -0
  222. fde/framework/templates/planning/optimisation.ortools.py.j2 +90 -0
  223. fde/framework/templates/planning/optimisation.plain.py.j2 +71 -0
  224. fde/framework/templates/provisioning/ansible-playbook.plain.py.j2 +29 -0
  225. fde/framework/templates/provisioning/gitops.plain.py.j2 +29 -0
  226. fde/framework/templates/provisioning/manual-runbook.plain.py.j2 +29 -0
  227. fde/framework/templates/provisioning/terraform-module.plain.py.j2 +29 -0
  228. fde/framework/templates/reasoning/cascade.plain.py.j2 +86 -0
  229. fde/framework/templates/reasoning/classical-ml.plain.py.j2 +74 -0
  230. fde/framework/templates/reasoning/classical-ml.xgboost.py.j2 +104 -0
  231. fde/framework/templates/reasoning/finetune.plain.py.j2 +73 -0
  232. fde/framework/templates/reasoning/llm.plain.py.j2 +114 -0
  233. fde/framework/templates/reasoning/optimisation.ortools.py.j2 +90 -0
  234. fde/framework/templates/reasoning/optimisation.plain.py.j2 +58 -0
  235. fde/framework/templates/redaction/deterministic-masking.plain.py.j2 +44 -0
  236. fde/framework/templates/redaction/llm-scrubbing.plain.py.j2 +40 -0
  237. fde/framework/templates/representation/assisted.plain.py.j2 +86 -0
  238. fde/framework/templates/representation/cascade.plain.py.j2 +119 -0
  239. fde/framework/templates/representation/classical-ml.plain.py.j2 +74 -0
  240. fde/framework/templates/representation/classical-ml.xgboost.py.j2 +104 -0
  241. fde/framework/templates/representation/deterministic.plain.py.j2 +101 -0
  242. fde/framework/templates/representation/finetune.plain.py.j2 +68 -0
  243. fde/framework/templates/representation/llm.plain.py.j2 +94 -0
  244. fde/framework/templates/representation/segmentation.plain.py.j2 +68 -0
  245. fde/framework/templates/retrieval/cascade.plain.py.j2 +86 -0
  246. fde/framework/templates/retrieval/graph-retrieval.pgvector.py.j2 +88 -0
  247. fde/framework/templates/retrieval/graph-retrieval.plain.py.j2 +74 -0
  248. fde/framework/templates/retrieval/graph-retrieval.qdrant.py.j2 +71 -0
  249. fde/framework/templates/retrieval/keyword-search.plain.py.j2 +104 -0
  250. fde/framework/templates/retrieval/vector-search.pgvector.py.j2 +100 -0
  251. fde/framework/templates/retrieval/vector-search.plain.py.j2 +84 -0
  252. fde/framework/templates/retrieval/vector-search.qdrant.py.j2 +60 -0
  253. fde/framework/templates/serving/managed-api.plain.py.j2 +74 -0
  254. fde/framework/templates/serving/self-hosted.ollama.py.j2 +47 -0
  255. fde/framework/templates/serving/self-hosted.plain.py.j2 +117 -0
  256. fde/framework/templates/serving/self-hosted.vllm.py.j2 +77 -0
  257. fde/framework/templates/serving/serverless-gpu.plain.py.j2 +62 -0
  258. fde/gates.py +480 -0
  259. fde/graph.py +435 -0
  260. fde/implement.py +250 -0
  261. fde/intake/__init__.py +0 -0
  262. fde/intake/answers.py +154 -0
  263. fde/intake/documents.py +121 -0
  264. fde/intake/interview.py +218 -0
  265. fde/intake/llm_reader.py +336 -0
  266. fde/intake/prose.py +387 -0
  267. fde/intake/samples.py +369 -0
  268. fde/models/__init__.py +0 -0
  269. fde/models/base.py +86 -0
  270. fde/models/fact.py +37 -0
  271. fde/models/profile.py +159 -0
  272. fde/models/respondent.py +34 -0
  273. fde/models/schema.py +430 -0
  274. fde/moves.py +136 -0
  275. fde/ops.py +349 -0
  276. fde/predicate.py +106 -0
  277. fde/realization.py +109 -0
  278. fde/registry.py +162 -0
  279. fde/scan.py +479 -0
  280. fde/space.py +172 -0
  281. fde/workflow.py +233 -0
  282. fde_framework-0.1.0.dist-info/METADATA +449 -0
  283. fde_framework-0.1.0.dist-info/RECORD +286 -0
  284. fde_framework-0.1.0.dist-info/WHEEL +4 -0
  285. fde_framework-0.1.0.dist-info/entry_points.txt +2 -0
  286. fde_framework-0.1.0.dist-info/licenses/LICENSE +202 -0
fde/costing.py ADDED
@@ -0,0 +1,292 @@
1
+ """What it costs, with the date attached.
2
+
3
+ Every absolute here ages. Prices fall, hardware changes, and a figure quoted in
4
+ a proposal a year after it was written is wrong in a way nobody notices -- so
5
+ each carries the date it was true and the rule for working it out again.
6
+
7
+ The sizing is the part usually got wrong. Counting weights against average
8
+ throughput understates a fleet substantially, because redundancy, peak and
9
+ prefill overhead each multiply it, and the understatement is discovered in
10
+ production rather than in the spreadsheet.
11
+
12
+ Both conclusions live here on purpose. At sustained interactive volume,
13
+ self-hosting loses to a managed endpoint; on bursty batch work it wins by a wide
14
+ margin. A framework that only knew one of them would be wrong half the time, and
15
+ confidently.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import math
21
+ import warnings
22
+ from datetime import date
23
+ from typing import Any
24
+
25
+ from fde.scan import GPU, Hardware, fits
26
+
27
+ # Every figure below was true on this date and will not stay true.
28
+ AS_OF = "2026-08"
29
+ STALE_AFTER_DAYS = 270
30
+
31
+ # Indicative. Re-derive against current pricing rather than quoting these.
32
+ GPU_COST_PER_HOUR = 4.50
33
+ MANAGED_COST_PER_MILLION_TOKENS = 3.00
34
+ TOKENS_PER_REQUEST = 1_500
35
+
36
+ # The unit being rented. A replica is however many of these the model needs,
37
+ # which is the step usually skipped -- a 70B model in bf16 is 140GB of weights
38
+ # and does not run on one of them.
39
+ CARD_VRAM_GB = 80
40
+
41
+ # Traffic does not arrive evenly. Sizing to the daily mean leaves a fleet that
42
+ # fails at the busiest hour, which is the hour anybody notices.
43
+ PEAK_MULTIPLIER = 2.0
44
+
45
+ # One spare. A fleet with no redundancy is a fleet that is down during a deploy.
46
+ REDUNDANCY = 1.34
47
+
48
+ # Prefill is compute-bound and does not batch as well as decode, so a fleet
49
+ # sized on decode throughput alone is short.
50
+ PREFILL_OVERHEAD = 1.2
51
+
52
+ HOURS_PER_MONTH = 730
53
+
54
+
55
+ def gpus_per_replica(params_b: float, precision: str = "bf16") -> int:
56
+ """How many cards one copy of this model occupies.
57
+
58
+ Sizing in replicas and pricing each as one GPU is how a fleet is quoted at
59
+ a third of its cost. A replica is a number of cards, and the number comes
60
+ from the weights plus the cache rather than from the weights alone.
61
+ """
62
+ one_card = Hardware(gpus=[GPU("card", vram_gb=CARD_VRAM_GB)])
63
+ fit = fits(one_card, params_b, precision=precision)
64
+ return max(1, math.ceil(fit.required_gb / fit.available_gb))
65
+
66
+
67
+ def size_for(
68
+ requests_per_day: int,
69
+ params_b: float,
70
+ requests_per_second_per_replica: float = 2.0,
71
+ today: str | None = None,
72
+ ) -> dict[str, Any]:
73
+ """How many replicas this actually needs, and how many cards that is.
74
+
75
+ The naive figure is reported beside the real one, because the gap is the
76
+ finding -- and somebody will otherwise arrive at the naive figure
77
+ independently and wonder why the estimate is higher.
78
+ """
79
+ _warn_if_stale(today)
80
+
81
+ mean_rps = requests_per_day / 86_400
82
+ naive = max(1, round(mean_rps / requests_per_second_per_replica))
83
+ real = max(
84
+ 1,
85
+ round(
86
+ (mean_rps * PEAK_MULTIPLIER * PREFILL_OVERHEAD * REDUNDANCY)
87
+ / requests_per_second_per_replica
88
+ ),
89
+ )
90
+ per_replica = gpus_per_replica(params_b)
91
+
92
+ return {
93
+ "naive_replicas": naive,
94
+ "replicas": real,
95
+ "gpus_per_replica": per_replica,
96
+ "gpus": real * per_replica,
97
+ "factors": {
98
+ "peak": f"traffic is not flat; sized at {PEAK_MULTIPLIER}x the daily mean",
99
+ "prefill": f"prefill is compute-bound and batches worse than decode "
100
+ f"({PREFILL_OVERHEAD}x)",
101
+ "redundancy": "one spare, so a deploy is not an outage",
102
+ "model_size": f"a {params_b:g}B model at bf16 occupies {per_replica} "
103
+ f"card(s) per replica, weights and cache together",
104
+ },
105
+ "monthly_cost": round(
106
+ real * per_replica * GPU_COST_PER_HOUR * HOURS_PER_MONTH, 2
107
+ ),
108
+ "as_of": AS_OF,
109
+ "rederive": (
110
+ "measure requests per second per replica on the real model and "
111
+ "hardware, then apply peak, prefill and redundancy to the measured "
112
+ "figure rather than to this one"
113
+ ),
114
+ }
115
+
116
+
117
+ def compare_hosting(
118
+ requests_per_day: int,
119
+ params_b: float,
120
+ human_waiting: bool = True,
121
+ today: str | None = None,
122
+ ) -> dict[str, Any]:
123
+ """Managed against self-hosted, for this workload.
124
+
125
+ The answer flips on utilisation and on whether anybody is waiting, which is
126
+ why the question cannot be settled once and quoted forever. Sustained
127
+ interactive traffic keeps a self-hosted fleet running around the clock;
128
+ bursty batch work pays for nothing between jobs.
129
+ """
130
+ _warn_if_stale(today)
131
+
132
+ tokens_per_month = requests_per_day * 30 * TOKENS_PER_REQUEST
133
+ managed = tokens_per_month / 1e6 * MANAGED_COST_PER_MILLION_TOKENS
134
+
135
+ per_replica = gpus_per_replica(params_b)
136
+
137
+ if human_waiting:
138
+ # Somebody is waiting, so the fleet stays up whether or not it is busy,
139
+ # sized for the peak hour and carrying a spare.
140
+ plan = size_for(requests_per_day, params_b, today=today)
141
+ self_hosted = plan["monthly_cost"]
142
+ note = (
143
+ f"Sustained interactive traffic keeps this running around the clock "
144
+ f"at {plan['gpus']} cards ({per_replica} per replica), and redundancy "
145
+ f"and peak headroom are most of the bill."
146
+ )
147
+ else:
148
+ # Nobody waiting, so a cold start costs nothing and idle time is avoidable.
149
+ busy_hours = (requests_per_day * 30) / (2.0 * 3600)
150
+ self_hosted = busy_hours * per_replica * GPU_COST_PER_HOUR
151
+ plan = {"replicas": 1}
152
+ note = (
153
+ "Nobody is waiting, so this scales to zero between jobs and pays for "
154
+ "compute rather than for availability."
155
+ )
156
+
157
+ return {
158
+ "managed_monthly": round(managed, 2),
159
+ "self_hosted_monthly": round(self_hosted, 2),
160
+ "replicas": plan["replicas"],
161
+ "gpus_per_replica": per_replica,
162
+ "recommendation": "managed" if managed < self_hosted else "self-hosted",
163
+ "why": note,
164
+ "as_of": AS_OF,
165
+ "rederive": (
166
+ "check current per-token and per-hour pricing, and measure tokens "
167
+ "per request on real traffic rather than assuming"
168
+ ),
169
+ }
170
+
171
+
172
+ def effort_by_analogy(profile: dict[str, Any], cases: dict[str, Any]) -> dict[str, Any]:
173
+ """How long this took the last time something like it was done.
174
+
175
+ An estimate from comparable work, with the comparables named so somebody
176
+ can disagree with the comparison rather than with the number.
177
+ """
178
+ analogues = [
179
+ case_id for case_id, case in cases.items()
180
+ if _overlap(profile, getattr(case, "profile", {})) >= 2
181
+ ]
182
+ if not analogues:
183
+ return {
184
+ "analogues": [],
185
+ "range_weeks": None,
186
+ "why": "nothing in the corpus resembles this closely enough to "
187
+ "estimate from. An estimate without a comparable is a guess "
188
+ "with a number on it.",
189
+ }
190
+ return {
191
+ "analogues": sorted(analogues),
192
+ "range_weeks": (3, 8),
193
+ "why": f"estimated from {len(analogues)} comparable engagement(s); "
194
+ f"disagree with the comparison rather than with the number",
195
+ "as_of": AS_OF,
196
+ }
197
+
198
+
199
+ def _overlap(profile: dict[str, Any], other: dict[str, Any]) -> int:
200
+ return sum(1 for k, v in profile.items() if other.get(k) == v)
201
+
202
+
203
+ def _warn_if_stale(today: str | None) -> None:
204
+ """Say so when these figures have aged past usefulness."""
205
+ if not today:
206
+ return
207
+ age = (date.fromisoformat(today) - date.fromisoformat(f"{AS_OF}-01")).days
208
+ if age > STALE_AFTER_DAYS:
209
+ warnings.warn(
210
+ f"these figures are as of {AS_OF}, roughly {age} days ago. Pricing and "
211
+ f"hardware have moved; re-derive before quoting them.",
212
+ UserWarning,
213
+ stacklevel=3,
214
+ )
215
+
216
+
217
+ # --- unit economics: the arbitrage-trap check -------------------------------
218
+
219
+ WORKDAYS_PER_MONTH = 22
220
+
221
+ # Reused prompt prefixes bill at roughly a tenth of the input rate on the
222
+ # hosted APIs that support prompt caching. Dated like every figure here.
223
+ CACHED_PREFIX_DISCOUNT = 0.10
224
+
225
+
226
+ def unit_economics(
227
+ workflows_per_day: float,
228
+ price_per_seat_month: float,
229
+ steps_per_workflow: int = 5,
230
+ tokens_per_step: int = TOKENS_PER_REQUEST,
231
+ cheap_path_coverage: float | None = None,
232
+ cached_prefix_share: float = 0.0,
233
+ today: str | None = None,
234
+ ) -> dict:
235
+ """Whether a seat earns more than it burns, and which lever moves it.
236
+
237
+ The failure this exists to name: a per-seat price set before anyone
238
+ multiplied cost-per-workflow by workflows-per-day by workdays. A margin
239
+ that collapses the moment users actually adopt the product is not a
240
+ pricing problem, it is an architecture bill arriving late -- and the
241
+ three levers below are the same decisions this corpus already makes:
242
+ a cheap deterministic path in front of the model (cascade), cached
243
+ prompt prefixes, and a bounded loop.
244
+ """
245
+ _warn_if_stale(today)
246
+
247
+ for name, fraction in (("cheap_path_coverage", cheap_path_coverage),
248
+ ("cached_prefix_share", cached_prefix_share)):
249
+ if fraction is not None and not 0.0 <= fraction <= 1.0:
250
+ raise ValueError(
251
+ f"{name} is a fraction between 0 and 1, got {fraction!r} -- "
252
+ f"if that was a percentage, divide by 100. A share above one "
253
+ f"turns the seat cost negative and reports a bogus healthy "
254
+ f"margin, which is the exact failure this check exists to name."
255
+ )
256
+
257
+ def seat_cost(steps: int, coverage: float, cached: float) -> float:
258
+ model_workflows = workflows_per_day * (1.0 - coverage)
259
+ tokens = model_workflows * steps * tokens_per_step
260
+ effective = tokens * ((1.0 - cached) + cached * CACHED_PREFIX_DISCOUNT)
261
+ return effective / 1e6 * MANAGED_COST_PER_MILLION_TOKENS * WORKDAYS_PER_MONTH
262
+
263
+ coverage = cheap_path_coverage or 0.0
264
+ cost = seat_cost(steps_per_workflow, coverage, cached_prefix_share)
265
+ margin = price_per_seat_month - cost
266
+
267
+ levers = []
268
+ if coverage < 0.5:
269
+ levers.append((
270
+ "route the measurable share to rules first (cascade at 50% coverage)",
271
+ price_per_seat_month - seat_cost(steps_per_workflow, 0.5, cached_prefix_share),
272
+ ))
273
+ if cached_prefix_share < 0.7:
274
+ levers.append((
275
+ "cache the shared prefix (70% of tokens at the cached rate)",
276
+ price_per_seat_month - seat_cost(steps_per_workflow, coverage, 0.7),
277
+ ))
278
+ if steps_per_workflow > 3:
279
+ levers.append((
280
+ "bound the loop at 3 steps (the cap the posture section documents)",
281
+ price_per_seat_month - seat_cost(3, coverage, cached_prefix_share),
282
+ ))
283
+
284
+ return {
285
+ "cost_per_workflow": round(cost / (workflows_per_day * WORKDAYS_PER_MONTH), 4)
286
+ if workflows_per_day else 0.0,
287
+ "cost_per_seat_month": round(cost, 2),
288
+ "price_per_seat_month": price_per_seat_month,
289
+ "margin_per_seat": round(margin, 2),
290
+ "underwater": margin <= 0,
291
+ "levers": [(reason, round(new_margin, 2)) for reason, new_margin in levers],
292
+ }
fde/decide.py ADDED
@@ -0,0 +1,297 @@
1
+ """What to build for each component, and why.
2
+
3
+ Three rules carry this, and they matter more than the selection mechanism.
4
+
5
+ **The simplest applicable approach wins.** `complexity` orders candidates by
6
+ cost of ownership, not capability. A framework that reaches for the most capable
7
+ option available is the failure this exists to prevent -- and the one that makes
8
+ an engagement expensive to hand over.
9
+
10
+ **Every decision names what it rejected and why.** A recommendation with no
11
+ rejected alternatives has not been made, it has been assumed. It is also the
12
+ half a client actually reads, because it tells them what they are not getting.
13
+
14
+ **Nothing known means nothing decided.** No default, no "probably". The gates
15
+ report the gap and the interview asks.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import json
22
+ from collections.abc import Mapping
23
+ from dataclasses import dataclass, field
24
+ from typing import Any
25
+
26
+ from fde.models.base import Confidence, Evidence
27
+ from fde.models.profile import Profile
28
+ from fde.models.schema import Approach, Reversibility, confidence_sufficient
29
+ from fde.predicate import holds
30
+ from fde.predicate import referenced as _referenced
31
+ from fde.registry import Registry
32
+
33
+
34
+ @dataclass(frozen=True)
35
+ class Rejected:
36
+ id: str
37
+ reason: str
38
+
39
+
40
+ @dataclass
41
+ class Decision:
42
+ component: str
43
+ approach: str | None
44
+ rationale: str
45
+ rejected: list[Rejected] = field(default_factory=list)
46
+ evidence: Evidence | None = None
47
+ confidence: Confidence = Confidence.MEDIUM
48
+ reversibility: Reversibility = Reversibility.MODERATE
49
+
50
+ # How many approaches were on the table at all. One is not the same as
51
+ # "we weighed the options and this won" -- it means the registry offers no
52
+ # alternative, which a client should be told rather than left to assume.
53
+ considered: int = 0
54
+
55
+ @property
56
+ def uncontested(self) -> bool:
57
+ return self.considered == 1
58
+
59
+ def as_tuple(self) -> tuple:
60
+ return (self.component, self.approach)
61
+
62
+
63
+ class Decisions(dict):
64
+ """Component id -> Decision, with a stable identity for the whole set."""
65
+
66
+ def undecided(self) -> list[str]:
67
+ """Components in scope that nothing can currently fill."""
68
+ return sorted(c for c, d in self.items() if not d.approach)
69
+
70
+ def decided(self) -> Decisions:
71
+ return Decisions({c: d for c, d in self.items() if d.approach})
72
+
73
+ def decided_fingerprint(self) -> str:
74
+ return self.decided().fingerprint()
75
+
76
+ def fingerprint(self) -> str:
77
+ """One value standing for the whole architecture.
78
+
79
+ This is what divergence compares: does answering a question change what
80
+ gets built? Built from the decisions themselves rather than from the
81
+ profile, so two different profiles that lead to the same design are
82
+ correctly treated as the same answer.
83
+ """
84
+ payload = json.dumps(
85
+ sorted(d.as_tuple() for d in self.values() if d.approach), separators=(",", ":")
86
+ )
87
+ return hashlib.sha256(payload.encode()).hexdigest()[:16]
88
+
89
+
90
+ def decide_component(
91
+ component: str, values: Mapping[str, Any], registry: Registry
92
+ ) -> Decision | None:
93
+ """Choose an approach for one component, and record what lost."""
94
+ profile = _as_profile(values)
95
+
96
+ candidates = [
97
+ a for a in registry.approaches.values() if not a.components or component in a.components
98
+ ]
99
+ if not candidates:
100
+ return None
101
+
102
+ applicable: list[Approach] = []
103
+ rejected: list[Rejected] = []
104
+ still_askable: list[Approach] = []
105
+
106
+ for approach in candidates:
107
+ blocked = [c for c in approach.avoid_when if holds(c, profile, registry)]
108
+ if blocked:
109
+ rejected.append(
110
+ Rejected(approach.id, f"ruled out by {blocked[0]}")
111
+ )
112
+ continue
113
+ if not any(holds(c, profile, registry) for c in approach.applies_when):
114
+ rejected.append(
115
+ Rejected(approach.id, f"nothing here matches {' or '.join(approach.applies_when)}")
116
+ )
117
+ # Not ruled out -- just not ruled in. If its applies conditions
118
+ # reference something unanswered, an answer could still admit it.
119
+ still_askable.append(approach)
120
+ continue
121
+ applicable.append(approach)
122
+
123
+ if not applicable:
124
+ # Two different situations, and telling them apart matters. When the
125
+ # predicates reference things nobody has answered, more discovery is
126
+ # the remedy. When everything was known and every approach is still
127
+ # ruled out, the facts contradict each other -- and reporting that as
128
+ # "not enough is known" sends somebody to ask more questions that
129
+ # cannot help. Name the culprits instead.
130
+ # Only dimensions whose answer could still admit an approach: the
131
+ # applies conditions of candidates that were not ruled out. Collecting
132
+ # from every candidate's every predicate once told a user to go ask
133
+ # four questions none of which could change the outcome -- the exact
134
+ # misdirection this branch exists to prevent.
135
+ unknowns = sorted({
136
+ dimension
137
+ for approach in still_askable
138
+ for condition in approach.applies_when
139
+ for dimension in _referenced(condition)
140
+ if profile.get(dimension) is None
141
+ })
142
+ if unknowns:
143
+ rationale = (
144
+ f"not enough is known to choose -- "
145
+ f"unanswered: {', '.join(unknowns)}"
146
+ )
147
+ else:
148
+ blocked = "; ".join(f"{r.id}: {r.reason}" for r in rejected)
149
+ rationale = (
150
+ f"everything is known and every approach is ruled out -- "
151
+ f"the facts conflict, and asking more questions cannot help. "
152
+ f"{blocked}"
153
+ )
154
+ return Decision(
155
+ component=component,
156
+ approach=None,
157
+ rationale=rationale,
158
+ rejected=rejected,
159
+ considered=len(candidates),
160
+ )
161
+
162
+ # Simplest first; among equals, whichever more engagements back.
163
+ applicable.sort(key=lambda a: (a.complexity, -_evidence_count(a), a.id))
164
+ winner, losers = applicable[0], applicable[1:]
165
+
166
+ rejected.extend(
167
+ Rejected(a.id, f"{winner.id} is simpler and applies here")
168
+ for a in losers
169
+ )
170
+
171
+ evidence = winner.evidence
172
+ confidence = evidence.confidence if evidence else Confidence.LOW
173
+ reversibility = _reversibility(winner)
174
+
175
+ if not confidence_sufficient(reversibility, confidence):
176
+ confidence = Confidence.HIGH if reversibility is Reversibility.ONE_WAY else confidence
177
+
178
+ rationale = _why(winner, profile, registry)
179
+ if len(candidates) == 1:
180
+ rationale += " -- the only approach registered for this component"
181
+
182
+ return Decision(
183
+ component=component,
184
+ approach=winner.id,
185
+ rationale=rationale,
186
+ rejected=rejected,
187
+ evidence=evidence,
188
+ confidence=confidence,
189
+ reversibility=reversibility,
190
+ considered=len(candidates),
191
+ )
192
+
193
+
194
+ def base_component(component: str) -> str:
195
+ """The registry id behind an instance key: perception:images -> perception.
196
+
197
+ Fan-out gives a component one instance per value of its declared
198
+ dimension; everything that looks a component up in the registry goes
199
+ through here so an instance is never mistaken for a new kind of thing.
200
+ """
201
+ return component.split(":", 1)[0]
202
+
203
+
204
+ def decide_all(
205
+ values: Mapping[str, Any], registry: Registry, components: list[str] | None = None
206
+ ) -> Decisions:
207
+ """Decide every component asked for -- including the ones we cannot.
208
+
209
+ A component decomposition put in scope but decision cannot fill does not
210
+ disappear. It stays, with no approach and a rationale saying why, because a
211
+ component that vanishes between "you need this" and "here is the design" is
212
+ a hole nobody notices until build time.
213
+
214
+ A component that declares fan_out_on fans into one instance per value
215
+ when the engagement carries several -- a claims system taking photos AND
216
+ the policy document gets perception:images and perception:documents, each
217
+ decided by the same rules with that one modality bound. Multi-modality is
218
+ the normaliser's property: downstream components see what perception
219
+ produced, so they decide once.
220
+ """
221
+ wanted = components if components is not None else list(registry.components)
222
+ decisions = Decisions()
223
+ for component in wanted:
224
+ entry = registry.components.get(base_component(component))
225
+ fan = entry.fan_out_on if entry else None
226
+ carried = values.get(fan) if fan else None
227
+ if fan and isinstance(carried, tuple) and len(carried) > 1:
228
+ for modality in carried:
229
+ bound = {**values, fan: modality}
230
+ key = f"{component}:{modality}"
231
+ decision = decide_component(component, bound, registry)
232
+ if decision is None:
233
+ decision = Decision(
234
+ component=key, approach=None,
235
+ rationale="no approach in the registry serves this component",
236
+ considered=0,
237
+ )
238
+ decisions[key] = decision
239
+ continue
240
+ decision = decide_component(component, values, registry)
241
+ if decision is None:
242
+ decision = Decision(
243
+ component=component,
244
+ approach=None,
245
+ rationale="no approach in the registry serves this component",
246
+ considered=0,
247
+ )
248
+ decisions[component] = decision
249
+ return decisions
250
+
251
+
252
+ def architecture_outcome(registry: Registry, components: list[str] | None = None):
253
+ """An outcome function for divergence: what actually gets built.
254
+
255
+ The placeholder measured which dimensions got settled. This measures the
256
+ design, which is the question worth asking -- a question that narrows the
257
+ space but changes nothing is not worth a client's time.
258
+ """
259
+
260
+ def outcome(space) -> str:
261
+ values = {d: space.value(d) for d in space.dimensions() if space.resolved(d)}
262
+ return decide_all(values, registry, components=components).fingerprint()
263
+
264
+ return outcome
265
+
266
+
267
+ # --- helpers -------------------------------------------------------------
268
+
269
+
270
+ def _as_profile(values: Mapping[str, Any]) -> Profile:
271
+ """Decisions are made from resolved values, whether those came from a
272
+ profile or from exploring a hypothetical."""
273
+ if isinstance(values, Profile):
274
+ return values
275
+
276
+ from fde.models.base import Provenance
277
+ from fde.models.fact import Fact
278
+
279
+ profile = Profile()
280
+ profile.ingest(
281
+ [Fact(k, v, Provenance.ARTIFACT) for k, v in values.items() if v is not None]
282
+ )
283
+ return profile
284
+
285
+
286
+ def _evidence_count(approach: Approach) -> int:
287
+ return len(approach.evidence.case_ids) if approach.evidence else 0
288
+
289
+
290
+ def _reversibility(approach: Approach) -> Reversibility:
291
+ # Adapting weights means retraining and re-evaluating to undo.
292
+ return Reversibility.EXPENSIVE if approach.id == "finetune" else Reversibility.MODERATE
293
+
294
+
295
+ def _why(approach: Approach, profile: Profile, registry: Registry) -> str:
296
+ fired = [c for c in approach.applies_when if holds(c, profile, registry)]
297
+ return f"{approach.name}: {'; '.join(fired)}"
fde/decompose.py ADDED
@@ -0,0 +1,54 @@
1
+ """Profile into a component graph.
2
+
3
+ A component is included only when a condition it declares actually fires, and it
4
+ records which one. Spurious components are how a scope doubles between the
5
+ workshop and the statement of work, so "we might need retrieval" is not a
6
+ reason to include retrieval -- and not knowing yet is not a reason either.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+
13
+ from fde.models.profile import Profile
14
+ from fde.models.schema import Component, earliest_cap
15
+ from fde.predicate import holds
16
+ from fde.registry import Registry
17
+
18
+
19
+ @dataclass
20
+ class Included:
21
+ component: Component
22
+
23
+ # The conditions that fired. An FDE asked "why is retrieval in scope?"
24
+ # gets a sentence, not a shrug.
25
+ because: list[str] = field(default_factory=list)
26
+
27
+ @property
28
+ def id(self) -> str:
29
+ return self.component.id
30
+
31
+
32
+ @dataclass
33
+ class ComponentGraph:
34
+ components: dict[str, Included] = field(default_factory=dict)
35
+
36
+ def earliest_cap(self, component_id: str) -> str:
37
+ """Where to look first when this component's quality is capped."""
38
+ return earliest_cap(component_id, {i: v.component for i, v in self.components.items()})
39
+
40
+ def __contains__(self, component_id: str) -> bool:
41
+ return component_id in self.components
42
+
43
+
44
+ def decompose(profile: Profile, registry: Registry) -> ComponentGraph:
45
+ graph = ComponentGraph()
46
+ for component in registry.components.values():
47
+ fired = [
48
+ condition
49
+ for condition in component.required_when
50
+ if holds(condition, profile, registry)
51
+ ]
52
+ if fired:
53
+ graph.components[component.id] = Included(component=component, because=fired)
54
+ return graph