foretoken 0.0.2__tar.gz → 0.0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. foretoken-0.0.4/PKG-INFO +249 -0
  2. foretoken-0.0.4/README.md +205 -0
  3. foretoken-0.0.4/benchmarks/__init__.py +4 -0
  4. foretoken-0.0.4/benchmarks/config/__init__.py +4 -0
  5. foretoken-0.0.4/benchmarks/config/benchmark.py +645 -0
  6. foretoken-0.0.4/benchmarks/config/cli.py +602 -0
  7. foretoken-0.0.4/benchmarks/config/distribution_comparison.py +255 -0
  8. foretoken-0.0.4/benchmarks/config/evaluation.py +218 -0
  9. foretoken-0.0.4/benchmarks/config/video.py +220 -0
  10. foretoken-0.0.4/benchmarks/config/video_cli.py +142 -0
  11. foretoken-0.0.4/benchmarks/datasets/__init__.py +4 -0
  12. foretoken-0.0.4/benchmarks/datasets/conversations.py +497 -0
  13. foretoken-0.0.4/benchmarks/datasets/huggingface.py +209 -0
  14. foretoken-0.0.4/benchmarks/datasets/multimodal.py +4 -0
  15. foretoken-0.0.4/benchmarks/datasets/synthetic.py +184 -0
  16. foretoken-0.0.4/benchmarks/datasets/traces.py +236 -0
  17. foretoken-0.0.4/benchmarks/datasets/video.py +471 -0
  18. foretoken-0.0.4/benchmarks/evaluation.py +78 -0
  19. foretoken-0.0.4/benchmarks/integrations/__init__.py +4 -0
  20. foretoken-0.0.4/benchmarks/integrations/distributions.py +151 -0
  21. foretoken-0.0.4/benchmarks/integrations/evalscope/__init__.py +4 -0
  22. foretoken-0.0.4/benchmarks/integrations/evalscope/evaluation.py +92 -0
  23. foretoken-0.0.4/benchmarks/integrations/evalscope/performance.py +643 -0
  24. foretoken-0.0.4/benchmarks/integrations/evalscope/slo.py +46 -0
  25. foretoken-0.0.4/benchmarks/integrations/hermes.py +4 -0
  26. foretoken-0.0.4/benchmarks/integrations/lm_eval/__init__.py +4 -0
  27. foretoken-0.0.4/benchmarks/integrations/lm_eval/model.py +221 -0
  28. foretoken-0.0.4/benchmarks/integrations/lm_eval/responses.py +117 -0
  29. foretoken-0.0.4/benchmarks/integrations/lm_eval/runner.py +118 -0
  30. foretoken-0.0.4/benchmarks/integrations/openai.py +218 -0
  31. foretoken-0.0.4/benchmarks/integrations/pi.py +4 -0
  32. foretoken-0.0.4/benchmarks/integrations/quality.py +52 -0
  33. foretoken-0.0.4/benchmarks/integrations/streaming.py +32 -0
  34. foretoken-0.0.4/benchmarks/integrations/verifiers.py +4 -0
  35. foretoken-0.0.4/benchmarks/integrations/video.py +502 -0
  36. foretoken-0.0.4/benchmarks/main.py +88 -0
  37. foretoken-0.0.4/benchmarks/model_service.py +402 -0
  38. foretoken-0.0.4/benchmarks/plot.py +59 -0
  39. foretoken-0.0.4/benchmarks/profiling/__init__.py +10 -0
  40. foretoken-0.0.4/benchmarks/profiling/capture.py +134 -0
  41. {foretoken-0.0.2/benchmarks/logger → foretoken-0.0.4/benchmarks/results}/__init__.py +1 -2
  42. foretoken-0.0.4/benchmarks/results/console.py +514 -0
  43. foretoken-0.0.4/benchmarks/results/distribution_comparison.py +324 -0
  44. foretoken-0.0.4/benchmarks/results/distribution_comparison_checkpoint.py +102 -0
  45. foretoken-0.0.4/benchmarks/results/environment.py +126 -0
  46. foretoken-0.0.4/benchmarks/results/evaluation.py +366 -0
  47. foretoken-0.0.4/benchmarks/results/metrics.py +350 -0
  48. foretoken-0.0.4/benchmarks/results/output.py +783 -0
  49. foretoken-0.0.4/benchmarks/results/plots/__init__.py +9 -0
  50. foretoken-0.0.4/benchmarks/results/plots/data.py +327 -0
  51. foretoken-0.0.4/benchmarks/results/plots/figures.py +578 -0
  52. foretoken-0.0.4/benchmarks/results/plots/measurements.py +720 -0
  53. foretoken-0.0.4/benchmarks/results/prometheus.py +269 -0
  54. foretoken-0.0.4/benchmarks/results/replicas.py +458 -0
  55. foretoken-0.0.4/benchmarks/results/timeseries.py +219 -0
  56. foretoken-0.0.4/benchmarks/results/video.py +261 -0
  57. foretoken-0.0.4/benchmarks/results/video_wandb.py +138 -0
  58. foretoken-0.0.4/benchmarks/results/wandb.py +460 -0
  59. {foretoken-0.0.2/benchmarks/metrics → foretoken-0.0.4/benchmarks/runs}/__init__.py +2 -0
  60. foretoken-0.0.2/benchmarks/__init__.py → foretoken-0.0.4/benchmarks/runs/agent.py +2 -0
  61. foretoken-0.0.4/benchmarks/runs/dispatch.py +76 -0
  62. foretoken-0.0.4/benchmarks/runs/distribution_comparison.py +413 -0
  63. foretoken-0.0.4/benchmarks/runs/evaluation.py +177 -0
  64. foretoken-0.0.4/benchmarks/runs/executor.py +444 -0
  65. foretoken-0.0.4/benchmarks/runs/http.py +77 -0
  66. foretoken-0.0.4/benchmarks/runs/slo.py +319 -0
  67. foretoken-0.0.4/benchmarks/runs/trace.py +458 -0
  68. foretoken-0.0.4/benchmarks/runs/video.py +134 -0
  69. foretoken-0.0.4/benchmarks/scoring/__init__.py +4 -0
  70. foretoken-0.0.4/benchmarks/sweeps/__init__.py +4 -0
  71. foretoken-0.0.4/benchmarks/sweeps/core.py +428 -0
  72. foretoken-0.0.4/benchmarks/sweeps/http.py +247 -0
  73. foretoken-0.0.4/benchmarks/sweeps/video.py +170 -0
  74. foretoken-0.0.4/cli/README.md +222 -0
  75. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/__init__.py +8 -3
  76. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/accelerators/_exporter.py +62 -15
  77. foretoken-0.0.4/cli/foretoken/accelerators/config.py +13 -0
  78. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/accelerators/discovery.py +224 -29
  79. foretoken-0.0.4/cli/foretoken/accelerators/metax.py +212 -0
  80. foretoken-0.0.4/cli/foretoken/accelerators/mx-exporter.yaml +119 -0
  81. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/accelerators/nvidia.py +11 -1
  82. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/arguments.py +216 -16
  83. foretoken-0.0.4/cli/foretoken/cluster.py +147 -0
  84. foretoken-0.0.4/cli/foretoken/cluster_build.py +827 -0
  85. foretoken-0.0.4/cli/foretoken/editable.py +1221 -0
  86. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/kubernetes.py +138 -20
  87. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/main.py +99 -44
  88. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/manifest.py +51 -4
  89. foretoken-0.0.4/cli/foretoken/network_sources.py +379 -0
  90. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/observability.py +79 -0
  91. foretoken-0.0.4/cli/foretoken/platform/config.py +347 -0
  92. foretoken-0.0.4/cli/foretoken/platform/helm.py +959 -0
  93. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/platform/helm_client.py +29 -18
  94. foretoken-0.0.4/cli/foretoken/platform/leader_worker.py +136 -0
  95. foretoken-0.0.4/cli/foretoken/platform/lifecycle.py +762 -0
  96. foretoken-0.0.4/cli/foretoken/platform/load_balancer.py +442 -0
  97. foretoken-0.0.4/cli/foretoken/platform/logs.py +279 -0
  98. foretoken-0.0.4/cli/foretoken/platform/model_distribution.py +206 -0
  99. foretoken-0.0.4/cli/foretoken/platform/rdma.py +253 -0
  100. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/platform/types.py +16 -0
  101. foretoken-0.0.4/cli/foretoken/profiling/__init__.py +8 -0
  102. foretoken-0.0.4/cli/foretoken/profiling/capture.py +126 -0
  103. foretoken-0.0.4/cli/foretoken/profiling/reader.py +219 -0
  104. foretoken-0.0.4/cli/foretoken/profiling/storage.py +778 -0
  105. foretoken-0.0.4/cli/foretoken/profiling/viewer.html +512 -0
  106. foretoken-0.0.4/cli/foretoken/profiling/viewer.py +302 -0
  107. foretoken-0.0.4/cli/foretoken/progress.py +256 -0
  108. foretoken-0.0.4/cli/foretoken/source.py +499 -0
  109. foretoken-0.0.4/cli/foretoken/storage.py +393 -0
  110. foretoken-0.0.4/foretoken.egg-info/PKG-INFO +249 -0
  111. foretoken-0.0.4/foretoken.egg-info/SOURCES.txt +120 -0
  112. {foretoken-0.0.2 → foretoken-0.0.4}/foretoken.egg-info/requires.txt +7 -8
  113. {foretoken-0.0.2 → foretoken-0.0.4}/pyproject.toml +27 -21
  114. foretoken-0.0.2/PKG-INFO +0 -201
  115. foretoken-0.0.2/README.md +0 -168
  116. foretoken-0.0.2/benchmarks/arguments.py +0 -444
  117. foretoken-0.0.2/benchmarks/client/__init__.py +0 -2
  118. foretoken-0.0.2/benchmarks/client/openai_client.py +0 -148
  119. foretoken-0.0.2/benchmarks/config.py +0 -358
  120. foretoken-0.0.2/benchmarks/deployment/__init__.py +0 -9
  121. foretoken-0.0.2/benchmarks/deployment/discovery.py +0 -166
  122. foretoken-0.0.2/benchmarks/deployment/lifecycle.py +0 -111
  123. foretoken-0.0.2/benchmarks/logger/cli.py +0 -27
  124. foretoken-0.0.2/benchmarks/logger/wandb.py +0 -261
  125. foretoken-0.0.2/benchmarks/main.py +0 -74
  126. foretoken-0.0.2/benchmarks/metrics/aggregator.py +0 -161
  127. foretoken-0.0.2/benchmarks/report/__init__.py +0 -2
  128. foretoken-0.0.2/benchmarks/report/pareto.py +0 -145
  129. foretoken-0.0.2/benchmarks/report/summary.py +0 -156
  130. foretoken-0.0.2/benchmarks/runner/__init__.py +0 -2
  131. foretoken-0.0.2/benchmarks/runner/base.py +0 -284
  132. foretoken-0.0.2/benchmarks/runner/multi_dataset.py +0 -105
  133. foretoken-0.0.2/benchmarks/runner/run_benchmark.py +0 -79
  134. foretoken-0.0.2/benchmarks/runner/run_spec.py +0 -22
  135. foretoken-0.0.2/benchmarks/runner/select_runner.py +0 -27
  136. foretoken-0.0.2/benchmarks/runner/sweep.py +0 -141
  137. foretoken-0.0.2/benchmarks/runner/trace_runner.py +0 -225
  138. foretoken-0.0.2/benchmarks/storage/__init__.py +0 -2
  139. foretoken-0.0.2/benchmarks/storage/result_writer.py +0 -37
  140. foretoken-0.0.2/benchmarks/utils/__init__.py +0 -2
  141. foretoken-0.0.2/benchmarks/utils/bench_params.py +0 -223
  142. foretoken-0.0.2/benchmarks/workload/__init__.py +0 -2
  143. foretoken-0.0.2/benchmarks/workload/hf_dataset.py +0 -154
  144. foretoken-0.0.2/benchmarks/workload/loader.py +0 -205
  145. foretoken-0.0.2/benchmarks/workload/random_dataset.py +0 -273
  146. foretoken-0.0.2/benchmarks/workload/trace_loader.py +0 -275
  147. foretoken-0.0.2/benchmarks/workload/trace_workload.py +0 -109
  148. foretoken-0.0.2/cli/README.md +0 -174
  149. foretoken-0.0.2/cli/foretoken/accelerators/metax.py +0 -61
  150. foretoken-0.0.2/cli/foretoken/platform/config.py +0 -119
  151. foretoken-0.0.2/cli/foretoken/platform/helm.py +0 -422
  152. foretoken-0.0.2/cli/foretoken/platform/lifecycle.py +0 -345
  153. foretoken-0.0.2/cli/foretoken/source.py +0 -172
  154. foretoken-0.0.2/foretoken.egg-info/PKG-INFO +0 -201
  155. foretoken-0.0.2/foretoken.egg-info/SOURCES.txt +0 -65
  156. {foretoken-0.0.2 → foretoken-0.0.4}/LICENSE +0 -0
  157. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/accelerators/__init__.py +0 -0
  158. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/platform/__init__.py +0 -0
  159. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/platform/gateway.py +0 -0
  160. {foretoken-0.0.2 → foretoken-0.0.4}/cli/foretoken/platform/gateway_resources.py +0 -0
  161. {foretoken-0.0.2 → foretoken-0.0.4}/foretoken.egg-info/dependency_links.txt +0 -0
  162. {foretoken-0.0.2 → foretoken-0.0.4}/foretoken.egg-info/entry_points.txt +0 -0
  163. {foretoken-0.0.2 → foretoken-0.0.4}/foretoken.egg-info/top_level.txt +0 -0
  164. {foretoken-0.0.2 → foretoken-0.0.4}/setup.cfg +0 -0
@@ -0,0 +1,249 @@
1
+ Metadata-Version: 2.4
2
+ Name: foretoken
3
+ Version: 0.0.4
4
+ Summary: Command-line tools for installing and operating Foretoken
5
+ Author: Foretoken contributors
6
+ License-Expression: Apache-2.0
7
+ Requires-Python: >=3.11
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: datasets>=2.14
11
+ Requires-Dist: evalscope[perf]==1.11.1
12
+ Requires-Dist: httpx>=0.27
13
+ Requires-Dist: huggingface_hub>=0.23
14
+ Requires-Dist: ijson>=3.2
15
+ Requires-Dist: lm-eval[api]==0.4.13
16
+ Requires-Dist: matplotlib>=3.7
17
+ Requires-Dist: numpy>=1.24
18
+ Requires-Dist: openai>=1.40
19
+ Requires-Dist: packaging>=24.0
20
+ Requires-Dist: PyYAML>=6.0
21
+ Requires-Dist: tqdm>=4.65
22
+ Requires-Dist: transformers>=4.40
23
+ Requires-Dist: wandb>=0.19.10
24
+ Provides-Extra: dev
25
+ Requires-Dist: grafana-foundation-sdk==1769699379!11.6.0; extra == "dev"
26
+ Dynamic: license-file
27
+
28
+ <!--
29
+ SPDX-License-Identifier: Apache-2.0
30
+ SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
31
+ -->
32
+
33
+ # Foretoken command-line tool
34
+
35
+ English | [简体中文](README_zh.md)
36
+
37
+ The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend URLs, and runs benchmarks through one `foretoken` entry point.
38
+
39
+ ## Before you start
40
+
41
+ You need Python 3.11 or later, an active Kubernetes context, `kubectl`, and Helm. GPU nodes must already have their vendor driver and Kubernetes device plugin.
42
+
43
+ ## Install the command-line tool
44
+
45
+ Install the published command-line tool with pip:
46
+
47
+ ```bash
48
+ pip install foretoken
49
+
50
+ # From a source checkout:
51
+ # pip install -e .
52
+ ```
53
+
54
+ Or create and activate a virtual environment with uv:
55
+
56
+ ```bash
57
+ uv venv
58
+ source .venv/bin/activate
59
+ uv pip install foretoken
60
+ ```
61
+
62
+ Run `foretoken --version` to check the installed CLI version.
63
+
64
+ ## Create a local cluster
65
+
66
+ On a Linux GPU host with Docker, NVIDIA Container Toolkit, and k3d installed:
67
+
68
+ ```bash
69
+ # Name the local cluster and use GPU index 0 from nvidia-smi.
70
+ # To use two GPUs, pass --gpus 0,1.
71
+ foretoken cluster create k3d --name foretoken-dev --gpus 0
72
+ ```
73
+
74
+ For a local kind development cluster:
75
+
76
+ ```bash
77
+ foretoken cluster create kind --name foretoken-dev
78
+ ```
79
+
80
+ Remove a cluster created by the CLI with:
81
+
82
+ ```bash
83
+ foretoken cluster delete k3d --name foretoken-dev
84
+ ```
85
+
86
+ ## Install the Kubernetes platform
87
+
88
+ `foretoken install` installs the Foretoken CRDs and controller in the active Kubernetes context. Platform resources use the `foretoken-platform` namespace. The command also configures monitoring and, in Gateway mode, the Gateway resources. Deploy model services separately with `foretoken deploy`.
89
+
90
+ ### Default installation
91
+
92
+ The default uses release images and local access through a `LoadBalancer` Service:
93
+
94
+ ```bash
95
+ foretoken install
96
+ ```
97
+
98
+ Installation selects the NVIDIA or MetaX runtime and automatically reuses or installs LeaderWorkerSet and the shared RDMA device plugin. Explicit runtime settings in `--values` take precedence; in a mixed-GPU cluster, select a resource with `runtime.vllm.gpu.resourceName` or restrict the nodes with `runtime.vllm.gpu.nodeSelector`.
99
+
100
+ Log collection and persistence are enabled by default. See [Observability](../observability/README.md) for configuration, log queries, dashboards, and alerts.
101
+
102
+ ### Gateway mode
103
+
104
+ Gateway mode creates a dedicated `GatewayClass` and `Gateway`, installing Envoy Gateway if no compatible controller is available:
105
+
106
+ ```bash
107
+ foretoken install --frontend-mode gateway
108
+ ```
109
+
110
+ With another Gateway Controller, reuse a Gateway managed by that controller:
111
+
112
+ ```bash
113
+ foretoken install \
114
+ --frontend-mode gateway \
115
+ --gateway-name inference-gateway \
116
+ --gateway-namespace gateway-system
117
+ ```
118
+
119
+ Add `--gateway-section-name LISTENER` only when more than one listener matches.
120
+
121
+ ### Current source
122
+
123
+ Build and install from the repository root. The cluster needs a default StorageClass for compiler caches; see the [source deployment guide](../docs/custom-deployment.md) for storage overrides.
124
+
125
+ ```bash
126
+ foretoken install -e .
127
+ ```
128
+
129
+ This builds the platform in dedicated Pods and binds the checkout to the target cluster.
130
+
131
+ After editing it, use `foretoken deploy` to [redeploy source changes](../docs/custom-deployment.md#deploy-and-update-code). Use `--engine-source PATH` to also bind a [vLLM engine checkout](../docs/custom-deployment.md#edit-an-inference-engine).
132
+
133
+ A standard active kind or k3d context loads the built images directly into its nodes. Other Kubernetes contexts need a registry reachable by the Build Pods and nodes. For an internal registry without authentication:
134
+
135
+ ```bash
136
+ foretoken install -e . --registry registry.example.com:5000/foretoken
137
+ ```
138
+
139
+ If the registry requires authentication, follow the [source deployment guide](../docs/kubernetes-deployment.md) before installation.
140
+
141
+ ### Model distribution
142
+
143
+ To share public model downloads between nodes through Dragonfly, save this in `deploy/platform-values.yaml`:
144
+
145
+ ```yaml
146
+ modelDistribution:
147
+ dragonfly:
148
+ enabled: true
149
+ ```
150
+
151
+ For a published platform installation, apply the values with:
152
+
153
+ ```bash
154
+ foretoken install --values deploy/platform-values.yaml
155
+ ```
156
+
157
+ For a source installation, run `foretoken install -e . --values deploy/platform-values.yaml` from the repository root, retaining the original registry and engine-source options.
158
+
159
+ Installation prepares Dragonfly or reuses an existing installation. Models that require authentication and custom model endpoints download directly from their provider. To select a particular Dragonfly Helm release, set `existingRelease: {name: dragonfly, namespace: dragonfly-system}` under `modelDistribution.dragonfly`.
160
+
161
+ On NVIDIA clusters with RDMA, ModelExpress can load weights from running replicas. Add this alongside `dragonfly` to enable it:
162
+
163
+ ```yaml
164
+ modelDistribution:
165
+ modelexpress:
166
+ enabled: true
167
+ ```
168
+
169
+ Automatic weight transfer uses remote models with a persistent cache, data parallelism of one, and fixed expert placement. An explicit `load-format` remains unchanged. Each GPU worker selects a nearby available RDMA interface; replicas without a compatible source load the prepared files.
170
+
171
+ Reapply the installation command after changing either setting. Set `enabled: false` to disable it. `foretoken uninstall` removes managed Dragonfly resources once no workloads use them; reused installations are retained.
172
+
173
+ ### Installation options
174
+
175
+ Use `--values` only to override platform image, runtime, or hardware settings. Without an override, installation compares supported public sources for default platform images and OCI charts. Use `--oci-registry` to select a registry explicitly; image references supplied through values remain unchanged. Source selection runs on the CLI host, so the selected registry must also be reachable from the cluster nodes.
176
+
177
+ Model services are reached through an IP address outside the cluster. k3d, k3s, and cloud clusters assign one automatically. Clusters built with kubeadm, RKE2, or kubespray have no address assignment by default, so installation there ends with `LoadBalancer support Not verified`. Give Foretoken a range of unused addresses in the nodes' subnet, confirmed with the cluster administrator, and it assigns them to services:
178
+
179
+ ```yaml
180
+ loadBalancer:
181
+ managedAddresses:
182
+ - 192.168.1.240-192.168.1.250
183
+ ```
184
+
185
+ ## Deploy and operate model services
186
+
187
+ Run from the repository checkout prepared in the [Quick Start](../README.md). Deploy one frontend and all models rendered by a Kustomize root.
188
+
189
+ See the [multi-model example](../examples/multi-model-quickstart/README.md) for resources and [model storage](../docs/model-storage.md) for directory or PVC configuration. Use `examples/quickstart` for a single model.
190
+
191
+ ```bash
192
+ foretoken deploy examples/multi-model-quickstart --timeout 20m
193
+ ```
194
+
195
+ The command applies the configuration, shows service status, and streams Pod and container logs with source prefixes while waiting. It exits when every service reports Ready and its selected alerts are configured. Without `--timeout`, it waits up to ten minutes. Configure service alerts in the Kustomize deployment; see [service observability](../examples/observability/README.md).
196
+
197
+ Inspect the same deployment without applying it:
198
+
199
+ ```bash
200
+ foretoken status examples/multi-model-quickstart
201
+ ```
202
+
203
+ Inspect every Foretoken service in a namespace. With `--watch`, follow service state changes and Pod/container logs until Ctrl+C:
204
+
205
+ ```bash
206
+ foretoken status -n foretoken-multi-model-demo
207
+ foretoken status -n foretoken-multi-model-demo --watch
208
+ ```
209
+
210
+ Resolve the public frontend URL after deployment:
211
+
212
+ ```bash
213
+ FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/multi-model-quickstart)"
214
+ ```
215
+
216
+ For an HTTP Gateway, resolve its request `Host` separately:
217
+
218
+ ```bash
219
+ FORETOKEN_REQUEST_HOST="$(foretoken endpoint examples/multi-model-quickstart --host)"
220
+ ```
221
+
222
+ `--host` returns the host and optional port for direct access, or the configured routing hostname for an HTTP Gateway. `foretoken endpoint` waits for the LoadBalancer or Gateway address; use `foretoken deploy` to wait for the services to become ready.
223
+
224
+ ## Measure serving performance
225
+
226
+ Use `foretoken perf` to measure response latency and request or token throughput. Pass a Kustomize directory, or `--url` with `--model` for an existing endpoint. Choose a workload in [Performance examples](../benchmarks/docs/perf/README.md).
227
+
228
+ ## Evaluate and compare models
229
+
230
+ Use `foretoken eval` to score model answers with lm-evaluation-harness or EvalScope. It accepts the same service selection options; task and scoring parameters use the selected framework's syntax. See [Quality evaluation](../benchmarks/docs/eval/README.md). Add `--reference` to [compare a candidate's probabilities against a reference](../benchmarks/docs/eval/distribution-comparison.md).
231
+
232
+ ## Export figures
233
+
234
+ Use `--output local,wandb,plot` with a benchmark, or `foretoken plot RESULT_DIR` to redraw a saved run or sweep without running inference. Export options and comparisons are in [parameter sweeps](../benchmarks/docs/perf/sweep.md).
235
+
236
+ ## Find execution bottlenecks
237
+
238
+ Add `--profile` to `foretoken deploy` or `foretoken perf` to record CPU/GPU execution, then browse captures with `foretoken profile view`. Setup and commands are in [Profiling](../benchmarks/docs/profile/README.md).
239
+
240
+ ## Clean up
241
+
242
+ Delete the deployed services before uninstalling the platform:
243
+
244
+ ```bash
245
+ foretoken delete examples/multi-model-quickstart
246
+ foretoken uninstall
247
+ ```
248
+
249
+ CRDs and reused cluster components are retained. Managed LeaderWorkerSet and MetalLB controllers are also retained while workloads still depend on them.
@@ -0,0 +1,205 @@
1
+ # Foretoken
2
+
3
+ English | [简体中文](README_zh.md)
4
+
5
+ Foretoken is a generative inference orchestration framework built for SLO/SLA targets and heterogeneous accelerators.
6
+
7
+ Built on inference engines such as vLLM and SGLang, Foretoken organizes multiple generation instances into a cluster service for request routing, autoscaling, instance management, and benchmarking.
8
+ We aim to turn an inference cluster into a token factory that continuously converts compute into tokens while meeting latency and quality requirements.
9
+
10
+ ## When to Use Foretoken
11
+
12
+ - Serve one or more models across multiple GPUs or nodes.
13
+ - Route requests based on load, queue depth, or KV cache state.
14
+ - Autoscale inference instances based on traffic and SLO targets.
15
+ - Compare aggregated serving, Prefill/Decode disaggregation, and different parallelism strategies.
16
+ - Use the same orchestration stack across NVIDIA and MetaX accelerators.
17
+
18
+ If you only need to serve a single model on one GPU, using an inference engine such as vLLM directly is usually enough.
19
+
20
+ ## Features and Status
21
+
22
+ | Feature | Description | Status |
23
+ |---|---|---|
24
+ | [Evaluation](benchmarks/README.md) | Measure service performance and model quality | In development |
25
+ | [Profiling](benchmarks/docs/profile/README.md) | Capture PyTorch, NVIDIA Nsight Systems, or MetaX mcTracer timelines for a model service | In development |
26
+ | Hardware support | Common interfaces for device capabilities, runtimes, communication, and metrics; see [MetaX deployment](docs/metax-deployment.md) | In development |
27
+ | Request routing | Select instances based on load, queues, KV reuse, and service levels | Research |
28
+ | Distributed inference | Aggregated serving, Prefill/Decode disaggregation, and WideEP parallelism | Research |
29
+ | Control plane | Model services, replica management, autoscaling, updates, and failure recovery | In development |
30
+ | [Observability](observability/README.md) | Collect metrics and persistent service logs, evaluate alerts, and inspect the system Dashboard | In development |
31
+
32
+ ## Quick Start
33
+
34
+ Choose the deployment path before running the common steps:
35
+
36
+ | Situation | Guide |
37
+ |---|---|
38
+ | Single-host local deployment | [k3d deployment](docs/k3d-deployment.md) · [kind deployment](docs/kind-deployment.md) |
39
+ | Kubernetes deployment with K3s, RKE2, KubeSphere, cloud, or another cluster | [Kubernetes deployment](docs/kubernetes-deployment.md) |
40
+ | MetaX GPU deployment | [MetaX deployment](docs/metax-deployment.md) |
41
+
42
+ The steps below use k3d as the example.
43
+
44
+ ### 1. Get the examples and install the command-line tool
45
+
46
+ ```bash
47
+ git clone https://github.com/shiweijiezero/foretoken.git
48
+ cd foretoken
49
+ pip install -e .
50
+
51
+ # For the published CLI instead:
52
+ # pip install foretoken
53
+ ```
54
+
55
+ ### 2. Install the Kubernetes platform
56
+
57
+ Create a local k3d cluster named `foretoken-dev` and install Foretoken:
58
+
59
+ ```bash
60
+ # Use GPU index 0 from nvidia-smi. To use two GPUs, pass --gpus 0,1.
61
+ foretoken cluster create k3d --name foretoken-dev --gpus 0
62
+
63
+ # Build from the current source checkout:
64
+ foretoken install -e .
65
+
66
+ # Use published images instead:
67
+ # foretoken install
68
+ ```
69
+
70
+ The host must have Docker, NVIDIA Container Toolkit, k3d, kubectl, and Helm, and your user must be able to run `docker info` without `sudo`. Install host dependencies separately when needed. For kind or an existing Kubernetes cluster, use the corresponding guide in the table above.
71
+
72
+ ### 3. Deploy the Quick Start
73
+
74
+ ```bash
75
+ foretoken deploy examples/quickstart --timeout 20m
76
+ ```
77
+
78
+ This example deploys one frontend service and one `Qwen/Qwen3-0.6B` model replica. The model requests 1 GPU, 4 CPU, and 48 GiB memory, with limits of 8 CPU and 64 GiB. The example uses the repository-root `./data` directory for model files and runtime cache. More deployments are available in [`examples/`](examples/).
79
+
80
+ ### 4. Send a test request
81
+
82
+ ```bash
83
+ FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/quickstart)"
84
+
85
+ curl --fail-with-body --no-buffer \
86
+ "$FORETOKEN_FRONTEND_URL/v1/chat/completions" \
87
+ -H "Content-Type: application/json" \
88
+ -d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
89
+ ```
90
+
91
+ ### Iterate on source
92
+
93
+ After editing the checkout, run the same deploy command again to apply the change without recreating the cluster or manually importing the runtime image:
94
+
95
+ ```bash
96
+ foretoken deploy examples/quickstart --timeout 20m
97
+ ```
98
+
99
+ Python, Triton, Rust, CUDA, C/C++, and vLLM source changes use the cluster build caches and reuse the runtime environment when dependencies and startup code are unchanged. See [Deploy Foretoken from Source](docs/custom-deployment.md) for engine checkouts and runtime changes.
100
+
101
+ ### 5. Evaluate and profile the service
102
+
103
+ The examples save results locally and to W&B. Run `wandb login` once before using W&B.
104
+
105
+ #### Performance: latency and throughput
106
+
107
+ ```bash
108
+ foretoken perf examples/quickstart --num-prompts 20 --output local,wandb
109
+ ```
110
+
111
+ Read request success, latency, and throughput in the summary. [Performance examples](benchmarks/docs/perf/README.md) cover other workloads and load settings.
112
+
113
+ #### Quality: score model answers
114
+
115
+ ```bash
116
+ foretoken eval examples/quickstart \
117
+ --evaluator lm-eval --tasks gsm8k --limit 100 --output local,wandb
118
+ ```
119
+
120
+ This scores 100 GSM8K math problems. See [Quality evaluation](benchmarks/docs/eval/README.md) for EvalScope, task parameters, and saved scores.
121
+
122
+ #### Profiling: inspect execution bottlenecks
123
+
124
+ Use a source-installed CLI and platform for profiling, as described in the [profiling guide](benchmarks/docs/profile/README.md). The Quick Start already configures persistent capture storage.
125
+
126
+ ```bash
127
+ foretoken perf examples/quickstart \
128
+ --profile --profile-engine pytorch --profile-duration 15s \
129
+ --num-prompts 2 --max-tokens 128 --output local,wandb
130
+ foretoken profile view
131
+ ```
132
+
133
+ Open the printed URL to inspect the capture. Press Ctrl+C to close the viewer; the model service remains running.
134
+
135
+ ## Gateway Mode
136
+
137
+ Gateway mode provides a shared entry point through Kubernetes Gateway and a hostname. It suits clusters that already use Gateway or manage external traffic centrally.
138
+
139
+ Add the public hostname under `spec` in `examples/quickstart/frontend.yaml`:
140
+
141
+ ```yaml
142
+ spec:
143
+ hostname: foretoken.example.com
144
+ ```
145
+
146
+ Then run:
147
+
148
+ ```bash
149
+ # Install the platform in Gateway mode
150
+ foretoken install --frontend-mode gateway
151
+ # For a source-installed platform:
152
+ # foretoken install -e . --frontend-mode gateway
153
+
154
+ # Deploy the Quick Start
155
+ foretoken deploy examples/quickstart --timeout 20m
156
+
157
+ # Resolve the Gateway address and request hostname
158
+ FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/quickstart)"
159
+ FORETOKEN_REQUEST_HOST="$(foretoken endpoint examples/quickstart --host)"
160
+
161
+ # Send a test request
162
+ curl --fail-with-body --no-buffer \
163
+ "$FORETOKEN_FRONTEND_URL/v1/chat/completions" \
164
+ -H "Host: $FORETOKEN_REQUEST_HOST" \
165
+ -H "Content-Type: application/json" \
166
+ -d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
167
+ ```
168
+
169
+ The command installs Envoy Gateway when needed. See the [command-line tool guide](cli/README.md) to reuse an existing Gateway or select a listener.
170
+
171
+ ## Stop and Uninstall
172
+
173
+ ```bash
174
+ # Delete the Quick Start resources, including its namespace and runtime cache PVC
175
+ foretoken delete examples/quickstart
176
+
177
+ # Uninstall the Foretoken platform
178
+ foretoken uninstall
179
+ ```
180
+
181
+ The uninstall command preserves Foretoken CRDs, log storage, and reused cluster components. It removes the platform and the monitoring or Gateway resources managed by the command-line tool.
182
+
183
+ ## Related Projects
184
+
185
+ - [vLLM](https://github.com/vllm-project/vllm)
186
+ - [NVIDIA Dynamo](https://github.com/ai-dynamo/dynamo)
187
+ - [llm-d](https://github.com/llm-d/llm-d)
188
+ - [AIBrix](https://github.com/vllm-project/aibrix)
189
+ - [vLLM Production Stack](https://github.com/vllm-project/production-stack)
190
+
191
+ ## Contributing
192
+
193
+ Contributions of all kinds are welcome, including code, documentation, tests, design discussions, issue reports, and improvements to deployment, hardware, benchmarking, routing, and autoscaling.
194
+ Performance-related changes should include the test setup, raw results, and reproducible commands.
195
+ See [Contributing to Foretoken](CONTRIBUTING.md) for development principles, collaboration expectations, and the pull request workflow.
196
+
197
+ Thank you to everyone who has contributed to Foretoken.
198
+
199
+ <a href="https://github.com/shiweijiezero/foretoken/graphs/contributors">
200
+ <img src="https://contrib.rocks/image?repo=shiweijiezero/foretoken" width="256" alt="Foretoken contributors" />
201
+ </a>
202
+
203
+ ## License
204
+
205
+ This project is licensed under the [Apache License 2.0](LICENSE).
@@ -0,0 +1,4 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
3
+
4
+ """Foretoken benchmark command family; each domain owns an independent lifecycle."""
@@ -0,0 +1,4 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
3
+
4
+ """Benchmark configuration types and command-line parsing."""