foretoken 0.0.3__tar.gz → 0.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- foretoken-0.0.4/PKG-INFO +249 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/README.md +64 -28
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/config/benchmark.py +240 -49
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/config/cli.py +275 -121
- foretoken-0.0.4/benchmarks/config/distribution_comparison.py +255 -0
- foretoken-0.0.4/benchmarks/config/evaluation.py +218 -0
- foretoken-0.0.4/benchmarks/config/video.py +220 -0
- foretoken-0.0.4/benchmarks/config/video_cli.py +142 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/conversations.py +128 -54
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/huggingface.py +16 -10
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/synthetic.py +27 -10
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/traces.py +4 -2
- foretoken-0.0.4/benchmarks/datasets/video.py +471 -0
- foretoken-0.0.4/benchmarks/evaluation.py +78 -0
- foretoken-0.0.4/benchmarks/integrations/distributions.py +151 -0
- {foretoken-0.0.3/benchmarks/profiling → foretoken-0.0.4/benchmarks/integrations/evalscope}/__init__.py +1 -1
- foretoken-0.0.4/benchmarks/integrations/evalscope/evaluation.py +92 -0
- foretoken-0.0.3/benchmarks/integrations/evalscope.py → foretoken-0.0.4/benchmarks/integrations/evalscope/performance.py +226 -262
- foretoken-0.0.4/benchmarks/integrations/evalscope/slo.py +46 -0
- foretoken-0.0.4/benchmarks/integrations/lm_eval/__init__.py +4 -0
- foretoken-0.0.4/benchmarks/integrations/lm_eval/model.py +221 -0
- foretoken-0.0.4/benchmarks/integrations/lm_eval/responses.py +117 -0
- foretoken-0.0.4/benchmarks/integrations/lm_eval/runner.py +118 -0
- foretoken-0.0.4/benchmarks/integrations/openai.py +218 -0
- foretoken-0.0.4/benchmarks/integrations/quality.py +52 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/integrations/streaming.py +1 -1
- foretoken-0.0.4/benchmarks/integrations/video.py +502 -0
- foretoken-0.0.4/benchmarks/main.py +88 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/model_service.py +155 -25
- foretoken-0.0.4/benchmarks/plot.py +59 -0
- foretoken-0.0.4/benchmarks/profiling/__init__.py +10 -0
- foretoken-0.0.4/benchmarks/profiling/capture.py +134 -0
- foretoken-0.0.4/benchmarks/results/console.py +514 -0
- foretoken-0.0.4/benchmarks/results/distribution_comparison.py +324 -0
- foretoken-0.0.4/benchmarks/results/distribution_comparison_checkpoint.py +102 -0
- foretoken-0.0.4/benchmarks/results/environment.py +126 -0
- foretoken-0.0.4/benchmarks/results/evaluation.py +366 -0
- foretoken-0.0.4/benchmarks/results/metrics.py +350 -0
- foretoken-0.0.4/benchmarks/results/output.py +783 -0
- foretoken-0.0.4/benchmarks/results/plots/__init__.py +9 -0
- foretoken-0.0.4/benchmarks/results/plots/data.py +327 -0
- foretoken-0.0.4/benchmarks/results/plots/figures.py +578 -0
- foretoken-0.0.4/benchmarks/results/plots/measurements.py +720 -0
- foretoken-0.0.4/benchmarks/results/prometheus.py +269 -0
- foretoken-0.0.4/benchmarks/results/replicas.py +458 -0
- foretoken-0.0.4/benchmarks/results/timeseries.py +219 -0
- foretoken-0.0.4/benchmarks/results/video.py +261 -0
- foretoken-0.0.4/benchmarks/results/video_wandb.py +138 -0
- foretoken-0.0.4/benchmarks/results/wandb.py +460 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/runs/__init__.py +1 -1
- foretoken-0.0.4/benchmarks/runs/dispatch.py +76 -0
- foretoken-0.0.4/benchmarks/runs/distribution_comparison.py +413 -0
- foretoken-0.0.4/benchmarks/runs/evaluation.py +177 -0
- foretoken-0.0.4/benchmarks/runs/executor.py +444 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/runs/http.py +22 -23
- foretoken-0.0.4/benchmarks/runs/slo.py +319 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/runs/trace.py +169 -73
- foretoken-0.0.4/benchmarks/runs/video.py +134 -0
- foretoken-0.0.4/benchmarks/sweeps/__init__.py +4 -0
- foretoken-0.0.4/benchmarks/sweeps/core.py +428 -0
- foretoken-0.0.4/benchmarks/sweeps/http.py +247 -0
- foretoken-0.0.4/benchmarks/sweeps/video.py +170 -0
- foretoken-0.0.4/cli/README.md +222 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/__init__.py +8 -3
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/accelerators/_exporter.py +62 -15
- foretoken-0.0.4/cli/foretoken/accelerators/config.py +13 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/accelerators/discovery.py +224 -29
- foretoken-0.0.4/cli/foretoken/accelerators/metax.py +212 -0
- foretoken-0.0.4/cli/foretoken/accelerators/mx-exporter.yaml +119 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/accelerators/nvidia.py +10 -3
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/arguments.py +201 -13
- foretoken-0.0.4/cli/foretoken/cluster.py +147 -0
- foretoken-0.0.4/cli/foretoken/cluster_build.py +827 -0
- foretoken-0.0.4/cli/foretoken/editable.py +1221 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/kubernetes.py +114 -18
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/main.py +96 -40
- foretoken-0.0.4/cli/foretoken/network_sources.py +379 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/observability.py +79 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/config.py +130 -8
- foretoken-0.0.4/cli/foretoken/platform/helm.py +959 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/helm_client.py +9 -4
- foretoken-0.0.4/cli/foretoken/platform/leader_worker.py +136 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/lifecycle.py +291 -66
- foretoken-0.0.4/cli/foretoken/platform/logs.py +279 -0
- foretoken-0.0.4/cli/foretoken/platform/model_distribution.py +206 -0
- foretoken-0.0.4/cli/foretoken/platform/rdma.py +253 -0
- foretoken-0.0.4/cli/foretoken/profiling/__init__.py +8 -0
- foretoken-0.0.4/cli/foretoken/profiling/capture.py +126 -0
- foretoken-0.0.4/cli/foretoken/profiling/reader.py +219 -0
- foretoken-0.0.4/cli/foretoken/profiling/storage.py +778 -0
- foretoken-0.0.4/cli/foretoken/profiling/viewer.html +512 -0
- foretoken-0.0.4/cli/foretoken/profiling/viewer.py +302 -0
- foretoken-0.0.4/cli/foretoken/progress.py +256 -0
- foretoken-0.0.4/cli/foretoken/source.py +499 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/storage.py +43 -9
- foretoken-0.0.4/foretoken.egg-info/PKG-INFO +249 -0
- foretoken-0.0.4/foretoken.egg-info/SOURCES.txt +120 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/foretoken.egg-info/requires.txt +4 -5
- {foretoken-0.0.3 → foretoken-0.0.4}/pyproject.toml +16 -8
- foretoken-0.0.3/PKG-INFO +0 -184
- foretoken-0.0.3/benchmarks/datasets/multi_dataset.py +0 -224
- foretoken-0.0.3/benchmarks/integrations/openai.py +0 -135
- foretoken-0.0.3/benchmarks/main.py +0 -67
- foretoken-0.0.3/benchmarks/results/console.py +0 -315
- foretoken-0.0.3/benchmarks/results/metrics.py +0 -157
- foretoken-0.0.3/benchmarks/results/output.py +0 -309
- foretoken-0.0.3/benchmarks/results/pareto.py +0 -151
- foretoken-0.0.3/benchmarks/results/timeseries.py +0 -104
- foretoken-0.0.3/benchmarks/results/wandb.py +0 -302
- foretoken-0.0.3/benchmarks/runs/evaluation.py +0 -4
- foretoken-0.0.3/benchmarks/runs/sweep.py +0 -386
- foretoken-0.0.3/cli/README.md +0 -157
- foretoken-0.0.3/cli/foretoken/accelerators/metax.py +0 -64
- foretoken-0.0.3/cli/foretoken/platform/helm.py +0 -529
- foretoken-0.0.3/cli/foretoken/source.py +0 -174
- foretoken-0.0.3/foretoken.egg-info/PKG-INFO +0 -184
- foretoken-0.0.3/foretoken.egg-info/SOURCES.txt +0 -67
- {foretoken-0.0.3 → foretoken-0.0.4}/LICENSE +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/config/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/datasets/multimodal.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/integrations/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/integrations/hermes.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/integrations/pi.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/integrations/verifiers.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/results/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/runs/agent.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/benchmarks/scoring/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/accelerators/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/manifest.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/__init__.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/gateway.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/gateway_resources.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/load_balancer.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/cli/foretoken/platform/types.py +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/foretoken.egg-info/dependency_links.txt +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/foretoken.egg-info/entry_points.txt +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/foretoken.egg-info/top_level.txt +0 -0
- {foretoken-0.0.3 → foretoken-0.0.4}/setup.cfg +0 -0
foretoken-0.0.4/PKG-INFO
ADDED
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: foretoken
|
|
3
|
+
Version: 0.0.4
|
|
4
|
+
Summary: Command-line tools for installing and operating Foretoken
|
|
5
|
+
Author: Foretoken contributors
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: datasets>=2.14
|
|
11
|
+
Requires-Dist: evalscope[perf]==1.11.1
|
|
12
|
+
Requires-Dist: httpx>=0.27
|
|
13
|
+
Requires-Dist: huggingface_hub>=0.23
|
|
14
|
+
Requires-Dist: ijson>=3.2
|
|
15
|
+
Requires-Dist: lm-eval[api]==0.4.13
|
|
16
|
+
Requires-Dist: matplotlib>=3.7
|
|
17
|
+
Requires-Dist: numpy>=1.24
|
|
18
|
+
Requires-Dist: openai>=1.40
|
|
19
|
+
Requires-Dist: packaging>=24.0
|
|
20
|
+
Requires-Dist: PyYAML>=6.0
|
|
21
|
+
Requires-Dist: tqdm>=4.65
|
|
22
|
+
Requires-Dist: transformers>=4.40
|
|
23
|
+
Requires-Dist: wandb>=0.19.10
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: grafana-foundation-sdk==1769699379!11.6.0; extra == "dev"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
<!--
|
|
29
|
+
SPDX-License-Identifier: Apache-2.0
|
|
30
|
+
SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
|
|
31
|
+
-->
|
|
32
|
+
|
|
33
|
+
# Foretoken command-line tool
|
|
34
|
+
|
|
35
|
+
English | [简体中文](README_zh.md)
|
|
36
|
+
|
|
37
|
+
The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend URLs, and runs benchmarks through one `foretoken` entry point.
|
|
38
|
+
|
|
39
|
+
## Before you start
|
|
40
|
+
|
|
41
|
+
You need Python 3.11 or later, an active Kubernetes context, `kubectl`, and Helm. GPU nodes must already have their vendor driver and Kubernetes device plugin.
|
|
42
|
+
|
|
43
|
+
## Install the command-line tool
|
|
44
|
+
|
|
45
|
+
Install the published command-line tool with pip:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install foretoken
|
|
49
|
+
|
|
50
|
+
# From a source checkout:
|
|
51
|
+
# pip install -e .
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Or create and activate a virtual environment with uv:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
uv venv
|
|
58
|
+
source .venv/bin/activate
|
|
59
|
+
uv pip install foretoken
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Run `foretoken --version` to check the installed CLI version.
|
|
63
|
+
|
|
64
|
+
## Create a local cluster
|
|
65
|
+
|
|
66
|
+
On a Linux GPU host with Docker, NVIDIA Container Toolkit, and k3d installed:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
# Name the local cluster and use GPU index 0 from nvidia-smi.
|
|
70
|
+
# To use two GPUs, pass --gpus 0,1.
|
|
71
|
+
foretoken cluster create k3d --name foretoken-dev --gpus 0
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
For a local kind development cluster:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
foretoken cluster create kind --name foretoken-dev
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Remove a cluster created by the CLI with:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
foretoken cluster delete k3d --name foretoken-dev
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Install the Kubernetes platform
|
|
87
|
+
|
|
88
|
+
`foretoken install` installs the Foretoken CRDs and controller in the active Kubernetes context. Platform resources use the `foretoken-platform` namespace. The command also configures monitoring and, in Gateway mode, the Gateway resources. Deploy model services separately with `foretoken deploy`.
|
|
89
|
+
|
|
90
|
+
### Default installation
|
|
91
|
+
|
|
92
|
+
The default uses release images and local access through a `LoadBalancer` Service:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
foretoken install
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Installation selects the NVIDIA or MetaX runtime and automatically reuses or installs LeaderWorkerSet and the shared RDMA device plugin. Explicit runtime settings in `--values` take precedence; in a mixed-GPU cluster, select a resource with `runtime.vllm.gpu.resourceName` or restrict the nodes with `runtime.vllm.gpu.nodeSelector`.
|
|
99
|
+
|
|
100
|
+
Log collection and persistence are enabled by default. See [Observability](../observability/README.md) for configuration, log queries, dashboards, and alerts.
|
|
101
|
+
|
|
102
|
+
### Gateway mode
|
|
103
|
+
|
|
104
|
+
Gateway mode creates a dedicated `GatewayClass` and `Gateway`, installing Envoy Gateway if no compatible controller is available:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
foretoken install --frontend-mode gateway
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
With another Gateway Controller, reuse a Gateway managed by that controller:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
foretoken install \
|
|
114
|
+
--frontend-mode gateway \
|
|
115
|
+
--gateway-name inference-gateway \
|
|
116
|
+
--gateway-namespace gateway-system
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Add `--gateway-section-name LISTENER` only when more than one listener matches.
|
|
120
|
+
|
|
121
|
+
### Current source
|
|
122
|
+
|
|
123
|
+
Build and install from the repository root. The cluster needs a default StorageClass for compiler caches; see the [source deployment guide](../docs/custom-deployment.md) for storage overrides.
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
foretoken install -e .
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
This builds the platform in dedicated Pods and binds the checkout to the target cluster.
|
|
130
|
+
|
|
131
|
+
After editing it, use `foretoken deploy` to [redeploy source changes](../docs/custom-deployment.md#deploy-and-update-code). Use `--engine-source PATH` to also bind a [vLLM engine checkout](../docs/custom-deployment.md#edit-an-inference-engine).
|
|
132
|
+
|
|
133
|
+
A standard active kind or k3d context loads the built images directly into its nodes. Other Kubernetes contexts need a registry reachable by the Build Pods and nodes. For an internal registry without authentication:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
foretoken install -e . --registry registry.example.com:5000/foretoken
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
If the registry requires authentication, follow the [source deployment guide](../docs/kubernetes-deployment.md) before installation.
|
|
140
|
+
|
|
141
|
+
### Model distribution
|
|
142
|
+
|
|
143
|
+
To share public model downloads between nodes through Dragonfly, save this in `deploy/platform-values.yaml`:
|
|
144
|
+
|
|
145
|
+
```yaml
|
|
146
|
+
modelDistribution:
|
|
147
|
+
dragonfly:
|
|
148
|
+
enabled: true
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
For a published platform installation, apply the values with:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
foretoken install --values deploy/platform-values.yaml
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
For a source installation, run `foretoken install -e . --values deploy/platform-values.yaml` from the repository root, retaining the original registry and engine-source options.
|
|
158
|
+
|
|
159
|
+
Installation prepares Dragonfly or reuses an existing installation. Models that require authentication and custom model endpoints download directly from their provider. To select a particular Dragonfly Helm release, set `existingRelease: {name: dragonfly, namespace: dragonfly-system}` under `modelDistribution.dragonfly`.
|
|
160
|
+
|
|
161
|
+
On NVIDIA clusters with RDMA, ModelExpress can load weights from running replicas. Add this alongside `dragonfly` to enable it:
|
|
162
|
+
|
|
163
|
+
```yaml
|
|
164
|
+
modelDistribution:
|
|
165
|
+
modelexpress:
|
|
166
|
+
enabled: true
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Automatic weight transfer uses remote models with a persistent cache, data parallelism of one, and fixed expert placement. An explicit `load-format` remains unchanged. Each GPU worker selects a nearby available RDMA interface; replicas without a compatible source load the prepared files.
|
|
170
|
+
|
|
171
|
+
Reapply the installation command after changing either setting. Set `enabled: false` to disable it. `foretoken uninstall` removes managed Dragonfly resources once no workloads use them; reused installations are retained.
|
|
172
|
+
|
|
173
|
+
### Installation options
|
|
174
|
+
|
|
175
|
+
Use `--values` only to override platform image, runtime, or hardware settings. Without an override, installation compares supported public sources for default platform images and OCI charts. Use `--oci-registry` to select a registry explicitly; image references supplied through values remain unchanged. Source selection runs on the CLI host, so the selected registry must also be reachable from the cluster nodes.
|
|
176
|
+
|
|
177
|
+
Model services are reached through an IP address outside the cluster. k3d, k3s, and cloud clusters assign one automatically. Clusters built with kubeadm, RKE2, or kubespray have no address assignment by default, so installation there ends with `LoadBalancer support Not verified`. Give Foretoken a range of unused addresses in the nodes' subnet, confirmed with the cluster administrator, and it assigns them to services:
|
|
178
|
+
|
|
179
|
+
```yaml
|
|
180
|
+
loadBalancer:
|
|
181
|
+
managedAddresses:
|
|
182
|
+
- 192.168.1.240-192.168.1.250
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
## Deploy and operate model services
|
|
186
|
+
|
|
187
|
+
Run from the repository checkout prepared in the [Quick Start](../README.md). Deploy one frontend and all models rendered by a Kustomize root.
|
|
188
|
+
|
|
189
|
+
See the [multi-model example](../examples/multi-model-quickstart/README.md) for resources and [model storage](../docs/model-storage.md) for directory or PVC configuration. Use `examples/quickstart` for a single model.
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
foretoken deploy examples/multi-model-quickstart --timeout 20m
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
The command applies the configuration, shows service status, and streams Pod and container logs with source prefixes while waiting. It exits when every service reports Ready and its selected alerts are configured. Without `--timeout`, it waits up to ten minutes. Configure service alerts in the Kustomize deployment; see [service observability](../examples/observability/README.md).
|
|
196
|
+
|
|
197
|
+
Inspect the same deployment without applying it:
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
foretoken status examples/multi-model-quickstart
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Inspect every Foretoken service in a namespace. With `--watch`, follow service state changes and Pod/container logs until Ctrl+C:
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
foretoken status -n foretoken-multi-model-demo
|
|
207
|
+
foretoken status -n foretoken-multi-model-demo --watch
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Resolve the public frontend URL after deployment:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/multi-model-quickstart)"
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
For an HTTP Gateway, resolve its request `Host` separately:
|
|
217
|
+
|
|
218
|
+
```bash
|
|
219
|
+
FORETOKEN_REQUEST_HOST="$(foretoken endpoint examples/multi-model-quickstart --host)"
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
`--host` returns the host and optional port for direct access, or the configured routing hostname for an HTTP Gateway. `foretoken endpoint` waits for the LoadBalancer or Gateway address; use `foretoken deploy` to wait for the services to become ready.
|
|
223
|
+
|
|
224
|
+
## Measure serving performance
|
|
225
|
+
|
|
226
|
+
Use `foretoken perf` to measure response latency and request or token throughput. Pass a Kustomize directory, or `--url` with `--model` for an existing endpoint. Choose a workload in [Performance examples](../benchmarks/docs/perf/README.md).
|
|
227
|
+
|
|
228
|
+
## Evaluate and compare models
|
|
229
|
+
|
|
230
|
+
Use `foretoken eval` to score model answers with lm-evaluation-harness or EvalScope. It accepts the same service selection options; task and scoring parameters use the selected framework's syntax. See [Quality evaluation](../benchmarks/docs/eval/README.md). Add `--reference` to [compare a candidate's probabilities against a reference](../benchmarks/docs/eval/distribution-comparison.md).
|
|
231
|
+
|
|
232
|
+
## Export figures
|
|
233
|
+
|
|
234
|
+
Use `--output local,wandb,plot` with a benchmark, or `foretoken plot RESULT_DIR` to redraw a saved run or sweep without running inference. Export options and comparisons are in [parameter sweeps](../benchmarks/docs/perf/sweep.md).
|
|
235
|
+
|
|
236
|
+
## Find execution bottlenecks
|
|
237
|
+
|
|
238
|
+
Add `--profile` to `foretoken deploy` or `foretoken perf` to record CPU/GPU execution, then browse captures with `foretoken profile view`. Setup and commands are in [Profiling](../benchmarks/docs/profile/README.md).
|
|
239
|
+
|
|
240
|
+
## Clean up
|
|
241
|
+
|
|
242
|
+
Delete the deployed services before uninstalling the platform:
|
|
243
|
+
|
|
244
|
+
```bash
|
|
245
|
+
foretoken delete examples/multi-model-quickstart
|
|
246
|
+
foretoken uninstall
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
CRDs and reused cluster components are retained. Managed LeaderWorkerSet and MetalLB controllers are also retained while workloads still depend on them.
|
|
@@ -21,42 +21,53 @@ If you only need to serve a single model on one GPU, using an inference engine s
|
|
|
21
21
|
|
|
22
22
|
| Feature | Description | Status |
|
|
23
23
|
|---|---|---|
|
|
24
|
-
| [
|
|
25
|
-
| Profiling |
|
|
24
|
+
| [Evaluation](benchmarks/README.md) | Measure service performance and model quality | In development |
|
|
25
|
+
| [Profiling](benchmarks/docs/profile/README.md) | Capture PyTorch, NVIDIA Nsight Systems, or MetaX mcTracer timelines for a model service | In development |
|
|
26
26
|
| Hardware support | Common interfaces for device capabilities, runtimes, communication, and metrics; see [MetaX deployment](docs/metax-deployment.md) | In development |
|
|
27
27
|
| Request routing | Select instances based on load, queues, KV reuse, and service levels | Research |
|
|
28
28
|
| Distributed inference | Aggregated serving, Prefill/Decode disaggregation, and WideEP parallelism | Research |
|
|
29
29
|
| Control plane | Model services, replica management, autoscaling, updates, and failure recovery | In development |
|
|
30
|
-
| [Observability](observability/README.md) | Collect
|
|
30
|
+
| [Observability](observability/README.md) | Collect metrics and persistent service logs, evaluate alerts, and inspect the system Dashboard | In development |
|
|
31
31
|
|
|
32
32
|
## Quick Start
|
|
33
33
|
|
|
34
|
-
|
|
34
|
+
Choose the deployment path before running the common steps:
|
|
35
|
+
|
|
36
|
+
| Situation | Guide |
|
|
37
|
+
|---|---|
|
|
38
|
+
| Single-host local deployment | [k3d deployment](docs/k3d-deployment.md) · [kind deployment](docs/kind-deployment.md) |
|
|
39
|
+
| Kubernetes deployment with K3s, RKE2, KubeSphere, cloud, or another cluster | [Kubernetes deployment](docs/kubernetes-deployment.md) |
|
|
40
|
+
| MetaX GPU deployment | [MetaX deployment](docs/metax-deployment.md) |
|
|
41
|
+
|
|
42
|
+
The steps below use k3d as the example.
|
|
35
43
|
|
|
36
44
|
### 1. Get the examples and install the command-line tool
|
|
37
45
|
|
|
38
46
|
```bash
|
|
39
47
|
git clone https://github.com/shiweijiezero/foretoken.git
|
|
40
48
|
cd foretoken
|
|
41
|
-
pip install
|
|
49
|
+
pip install -e .
|
|
42
50
|
|
|
43
|
-
#
|
|
44
|
-
# pip install
|
|
51
|
+
# For the published CLI instead:
|
|
52
|
+
# pip install foretoken
|
|
45
53
|
```
|
|
46
54
|
|
|
47
55
|
### 2. Install the Kubernetes platform
|
|
48
56
|
|
|
57
|
+
Create a local k3d cluster named `foretoken-dev` and install Foretoken:
|
|
58
|
+
|
|
49
59
|
```bash
|
|
50
|
-
# Use
|
|
51
|
-
foretoken
|
|
60
|
+
# Use GPU index 0 from nvidia-smi. To use two GPUs, pass --gpus 0,1.
|
|
61
|
+
foretoken cluster create k3d --name foretoken-dev --gpus 0
|
|
52
62
|
|
|
53
|
-
# Build
|
|
54
|
-
|
|
55
|
-
```
|
|
63
|
+
# Build from the current source checkout:
|
|
64
|
+
foretoken install -e .
|
|
56
65
|
|
|
57
|
-
|
|
66
|
+
# Use published images instead:
|
|
67
|
+
# foretoken install
|
|
68
|
+
```
|
|
58
69
|
|
|
59
|
-
|
|
70
|
+
The host must have Docker, NVIDIA Container Toolkit, k3d, kubectl, and Helm, and your user must be able to run `docker info` without `sudo`. Install host dependencies separately when needed. For kind or an existing Kubernetes cluster, use the corresponding guide in the table above.
|
|
60
71
|
|
|
61
72
|
### 3. Deploy the Quick Start
|
|
62
73
|
|
|
@@ -64,7 +75,7 @@ See the [source deployment guide](docs/custom-deployment.md) for build tools and
|
|
|
64
75
|
foretoken deploy examples/quickstart --timeout 20m
|
|
65
76
|
```
|
|
66
77
|
|
|
67
|
-
This example deploys one frontend service and one `Qwen/Qwen3-0.6B` model replica
|
|
78
|
+
This example deploys one frontend service and one `Qwen/Qwen3-0.6B` model replica. The model requests 1 GPU, 4 CPU, and 48 GiB memory, with limits of 8 CPU and 64 GiB. The example uses the repository-root `./data` directory for model files and runtime cache. More deployments are available in [`examples/`](examples/).
|
|
68
79
|
|
|
69
80
|
### 4. Send a test request
|
|
70
81
|
|
|
@@ -77,18 +88,49 @@ curl --fail-with-body --no-buffer \
|
|
|
77
88
|
-d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
|
|
78
89
|
```
|
|
79
90
|
|
|
80
|
-
###
|
|
91
|
+
### Iterate on source
|
|
92
|
+
|
|
93
|
+
After editing the checkout, run the same deploy command again to apply the change without recreating the cluster or manually importing the runtime image:
|
|
81
94
|
|
|
82
95
|
```bash
|
|
83
|
-
|
|
96
|
+
foretoken deploy examples/quickstart --timeout 20m
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Python, Triton, Rust, CUDA, C/C++, and vLLM source changes use the cluster build caches and reuse the runtime environment when dependencies and startup code are unchanged. See [Deploy Foretoken from Source](docs/custom-deployment.md) for engine checkouts and runtime changes.
|
|
100
|
+
|
|
101
|
+
### 5. Evaluate and profile the service
|
|
84
102
|
|
|
85
|
-
|
|
86
|
-
# pip install -e '.[bench]'
|
|
103
|
+
The examples save results locally and to W&B. Run `wandb login` once before using W&B.
|
|
87
104
|
|
|
88
|
-
|
|
105
|
+
#### Performance: latency and throughput
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
foretoken perf examples/quickstart --num-prompts 20 --output local,wandb
|
|
89
109
|
```
|
|
90
110
|
|
|
91
|
-
|
|
111
|
+
Read request success, latency, and throughput in the summary. [Performance examples](benchmarks/docs/perf/README.md) cover other workloads and load settings.
|
|
112
|
+
|
|
113
|
+
#### Quality: score model answers
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
foretoken eval examples/quickstart \
|
|
117
|
+
--evaluator lm-eval --tasks gsm8k --limit 100 --output local,wandb
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
This scores 100 GSM8K math problems. See [Quality evaluation](benchmarks/docs/eval/README.md) for EvalScope, task parameters, and saved scores.
|
|
121
|
+
|
|
122
|
+
#### Profiling: inspect execution bottlenecks
|
|
123
|
+
|
|
124
|
+
Use a source-installed CLI and platform for profiling, as described in the [profiling guide](benchmarks/docs/profile/README.md). The Quick Start already configures persistent capture storage.
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
foretoken perf examples/quickstart \
|
|
128
|
+
--profile --profile-engine pytorch --profile-duration 15s \
|
|
129
|
+
--num-prompts 2 --max-tokens 128 --output local,wandb
|
|
130
|
+
foretoken profile view
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
Open the printed URL to inspect the capture. Press Ctrl+C to close the viewer; the model service remains running.
|
|
92
134
|
|
|
93
135
|
## Gateway Mode
|
|
94
136
|
|
|
@@ -136,13 +178,7 @@ foretoken delete examples/quickstart
|
|
|
136
178
|
foretoken uninstall
|
|
137
179
|
```
|
|
138
180
|
|
|
139
|
-
The uninstall command preserves Foretoken CRDs and reused cluster components. It removes the platform and the monitoring or Gateway resources managed by the command-line tool.
|
|
140
|
-
|
|
141
|
-
## Deployment Guides
|
|
142
|
-
|
|
143
|
-
- [Source builds and private registries](docs/custom-deployment.md)
|
|
144
|
-
- [Single-machine GPU clusters with k3d](docs/k3d-deployment.md)
|
|
145
|
-
- [MetaX GPUs](docs/metax-deployment.md)
|
|
181
|
+
The uninstall command preserves Foretoken CRDs, log storage, and reused cluster components. It removes the platform and the monitoring or Gateway resources managed by the command-line tool.
|
|
146
182
|
|
|
147
183
|
## Related Projects
|
|
148
184
|
|