foretoken 0.0.2__tar.gz → 0.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {foretoken-0.0.2/foretoken.egg-info → foretoken-0.0.3}/PKG-INFO +33 -50
  2. {foretoken-0.0.2 → foretoken-0.0.3}/README.md +31 -30
  3. foretoken-0.0.3/benchmarks/__init__.py +4 -0
  4. foretoken-0.0.3/benchmarks/config/__init__.py +4 -0
  5. foretoken-0.0.3/benchmarks/config/benchmark.py +454 -0
  6. foretoken-0.0.2/benchmarks/arguments.py → foretoken-0.0.3/benchmarks/config/cli.py +156 -152
  7. foretoken-0.0.3/benchmarks/datasets/__init__.py +4 -0
  8. foretoken-0.0.3/benchmarks/datasets/conversations.py +423 -0
  9. foretoken-0.0.3/benchmarks/datasets/huggingface.py +203 -0
  10. foretoken-0.0.3/benchmarks/datasets/multi_dataset.py +224 -0
  11. foretoken-0.0.3/benchmarks/datasets/multimodal.py +4 -0
  12. foretoken-0.0.3/benchmarks/datasets/synthetic.py +167 -0
  13. foretoken-0.0.3/benchmarks/datasets/traces.py +234 -0
  14. foretoken-0.0.3/benchmarks/integrations/__init__.py +4 -0
  15. foretoken-0.0.3/benchmarks/integrations/evalscope.py +679 -0
  16. foretoken-0.0.3/benchmarks/integrations/hermes.py +4 -0
  17. foretoken-0.0.3/benchmarks/integrations/openai.py +135 -0
  18. foretoken-0.0.3/benchmarks/integrations/pi.py +4 -0
  19. foretoken-0.0.3/benchmarks/integrations/streaming.py +32 -0
  20. foretoken-0.0.3/benchmarks/integrations/verifiers.py +4 -0
  21. foretoken-0.0.3/benchmarks/main.py +67 -0
  22. foretoken-0.0.3/benchmarks/model_service.py +272 -0
  23. foretoken-0.0.3/benchmarks/profiling/__init__.py +4 -0
  24. {foretoken-0.0.2/benchmarks/logger → foretoken-0.0.3/benchmarks/results}/__init__.py +1 -2
  25. foretoken-0.0.3/benchmarks/results/console.py +315 -0
  26. foretoken-0.0.3/benchmarks/results/metrics.py +157 -0
  27. foretoken-0.0.3/benchmarks/results/output.py +309 -0
  28. {foretoken-0.0.2/benchmarks/report → foretoken-0.0.3/benchmarks/results}/pareto.py +23 -17
  29. foretoken-0.0.3/benchmarks/results/timeseries.py +104 -0
  30. {foretoken-0.0.2/benchmarks/logger → foretoken-0.0.3/benchmarks/results}/wandb.py +97 -56
  31. {foretoken-0.0.2/benchmarks/metrics → foretoken-0.0.3/benchmarks/runs}/__init__.py +2 -0
  32. foretoken-0.0.2/benchmarks/__init__.py → foretoken-0.0.3/benchmarks/runs/agent.py +2 -0
  33. foretoken-0.0.3/benchmarks/runs/evaluation.py +4 -0
  34. foretoken-0.0.3/benchmarks/runs/http.py +78 -0
  35. foretoken-0.0.3/benchmarks/runs/sweep.py +386 -0
  36. foretoken-0.0.3/benchmarks/runs/trace.py +362 -0
  37. foretoken-0.0.3/benchmarks/scoring/__init__.py +4 -0
  38. {foretoken-0.0.2 → foretoken-0.0.3}/cli/README.md +28 -45
  39. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/metax.py +4 -1
  40. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/nvidia.py +4 -1
  41. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/arguments.py +19 -7
  42. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/kubernetes.py +24 -2
  43. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/main.py +4 -5
  44. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/manifest.py +51 -4
  45. foretoken-0.0.3/cli/foretoken/platform/config.py +225 -0
  46. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/helm.py +110 -3
  47. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/helm_client.py +20 -14
  48. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/lifecycle.py +197 -5
  49. foretoken-0.0.3/cli/foretoken/platform/load_balancer.py +442 -0
  50. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/types.py +16 -0
  51. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/source.py +4 -2
  52. foretoken-0.0.3/cli/foretoken/storage.py +359 -0
  53. {foretoken-0.0.2 → foretoken-0.0.3/foretoken.egg-info}/PKG-INFO +33 -50
  54. foretoken-0.0.3/foretoken.egg-info/SOURCES.txt +67 -0
  55. {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/requires.txt +3 -3
  56. {foretoken-0.0.2 → foretoken-0.0.3}/pyproject.toml +12 -14
  57. foretoken-0.0.2/benchmarks/client/__init__.py +0 -2
  58. foretoken-0.0.2/benchmarks/client/openai_client.py +0 -148
  59. foretoken-0.0.2/benchmarks/config.py +0 -358
  60. foretoken-0.0.2/benchmarks/deployment/__init__.py +0 -9
  61. foretoken-0.0.2/benchmarks/deployment/discovery.py +0 -166
  62. foretoken-0.0.2/benchmarks/deployment/lifecycle.py +0 -111
  63. foretoken-0.0.2/benchmarks/logger/cli.py +0 -27
  64. foretoken-0.0.2/benchmarks/main.py +0 -74
  65. foretoken-0.0.2/benchmarks/metrics/aggregator.py +0 -161
  66. foretoken-0.0.2/benchmarks/report/__init__.py +0 -2
  67. foretoken-0.0.2/benchmarks/report/summary.py +0 -156
  68. foretoken-0.0.2/benchmarks/runner/__init__.py +0 -2
  69. foretoken-0.0.2/benchmarks/runner/base.py +0 -284
  70. foretoken-0.0.2/benchmarks/runner/multi_dataset.py +0 -105
  71. foretoken-0.0.2/benchmarks/runner/run_benchmark.py +0 -79
  72. foretoken-0.0.2/benchmarks/runner/run_spec.py +0 -22
  73. foretoken-0.0.2/benchmarks/runner/select_runner.py +0 -27
  74. foretoken-0.0.2/benchmarks/runner/sweep.py +0 -141
  75. foretoken-0.0.2/benchmarks/runner/trace_runner.py +0 -225
  76. foretoken-0.0.2/benchmarks/storage/__init__.py +0 -2
  77. foretoken-0.0.2/benchmarks/storage/result_writer.py +0 -37
  78. foretoken-0.0.2/benchmarks/utils/__init__.py +0 -2
  79. foretoken-0.0.2/benchmarks/utils/bench_params.py +0 -223
  80. foretoken-0.0.2/benchmarks/workload/__init__.py +0 -2
  81. foretoken-0.0.2/benchmarks/workload/hf_dataset.py +0 -154
  82. foretoken-0.0.2/benchmarks/workload/loader.py +0 -205
  83. foretoken-0.0.2/benchmarks/workload/random_dataset.py +0 -273
  84. foretoken-0.0.2/benchmarks/workload/trace_loader.py +0 -275
  85. foretoken-0.0.2/benchmarks/workload/trace_workload.py +0 -109
  86. foretoken-0.0.2/cli/foretoken/platform/config.py +0 -119
  87. foretoken-0.0.2/foretoken.egg-info/SOURCES.txt +0 -65
  88. {foretoken-0.0.2 → foretoken-0.0.3}/LICENSE +0 -0
  89. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/__init__.py +0 -0
  90. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/__init__.py +0 -0
  91. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/_exporter.py +0 -0
  92. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/discovery.py +0 -0
  93. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/observability.py +0 -0
  94. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/__init__.py +0 -0
  95. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/gateway.py +0 -0
  96. {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/gateway_resources.py +0 -0
  97. {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/dependency_links.txt +0 -0
  98. {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/entry_points.txt +0 -0
  99. {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/top_level.txt +0 -0
  100. {foretoken-0.0.2 → foretoken-0.0.3}/setup.cfg +0 -0
@@ -1,17 +1,17 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: foretoken
3
- Version: 0.0.2
3
+ Version: 0.0.3
4
4
  Summary: Command-line tools for installing and operating Foretoken
5
5
  Author: Foretoken contributors
6
6
  License-Expression: Apache-2.0
7
- Requires-Python: >=3.10
7
+ Requires-Python: >=3.11
8
8
  Description-Content-Type: text/markdown
9
9
  License-File: LICENSE
10
10
  Requires-Dist: packaging>=24.0
11
11
  Requires-Dist: PyYAML>=6.0
12
12
  Provides-Extra: bench
13
13
  Requires-Dist: datasets>=2.14; extra == "bench"
14
- Requires-Dist: evalscope>=1.10; extra == "bench"
14
+ Requires-Dist: evalscope[perf]==1.11.1; extra == "bench"
15
15
  Requires-Dist: httpx>=0.27; extra == "bench"
16
16
  Requires-Dist: huggingface_hub>=0.23; extra == "bench"
17
17
  Requires-Dist: matplotlib>=3.7; extra == "bench"
@@ -19,9 +19,9 @@ Requires-Dist: numpy>=1.24; extra == "bench"
19
19
  Requires-Dist: openai>=1.40; extra == "bench"
20
20
  Requires-Dist: tqdm>=4.65; extra == "bench"
21
21
  Requires-Dist: transformers>=4.40; extra == "bench"
22
- Requires-Dist: vllm>=0.18; extra == "bench"
23
- Requires-Dist: wandb>=0.16; extra == "bench"
22
+ Requires-Dist: wandb>=0.19.10; extra == "bench"
24
23
  Provides-Extra: dev
24
+ Requires-Dist: grafana-foundation-sdk==1769699379!11.6.0; extra == "dev"
25
25
  Requires-Dist: pytest>=7.0; extra == "dev"
26
26
  Dynamic: license-file
27
27
 
@@ -34,13 +34,11 @@ SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
34
34
 
35
35
  English | [简体中文](README_zh.md)
36
36
 
37
- The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend endpoints, and runs benchmarks through one `foretoken` entry point.
38
-
39
- For a new cluster, start by installing the command-line tool. If `foretoken --version` already works, go straight to platform installation. If the cluster already has the Foretoken platform, start with model deployment.
37
+ The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend URLs, and runs benchmarks through one `foretoken` entry point.
40
38
 
41
39
  ## Before you start
42
40
 
43
- You need Python 3.10 or later, an active Kubernetes context, `kubectl`, and Helm. GPU nodes must already have their vendor driver and Kubernetes device plugin. Source installation also requires Docker and Make, plus either a local kind/k3d cluster or an OCI registry reachable by every target node.
41
+ You need Python 3.11 or later, an active Kubernetes context, `kubectl`, and Helm. GPU nodes must already have their vendor driver and Kubernetes device plugin.
44
42
 
45
43
  ## Install the command-line tool
46
44
 
@@ -61,7 +59,7 @@ source .venv/bin/activate
61
59
  uv pip install foretoken
62
60
  ```
63
61
 
64
- This step only installs the `foretoken` command in the current Python environment; it does not change the Kubernetes cluster. Run `foretoken --version` to see the command-line tool and corresponding platform version.
62
+ Run `foretoken --version` to check the installed command-line tool version.
65
63
 
66
64
  ## Install the Kubernetes platform
67
65
 
@@ -75,11 +73,13 @@ The default uses release images and local access through a `LoadBalancer` Servic
75
73
  foretoken install
76
74
  ```
77
75
 
78
- During installation, the command-line tool discovers Prometheus and accelerator metric exporters. It reuses compatible shared instances, installs managed Prometheus and NVIDIA DCGM Exporter releases when needed, and connects to the mxExporter already provided by a MetaX cluster. See [Observability](../observability/README.md) for monitoring selection and configuration.
76
+ Installation selects the NVIDIA or MetaX runtime from the cluster's GPU resources. Explicit runtime settings in `--values` take precedence; in a mixed-GPU cluster, select a resource with `runtime.vllm.gpu.resourceName` or restrict the nodes with `runtime.vllm.gpu.nodeSelector`.
77
+
78
+ Installation also sets up monitoring, reusing a Prometheus and GPU metrics exporter already in the cluster when they exist. See [Observability](../observability/README.md).
79
79
 
80
80
  ### Gateway mode
81
81
 
82
- The command-line tool creates a dedicated `GatewayClass` and `Gateway` only when the cluster runs Envoy Gateway:
82
+ Gateway mode creates a dedicated `GatewayClass` and `Gateway`, installing Envoy Gateway if no compatible controller is available:
83
83
 
84
84
  ```bash
85
85
  foretoken install --frontend-mode gateway
@@ -98,7 +98,7 @@ Add `--gateway-section-name LISTENER` only when more than one listener matches.
98
98
 
99
99
  ### Current source
100
100
 
101
- Build Foretoken images from the current source tree and configure the platform to use them:
101
+ Prepare the build tools listed in the [source deployment guide](../docs/custom-deployment.md), then build and install from the repository root:
102
102
 
103
103
  ```bash
104
104
  foretoken install -e .
@@ -115,21 +115,31 @@ Registry login authorizes the local image push. Private registries also need `im
115
115
 
116
116
  ### Installation options
117
117
 
118
- Repeatable `--values` files provide platform image, runtime, and hardware settings. Release and source installs record their mode in Helm metadata and cannot switch silently. Releases originally installed directly with Helm remain under their existing Helm lifecycle and are not adopted automatically.
118
+ Repeatable `--values` files provide platform image, runtime, and hardware settings.
119
119
 
120
- ### Persistent runtime cache
120
+ Model services are reached through an IP address outside the cluster. k3d, k3s, and cloud clusters assign one automatically. Clusters built with kubeadm, RKE2, or kubespray have no address assignment by default, so installation there ends with `LoadBalancer support Not verified`. Give Foretoken a range of unused addresses in the nodes' subnet, confirmed with the cluster administrator, and it assigns them to services:
121
121
 
122
- Create one `RuntimeCache` in a workload namespace to let Foretoken provision and manage a shared cache PVC. Existing PVCs remain supported through `workload.cache.claimName`. See [Persistent Runtime Cache](../docs/development/runtime-cache.md).
122
+ ```yaml
123
+ loadBalancer:
124
+ managedAddresses:
125
+ - 192.168.1.240-192.168.1.250
126
+ ```
127
+
128
+ ```bash
129
+ foretoken install --values platform-values.yaml
130
+ ```
123
131
 
124
132
  ## Deploy and operate model services
125
133
 
126
- Deploy one frontend and all models rendered by a Kustomize root. The multi-model example needs up to four GPUs, 12 CPU cores, 100 GiB memory, and a default `ReadWriteMany` StorageClass with online expansion; use `examples/quickstart` for the smallest path.
134
+ Run from the repository checkout prepared in the [Quick Start](../README.md). Deploy one frontend and all models rendered by a Kustomize root.
135
+
136
+ See the [multi-model example](../examples/multi-model-quickstart/README.md) for resources and [model storage](../docs/model-storage.md) for directory or PVC configuration. Use `examples/quickstart` for a single model.
127
137
 
128
138
  ```bash
129
- foretoken deploy examples/multi-model-quickstart
139
+ foretoken deploy examples/multi-model-quickstart --timeout 20m
130
140
  ```
131
141
 
132
- The command applies the configuration, reports each `FrontendService` and `ModelService` state when it changes, and exits when every service reports Ready for its current configuration. Change the default ten-minute deadline with `--timeout`.
142
+ The command applies the configuration, reports service state changes, and exits when every service is Ready. Without `--timeout`, it waits up to ten minutes.
133
143
 
134
144
  Inspect the same deployment without applying it:
135
145
 
@@ -158,44 +168,17 @@ FORETOKEN_REQUEST_HOST="$(foretoken endpoint examples/multi-model-quickstart --h
158
168
 
159
169
  `--host` returns the host and optional port for direct access, or the configured routing hostname for an HTTP Gateway. `foretoken endpoint` waits for the LoadBalancer or Gateway address; use `foretoken deploy` to wait for the services to become ready.
160
170
 
161
- ## Run benchmarks
162
-
163
- Install the optional benchmark dependencies with pip:
164
-
165
- ```bash
166
- pip install 'foretoken[bench]'
167
-
168
- # For source installation from the repository:
169
- # pip install -e .
170
- # pip install -e '.[bench]'
171
- ```
172
-
173
- Or install the benchmark dependencies in the activated uv environment:
171
+ ## Benchmark model services
174
172
 
175
- ```bash
176
- uv pip install 'foretoken[bench]'
177
- ```
178
-
179
- Then run the benchmark:
180
-
181
- ```bash
182
- foretoken bench examples/multi-model-quickstart --model Qwen/Qwen3-0.6B
183
- ```
184
-
185
- The command-line tool uses the active `kubectl` context and honors standard Kubernetes configuration such as `KUBECONFIG`.
173
+ Use `foretoken bench` to measure model-service performance. Commands and examples are in [Model Service Benchmarks](../benchmarks/README.md).
186
174
 
187
175
  ## Clean up
188
176
 
189
- Delete the resources rendered by the same configuration:
177
+ Delete the deployed services before uninstalling the platform:
190
178
 
191
179
  ```bash
192
180
  foretoken delete examples/multi-model-quickstart
193
- ```
194
-
195
- The command waits for deletion and ignores resources that are already absent. After deleting all Foretoken services, remove the platform release:
196
-
197
- ```bash
198
181
  foretoken uninstall
199
182
  ```
200
183
 
201
- The command preserves Foretoken CRDs and refuses to uninstall while user-owned services remain. It removes monitoring and Gateway resources managed by the command-line tool with the platform, while reused cluster components remain unchanged.
184
+ Foretoken CRDs and reused cluster components are retained. Managed MetalLB is also retained while other services depend on it.
@@ -21,50 +21,50 @@ If you only need to serve a single model on one GPU, using an inference engine s
21
21
 
22
22
  | Feature | Description | Status |
23
23
  |---|---|---|
24
- | Benchmarking | Performance benchmarks and parameter sweeps, correctness evaluation, and SLO simulation | In development |
24
+ | [Benchmarking](benchmarks/README.md) | Measure model-service performance | In development |
25
25
  | Profiling | Use PyTorch Profiler and Nsight to identify compute, communication, and CPU/GPU bottlenecks | Planned |
26
26
  | Hardware support | Common interfaces for device capabilities, runtimes, communication, and metrics; see [MetaX deployment](docs/metax-deployment.md) | In development |
27
27
  | Request routing | Select instances based on load, queues, KV reuse, and service levels | Research |
28
28
  | Distributed inference | Aggregated serving, Prefill/Decode disaggregation, and WideEP parallelism | Research |
29
29
  | Control plane | Model services, replica management, autoscaling, updates, and failure recovery | In development |
30
- | [Observability](observability/README.md) | Collect runtime metrics, evaluate alerts, and profile CPU/GPU bottlenecks | In development |
30
+ | [Observability](observability/README.md) | Collect service and accelerator metrics, evaluate alerts, and inspect the system Dashboard | In development |
31
31
 
32
32
  ## Quick Start
33
33
 
34
- This Quick Start requires Python 3.10+, Kubernetes with an expandable default `StorageClass`, `kubectl`, Helm, one GPU, and a working `LoadBalancer` (k3s ServiceLB is sufficient for k3d). See the [k3d guide](docs/k3d-deployment.md) for a single-machine test cluster.
34
+ Start with a GPU-enabled Kubernetes cluster and Python 3.11+, `kubectl`, and Helm installed locally.
35
35
 
36
- ### 1. Install the command-line tool
37
-
38
- Install the published command-line tool package:
36
+ ### 1. Get the examples and install the command-line tool
39
37
 
40
38
  ```bash
39
+ git clone https://github.com/shiweijiezero/foretoken.git
40
+ cd foretoken
41
41
  pip install foretoken
42
42
 
43
- # For source installation from the repository:
43
+ # From a source checkout:
44
44
  # pip install -e .
45
45
  ```
46
46
 
47
47
  ### 2. Install the Kubernetes platform
48
48
 
49
- By default, installation uses the Foretoken images published on GHCR:
50
-
51
49
  ```bash
52
- # Release images:
50
+ # Use release images from GHCR:
53
51
  foretoken install
54
52
 
55
- # Source installation from the repository:
53
+ # Build and install from a source checkout:
56
54
  # foretoken install -e .
57
55
  ```
58
56
 
59
- This installs the Foretoken CRDs and controller in the `foretoken-platform` namespace and waits for the controller to become ready. The default mode exposes the frontend through a `LoadBalancer` Service. Source installation rebuilds the images and updates the cluster; to deploy the current source to a remote cluster, see the [source deployment guide](docs/custom-deployment.md).
57
+ For deployment on MetaX GPUs, follow the [MetaX deployment guide](docs/metax-deployment.md).
58
+
59
+ See the [source deployment guide](docs/custom-deployment.md) for build tools and remote clusters.
60
60
 
61
61
  ### 3. Deploy the Quick Start
62
62
 
63
63
  ```bash
64
- foretoken deploy examples/quickstart
64
+ foretoken deploy examples/quickstart --timeout 20m
65
65
  ```
66
66
 
67
- This example deploys one frontend service, one `Qwen/Qwen3-0.6B` model replica, and an automatically expanding runtime cache PVC starting at 10 GiB. The workload requests one GPU, 8 CPU, and 52 GiB memory; allow additional capacity for the platform. See the [single-model example](examples/quickstart/README.md) for its resource configuration and [`examples/`](examples/) for more deployments.
67
+ This example deploys one frontend service and one `Qwen/Qwen3-0.6B` model replica, requesting one GPU, 8 CPU, and 52 GiB memory. More deployments are available in [`examples/`](examples/).
68
68
 
69
69
  ### 4. Send a test request
70
70
 
@@ -77,24 +77,24 @@ curl --fail-with-body --no-buffer \
77
77
  -d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
78
78
  ```
79
79
 
80
- ### 5. Run a benchmark
80
+ ### 5. Measure the model service
81
81
 
82
82
  ```bash
83
83
  pip install 'foretoken[bench]'
84
84
 
85
- # For source installation from the repository:
86
- # pip install -e .
85
+ # From a source checkout:
87
86
  # pip install -e '.[bench]'
88
- foretoken bench examples/quickstart
87
+
88
+ foretoken bench examples/quickstart --output local,wandb
89
89
  ```
90
90
 
91
- See [Benchmarking](benchmarks/README.md) for datasets, remote endpoints, result storage, and parameter sweeps.
91
+ Run `wandb login` before first using W&B. See [Model Service Benchmarks](benchmarks/README.md) for more examples.
92
92
 
93
93
  ## Gateway Mode
94
94
 
95
95
  Gateway mode provides a shared entry point through Kubernetes Gateway and a hostname. It suits clusters that already use Gateway or manage external traffic centrally.
96
96
 
97
- Foretoken creates its default Gateway for Envoy Gateway. Install Envoy Gateway, then add the public hostname under `spec` in `examples/quickstart/frontend.yaml`:
97
+ Add the public hostname under `spec` in `examples/quickstart/frontend.yaml`:
98
98
 
99
99
  ```yaml
100
100
  spec:
@@ -104,18 +104,13 @@ spec:
104
104
  Then run:
105
105
 
106
106
  ```bash
107
- # Install Envoy Gateway
108
- helm upgrade --install envoy-gateway \
109
- oci://docker.io/envoyproxy/gateway-helm \
110
- --namespace envoy-gateway-system \
111
- --create-namespace \
112
- --wait
113
-
114
107
  # Install the platform in Gateway mode
115
108
  foretoken install --frontend-mode gateway
109
+ # For a source-installed platform:
110
+ # foretoken install -e . --frontend-mode gateway
116
111
 
117
112
  # Deploy the Quick Start
118
- foretoken deploy examples/quickstart
113
+ foretoken deploy examples/quickstart --timeout 20m
119
114
 
120
115
  # Resolve the Gateway address and request hostname
121
116
  FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/quickstart)"
@@ -129,7 +124,7 @@ curl --fail-with-body --no-buffer \
129
124
  -d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
130
125
  ```
131
126
 
132
- See the [command-line tool guide](cli/README.md) to reuse a Gateway from another controller, select a listener, or configure TLS.
127
+ The command installs Envoy Gateway when needed. See the [command-line tool guide](cli/README.md) to reuse an existing Gateway or select a listener.
133
128
 
134
129
  ## Stop and Uninstall
135
130
 
@@ -143,6 +138,12 @@ foretoken uninstall
143
138
 
144
139
  The uninstall command preserves Foretoken CRDs and reused cluster components. It removes the platform and the monitoring or Gateway resources managed by the command-line tool.
145
140
 
141
+ ## Deployment Guides
142
+
143
+ - [Source builds and private registries](docs/custom-deployment.md)
144
+ - [Single-machine GPU clusters with k3d](docs/k3d-deployment.md)
145
+ - [MetaX GPUs](docs/metax-deployment.md)
146
+
146
147
  ## Related Projects
147
148
 
148
149
  - [vLLM](https://github.com/vllm-project/vllm)
@@ -160,7 +161,7 @@ See [Contributing to Foretoken](CONTRIBUTING.md) for development principles, col
160
161
  Thank you to everyone who has contributed to Foretoken.
161
162
 
162
163
  <a href="https://github.com/shiweijiezero/foretoken/graphs/contributors">
163
- <img src="https://contrib.rocks/image?repo=shiweijiezero/foretoken" alt="Foretoken contributors" />
164
+ <img src="https://contrib.rocks/image?repo=shiweijiezero/foretoken" width="256" alt="Foretoken contributors" />
164
165
  </a>
165
166
 
166
167
  ## License
@@ -0,0 +1,4 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
3
+
4
+ """Foretoken benchmark command family; each domain owns an independent lifecycle."""
@@ -0,0 +1,4 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
3
+
4
+ """Benchmark configuration types and command-line parsing."""