foretoken 0.0.2__tar.gz → 0.0.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {foretoken-0.0.2/foretoken.egg-info → foretoken-0.0.3}/PKG-INFO +33 -50
- {foretoken-0.0.2 → foretoken-0.0.3}/README.md +31 -30
- foretoken-0.0.3/benchmarks/__init__.py +4 -0
- foretoken-0.0.3/benchmarks/config/__init__.py +4 -0
- foretoken-0.0.3/benchmarks/config/benchmark.py +454 -0
- foretoken-0.0.2/benchmarks/arguments.py → foretoken-0.0.3/benchmarks/config/cli.py +156 -152
- foretoken-0.0.3/benchmarks/datasets/__init__.py +4 -0
- foretoken-0.0.3/benchmarks/datasets/conversations.py +423 -0
- foretoken-0.0.3/benchmarks/datasets/huggingface.py +203 -0
- foretoken-0.0.3/benchmarks/datasets/multi_dataset.py +224 -0
- foretoken-0.0.3/benchmarks/datasets/multimodal.py +4 -0
- foretoken-0.0.3/benchmarks/datasets/synthetic.py +167 -0
- foretoken-0.0.3/benchmarks/datasets/traces.py +234 -0
- foretoken-0.0.3/benchmarks/integrations/__init__.py +4 -0
- foretoken-0.0.3/benchmarks/integrations/evalscope.py +679 -0
- foretoken-0.0.3/benchmarks/integrations/hermes.py +4 -0
- foretoken-0.0.3/benchmarks/integrations/openai.py +135 -0
- foretoken-0.0.3/benchmarks/integrations/pi.py +4 -0
- foretoken-0.0.3/benchmarks/integrations/streaming.py +32 -0
- foretoken-0.0.3/benchmarks/integrations/verifiers.py +4 -0
- foretoken-0.0.3/benchmarks/main.py +67 -0
- foretoken-0.0.3/benchmarks/model_service.py +272 -0
- foretoken-0.0.3/benchmarks/profiling/__init__.py +4 -0
- {foretoken-0.0.2/benchmarks/logger → foretoken-0.0.3/benchmarks/results}/__init__.py +1 -2
- foretoken-0.0.3/benchmarks/results/console.py +315 -0
- foretoken-0.0.3/benchmarks/results/metrics.py +157 -0
- foretoken-0.0.3/benchmarks/results/output.py +309 -0
- {foretoken-0.0.2/benchmarks/report → foretoken-0.0.3/benchmarks/results}/pareto.py +23 -17
- foretoken-0.0.3/benchmarks/results/timeseries.py +104 -0
- {foretoken-0.0.2/benchmarks/logger → foretoken-0.0.3/benchmarks/results}/wandb.py +97 -56
- {foretoken-0.0.2/benchmarks/metrics → foretoken-0.0.3/benchmarks/runs}/__init__.py +2 -0
- foretoken-0.0.2/benchmarks/__init__.py → foretoken-0.0.3/benchmarks/runs/agent.py +2 -0
- foretoken-0.0.3/benchmarks/runs/evaluation.py +4 -0
- foretoken-0.0.3/benchmarks/runs/http.py +78 -0
- foretoken-0.0.3/benchmarks/runs/sweep.py +386 -0
- foretoken-0.0.3/benchmarks/runs/trace.py +362 -0
- foretoken-0.0.3/benchmarks/scoring/__init__.py +4 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/README.md +28 -45
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/metax.py +4 -1
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/nvidia.py +4 -1
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/arguments.py +19 -7
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/kubernetes.py +24 -2
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/main.py +4 -5
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/manifest.py +51 -4
- foretoken-0.0.3/cli/foretoken/platform/config.py +225 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/helm.py +110 -3
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/helm_client.py +20 -14
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/lifecycle.py +197 -5
- foretoken-0.0.3/cli/foretoken/platform/load_balancer.py +442 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/types.py +16 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/source.py +4 -2
- foretoken-0.0.3/cli/foretoken/storage.py +359 -0
- {foretoken-0.0.2 → foretoken-0.0.3/foretoken.egg-info}/PKG-INFO +33 -50
- foretoken-0.0.3/foretoken.egg-info/SOURCES.txt +67 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/requires.txt +3 -3
- {foretoken-0.0.2 → foretoken-0.0.3}/pyproject.toml +12 -14
- foretoken-0.0.2/benchmarks/client/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/client/openai_client.py +0 -148
- foretoken-0.0.2/benchmarks/config.py +0 -358
- foretoken-0.0.2/benchmarks/deployment/__init__.py +0 -9
- foretoken-0.0.2/benchmarks/deployment/discovery.py +0 -166
- foretoken-0.0.2/benchmarks/deployment/lifecycle.py +0 -111
- foretoken-0.0.2/benchmarks/logger/cli.py +0 -27
- foretoken-0.0.2/benchmarks/main.py +0 -74
- foretoken-0.0.2/benchmarks/metrics/aggregator.py +0 -161
- foretoken-0.0.2/benchmarks/report/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/report/summary.py +0 -156
- foretoken-0.0.2/benchmarks/runner/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/runner/base.py +0 -284
- foretoken-0.0.2/benchmarks/runner/multi_dataset.py +0 -105
- foretoken-0.0.2/benchmarks/runner/run_benchmark.py +0 -79
- foretoken-0.0.2/benchmarks/runner/run_spec.py +0 -22
- foretoken-0.0.2/benchmarks/runner/select_runner.py +0 -27
- foretoken-0.0.2/benchmarks/runner/sweep.py +0 -141
- foretoken-0.0.2/benchmarks/runner/trace_runner.py +0 -225
- foretoken-0.0.2/benchmarks/storage/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/storage/result_writer.py +0 -37
- foretoken-0.0.2/benchmarks/utils/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/utils/bench_params.py +0 -223
- foretoken-0.0.2/benchmarks/workload/__init__.py +0 -2
- foretoken-0.0.2/benchmarks/workload/hf_dataset.py +0 -154
- foretoken-0.0.2/benchmarks/workload/loader.py +0 -205
- foretoken-0.0.2/benchmarks/workload/random_dataset.py +0 -273
- foretoken-0.0.2/benchmarks/workload/trace_loader.py +0 -275
- foretoken-0.0.2/benchmarks/workload/trace_workload.py +0 -109
- foretoken-0.0.2/cli/foretoken/platform/config.py +0 -119
- foretoken-0.0.2/foretoken.egg-info/SOURCES.txt +0 -65
- {foretoken-0.0.2 → foretoken-0.0.3}/LICENSE +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/__init__.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/__init__.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/_exporter.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/accelerators/discovery.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/observability.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/__init__.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/gateway.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/cli/foretoken/platform/gateway_resources.py +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/dependency_links.txt +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/entry_points.txt +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/foretoken.egg-info/top_level.txt +0 -0
- {foretoken-0.0.2 → foretoken-0.0.3}/setup.cfg +0 -0
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: foretoken
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.3
|
|
4
4
|
Summary: Command-line tools for installing and operating Foretoken
|
|
5
5
|
Author: Foretoken contributors
|
|
6
6
|
License-Expression: Apache-2.0
|
|
7
|
-
Requires-Python: >=3.
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
8
|
Description-Content-Type: text/markdown
|
|
9
9
|
License-File: LICENSE
|
|
10
10
|
Requires-Dist: packaging>=24.0
|
|
11
11
|
Requires-Dist: PyYAML>=6.0
|
|
12
12
|
Provides-Extra: bench
|
|
13
13
|
Requires-Dist: datasets>=2.14; extra == "bench"
|
|
14
|
-
Requires-Dist: evalscope
|
|
14
|
+
Requires-Dist: evalscope[perf]==1.11.1; extra == "bench"
|
|
15
15
|
Requires-Dist: httpx>=0.27; extra == "bench"
|
|
16
16
|
Requires-Dist: huggingface_hub>=0.23; extra == "bench"
|
|
17
17
|
Requires-Dist: matplotlib>=3.7; extra == "bench"
|
|
@@ -19,9 +19,9 @@ Requires-Dist: numpy>=1.24; extra == "bench"
|
|
|
19
19
|
Requires-Dist: openai>=1.40; extra == "bench"
|
|
20
20
|
Requires-Dist: tqdm>=4.65; extra == "bench"
|
|
21
21
|
Requires-Dist: transformers>=4.40; extra == "bench"
|
|
22
|
-
Requires-Dist:
|
|
23
|
-
Requires-Dist: wandb>=0.16; extra == "bench"
|
|
22
|
+
Requires-Dist: wandb>=0.19.10; extra == "bench"
|
|
24
23
|
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: grafana-foundation-sdk==1769699379!11.6.0; extra == "dev"
|
|
25
25
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
26
26
|
Dynamic: license-file
|
|
27
27
|
|
|
@@ -34,13 +34,11 @@ SPDX-FileCopyrightText: Copyright contributors to the Foretoken project
|
|
|
34
34
|
|
|
35
35
|
English | [简体中文](README_zh.md)
|
|
36
36
|
|
|
37
|
-
The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend
|
|
38
|
-
|
|
39
|
-
For a new cluster, start by installing the command-line tool. If `foretoken --version` already works, go straight to platform installation. If the cluster already has the Foretoken platform, start with model deployment.
|
|
37
|
+
The Foretoken command-line tool installs the shared Kubernetes platform, deploys model services from Kustomize configurations, reports serving readiness, resolves frontend URLs, and runs benchmarks through one `foretoken` entry point.
|
|
40
38
|
|
|
41
39
|
## Before you start
|
|
42
40
|
|
|
43
|
-
You need Python 3.
|
|
41
|
+
You need Python 3.11 or later, an active Kubernetes context, `kubectl`, and Helm. GPU nodes must already have their vendor driver and Kubernetes device plugin.
|
|
44
42
|
|
|
45
43
|
## Install the command-line tool
|
|
46
44
|
|
|
@@ -61,7 +59,7 @@ source .venv/bin/activate
|
|
|
61
59
|
uv pip install foretoken
|
|
62
60
|
```
|
|
63
61
|
|
|
64
|
-
|
|
62
|
+
Run `foretoken --version` to check the installed command-line tool version.
|
|
65
63
|
|
|
66
64
|
## Install the Kubernetes platform
|
|
67
65
|
|
|
@@ -75,11 +73,13 @@ The default uses release images and local access through a `LoadBalancer` Servic
|
|
|
75
73
|
foretoken install
|
|
76
74
|
```
|
|
77
75
|
|
|
78
|
-
|
|
76
|
+
Installation selects the NVIDIA or MetaX runtime from the cluster's GPU resources. Explicit runtime settings in `--values` take precedence; in a mixed-GPU cluster, select a resource with `runtime.vllm.gpu.resourceName` or restrict the nodes with `runtime.vllm.gpu.nodeSelector`.
|
|
77
|
+
|
|
78
|
+
Installation also sets up monitoring, reusing a Prometheus and GPU metrics exporter already in the cluster when they exist. See [Observability](../observability/README.md).
|
|
79
79
|
|
|
80
80
|
### Gateway mode
|
|
81
81
|
|
|
82
|
-
|
|
82
|
+
Gateway mode creates a dedicated `GatewayClass` and `Gateway`, installing Envoy Gateway if no compatible controller is available:
|
|
83
83
|
|
|
84
84
|
```bash
|
|
85
85
|
foretoken install --frontend-mode gateway
|
|
@@ -98,7 +98,7 @@ Add `--gateway-section-name LISTENER` only when more than one listener matches.
|
|
|
98
98
|
|
|
99
99
|
### Current source
|
|
100
100
|
|
|
101
|
-
|
|
101
|
+
Prepare the build tools listed in the [source deployment guide](../docs/custom-deployment.md), then build and install from the repository root:
|
|
102
102
|
|
|
103
103
|
```bash
|
|
104
104
|
foretoken install -e .
|
|
@@ -115,21 +115,31 @@ Registry login authorizes the local image push. Private registries also need `im
|
|
|
115
115
|
|
|
116
116
|
### Installation options
|
|
117
117
|
|
|
118
|
-
Repeatable `--values` files provide platform image, runtime, and hardware settings.
|
|
118
|
+
Repeatable `--values` files provide platform image, runtime, and hardware settings.
|
|
119
119
|
|
|
120
|
-
|
|
120
|
+
Model services are reached through an IP address outside the cluster. k3d, k3s, and cloud clusters assign one automatically. Clusters built with kubeadm, RKE2, or kubespray have no address assignment by default, so installation there ends with `LoadBalancer support Not verified`. Give Foretoken a range of unused addresses in the nodes' subnet, confirmed with the cluster administrator, and it assigns them to services:
|
|
121
121
|
|
|
122
|
-
|
|
122
|
+
```yaml
|
|
123
|
+
loadBalancer:
|
|
124
|
+
managedAddresses:
|
|
125
|
+
- 192.168.1.240-192.168.1.250
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
foretoken install --values platform-values.yaml
|
|
130
|
+
```
|
|
123
131
|
|
|
124
132
|
## Deploy and operate model services
|
|
125
133
|
|
|
126
|
-
|
|
134
|
+
Run from the repository checkout prepared in the [Quick Start](../README.md). Deploy one frontend and all models rendered by a Kustomize root.
|
|
135
|
+
|
|
136
|
+
See the [multi-model example](../examples/multi-model-quickstart/README.md) for resources and [model storage](../docs/model-storage.md) for directory or PVC configuration. Use `examples/quickstart` for a single model.
|
|
127
137
|
|
|
128
138
|
```bash
|
|
129
|
-
foretoken deploy examples/multi-model-quickstart
|
|
139
|
+
foretoken deploy examples/multi-model-quickstart --timeout 20m
|
|
130
140
|
```
|
|
131
141
|
|
|
132
|
-
The command applies the configuration, reports
|
|
142
|
+
The command applies the configuration, reports service state changes, and exits when every service is Ready. Without `--timeout`, it waits up to ten minutes.
|
|
133
143
|
|
|
134
144
|
Inspect the same deployment without applying it:
|
|
135
145
|
|
|
@@ -158,44 +168,17 @@ FORETOKEN_REQUEST_HOST="$(foretoken endpoint examples/multi-model-quickstart --h
|
|
|
158
168
|
|
|
159
169
|
`--host` returns the host and optional port for direct access, or the configured routing hostname for an HTTP Gateway. `foretoken endpoint` waits for the LoadBalancer or Gateway address; use `foretoken deploy` to wait for the services to become ready.
|
|
160
170
|
|
|
161
|
-
##
|
|
162
|
-
|
|
163
|
-
Install the optional benchmark dependencies with pip:
|
|
164
|
-
|
|
165
|
-
```bash
|
|
166
|
-
pip install 'foretoken[bench]'
|
|
167
|
-
|
|
168
|
-
# For source installation from the repository:
|
|
169
|
-
# pip install -e .
|
|
170
|
-
# pip install -e '.[bench]'
|
|
171
|
-
```
|
|
172
|
-
|
|
173
|
-
Or install the benchmark dependencies in the activated uv environment:
|
|
171
|
+
## Benchmark model services
|
|
174
172
|
|
|
175
|
-
|
|
176
|
-
uv pip install 'foretoken[bench]'
|
|
177
|
-
```
|
|
178
|
-
|
|
179
|
-
Then run the benchmark:
|
|
180
|
-
|
|
181
|
-
```bash
|
|
182
|
-
foretoken bench examples/multi-model-quickstart --model Qwen/Qwen3-0.6B
|
|
183
|
-
```
|
|
184
|
-
|
|
185
|
-
The command-line tool uses the active `kubectl` context and honors standard Kubernetes configuration such as `KUBECONFIG`.
|
|
173
|
+
Use `foretoken bench` to measure model-service performance. Commands and examples are in [Model Service Benchmarks](../benchmarks/README.md).
|
|
186
174
|
|
|
187
175
|
## Clean up
|
|
188
176
|
|
|
189
|
-
Delete the
|
|
177
|
+
Delete the deployed services before uninstalling the platform:
|
|
190
178
|
|
|
191
179
|
```bash
|
|
192
180
|
foretoken delete examples/multi-model-quickstart
|
|
193
|
-
```
|
|
194
|
-
|
|
195
|
-
The command waits for deletion and ignores resources that are already absent. After deleting all Foretoken services, remove the platform release:
|
|
196
|
-
|
|
197
|
-
```bash
|
|
198
181
|
foretoken uninstall
|
|
199
182
|
```
|
|
200
183
|
|
|
201
|
-
|
|
184
|
+
Foretoken CRDs and reused cluster components are retained. Managed MetalLB is also retained while other services depend on it.
|
|
@@ -21,50 +21,50 @@ If you only need to serve a single model on one GPU, using an inference engine s
|
|
|
21
21
|
|
|
22
22
|
| Feature | Description | Status |
|
|
23
23
|
|---|---|---|
|
|
24
|
-
| Benchmarking |
|
|
24
|
+
| [Benchmarking](benchmarks/README.md) | Measure model-service performance | In development |
|
|
25
25
|
| Profiling | Use PyTorch Profiler and Nsight to identify compute, communication, and CPU/GPU bottlenecks | Planned |
|
|
26
26
|
| Hardware support | Common interfaces for device capabilities, runtimes, communication, and metrics; see [MetaX deployment](docs/metax-deployment.md) | In development |
|
|
27
27
|
| Request routing | Select instances based on load, queues, KV reuse, and service levels | Research |
|
|
28
28
|
| Distributed inference | Aggregated serving, Prefill/Decode disaggregation, and WideEP parallelism | Research |
|
|
29
29
|
| Control plane | Model services, replica management, autoscaling, updates, and failure recovery | In development |
|
|
30
|
-
| [Observability](observability/README.md) | Collect
|
|
30
|
+
| [Observability](observability/README.md) | Collect service and accelerator metrics, evaluate alerts, and inspect the system Dashboard | In development |
|
|
31
31
|
|
|
32
32
|
## Quick Start
|
|
33
33
|
|
|
34
|
-
|
|
34
|
+
Start with a GPU-enabled Kubernetes cluster and Python 3.11+, `kubectl`, and Helm installed locally.
|
|
35
35
|
|
|
36
|
-
### 1.
|
|
37
|
-
|
|
38
|
-
Install the published command-line tool package:
|
|
36
|
+
### 1. Get the examples and install the command-line tool
|
|
39
37
|
|
|
40
38
|
```bash
|
|
39
|
+
git clone https://github.com/shiweijiezero/foretoken.git
|
|
40
|
+
cd foretoken
|
|
41
41
|
pip install foretoken
|
|
42
42
|
|
|
43
|
-
#
|
|
43
|
+
# From a source checkout:
|
|
44
44
|
# pip install -e .
|
|
45
45
|
```
|
|
46
46
|
|
|
47
47
|
### 2. Install the Kubernetes platform
|
|
48
48
|
|
|
49
|
-
By default, installation uses the Foretoken images published on GHCR:
|
|
50
|
-
|
|
51
49
|
```bash
|
|
52
|
-
#
|
|
50
|
+
# Use release images from GHCR:
|
|
53
51
|
foretoken install
|
|
54
52
|
|
|
55
|
-
#
|
|
53
|
+
# Build and install from a source checkout:
|
|
56
54
|
# foretoken install -e .
|
|
57
55
|
```
|
|
58
56
|
|
|
59
|
-
|
|
57
|
+
For deployment on MetaX GPUs, follow the [MetaX deployment guide](docs/metax-deployment.md).
|
|
58
|
+
|
|
59
|
+
See the [source deployment guide](docs/custom-deployment.md) for build tools and remote clusters.
|
|
60
60
|
|
|
61
61
|
### 3. Deploy the Quick Start
|
|
62
62
|
|
|
63
63
|
```bash
|
|
64
|
-
foretoken deploy examples/quickstart
|
|
64
|
+
foretoken deploy examples/quickstart --timeout 20m
|
|
65
65
|
```
|
|
66
66
|
|
|
67
|
-
This example deploys one frontend service
|
|
67
|
+
This example deploys one frontend service and one `Qwen/Qwen3-0.6B` model replica, requesting one GPU, 8 CPU, and 52 GiB memory. More deployments are available in [`examples/`](examples/).
|
|
68
68
|
|
|
69
69
|
### 4. Send a test request
|
|
70
70
|
|
|
@@ -77,24 +77,24 @@ curl --fail-with-body --no-buffer \
|
|
|
77
77
|
-d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
### 5.
|
|
80
|
+
### 5. Measure the model service
|
|
81
81
|
|
|
82
82
|
```bash
|
|
83
83
|
pip install 'foretoken[bench]'
|
|
84
84
|
|
|
85
|
-
#
|
|
86
|
-
# pip install -e .
|
|
85
|
+
# From a source checkout:
|
|
87
86
|
# pip install -e '.[bench]'
|
|
88
|
-
|
|
87
|
+
|
|
88
|
+
foretoken bench examples/quickstart --output local,wandb
|
|
89
89
|
```
|
|
90
90
|
|
|
91
|
-
See [
|
|
91
|
+
Run `wandb login` before first using W&B. See [Model Service Benchmarks](benchmarks/README.md) for more examples.
|
|
92
92
|
|
|
93
93
|
## Gateway Mode
|
|
94
94
|
|
|
95
95
|
Gateway mode provides a shared entry point through Kubernetes Gateway and a hostname. It suits clusters that already use Gateway or manage external traffic centrally.
|
|
96
96
|
|
|
97
|
-
|
|
97
|
+
Add the public hostname under `spec` in `examples/quickstart/frontend.yaml`:
|
|
98
98
|
|
|
99
99
|
```yaml
|
|
100
100
|
spec:
|
|
@@ -104,18 +104,13 @@ spec:
|
|
|
104
104
|
Then run:
|
|
105
105
|
|
|
106
106
|
```bash
|
|
107
|
-
# Install Envoy Gateway
|
|
108
|
-
helm upgrade --install envoy-gateway \
|
|
109
|
-
oci://docker.io/envoyproxy/gateway-helm \
|
|
110
|
-
--namespace envoy-gateway-system \
|
|
111
|
-
--create-namespace \
|
|
112
|
-
--wait
|
|
113
|
-
|
|
114
107
|
# Install the platform in Gateway mode
|
|
115
108
|
foretoken install --frontend-mode gateway
|
|
109
|
+
# For a source-installed platform:
|
|
110
|
+
# foretoken install -e . --frontend-mode gateway
|
|
116
111
|
|
|
117
112
|
# Deploy the Quick Start
|
|
118
|
-
foretoken deploy examples/quickstart
|
|
113
|
+
foretoken deploy examples/quickstart --timeout 20m
|
|
119
114
|
|
|
120
115
|
# Resolve the Gateway address and request hostname
|
|
121
116
|
FORETOKEN_FRONTEND_URL="$(foretoken endpoint examples/quickstart)"
|
|
@@ -129,7 +124,7 @@ curl --fail-with-body --no-buffer \
|
|
|
129
124
|
-d '{"model":"Qwen/Qwen3-0.6B","messages":[{"role":"user","content":"Hello"}],"stream":true}'
|
|
130
125
|
```
|
|
131
126
|
|
|
132
|
-
See the [command-line tool guide](cli/README.md) to reuse
|
|
127
|
+
The command installs Envoy Gateway when needed. See the [command-line tool guide](cli/README.md) to reuse an existing Gateway or select a listener.
|
|
133
128
|
|
|
134
129
|
## Stop and Uninstall
|
|
135
130
|
|
|
@@ -143,6 +138,12 @@ foretoken uninstall
|
|
|
143
138
|
|
|
144
139
|
The uninstall command preserves Foretoken CRDs and reused cluster components. It removes the platform and the monitoring or Gateway resources managed by the command-line tool.
|
|
145
140
|
|
|
141
|
+
## Deployment Guides
|
|
142
|
+
|
|
143
|
+
- [Source builds and private registries](docs/custom-deployment.md)
|
|
144
|
+
- [Single-machine GPU clusters with k3d](docs/k3d-deployment.md)
|
|
145
|
+
- [MetaX GPUs](docs/metax-deployment.md)
|
|
146
|
+
|
|
146
147
|
## Related Projects
|
|
147
148
|
|
|
148
149
|
- [vLLM](https://github.com/vllm-project/vllm)
|
|
@@ -160,7 +161,7 @@ See [Contributing to Foretoken](CONTRIBUTING.md) for development principles, col
|
|
|
160
161
|
Thank you to everyone who has contributed to Foretoken.
|
|
161
162
|
|
|
162
163
|
<a href="https://github.com/shiweijiezero/foretoken/graphs/contributors">
|
|
163
|
-
<img src="https://contrib.rocks/image?repo=shiweijiezero/foretoken" alt="Foretoken contributors" />
|
|
164
|
+
<img src="https://contrib.rocks/image?repo=shiweijiezero/foretoken" width="256" alt="Foretoken contributors" />
|
|
164
165
|
</a>
|
|
165
166
|
|
|
166
167
|
## License
|