workingset 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. workingset-0.1.0/.gitignore +26 -0
  2. workingset-0.1.0/LICENSE +21 -0
  3. workingset-0.1.0/PKG-INFO +118 -0
  4. workingset-0.1.0/README.md +94 -0
  5. workingset-0.1.0/data/README.md +24 -0
  6. workingset-0.1.0/pyproject.toml +49 -0
  7. workingset-0.1.0/src/workingset/__init__.py +6 -0
  8. workingset-0.1.0/src/workingset/cli.py +412 -0
  9. workingset-0.1.0/src/workingset/config.py +411 -0
  10. workingset-0.1.0/src/workingset/hypotheses/__init__.py +152 -0
  11. workingset-0.1.0/src/workingset/hypotheses/base.py +230 -0
  12. workingset-0.1.0/src/workingset/hypotheses/burst.py +85 -0
  13. workingset-0.1.0/src/workingset/hypotheses/ceilings.py +226 -0
  14. workingset-0.1.0/src/workingset/hypotheses/context.py +339 -0
  15. workingset-0.1.0/src/workingset/hypotheses/gaps.py +646 -0
  16. workingset-0.1.0/src/workingset/metrics/__init__.py +59 -0
  17. workingset-0.1.0/src/workingset/metrics/adapter.py +340 -0
  18. workingset-0.1.0/src/workingset/metrics/cli.py +310 -0
  19. workingset-0.1.0/src/workingset/metrics/parse.py +384 -0
  20. workingset-0.1.0/src/workingset/metrics/sampler.py +1141 -0
  21. workingset-0.1.0/src/workingset/metrics/vllm.py +311 -0
  22. workingset-0.1.0/src/workingset/model.py +3720 -0
  23. workingset-0.1.0/src/workingset/predict.py +225 -0
  24. workingset-0.1.0/src/workingset/probe/__init__.py +49 -0
  25. workingset-0.1.0/src/workingset/probe/burst.py +168 -0
  26. workingset-0.1.0/src/workingset/probe/ladder.py +30 -0
  27. workingset-0.1.0/src/workingset/probe/options.py +45 -0
  28. workingset-0.1.0/src/workingset/probe/population.py +511 -0
  29. workingset-0.1.0/src/workingset/probe/request.py +470 -0
  30. workingset-0.1.0/src/workingset/probe/session.py +186 -0
  31. workingset-0.1.0/src/workingset/probe/stats.py +58 -0
  32. workingset-0.1.0/src/workingset/record.py +290 -0
  33. workingset-0.1.0/src/workingset/report.py +358 -0
  34. workingset-0.1.0/src/workingset/shared.py +2163 -0
  35. workingset-0.1.0/src/workingset/test_cmd.py +491 -0
  36. workingset-0.1.0/src/workingset/workload.py +2061 -0
  37. workingset-0.1.0/tests/golden/README.md +332 -0
@@ -0,0 +1,26 @@
1
+ # Data (real provider CSVs are private / not committed)
2
+ data/*.csv
3
+
4
+ # Think-time trace exports (session names + timestamps are private; only the
5
+ # MEASURED_* anchors in scripts/scenario_model.py are committed)
6
+ inter_event_gaps*.csv
7
+
8
+ # Python
9
+ __pycache__/
10
+ *.pyc
11
+ .venv/
12
+ venv/
13
+ .ipynb_checkpoints/
14
+
15
+ # OS / editor
16
+ .DS_Store
17
+
18
+ # Local agent configs for cross-review runs (installed per-machine by the
19
+ # delegate skills; not part of the study)
20
+ .opencode/
21
+
22
+ # Generated in CI by preview-explorer.yml (the committed wrangler.jsonc is
23
+ # production's); never committed
24
+ /wrangler.preview.jsonc
25
+ /wrangler.canary.jsonc
26
+ /.preview-canary/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Tom Vaucourt
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,118 @@
1
+ Metadata-Version: 2.5
2
+ Name: workingset
3
+ Version: 0.1.0
4
+ Summary: How many agentic-coding users a local LLM deployment serves: capacity model, CLI and live-endpoint hypothesis tests
5
+ Project-URL: Homepage, https://workingset.tomvaucourt.com
6
+ Project-URL: Repository, https://github.com/T0mSIlver/working-set
7
+ Author: Tom Vaucourt
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: benchmark,capacity-planning,kv-cache,llm,serving,vllm
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: System :: Benchmark
16
+ Requires-Python: >=3.11
17
+ Requires-Dist: httpx>=0.27
18
+ Requires-Dist: numpy>=2.0
19
+ Provides-Extra: study
20
+ Requires-Dist: matplotlib>=3.9; extra == 'study'
21
+ Requires-Dist: pandas>=2.0; extra == 'study'
22
+ Requires-Dist: scipy>=1.13; extra == 'study'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # Working Set
26
+
27
+ Tools for scaling **local LLM deployments**: how many concurrent users or
28
+ agents a given GPU configuration can keep warm, where KV cache, decode
29
+ bandwidth, and prefill compute each become the binding constraint, and which
30
+ knob (topology, dtypes, `max_num_seqs`, prompt caching) buys the most headroom
31
+ for agentic coding workloads.
32
+
33
+ **Start with the [interactive explorer](https://workingset.tomvaucourt.com/)** —
34
+ live sliders for the workload, model (Qwen3.8-27B / 35B-A3B /
35
+ Mistral-Medium-3.5 / GLM-5.3 / DeepSeek-V4-Flash / Qwen3.8-Flash-Next /
36
+ GLM-5.3-Flash), GPU (H200 / B300), weight & KV dtypes, and DP × TP
37
+ topology. It answers as a decision tool: a binding-constraint verdict, a
38
+ deploy recipe (vLLM flags), the **bill** (€/GPU-hour and €/kWh sliders:
39
+ hardware plus a duty-cycle power model), a **sensitivity panel** showing which assumption
40
+ would flip the decision, the **steady-state decode point** (how many
41
+ sessions are actually decoding at your load, and how fast each one runs —
42
+ Little's law, not the all-warm stress test), **shareable links** that
43
+ encode the whole configuration, and a **"Test these hypotheses" button**
44
+ that hands out the configuration on screen as a `workingset.toml` — feed it to
45
+ `ws test` below and measure the real limits on a live vLLM endpoint.
46
+
47
+ ## The `workingset` package
48
+
49
+ The model behind the explorer is a Python package (`src/workingset/`, the
50
+ source of truth; the explorer's JS mirrors it). It ships a CLI:
51
+
52
+ ```bash
53
+ uv run ws init --model Q38FN --gpu B300 --tp 8 --weight-dtype nvfp4 # writes workingset.toml
54
+ uv run ws predict workingset.toml # the four ceilings, which one binds, the operating point
55
+ uv run ws predict workingset.toml --json # the same as a run record
56
+ uv run ws hypotheses # the H-* and what each one needs
57
+ uv run ws test workingset.toml --dry-run # the plan, the sampler self-check, no requests
58
+ uv run ws test workingset.toml --exclusive --out run.json # measure it
59
+ uv run ws report run.json # re-print the verdicts
60
+ uv run ws models # model / GPU keys
61
+ uv run pytest # self-checks + config round-trips
62
+ ```
63
+
64
+ No checkout needed — the explorer's `workingset.toml` runs straight from git
65
+ (`--from` carries the package because `workingset` publishes one console
66
+ script, `ws`):
67
+
68
+ ```bash
69
+ uvx --from git+https://github.com/T0mSIlver/working-set ws predict workingset.toml
70
+ uvx --from git+https://github.com/T0mSIlver/working-set ws test workingset.toml --dry-run
71
+ uvx --from git+https://github.com/T0mSIlver/working-set ws test workingset.toml --all --exclusive --out run.json
72
+ ```
73
+
74
+ After the PyPI release the same commands shorten to `uvx --from workingset
75
+ ws …`.
76
+
77
+ Predictions live in no file: `ws predict` recomputes them from the config every
78
+ time, so a config can never carry a number the code did not produce. A harness
79
+ `.py` downloaded from the explorer before the package existed still loads (its
80
+ CONFIG block is extracted).
81
+
82
+ `ws test` puts the predictions to a live endpoint, one falsifiable hypothesis
83
+ at a time. Without `--exclusive` it runs only the hypotheses that need a
84
+ handful of requests (miss TTFT, the inter-token gap distribution, the steady
85
+ decode point) and lists the rest as skipped — a hypothesis that has to
86
+ generate its own population is never measured against someone else's load.
87
+ With `--exclusive` it drives the geometric load ladder once, and every ceiling
88
+ reads from it. `--burst N` adds the correlated-flush probe (B*).
89
+
90
+ ## Contents
91
+
92
+ - [docs/writeup.md](docs/writeup.md) — baseline study: KV-cache capacity and
93
+ the prompt-caching / offload / `max_num_seqs` trade-offs.
94
+ - [docs/scenarios.md](docs/scenarios.md) — extended scenario model: multi-GPU
95
+ topologies, MoE vs dense, subagent workloads, the cost of a cache miss, and
96
+ cold-spike tolerance.
97
+ - [scripts/](scripts/) — everything is reproducible:
98
+
99
+ ```bash
100
+ uv run ws selfcheck # the shared model's self-checks (src/workingset/model.py)
101
+ uv run scripts/scenarios.py # renders the scenario figures
102
+ uv run scripts/tables.py # regenerates every number in docs/scenarios.md
103
+ ```
104
+
105
+ - [research/](research/) — sourced constants for each model and GPU.
106
+ - [interactive/](interactive/) — the explorer, a dependency-free page
107
+ mirroring the Python model: `index.html` holds the markup and styles,
108
+ `src/*.js` the model and the charts as ES modules (`src/main.js` is the
109
+ entry and lists the layering). Browsers refuse module scripts from
110
+ `file://`, so serve the folder to open it locally:
111
+
112
+ ```sh
113
+ python3 -m http.server 8000 --directory interactive # then http://localhost:8000
114
+ ```
115
+
116
+ Method, calibration, and caveats are laid out in the docs above.
117
+
118
+ MIT licensed; see [LICENSE](LICENSE).
@@ -0,0 +1,94 @@
1
+ # Working Set
2
+
3
+ Tools for scaling **local LLM deployments**: how many concurrent users or
4
+ agents a given GPU configuration can keep warm, where KV cache, decode
5
+ bandwidth, and prefill compute each become the binding constraint, and which
6
+ knob (topology, dtypes, `max_num_seqs`, prompt caching) buys the most headroom
7
+ for agentic coding workloads.
8
+
9
+ **Start with the [interactive explorer](https://workingset.tomvaucourt.com/)** —
10
+ live sliders for the workload, model (Qwen3.8-27B / 35B-A3B /
11
+ Mistral-Medium-3.5 / GLM-5.3 / DeepSeek-V4-Flash / Qwen3.8-Flash-Next /
12
+ GLM-5.3-Flash), GPU (H200 / B300), weight & KV dtypes, and DP × TP
13
+ topology. It answers as a decision tool: a binding-constraint verdict, a
14
+ deploy recipe (vLLM flags), the **bill** (€/GPU-hour and €/kWh sliders:
15
+ hardware plus a duty-cycle power model), a **sensitivity panel** showing which assumption
16
+ would flip the decision, the **steady-state decode point** (how many
17
+ sessions are actually decoding at your load, and how fast each one runs —
18
+ Little's law, not the all-warm stress test), **shareable links** that
19
+ encode the whole configuration, and a **"Test these hypotheses" button**
20
+ that hands out the configuration on screen as a `workingset.toml` — feed it to
21
+ `ws test` below and measure the real limits on a live vLLM endpoint.
22
+
23
+ ## The `workingset` package
24
+
25
+ The model behind the explorer is a Python package (`src/workingset/`, the
26
+ source of truth; the explorer's JS mirrors it). It ships a CLI:
27
+
28
+ ```bash
29
+ uv run ws init --model Q38FN --gpu B300 --tp 8 --weight-dtype nvfp4 # writes workingset.toml
30
+ uv run ws predict workingset.toml # the four ceilings, which one binds, the operating point
31
+ uv run ws predict workingset.toml --json # the same as a run record
32
+ uv run ws hypotheses # the H-* and what each one needs
33
+ uv run ws test workingset.toml --dry-run # the plan, the sampler self-check, no requests
34
+ uv run ws test workingset.toml --exclusive --out run.json # measure it
35
+ uv run ws report run.json # re-print the verdicts
36
+ uv run ws models # model / GPU keys
37
+ uv run pytest # self-checks + config round-trips
38
+ ```
39
+
40
+ No checkout needed — the explorer's `workingset.toml` runs straight from git
41
+ (`--from` carries the package because `workingset` publishes one console
42
+ script, `ws`):
43
+
44
+ ```bash
45
+ uvx --from git+https://github.com/T0mSIlver/working-set ws predict workingset.toml
46
+ uvx --from git+https://github.com/T0mSIlver/working-set ws test workingset.toml --dry-run
47
+ uvx --from git+https://github.com/T0mSIlver/working-set ws test workingset.toml --all --exclusive --out run.json
48
+ ```
49
+
50
+ After the PyPI release the same commands shorten to `uvx --from workingset
51
+ ws …`.
52
+
53
+ Predictions live in no file: `ws predict` recomputes them from the config every
54
+ time, so a config can never carry a number the code did not produce. A harness
55
+ `.py` downloaded from the explorer before the package existed still loads (its
56
+ CONFIG block is extracted).
57
+
58
+ `ws test` puts the predictions to a live endpoint, one falsifiable hypothesis
59
+ at a time. Without `--exclusive` it runs only the hypotheses that need a
60
+ handful of requests (miss TTFT, the inter-token gap distribution, the steady
61
+ decode point) and lists the rest as skipped — a hypothesis that has to
62
+ generate its own population is never measured against someone else's load.
63
+ With `--exclusive` it drives the geometric load ladder once, and every ceiling
64
+ reads from it. `--burst N` adds the correlated-flush probe (B*).
65
+
66
+ ## Contents
67
+
68
+ - [docs/writeup.md](docs/writeup.md) — baseline study: KV-cache capacity and
69
+ the prompt-caching / offload / `max_num_seqs` trade-offs.
70
+ - [docs/scenarios.md](docs/scenarios.md) — extended scenario model: multi-GPU
71
+ topologies, MoE vs dense, subagent workloads, the cost of a cache miss, and
72
+ cold-spike tolerance.
73
+ - [scripts/](scripts/) — everything is reproducible:
74
+
75
+ ```bash
76
+ uv run ws selfcheck # the shared model's self-checks (src/workingset/model.py)
77
+ uv run scripts/scenarios.py # renders the scenario figures
78
+ uv run scripts/tables.py # regenerates every number in docs/scenarios.md
79
+ ```
80
+
81
+ - [research/](research/) — sourced constants for each model and GPU.
82
+ - [interactive/](interactive/) — the explorer, a dependency-free page
83
+ mirroring the Python model: `index.html` holds the markup and styles,
84
+ `src/*.js` the model and the charts as ES modules (`src/main.js` is the
85
+ entry and lists the layering). Browsers refuse module scripts from
86
+ `file://`, so serve the folder to open it locally:
87
+
88
+ ```sh
89
+ python3 -m http.server 8000 --directory interactive # then http://localhost:8000
90
+ ```
91
+
92
+ Method, calibration, and caveats are laid out in the docs above.
93
+
94
+ MIT licensed; see [LICENSE](LICENSE).
@@ -0,0 +1,24 @@
1
+ # data/
2
+
3
+ The real-workload scripts (`real_capacity.py`, `real_mns.py`, `warm_whisker.py`)
4
+ read two CSV files of observed prompt lengths from this directory:
5
+
6
+ | File | Column | Source |
7
+ | ------------------------------------------ | --------------- | ------------------------------- |
8
+ | `prompt_tokens_by_response_uid_igp.csv` | `prompt_tokens` | IGP coding-agent responses |
9
+ | `prompt_tokens_by_response_uid_watsonx.csv`| `prompt_tokens` | watsonX coding-agent responses |
10
+
11
+ Each file is one row per response, with a `prompt_tokens` integer column
12
+ (the prompt length in tokens). Prompts shorter than `MIN_TOKENS` (1000) are
13
+ dropped as junk during cleaning.
14
+
15
+ These CSVs contain personal usage data and are **not** committed to the repo
16
+ (see `.gitignore`). Drop your own copies here, or point the scripts at another
17
+ location with the `DATA_DIR` environment variable:
18
+
19
+ ```bash
20
+ DATA_DIR=/path/to/csvs python scripts/real_mns.py
21
+ ```
22
+
23
+ `warm_capacity.py` is fully synthetic (it sweeps hypothetical distributions in
24
+ its `HYPOTHESES` block) and needs no CSVs.
@@ -0,0 +1,49 @@
1
+ [project]
2
+ name = "workingset"
3
+ version = "0.1.0"
4
+ description = "How many agentic-coding users a local LLM deployment serves: capacity model, CLI and live-endpoint hypothesis tests"
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ authors = [{ name = "Tom Vaucourt" }]
8
+ license = "MIT"
9
+ license-files = ["LICENSE"]
10
+ keywords = ["vllm", "llm", "serving", "capacity-planning", "kv-cache", "benchmark"]
11
+ classifiers = [
12
+ "Development Status :: 3 - Alpha",
13
+ "Environment :: Console",
14
+ "Intended Audience :: Developers",
15
+ "Programming Language :: Python :: 3",
16
+ "Topic :: System :: Benchmark",
17
+ ]
18
+ dependencies = [
19
+ "numpy>=2.0",
20
+ "httpx>=0.27",
21
+ ]
22
+
23
+ [project.optional-dependencies]
24
+ # the study scripts under scripts/ (figures, tables, traces)
25
+ study = ["pandas>=2.0", "scipy>=1.13", "matplotlib>=3.9"]
26
+
27
+ [project.scripts]
28
+ ws = "workingset.cli:main"
29
+
30
+ [project.urls]
31
+ Homepage = "https://workingset.tomvaucourt.com"
32
+ Repository = "https://github.com/T0mSIlver/working-set"
33
+
34
+ [build-system]
35
+ requires = ["hatchling>=1.27"]
36
+ build-backend = "hatchling.build"
37
+
38
+ [tool.hatch.build.targets.wheel]
39
+ packages = ["src/workingset"]
40
+
41
+ [tool.hatch.build.targets.sdist]
42
+ include = ["src/workingset", "README.md", "LICENSE"]
43
+
44
+ [dependency-groups]
45
+ dev = ["pytest>=8", "pandas>=2.0", "scipy>=1.13", "matplotlib>=3.9"]
46
+
47
+ [tool.pytest.ini_options]
48
+ testpaths = ["tests"]
49
+ markers = ["slow: calibration self-checks (~seconds)"]
@@ -0,0 +1,6 @@
1
+ """workingset — serving-capacity model, CLI and measurement harness.
2
+
3
+ `workingset.model` is the source of truth for every number the study and the
4
+ explorer publish; the explorer's JS mirror is verified against it.
5
+ """
6
+ __version__ = "0.1.0"