logit-classifier 0.2.1__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/PKG-INFO +77 -3
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/README.md +75 -1
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/pyproject.toml +3 -3
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/_version.py +1 -1
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/comfy_clip.py +25 -1
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/classifier.py +25 -1
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/prompt.py +3 -3
- logit_classifier-0.3.0/src/logit_classifier/tags.py +36 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/__init__.py +5 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/__init__.py +50 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/classifier.py +39 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/generate.py +146 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/progress.py +86 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/scopes.py +77 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/comfyui/tagger.py +390 -0
- logit_classifier-0.3.0/src/logit_classifier/toolkit/tags.py +372 -0
- logit_classifier-0.3.0/tests/test_toolkit.py +1445 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/test_unit.py +294 -218
- logit_classifier-0.3.0/tests-AB/comfy_env.py +167 -0
- logit_classifier-0.2.1/src/logit_classifier/tags.py +0 -171
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/.gitignore +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/FINDINGS.md +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/LICENSE +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/compare_models.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/http_client.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/images.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/own_backend.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/question_types.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/examples/quickstart.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/__init__.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/__main__.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/__init__.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/_torch_window.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/base.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/hf.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/calibrate.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/cli.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/config.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/deps.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/errors.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/labels.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/py.typed +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/schema.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/scoring.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/service.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/vision.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/web/index.html +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/fixtures/banking77_test.json +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/fixtures/eval_set.json +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/fixtures/many_options_request.json +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/fixtures/quickstart_request.json +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests/test_model.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_branch_packing.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_determinism_scope.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_env.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_math_sdp_reduction.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_multi_label.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_noul_wording.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/ab_temperature.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/banking77.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/benchmark.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/evaluate.py +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/_full.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/_sheet.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/low-L.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/low-R.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/mid-L.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/mid-R.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/top-L.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/inputs/top-R.png +0 -0
- {logit_classifier-0.2.1 → logit_classifier-0.3.0}/tests-AB/tune_groups.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: logit-classifier
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Local zero-shot classifier for text and images. Declares options
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Local zero-shot classifier for text and images. Declares options and reads a calibrated probability for each from the logits. Accepts the TypeSafe System One request format, and ships a toolkit of image tag tools and ComfyUI node tools.
|
|
5
5
|
Project-URL: Repository, https://github.com/Blakeem/logit-classifier
|
|
6
6
|
Project-URL: Bug Tracker, https://github.com/Blakeem/logit-classifier/issues
|
|
7
7
|
Author: Blake
|
|
@@ -124,6 +124,14 @@ print(response.to_dict())
|
|
|
124
124
|
`load_model` loads the weights through the `hf` extra and returns a `Backend`.
|
|
125
125
|
`Classifier(Config())` calls it for you when you pass no backend.
|
|
126
126
|
|
|
127
|
+
`classifier.nouls` asks a list of statements about one state and returns one probability
|
|
128
|
+
per statement. It sends more than 256 statements as several requests.
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
probabilities = classifier.nouls(["The message is urgent", "The message is polite"],
|
|
132
|
+
state="the payment failed again")
|
|
133
|
+
```
|
|
134
|
+
|
|
127
135
|
Loading each model yourself is what lets one script compare several. The fitted
|
|
128
136
|
temperature follows the backend, so every model is scored with its own value.
|
|
129
137
|
|
|
@@ -170,7 +178,7 @@ the model.
|
|
|
170
178
|
|
|
171
179
|
```json
|
|
172
180
|
{
|
|
173
|
-
"model": "logit-classifier-0.
|
|
181
|
+
"model": "logit-classifier-0.3.0",
|
|
174
182
|
"answers": {
|
|
175
183
|
"department": {
|
|
176
184
|
"type": "choice",
|
|
@@ -342,6 +350,72 @@ shape `[1, H, W, 3]`. An empty state asks about the image alone. Questions share
|
|
|
342
350
|
pass of up to 4,096 tokens, and the image counts toward that limit. Any text encoder other
|
|
343
351
|
than Qwen3-VL raises `UnsupportedModelError`.
|
|
344
352
|
|
|
353
|
+
## Toolkit
|
|
354
|
+
|
|
355
|
+
`logit_classifier.toolkit` holds tools built on the classifier with the settings our
|
|
356
|
+
measurements found best. A project imports them to get good results without working out the
|
|
357
|
+
wording, thresholds and host details itself. The toolkit uses the classifier's public API,
|
|
358
|
+
and the classifier never imports the toolkit.
|
|
359
|
+
|
|
360
|
+
| Package | Runs in | Holds |
|
|
361
|
+
|---|---|---|
|
|
362
|
+
| `logit_classifier.toolkit.tags` | any Python process | text tools for image tags and prompts |
|
|
363
|
+
| `logit_classifier.toolkit.comfyui` | a ComfyUI node | a classifier over the workflow's CLIP, text generation, a progress display and an image tagger |
|
|
364
|
+
|
|
365
|
+
### Tag Text Tools
|
|
366
|
+
|
|
367
|
+
| Tool | Does |
|
|
368
|
+
|---|---|
|
|
369
|
+
| `parse_candidates` | splits a model's tag reply into clean tags with no duplicates |
|
|
370
|
+
| `split_prompt` | splits an image prompt into its fragments |
|
|
371
|
+
| `drop_subsets` | drops a tag whose words all appear in a longer tag |
|
|
372
|
+
| `prompt_word_forms`, `in_prompt` | test whether a tag's words appear in a prompt, singular or plural |
|
|
373
|
+
| `merge_candidates` | merges proposed tags with a prompt's tags under a cap |
|
|
374
|
+
| `presence_question` | writes "Is there a car in this image?" with the article the thing takes |
|
|
375
|
+
|
|
376
|
+
```python
|
|
377
|
+
from logit_classifier.toolkit.tags import drop_subsets, parse_candidates
|
|
378
|
+
|
|
379
|
+
tags = drop_subsets(parse_candidates("red car, car, wooden bench, bench"))
|
|
380
|
+
# ['red car', 'wooden bench']
|
|
381
|
+
```
|
|
382
|
+
|
|
383
|
+
`drop_subsets(tags, rule="head-noun")` drops a tag only into a longer tag with the same head
|
|
384
|
+
noun. The head noun is the last word before "of", "in", "on", "with" or "reading". So "cat"
|
|
385
|
+
stays beside "cat ear", and "signage" is dropped beside "signage in chinese".
|
|
386
|
+
|
|
387
|
+
### ComfyUI Tools
|
|
388
|
+
|
|
389
|
+
These run inside a ComfyUI node over the CLIP the workflow loaded. Each one imports comfy and
|
|
390
|
+
torch only when it is called.
|
|
391
|
+
|
|
392
|
+
| Tool | Does |
|
|
393
|
+
|---|---|
|
|
394
|
+
| `comfy_classifier` | builds a classifier over the CLIP, with errors that name the node and the Load CLIP fix |
|
|
395
|
+
| `generate_text` | generates one greedy reply with the Qwen chat template, and stops early on a condition |
|
|
396
|
+
| `shared_vision_encode` | runs the vision tower once per picture for every generate and classify inside the block |
|
|
397
|
+
| `skip_resident_loads` | skips core's model load while the CLIP is already on the GPU |
|
|
398
|
+
| `routed_progress_bars`, `send_status`, `run_outcome` | show one progress bar per run and a status line under the node |
|
|
399
|
+
| `fit_picture` | downscales an image to a pixel cap |
|
|
400
|
+
| `prompt_tags` | lists the physical things a prompt names |
|
|
401
|
+
| `tag_picture` | tags a picture, with a prompt's things added as candidates |
|
|
402
|
+
| `TagSettings` | holds every wording, budget, cap and threshold of the tagger |
|
|
403
|
+
|
|
404
|
+
The tagger generates candidate tags over the CLIP and verifies each one with the classifier.
|
|
405
|
+
|
|
406
|
+
```python
|
|
407
|
+
from logit_classifier.toolkit.comfyui import TagSettings, comfy_classifier, fit_picture, tag_picture
|
|
408
|
+
|
|
409
|
+
classifier = comfy_classifier(clip, node="My Tagger")
|
|
410
|
+
picture = fit_picture(image, 1024 * 1024)
|
|
411
|
+
trace = tag_picture(clip, classifier, picture, settings=TagSettings(), known={})
|
|
412
|
+
tags = trace.kept
|
|
413
|
+
```
|
|
414
|
+
|
|
415
|
+
`known` holds the thing check's answers across a run, so each tag is asked once. The
|
|
416
|
+
`TagSettings` defaults are the current measured values, and a later release may change them.
|
|
417
|
+
Pass every value your output depends on to keep it fixed.
|
|
418
|
+
|
|
345
419
|
## Differences From Jev
|
|
346
420
|
|
|
347
421
|
This project matches the System One request and response format and the documented limits.
|
|
@@ -91,6 +91,14 @@ print(response.to_dict())
|
|
|
91
91
|
`load_model` loads the weights through the `hf` extra and returns a `Backend`.
|
|
92
92
|
`Classifier(Config())` calls it for you when you pass no backend.
|
|
93
93
|
|
|
94
|
+
`classifier.nouls` asks a list of statements about one state and returns one probability
|
|
95
|
+
per statement. It sends more than 256 statements as several requests.
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
probabilities = classifier.nouls(["The message is urgent", "The message is polite"],
|
|
99
|
+
state="the payment failed again")
|
|
100
|
+
```
|
|
101
|
+
|
|
94
102
|
Loading each model yourself is what lets one script compare several. The fitted
|
|
95
103
|
temperature follows the backend, so every model is scored with its own value.
|
|
96
104
|
|
|
@@ -137,7 +145,7 @@ the model.
|
|
|
137
145
|
|
|
138
146
|
```json
|
|
139
147
|
{
|
|
140
|
-
"model": "logit-classifier-0.
|
|
148
|
+
"model": "logit-classifier-0.3.0",
|
|
141
149
|
"answers": {
|
|
142
150
|
"department": {
|
|
143
151
|
"type": "choice",
|
|
@@ -309,6 +317,72 @@ shape `[1, H, W, 3]`. An empty state asks about the image alone. Questions share
|
|
|
309
317
|
pass of up to 4,096 tokens, and the image counts toward that limit. Any text encoder other
|
|
310
318
|
than Qwen3-VL raises `UnsupportedModelError`.
|
|
311
319
|
|
|
320
|
+
## Toolkit
|
|
321
|
+
|
|
322
|
+
`logit_classifier.toolkit` holds tools built on the classifier with the settings our
|
|
323
|
+
measurements found best. A project imports them to get good results without working out the
|
|
324
|
+
wording, thresholds and host details itself. The toolkit uses the classifier's public API,
|
|
325
|
+
and the classifier never imports the toolkit.
|
|
326
|
+
|
|
327
|
+
| Package | Runs in | Holds |
|
|
328
|
+
|---|---|---|
|
|
329
|
+
| `logit_classifier.toolkit.tags` | any Python process | text tools for image tags and prompts |
|
|
330
|
+
| `logit_classifier.toolkit.comfyui` | a ComfyUI node | a classifier over the workflow's CLIP, text generation, a progress display and an image tagger |
|
|
331
|
+
|
|
332
|
+
### Tag Text Tools
|
|
333
|
+
|
|
334
|
+
| Tool | Does |
|
|
335
|
+
|---|---|
|
|
336
|
+
| `parse_candidates` | splits a model's tag reply into clean tags with no duplicates |
|
|
337
|
+
| `split_prompt` | splits an image prompt into its fragments |
|
|
338
|
+
| `drop_subsets` | drops a tag whose words all appear in a longer tag |
|
|
339
|
+
| `prompt_word_forms`, `in_prompt` | test whether a tag's words appear in a prompt, singular or plural |
|
|
340
|
+
| `merge_candidates` | merges proposed tags with a prompt's tags under a cap |
|
|
341
|
+
| `presence_question` | writes "Is there a car in this image?" with the article the thing takes |
|
|
342
|
+
|
|
343
|
+
```python
|
|
344
|
+
from logit_classifier.toolkit.tags import drop_subsets, parse_candidates
|
|
345
|
+
|
|
346
|
+
tags = drop_subsets(parse_candidates("red car, car, wooden bench, bench"))
|
|
347
|
+
# ['red car', 'wooden bench']
|
|
348
|
+
```
|
|
349
|
+
|
|
350
|
+
`drop_subsets(tags, rule="head-noun")` drops a tag only into a longer tag with the same head
|
|
351
|
+
noun. The head noun is the last word before "of", "in", "on", "with" or "reading". So "cat"
|
|
352
|
+
stays beside "cat ear", and "signage" is dropped beside "signage in chinese".
|
|
353
|
+
|
|
354
|
+
### ComfyUI Tools
|
|
355
|
+
|
|
356
|
+
These run inside a ComfyUI node over the CLIP the workflow loaded. Each one imports comfy and
|
|
357
|
+
torch only when it is called.
|
|
358
|
+
|
|
359
|
+
| Tool | Does |
|
|
360
|
+
|---|---|
|
|
361
|
+
| `comfy_classifier` | builds a classifier over the CLIP, with errors that name the node and the Load CLIP fix |
|
|
362
|
+
| `generate_text` | generates one greedy reply with the Qwen chat template, and stops early on a condition |
|
|
363
|
+
| `shared_vision_encode` | runs the vision tower once per picture for every generate and classify inside the block |
|
|
364
|
+
| `skip_resident_loads` | skips core's model load while the CLIP is already on the GPU |
|
|
365
|
+
| `routed_progress_bars`, `send_status`, `run_outcome` | show one progress bar per run and a status line under the node |
|
|
366
|
+
| `fit_picture` | downscales an image to a pixel cap |
|
|
367
|
+
| `prompt_tags` | lists the physical things a prompt names |
|
|
368
|
+
| `tag_picture` | tags a picture, with a prompt's things added as candidates |
|
|
369
|
+
| `TagSettings` | holds every wording, budget, cap and threshold of the tagger |
|
|
370
|
+
|
|
371
|
+
The tagger generates candidate tags over the CLIP and verifies each one with the classifier.
|
|
372
|
+
|
|
373
|
+
```python
|
|
374
|
+
from logit_classifier.toolkit.comfyui import TagSettings, comfy_classifier, fit_picture, tag_picture
|
|
375
|
+
|
|
376
|
+
classifier = comfy_classifier(clip, node="My Tagger")
|
|
377
|
+
picture = fit_picture(image, 1024 * 1024)
|
|
378
|
+
trace = tag_picture(clip, classifier, picture, settings=TagSettings(), known={})
|
|
379
|
+
tags = trace.kept
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
`known` holds the thing check's answers across a run, so each tag is asked once. The
|
|
383
|
+
`TagSettings` defaults are the current measured values, and a later release may change them.
|
|
384
|
+
Pass every value your output depends on to keep it fixed.
|
|
385
|
+
|
|
312
386
|
## Differences From Jev
|
|
313
387
|
|
|
314
388
|
This project matches the System One request and response format and the documented limits.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "logit-classifier"
|
|
3
3
|
dynamic = ["version"]
|
|
4
|
-
description = "Local zero-shot classifier for text and images. Declares options
|
|
4
|
+
description = "Local zero-shot classifier for text and images. Declares options and reads a calibrated probability for each from the logits. Accepts the TypeSafe System One request format, and ships a toolkit of image tag tools and ComfyUI node tools."
|
|
5
5
|
requires-python = ">=3.12"
|
|
6
6
|
license = "GPL-3.0-or-later"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -136,8 +136,8 @@ show_error_codes = true
|
|
|
136
136
|
[[tool.mypy.overrides]]
|
|
137
137
|
# transformers publishes no complete stubs, and accelerate none at all. torch is
|
|
138
138
|
# listed so the type gate still runs where the hf extra is absent, such as CI.
|
|
139
|
-
# The ComfyUI host supplies comfy, and neither this venv nor CI has
|
|
140
|
-
module = ["transformers.*", "accelerate.*", "torch.*", "comfy.*"]
|
|
139
|
+
# The ComfyUI host supplies comfy, server and comfy_execution, and neither this venv nor CI has them.
|
|
140
|
+
module = ["transformers.*", "accelerate.*", "torch.*", "comfy.*", "server", "server.*", "comfy_execution.*"]
|
|
141
141
|
ignore_missing_imports = true
|
|
142
142
|
|
|
143
143
|
[[tool.mypy.overrides]]
|
{logit_classifier-0.2.1 → logit_classifier-0.3.0}/src/logit_classifier/backends/comfy_clip.py
RENAMED
|
@@ -101,6 +101,29 @@ def _check_image(image: Any) -> None:
|
|
|
101
101
|
raise ImageError(f"a ComfyUI IMAGE of shape [1, H, W, 3] is required, got shape {shape}")
|
|
102
102
|
|
|
103
103
|
|
|
104
|
+
def _resident(clip: Any) -> bool:
|
|
105
|
+
"""Whether the CLIP sits where its own load_model call leaves it, so loading again is a no-op.
|
|
106
|
+
|
|
107
|
+
Core's load_models_gpu has no fast path for a loaded model and costs about 0.1 s per call.
|
|
108
|
+
"""
|
|
109
|
+
try:
|
|
110
|
+
import comfy.model_management as model_management
|
|
111
|
+
|
|
112
|
+
patcher = clip.patcher
|
|
113
|
+
loaded = model_management.current_loaded_models
|
|
114
|
+
# Another model's load always inserts at the head, and a changed LoRA patch set or an
|
|
115
|
+
# offloaded CLIP must reload.
|
|
116
|
+
return bool(
|
|
117
|
+
loaded
|
|
118
|
+
and loaded[0].model is patcher
|
|
119
|
+
and patcher.model.device == patcher.load_device
|
|
120
|
+
and patcher.model.current_weight_patches_uuid == patcher.patches_uuid
|
|
121
|
+
and patcher.model.model_loaded_weight_memory > 0
|
|
122
|
+
)
|
|
123
|
+
except (ImportError, AttributeError):
|
|
124
|
+
return False
|
|
125
|
+
|
|
126
|
+
|
|
104
127
|
def _pack_passes(prefix_length: int, suffix_lengths: list[int], batch_branches: bool) -> list[list[int]]:
|
|
105
128
|
"""Group suffixes in request order into passes that stay under PACKED_TOKEN_CEILING.
|
|
106
129
|
|
|
@@ -284,7 +307,8 @@ class ComfyClipBackend:
|
|
|
284
307
|
prefix_tokens[vision["pad_index"]] = {"type": "image", "data": vision["image"], "original_type": "image"}
|
|
285
308
|
with _determinism(), torch.inference_mode():
|
|
286
309
|
encoder.reset_clip_options()
|
|
287
|
-
clip
|
|
310
|
+
if not _resident(clip):
|
|
311
|
+
clip.load_model({self._tokens_key: [prefix_tokens]})
|
|
288
312
|
device = clip.patcher.load_device
|
|
289
313
|
encoder.set_clip_options({"layer": None, "execution_device": device})
|
|
290
314
|
# BaseGenerate.generate picks its execution dtype this way, at llama.py:1115-1120.
|
|
@@ -4,6 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
6
|
import random
|
|
7
|
+
from collections.abc import Sequence
|
|
7
8
|
from dataclasses import dataclass, field, replace
|
|
8
9
|
from typing import Any
|
|
9
10
|
|
|
@@ -11,7 +12,7 @@ import numpy as np
|
|
|
11
12
|
|
|
12
13
|
from .backends.base import Backend, BackendContractError, BranchLogits
|
|
13
14
|
from .calibrate import PriorStore
|
|
14
|
-
from .config import ANSWER_PREFILL, Config, fitted_temperature
|
|
15
|
+
from .config import ANSWER_PREFILL, JEV_MAX_QUESTIONS, Config, fitted_temperature
|
|
15
16
|
from .deps import MissingDependencyError
|
|
16
17
|
from .prompt import (
|
|
17
18
|
PROMPT_VERSION,
|
|
@@ -25,6 +26,7 @@ from .schema import (
|
|
|
25
26
|
Answer,
|
|
26
27
|
ChoiceAnswer,
|
|
27
28
|
ChoiceQuestion,
|
|
29
|
+
JSONContent,
|
|
28
30
|
NoulAnswer,
|
|
29
31
|
NoulQuestion,
|
|
30
32
|
Question,
|
|
@@ -279,6 +281,28 @@ class Classifier:
|
|
|
279
281
|
)
|
|
280
282
|
return response, diagnostics
|
|
281
283
|
|
|
284
|
+
def nouls(self, statements: Sequence[str], *, image: Any = None, state: JSONContent = "") -> list[float]:
|
|
285
|
+
"""Return each statement's noul about the state, in statement order.
|
|
286
|
+
|
|
287
|
+
The statements are split into requests of JEV_MAX_QUESTIONS, one classify call each, and no
|
|
288
|
+
statements make no request. An empty state with an image is the image-only state.
|
|
289
|
+
"""
|
|
290
|
+
keys = [f"s{index}" for index in range(len(statements))]
|
|
291
|
+
questions: dict[str, Question] = {}
|
|
292
|
+
nouls: list[float] = []
|
|
293
|
+
|
|
294
|
+
for start in range(0, len(statements), JEV_MAX_QUESTIONS):
|
|
295
|
+
window = slice(start, start + JEV_MAX_QUESTIONS)
|
|
296
|
+
questions = {key: NoulQuestion(instructions=statement)
|
|
297
|
+
for key, statement in zip(keys[window], statements[window], strict=True)}
|
|
298
|
+
response, _diagnostics = self.classify(SystemOneRequest(state=state, questions=questions), image=image)
|
|
299
|
+
for key in questions:
|
|
300
|
+
answer = response.answers[key]
|
|
301
|
+
if not isinstance(answer, NoulAnswer):
|
|
302
|
+
raise TypeError(f"question {key} got a {answer.type} answer, not a noul")
|
|
303
|
+
nouls.append(answer.noul)
|
|
304
|
+
return nouls
|
|
305
|
+
|
|
282
306
|
def _round_answers(self, questions: dict[str, Question], branches: list[Branch],
|
|
283
307
|
probabilities: list[np.ndarray]) -> dict[str, Answer]:
|
|
284
308
|
answers: dict[str, Answer] = {}
|
|
@@ -52,7 +52,7 @@ class Branch:
|
|
|
52
52
|
suffix_text: str
|
|
53
53
|
|
|
54
54
|
|
|
55
|
-
def
|
|
55
|
+
def inert(text: str) -> str:
|
|
56
56
|
# The backends' tokenizers read a <|...|> spelling in plain text as a control token, so
|
|
57
57
|
# client text could otherwise forge a chat turn or a second image pad.
|
|
58
58
|
return text.replace("<|", "<\u200b|")
|
|
@@ -67,7 +67,7 @@ def _option_lines(entries: list[tuple[str, str]]) -> str:
|
|
|
67
67
|
|
|
68
68
|
|
|
69
69
|
def _question_block(prompt_line: str, entries: list[tuple[str, str]]) -> str:
|
|
70
|
-
return
|
|
70
|
+
return inert(f"{prompt_line}\nOptions:\n{_option_lines(entries)}")
|
|
71
71
|
|
|
72
72
|
|
|
73
73
|
def _choice_branches(qid: str, question: ChoiceQuestion, plan: BranchPlan,
|
|
@@ -152,7 +152,7 @@ def build_branches(questions: dict[str, Question], score_method: str = "joint",
|
|
|
152
152
|
def prefix_content(state: Any, has_image: bool = False) -> str:
|
|
153
153
|
"""Build the user-message body every branch shares, image marker included."""
|
|
154
154
|
marker = f"{IMAGE_MARKER}\n" if has_image else ""
|
|
155
|
-
return f"Context:\n{marker}{
|
|
155
|
+
return f"Context:\n{marker}{inert(render_content(state))}\n\n"
|
|
156
156
|
|
|
157
157
|
|
|
158
158
|
def branch_content(state: Any, branch: Branch, has_image: bool = False) -> str:
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""The tag text tools, which moved to logit_classifier.toolkit.tags in 0.3.0.
|
|
2
|
+
|
|
3
|
+
This path stays for code built on 0.2.x, which PyPI published. It raises no warning, so that code keeps a quiet log.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from .toolkit.tags import (
|
|
9
|
+
MAX_CANDIDATES,
|
|
10
|
+
MAX_REPEAT_BLOCK,
|
|
11
|
+
MAX_TAG_CHARS,
|
|
12
|
+
MAX_TAG_WORDS,
|
|
13
|
+
clean_item,
|
|
14
|
+
complete_tags,
|
|
15
|
+
drop_subsets,
|
|
16
|
+
drop_unfinished_tag,
|
|
17
|
+
normalize_item,
|
|
18
|
+
parse_candidates,
|
|
19
|
+
repeated_block,
|
|
20
|
+
split_prompt,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"MAX_CANDIDATES",
|
|
25
|
+
"MAX_REPEAT_BLOCK",
|
|
26
|
+
"MAX_TAG_CHARS",
|
|
27
|
+
"MAX_TAG_WORDS",
|
|
28
|
+
"clean_item",
|
|
29
|
+
"complete_tags",
|
|
30
|
+
"drop_subsets",
|
|
31
|
+
"drop_unfinished_tag",
|
|
32
|
+
"normalize_item",
|
|
33
|
+
"parse_candidates",
|
|
34
|
+
"repeated_block",
|
|
35
|
+
"split_prompt",
|
|
36
|
+
]
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""The tools for a ComfyUI node that holds a loaded CLIP.
|
|
2
|
+
|
|
3
|
+
Every module here imports comfy, server, torch and comfy_execution inside the functions that need them, so importing
|
|
4
|
+
this package in a plain Python process loads none of them.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from ... import COMFY_SOCKET_TYPE
|
|
10
|
+
from ...backends.comfy_clip import PACKED_TOKEN_CEILING, ComfyClipBackend
|
|
11
|
+
from .classifier import comfy_classifier
|
|
12
|
+
from .generate import IMAGE_SLOT, chat, clip_generate, generate_text, stop_when
|
|
13
|
+
from .progress import routed_progress_bars, run_outcome, send_status, token_fill
|
|
14
|
+
from .scopes import shared_vision_encode, skip_resident_loads
|
|
15
|
+
from .tagger import (
|
|
16
|
+
PromptTags,
|
|
17
|
+
TagSettings,
|
|
18
|
+
TagTrace,
|
|
19
|
+
english_text,
|
|
20
|
+
fit_picture,
|
|
21
|
+
prompt_tags,
|
|
22
|
+
tag_picture,
|
|
23
|
+
thing_scores,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"COMFY_SOCKET_TYPE",
|
|
28
|
+
"IMAGE_SLOT",
|
|
29
|
+
"PACKED_TOKEN_CEILING",
|
|
30
|
+
"ComfyClipBackend",
|
|
31
|
+
"PromptTags",
|
|
32
|
+
"TagSettings",
|
|
33
|
+
"TagTrace",
|
|
34
|
+
"chat",
|
|
35
|
+
"clip_generate",
|
|
36
|
+
"comfy_classifier",
|
|
37
|
+
"english_text",
|
|
38
|
+
"fit_picture",
|
|
39
|
+
"generate_text",
|
|
40
|
+
"prompt_tags",
|
|
41
|
+
"routed_progress_bars",
|
|
42
|
+
"run_outcome",
|
|
43
|
+
"send_status",
|
|
44
|
+
"shared_vision_encode",
|
|
45
|
+
"skip_resident_loads",
|
|
46
|
+
"stop_when",
|
|
47
|
+
"tag_picture",
|
|
48
|
+
"thing_scores",
|
|
49
|
+
"token_fill",
|
|
50
|
+
]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""One Classifier over the CLIP a ComfyUI workflow already loaded."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from ...backends.base import UnsupportedModelError, VisionUnsupportedError
|
|
8
|
+
from ...backends.comfy_clip import ComfyClipBackend
|
|
9
|
+
from ...classifier import Classifier
|
|
10
|
+
from ...config import Config
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def comfy_classifier(
|
|
14
|
+
clip: Any, *, node: str | None = None, needs_images: bool = True, batch_branches: bool = True
|
|
15
|
+
) -> Classifier:
|
|
16
|
+
"""One Classifier over the workflow's CLIP, with no running prior and no calibration file.
|
|
17
|
+
|
|
18
|
+
Each error starts with the node's name when one is given, and names the Load CLIP fix.
|
|
19
|
+
"""
|
|
20
|
+
prefix = f"{node}: " if node else ""
|
|
21
|
+
backend: ComfyClipBackend | None = None
|
|
22
|
+
|
|
23
|
+
try:
|
|
24
|
+
backend = ComfyClipBackend(clip, batch_branches=batch_branches)
|
|
25
|
+
except UnsupportedModelError as error:
|
|
26
|
+
raise UnsupportedModelError(
|
|
27
|
+
f"{prefix}this CLIP is not a Qwen3-VL text encoder, so the node cannot read it. Load a Qwen3-VL text "
|
|
28
|
+
"encoder, such as qwen3vl_4b_fp8_scaled.safetensors, with a Load CLIP type that keeps its vision tower, "
|
|
29
|
+
"such as krea2."
|
|
30
|
+
) from error
|
|
31
|
+
if needs_images and not backend.sees_images:
|
|
32
|
+
raise VisionUnsupportedError(
|
|
33
|
+
f"{prefix}this CLIP was loaded without its vision tower, so the node cannot read images. Load the "
|
|
34
|
+
"Qwen3-VL text encoder with a Load CLIP type that keeps it, such as krea2. The flux and flux2 types "
|
|
35
|
+
"drop it."
|
|
36
|
+
)
|
|
37
|
+
# A node reuses one classifier across a batch, and the running prior would make an item's answer depend on the
|
|
38
|
+
# items before it.
|
|
39
|
+
return Classifier(Config(use_prior_debias=False, calibration_path=None), backend)
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Greedy replies from a loaded CLIP's own generate, for text a ComfyUI pack reads, such as a list of candidate tags.
|
|
2
|
+
|
|
3
|
+
The classifier core never calls these tools, so nothing a classifier answers is generated.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from collections.abc import Callable, Iterator
|
|
9
|
+
from contextlib import AbstractContextManager, contextmanager, nullcontext
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from ...backends.base import VisionUnsupportedError
|
|
13
|
+
from ...prompt import inert
|
|
14
|
+
from ..tags import complete_tags, drop_unfinished_tag
|
|
15
|
+
|
|
16
|
+
IMAGE_SLOT = "<|vision_start|><|image_pad|><|vision_end|>"
|
|
17
|
+
|
|
18
|
+
_NO_IMAGE_TOKENIZER = (
|
|
19
|
+
"this CLIP's tokenizer does not read images, so it cannot generate from one. Load a Qwen3-VL text encoder with a "
|
|
20
|
+
"Load CLIP type that keeps its vision tower, such as krea2."
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def chat(user: str, *, image: bool = False) -> str:
|
|
25
|
+
"""Return the Qwen chat text for one user turn, with the image slot first when image is true."""
|
|
26
|
+
# The tokenizer reads a <|...|> spelling in plain text as a control token, so an unescaped prompt could
|
|
27
|
+
# close the user turn.
|
|
28
|
+
safe_user = inert(user)
|
|
29
|
+
|
|
30
|
+
# A text starting with <|im_start|> skips core's template, whose empty think block made this model reply
|
|
31
|
+
# empty or refuse.
|
|
32
|
+
return f"<|im_start|>user\n{IMAGE_SLOT if image else ''}{safe_user}<|im_end|>\n<|im_start|>assistant\n"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def clip_generate(clip: Any, tokens: Any, **kwargs: Any) -> Any:
|
|
36
|
+
"""Run clip.generate with every keyword passed through, then core's cleanup of the CUDA graphs it captured.
|
|
37
|
+
|
|
38
|
+
Core's Qwen3 graph decode keeps each layer's graph bound to the freed KV cache and asserts on the next generate in
|
|
39
|
+
the same node (core issue #16441). The cleanup keeps graph decode on, which measured 100 to 16.5 ms per token
|
|
40
|
+
(ContextAnchoredTileRefine tests-AB/tags-bench-log.md, section 3).
|
|
41
|
+
"""
|
|
42
|
+
try:
|
|
43
|
+
return clip.generate(tokens, **kwargs)
|
|
44
|
+
finally:
|
|
45
|
+
_drop_decode_graphs()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _drop_decode_graphs() -> None:
|
|
49
|
+
try:
|
|
50
|
+
from comfy.model_prefetch import cleanup_prefetch_queues
|
|
51
|
+
except ImportError:
|
|
52
|
+
return
|
|
53
|
+
cleanup_prefetch_queues()
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _transformer(clip: Any) -> Any:
|
|
57
|
+
"""Core's generating transformer, or None for a CLIP without one."""
|
|
58
|
+
model = getattr(clip, "cond_stage_model", None)
|
|
59
|
+
encoder_name = getattr(model, "clip", None)
|
|
60
|
+
encoder = getattr(model, encoder_name, None) if isinstance(encoder_name, str) else None
|
|
61
|
+
|
|
62
|
+
return getattr(encoder, "transformer", None)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@contextmanager
|
|
66
|
+
def stop_when(clip: Any, should_stop: Callable[[list[str]], bool]) -> Iterator[None]:
|
|
67
|
+
"""End core's greedy decode on a stop token once should_stop(the complete tags so far) is true.
|
|
68
|
+
|
|
69
|
+
Core's generate loop takes no stop condition, so the token sample_token returns is replaced. Core copies that
|
|
70
|
+
token into the sequence and breaks on it, in comfy/text_encoders/llama.py BaseGenerate.generate. A CLIP without
|
|
71
|
+
core's sample_token or stop tokens passes through.
|
|
72
|
+
"""
|
|
73
|
+
transformer = _transformer(clip)
|
|
74
|
+
config = getattr(getattr(transformer, "model", None), "config", None)
|
|
75
|
+
stop_tokens = getattr(config, "stop_tokens", None)
|
|
76
|
+
history: list[int] = []
|
|
77
|
+
|
|
78
|
+
if not callable(getattr(transformer, "sample_token", None)) or not stop_tokens:
|
|
79
|
+
yield
|
|
80
|
+
return
|
|
81
|
+
stop_id = stop_tokens[0]
|
|
82
|
+
original = transformer.sample_token
|
|
83
|
+
shadowed = "sample_token" in vars(transformer)
|
|
84
|
+
previous = vars(transformer).get("sample_token")
|
|
85
|
+
|
|
86
|
+
def watch(*args: Any, **kwargs: Any) -> Any:
|
|
87
|
+
token = original(*args, **kwargs)
|
|
88
|
+
history.append(int(token.reshape(-1)[0]))
|
|
89
|
+
if should_stop(complete_tags(clip.decode(history))):
|
|
90
|
+
return token.new_full(token.shape, stop_id)
|
|
91
|
+
return token
|
|
92
|
+
|
|
93
|
+
transformer.sample_token = watch
|
|
94
|
+
try:
|
|
95
|
+
yield
|
|
96
|
+
finally:
|
|
97
|
+
if shadowed:
|
|
98
|
+
transformer.sample_token = previous
|
|
99
|
+
else:
|
|
100
|
+
del transformer.sample_token
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def generate_text(
|
|
104
|
+
clip: Any,
|
|
105
|
+
user: str,
|
|
106
|
+
*,
|
|
107
|
+
image: Any = None,
|
|
108
|
+
max_tokens: int,
|
|
109
|
+
should_stop: Callable[[list[str]], bool] | None = None,
|
|
110
|
+
whole_tags: bool = True,
|
|
111
|
+
) -> str:
|
|
112
|
+
"""One greedy reply to user. With whole_tags, a reply that fills max_tokens is cut back to its last whole tag.
|
|
113
|
+
|
|
114
|
+
A CLIP whose tokenizer does not read images raises VisionUnsupportedError for an image.
|
|
115
|
+
"""
|
|
116
|
+
text = chat(user, image=image is not None)
|
|
117
|
+
stopper: AbstractContextManager[None] = nullcontext() if should_stop is None else stop_when(clip, should_stop)
|
|
118
|
+
tokens: Any = None
|
|
119
|
+
ids: Any = None
|
|
120
|
+
reply: str = ""
|
|
121
|
+
|
|
122
|
+
tokens = clip.tokenize(text) if image is None else _tokenize_image(clip, text, image)
|
|
123
|
+
with stopper:
|
|
124
|
+
ids = clip_generate(clip, tokens, do_sample=False, max_length=max_tokens)
|
|
125
|
+
reply = clip.decode(ids)
|
|
126
|
+
# Core stops early only on a stop token, so a decode that fills the budget ends mid tag.
|
|
127
|
+
if whole_tags and len(ids) >= max_tokens:
|
|
128
|
+
return drop_unfinished_tag(reply)
|
|
129
|
+
return reply
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _tokenize_image(clip: Any, text: str, image: Any) -> Any:
|
|
133
|
+
"""Tokenize text with one image, or raise VisionUnsupportedError when the tokenizer placed no image entry."""
|
|
134
|
+
tokens: Any = None
|
|
135
|
+
holds_image = False
|
|
136
|
+
|
|
137
|
+
try:
|
|
138
|
+
tokens = clip.tokenize(text, images=[image])
|
|
139
|
+
except TypeError as error:
|
|
140
|
+
raise VisionUnsupportedError(_NO_IMAGE_TOKENIZER) from error
|
|
141
|
+
# A tokenizer that takes images=... through **kwargs and ignores it writes only text entries, and core's image
|
|
142
|
+
# entry is a dict in the token's first slot.
|
|
143
|
+
holds_image = any(isinstance(token[0], dict) for rows in tokens.values() for row in rows for token in row)
|
|
144
|
+
if not holds_image:
|
|
145
|
+
raise VisionUnsupportedError(_NO_IMAGE_TOKENIZER)
|
|
146
|
+
return tokens
|