llm-annotator 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_annotator-0.2.4/Makefile +11 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/PKG-INFO +3 -3
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/examples/sentiment.py +8 -13
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/pyproject.toml +2 -2
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/src/llm_annotator/annotator.py +24 -20
- llm_annotator-0.2.2/Makefile +0 -11
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/.gitignore +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/.pre-commit-config.yaml +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/LICENSE +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/README.md +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/src/llm_annotator/__init__.py +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/src/llm_annotator/utils.py +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.4}/tests/conftest.py +0 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
quality:
|
|
2
|
+
ruff check src/llm_annotator tests/ examples/
|
|
3
|
+
ruff format --check src/llm_annotator tests/ examples/
|
|
4
|
+
|
|
5
|
+
style:
|
|
6
|
+
ruff check src/llm_annotator tests/ examples/ --fix
|
|
7
|
+
ruff format src/llm_annotator tests/ examples/
|
|
8
|
+
|
|
9
|
+
setup:
|
|
10
|
+
uv sync --dev
|
|
11
|
+
pre-commit install --hook-type pre-push
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-annotator
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: An easy-to-extend LLM annotator for robust, resumable data annotation.
|
|
5
5
|
Author-email: Bram Vanroy <2779410+BramVanroy@users.noreply.github.com>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
7
7
|
License-File: LICENSE
|
|
8
8
|
Requires-Python: >=3.12
|
|
9
9
|
Requires-Dist: datasets<5,>=4.1.1
|
|
10
|
-
Requires-Dist: hf-transfer<
|
|
10
|
+
Requires-Dist: hf-transfer<1,>=0.1.9
|
|
11
11
|
Requires-Dist: hf-xet<2,>=1.1.10
|
|
12
|
-
Requires-Dist: vllm<0.
|
|
12
|
+
Requires-Dist: vllm<0.12,>=0.11
|
|
13
13
|
Description-Content-Type: text/markdown
|
|
14
14
|
|
|
15
15
|
# A simple, extensible LLM Annotator
|
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
|
|
2
|
-
import json
|
|
3
1
|
from huggingface_hub import HfApi
|
|
4
2
|
|
|
5
3
|
from llm_annotator import Annotator
|
|
@@ -11,7 +9,8 @@ def get_hf_username() -> str | None:
|
|
|
11
9
|
return whoami["name"]
|
|
12
10
|
else:
|
|
13
11
|
raise ValueError("No Hugging Face username found. Please login using `hf auth login`.")
|
|
14
|
-
|
|
12
|
+
|
|
13
|
+
|
|
15
14
|
def main():
|
|
16
15
|
hf_user = get_hf_username()
|
|
17
16
|
prompt_template = """Analyze the sentiment of the following movie review and classify it as positive or negative.
|
|
@@ -22,24 +21,19 @@ def main():
|
|
|
22
21
|
|
|
23
22
|
output_schema = {
|
|
24
23
|
"type": "object",
|
|
25
|
-
"properties": {
|
|
26
|
-
|
|
27
|
-
"type": "string",
|
|
28
|
-
"enum": ["positive", "negative", "neutral"]
|
|
29
|
-
}
|
|
30
|
-
},
|
|
31
|
-
"required": ["sentiment"]
|
|
24
|
+
"properties": {"sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}},
|
|
25
|
+
"required": ["sentiment"],
|
|
32
26
|
}
|
|
33
27
|
|
|
34
28
|
with Annotator(model_id="Qwen/Qwen2.5-0.5B-Instruct") as anno:
|
|
35
29
|
ds = anno.annotate_dataset(
|
|
36
|
-
"stanfordnlp/imdb",
|
|
37
30
|
output_dir="outputs/sentiment-imdb-qwen",
|
|
31
|
+
dataset_name="stanfordnlp/imdb",
|
|
38
32
|
dataset_split="test",
|
|
39
33
|
new_hub_id=f"{hf_user}/sentiment-imdb",
|
|
40
34
|
streaming=True,
|
|
41
35
|
max_num_samples=250,
|
|
42
|
-
cache_input_dataset=False, # `True` is generally useful, not for demo purposes
|
|
36
|
+
cache_input_dataset=False, # `True` is generally useful, not for demo purposes
|
|
43
37
|
prompt_template=prompt_template,
|
|
44
38
|
output_schema=output_schema,
|
|
45
39
|
# Backup to HF every 10 samples.
|
|
@@ -48,5 +42,6 @@ def main():
|
|
|
48
42
|
)
|
|
49
43
|
print(ds)
|
|
50
44
|
|
|
45
|
+
|
|
51
46
|
if __name__ == "__main__":
|
|
52
|
-
main()
|
|
47
|
+
main()
|
|
@@ -11,9 +11,9 @@ license-files = ["LICEN[CS]E*"]
|
|
|
11
11
|
requires-python = ">=3.12"
|
|
12
12
|
dependencies = [
|
|
13
13
|
"datasets>=4.1.1,<5",
|
|
14
|
-
"hf-transfer>=0.1.9,<
|
|
14
|
+
"hf-transfer>=0.1.9,<1",
|
|
15
15
|
"hf-xet>=1.1.10,<2",
|
|
16
|
-
"vllm>=0.
|
|
16
|
+
"vllm>=0.11,<0.12",
|
|
17
17
|
]
|
|
18
18
|
|
|
19
19
|
[dependency-groups]
|
|
@@ -62,7 +62,7 @@ class Annotator:
|
|
|
62
62
|
return self
|
|
63
63
|
|
|
64
64
|
def __exit__(self, exc_type, exc, tb):
|
|
65
|
-
self.
|
|
65
|
+
self.destroy_model()
|
|
66
66
|
|
|
67
67
|
def _get_skip_idxs(
|
|
68
68
|
self, *, pdout: Path, idx_column: str, dataset_split: str | None, dataset_config: str | None
|
|
@@ -406,29 +406,26 @@ class Annotator:
|
|
|
406
406
|
**result,
|
|
407
407
|
}
|
|
408
408
|
|
|
409
|
-
def
|
|
409
|
+
def destroy_model(self) -> None:
|
|
410
410
|
"""Clean up model resources to free memory.
|
|
411
411
|
|
|
412
|
-
Destroys the distributed environment, clears GPU cache, and resets internal state
|
|
412
|
+
Destroys the distributed environment, clears GPU cache, and resets internal state.
|
|
413
413
|
"""
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
# Remove nested attributes if present
|
|
419
|
-
if hasattr(self.pipe, "llm_engine") and hasattr(self.pipe.llm_engine, "model_executor"):
|
|
414
|
+
if isinstance(self.pipe, LLM):
|
|
415
|
+
destroy_model_parallel()
|
|
416
|
+
try:
|
|
417
|
+
self.pipe.llm_engine.model_executor.shutdown()
|
|
420
418
|
del self.pipe.llm_engine.model_executor
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
419
|
+
except Exception:
|
|
420
|
+
pass
|
|
421
|
+
del self.pipe.llm_engine
|
|
422
|
+
del self.pipe
|
|
423
|
+
gc.collect()
|
|
424
|
+
cuda.empty_cache()
|
|
425
|
+
destroy_distributed_environment()
|
|
426
|
+
gc.collect()
|
|
426
427
|
|
|
427
|
-
|
|
428
|
-
self.tokenizer = None
|
|
429
|
-
|
|
430
|
-
cuda.empty_cache()
|
|
431
|
-
gc.collect()
|
|
428
|
+
self.pipe = None
|
|
432
429
|
|
|
433
430
|
def _process_batch(
|
|
434
431
|
self,
|
|
@@ -598,7 +595,14 @@ class Annotator:
|
|
|
598
595
|
shutil.rmtree(pdout)
|
|
599
596
|
pdout.mkdir(exist_ok=True, parents=True)
|
|
600
597
|
|
|
601
|
-
self.
|
|
598
|
+
# We need to clear the model before doing self._load_dataset because the model
|
|
599
|
+
# cannot be be pickled (which is needed for multiprocessing in dataset.map)
|
|
600
|
+
if self.pipe is not None and self.num_proc is not None:
|
|
601
|
+
self.destroy_model()
|
|
602
|
+
|
|
603
|
+
if self.tokenizer is None:
|
|
604
|
+
self._load_tokenizer()
|
|
605
|
+
|
|
602
606
|
dataset, processed_n_samples = self._load_dataset(
|
|
603
607
|
prompt_template=prompt_template,
|
|
604
608
|
idx_column=idx_column,
|
llm_annotator-0.2.2/Makefile
DELETED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|