llm-annotator 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_annotator-0.2.3/Makefile +11 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/PKG-INFO +1 -1
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/examples/sentiment.py +8 -13
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/src/llm_annotator/annotator.py +20 -21
- llm_annotator-0.2.2/Makefile +0 -11
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/.gitignore +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/.pre-commit-config.yaml +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/LICENSE +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/README.md +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/pyproject.toml +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/src/llm_annotator/__init__.py +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/src/llm_annotator/utils.py +0 -0
- {llm_annotator-0.2.2 → llm_annotator-0.2.3}/tests/conftest.py +0 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
quality:
|
|
2
|
+
ruff check src/llm_annotator tests/ examples/
|
|
3
|
+
ruff format --check src/llm_annotator tests/ examples/
|
|
4
|
+
|
|
5
|
+
style:
|
|
6
|
+
ruff check src/llm_annotator tests/ examples/ --fix
|
|
7
|
+
ruff format src/llm_annotator tests/ examples/
|
|
8
|
+
|
|
9
|
+
setup:
|
|
10
|
+
uv sync --dev
|
|
11
|
+
pre-commit install --hook-type pre-push
|
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
|
|
2
|
-
import json
|
|
3
1
|
from huggingface_hub import HfApi
|
|
4
2
|
|
|
5
3
|
from llm_annotator import Annotator
|
|
@@ -11,7 +9,8 @@ def get_hf_username() -> str | None:
|
|
|
11
9
|
return whoami["name"]
|
|
12
10
|
else:
|
|
13
11
|
raise ValueError("No Hugging Face username found. Please login using `hf auth login`.")
|
|
14
|
-
|
|
12
|
+
|
|
13
|
+
|
|
15
14
|
def main():
|
|
16
15
|
hf_user = get_hf_username()
|
|
17
16
|
prompt_template = """Analyze the sentiment of the following movie review and classify it as positive or negative.
|
|
@@ -22,24 +21,19 @@ def main():
|
|
|
22
21
|
|
|
23
22
|
output_schema = {
|
|
24
23
|
"type": "object",
|
|
25
|
-
"properties": {
|
|
26
|
-
|
|
27
|
-
"type": "string",
|
|
28
|
-
"enum": ["positive", "negative", "neutral"]
|
|
29
|
-
}
|
|
30
|
-
},
|
|
31
|
-
"required": ["sentiment"]
|
|
24
|
+
"properties": {"sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}},
|
|
25
|
+
"required": ["sentiment"],
|
|
32
26
|
}
|
|
33
27
|
|
|
34
28
|
with Annotator(model_id="Qwen/Qwen2.5-0.5B-Instruct") as anno:
|
|
35
29
|
ds = anno.annotate_dataset(
|
|
36
|
-
"stanfordnlp/imdb",
|
|
37
30
|
output_dir="outputs/sentiment-imdb-qwen",
|
|
31
|
+
dataset_name="stanfordnlp/imdb",
|
|
38
32
|
dataset_split="test",
|
|
39
33
|
new_hub_id=f"{hf_user}/sentiment-imdb",
|
|
40
34
|
streaming=True,
|
|
41
35
|
max_num_samples=250,
|
|
42
|
-
cache_input_dataset=False, # `True` is generally useful, not for demo purposes
|
|
36
|
+
cache_input_dataset=False, # `True` is generally useful, not for demo purposes
|
|
43
37
|
prompt_template=prompt_template,
|
|
44
38
|
output_schema=output_schema,
|
|
45
39
|
# Backup to HF every 10 samples.
|
|
@@ -48,5 +42,6 @@ def main():
|
|
|
48
42
|
)
|
|
49
43
|
print(ds)
|
|
50
44
|
|
|
45
|
+
|
|
51
46
|
if __name__ == "__main__":
|
|
52
|
-
main()
|
|
47
|
+
main()
|
|
@@ -62,7 +62,7 @@ class Annotator:
|
|
|
62
62
|
return self
|
|
63
63
|
|
|
64
64
|
def __exit__(self, exc_type, exc, tb):
|
|
65
|
-
self.
|
|
65
|
+
self.destroy_model()
|
|
66
66
|
|
|
67
67
|
def _get_skip_idxs(
|
|
68
68
|
self, *, pdout: Path, idx_column: str, dataset_split: str | None, dataset_config: str | None
|
|
@@ -406,29 +406,21 @@ class Annotator:
|
|
|
406
406
|
**result,
|
|
407
407
|
}
|
|
408
408
|
|
|
409
|
-
def
|
|
409
|
+
def destroy_model(self) -> None:
|
|
410
410
|
"""Clean up model resources to free memory.
|
|
411
411
|
|
|
412
|
-
Destroys the distributed environment, clears GPU cache, and resets internal state
|
|
412
|
+
Destroys the distributed environment, clears GPU cache, and resets internal state.
|
|
413
413
|
"""
|
|
414
|
-
|
|
415
|
-
|
|
414
|
+
if isinstance(self.pipe, LLM):
|
|
415
|
+
destroy_model_parallel()
|
|
416
|
+
del self.pipe.llm_engine
|
|
417
|
+
del self.pipe
|
|
418
|
+
gc.collect()
|
|
419
|
+
cuda.empty_cache()
|
|
420
|
+
destroy_distributed_environment()
|
|
421
|
+
gc.collect()
|
|
416
422
|
|
|
417
|
-
|
|
418
|
-
# Remove nested attributes if present
|
|
419
|
-
if hasattr(self.pipe, "llm_engine") and hasattr(self.pipe.llm_engine, "model_executor"):
|
|
420
|
-
del self.pipe.llm_engine.model_executor
|
|
421
|
-
except Exception:
|
|
422
|
-
pass
|
|
423
|
-
|
|
424
|
-
del self.pipe
|
|
425
|
-
del self.tokenizer
|
|
426
|
-
|
|
427
|
-
self.pipe = None
|
|
428
|
-
self.tokenizer = None
|
|
429
|
-
|
|
430
|
-
cuda.empty_cache()
|
|
431
|
-
gc.collect()
|
|
423
|
+
self.pipe = None
|
|
432
424
|
|
|
433
425
|
def _process_batch(
|
|
434
426
|
self,
|
|
@@ -598,7 +590,14 @@ class Annotator:
|
|
|
598
590
|
shutil.rmtree(pdout)
|
|
599
591
|
pdout.mkdir(exist_ok=True, parents=True)
|
|
600
592
|
|
|
601
|
-
self.
|
|
593
|
+
# We need to clear the model before doing self._load_dataset because the model
|
|
594
|
+
# cannot be be pickled (which is needed for multiprocessing in dataset.map)
|
|
595
|
+
if self.pipe is not None and self.num_proc is not None:
|
|
596
|
+
self.destroy_model()
|
|
597
|
+
|
|
598
|
+
if self.tokenizer is None:
|
|
599
|
+
self._load_tokenizer()
|
|
600
|
+
|
|
602
601
|
dataset, processed_n_samples = self._load_dataset(
|
|
603
602
|
prompt_template=prompt_template,
|
|
604
603
|
idx_column=idx_column,
|
llm_annotator-0.2.2/Makefile
DELETED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|