llm-annotator 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,11 @@
1
+ quality:
2
+ ruff check src/llm_annotator tests/ examples/
3
+ ruff format --check src/llm_annotator tests/ examples/
4
+
5
+ style:
6
+ ruff check src/llm_annotator tests/ examples/ --fix
7
+ ruff format src/llm_annotator tests/ examples/
8
+
9
+ setup:
10
+ uv sync --dev
11
+ pre-commit install --hook-type pre-push
@@ -1,15 +1,15 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-annotator
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: An easy-to-extend LLM annotator for robust, resumable data annotation.
5
5
  Author-email: Bram Vanroy <2779410+BramVanroy@users.noreply.github.com>
6
6
  License-Expression: Apache-2.0
7
7
  License-File: LICENSE
8
8
  Requires-Python: >=3.12
9
9
  Requires-Dist: datasets<5,>=4.1.1
10
- Requires-Dist: hf-transfer<2,>=0.1.9
10
+ Requires-Dist: hf-transfer<1,>=0.1.9
11
11
  Requires-Dist: hf-xet<2,>=1.1.10
12
- Requires-Dist: vllm<0.11,>=0.10.2
12
+ Requires-Dist: vllm<0.12,>=0.11
13
13
  Description-Content-Type: text/markdown
14
14
 
15
15
  # A simple, extensible LLM Annotator
@@ -1,5 +1,3 @@
1
-
2
- import json
3
1
  from huggingface_hub import HfApi
4
2
 
5
3
  from llm_annotator import Annotator
@@ -11,7 +9,8 @@ def get_hf_username() -> str | None:
11
9
  return whoami["name"]
12
10
  else:
13
11
  raise ValueError("No Hugging Face username found. Please login using `hf auth login`.")
14
-
12
+
13
+
15
14
  def main():
16
15
  hf_user = get_hf_username()
17
16
  prompt_template = """Analyze the sentiment of the following movie review and classify it as positive or negative.
@@ -22,24 +21,19 @@ def main():
22
21
 
23
22
  output_schema = {
24
23
  "type": "object",
25
- "properties": {
26
- "sentiment": {
27
- "type": "string",
28
- "enum": ["positive", "negative", "neutral"]
29
- }
30
- },
31
- "required": ["sentiment"]
24
+ "properties": {"sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}},
25
+ "required": ["sentiment"],
32
26
  }
33
27
 
34
28
  with Annotator(model_id="Qwen/Qwen2.5-0.5B-Instruct") as anno:
35
29
  ds = anno.annotate_dataset(
36
- "stanfordnlp/imdb",
37
30
  output_dir="outputs/sentiment-imdb-qwen",
31
+ dataset_name="stanfordnlp/imdb",
38
32
  dataset_split="test",
39
33
  new_hub_id=f"{hf_user}/sentiment-imdb",
40
34
  streaming=True,
41
35
  max_num_samples=250,
42
- cache_input_dataset=False, # `True` is generally useful, not for demo purposes
36
+ cache_input_dataset=False, # `True` is generally useful, not for demo purposes
43
37
  prompt_template=prompt_template,
44
38
  output_schema=output_schema,
45
39
  # Backup to HF every 10 samples.
@@ -48,5 +42,6 @@ def main():
48
42
  )
49
43
  print(ds)
50
44
 
45
+
51
46
  if __name__ == "__main__":
52
- main()
47
+ main()
@@ -11,9 +11,9 @@ license-files = ["LICEN[CS]E*"]
11
11
  requires-python = ">=3.12"
12
12
  dependencies = [
13
13
  "datasets>=4.1.1,<5",
14
- "hf-transfer>=0.1.9,<2",
14
+ "hf-transfer>=0.1.9,<1",
15
15
  "hf-xet>=1.1.10,<2",
16
- "vllm>=0.10.2,<0.11",
16
+ "vllm>=0.11,<0.12",
17
17
  ]
18
18
 
19
19
  [dependency-groups]
@@ -62,7 +62,7 @@ class Annotator:
62
62
  return self
63
63
 
64
64
  def __exit__(self, exc_type, exc, tb):
65
- self.reset_model()
65
+ self.destroy_model()
66
66
 
67
67
  def _get_skip_idxs(
68
68
  self, *, pdout: Path, idx_column: str, dataset_split: str | None, dataset_config: str | None
@@ -406,29 +406,26 @@ class Annotator:
406
406
  **result,
407
407
  }
408
408
 
409
- def reset_model(self) -> None:
409
+ def destroy_model(self) -> None:
410
410
  """Clean up model resources to free memory.
411
411
 
412
- Destroys the distributed environment, clears GPU cache, and resets internal state..
412
+ Destroys the distributed environment, clears GPU cache, and resets internal state.
413
413
  """
414
- destroy_model_parallel()
415
- destroy_distributed_environment()
416
-
417
- try:
418
- # Remove nested attributes if present
419
- if hasattr(self.pipe, "llm_engine") and hasattr(self.pipe.llm_engine, "model_executor"):
414
+ if isinstance(self.pipe, LLM):
415
+ destroy_model_parallel()
416
+ try:
417
+ self.pipe.llm_engine.model_executor.shutdown()
420
418
  del self.pipe.llm_engine.model_executor
421
- except Exception:
422
- pass
423
-
424
- del self.pipe
425
- del self.tokenizer
419
+ except Exception:
420
+ pass
421
+ del self.pipe.llm_engine
422
+ del self.pipe
423
+ gc.collect()
424
+ cuda.empty_cache()
425
+ destroy_distributed_environment()
426
+ gc.collect()
426
427
 
427
- self.pipe = None
428
- self.tokenizer = None
429
-
430
- cuda.empty_cache()
431
- gc.collect()
428
+ self.pipe = None
432
429
 
433
430
  def _process_batch(
434
431
  self,
@@ -598,7 +595,14 @@ class Annotator:
598
595
  shutil.rmtree(pdout)
599
596
  pdout.mkdir(exist_ok=True, parents=True)
600
597
 
601
- self._load_tokenizer()
598
+ # We need to clear the model before doing self._load_dataset because the model
599
+ # cannot be be pickled (which is needed for multiprocessing in dataset.map)
600
+ if self.pipe is not None and self.num_proc is not None:
601
+ self.destroy_model()
602
+
603
+ if self.tokenizer is None:
604
+ self._load_tokenizer()
605
+
602
606
  dataset, processed_n_samples = self._load_dataset(
603
607
  prompt_template=prompt_template,
604
608
  idx_column=idx_column,
@@ -1,11 +0,0 @@
1
- quality:
2
- ruff check src/llm_annotator tests/
3
- ruff format --check src/llm_annotator tests/
4
-
5
- style:
6
- ruff check src/llm_annotator tests/ --fix
7
- ruff format src/llm_annotator tests/
8
-
9
- setup:
10
- uv sync --dev
11
- pre-commit install --hook-type pre-push
File without changes
File without changes
File without changes