llm-annotator 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,11 @@
1
+ quality:
2
+ ruff check src/llm_annotator tests/ examples/
3
+ ruff format --check src/llm_annotator tests/ examples/
4
+
5
+ style:
6
+ ruff check src/llm_annotator tests/ examples/ --fix
7
+ ruff format src/llm_annotator tests/ examples/
8
+
9
+ setup:
10
+ uv sync --dev
11
+ pre-commit install --hook-type pre-push
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-annotator
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: An easy-to-extend LLM annotator for robust, resumable data annotation.
5
5
  Author-email: Bram Vanroy <2779410+BramVanroy@users.noreply.github.com>
6
6
  License-Expression: Apache-2.0
@@ -1,5 +1,3 @@
1
-
2
- import json
3
1
  from huggingface_hub import HfApi
4
2
 
5
3
  from llm_annotator import Annotator
@@ -11,7 +9,8 @@ def get_hf_username() -> str | None:
11
9
  return whoami["name"]
12
10
  else:
13
11
  raise ValueError("No Hugging Face username found. Please login using `hf auth login`.")
14
-
12
+
13
+
15
14
  def main():
16
15
  hf_user = get_hf_username()
17
16
  prompt_template = """Analyze the sentiment of the following movie review and classify it as positive or negative.
@@ -22,24 +21,19 @@ def main():
22
21
 
23
22
  output_schema = {
24
23
  "type": "object",
25
- "properties": {
26
- "sentiment": {
27
- "type": "string",
28
- "enum": ["positive", "negative", "neutral"]
29
- }
30
- },
31
- "required": ["sentiment"]
24
+ "properties": {"sentiment": {"type": "string", "enum": ["positive", "negative", "neutral"]}},
25
+ "required": ["sentiment"],
32
26
  }
33
27
 
34
28
  with Annotator(model_id="Qwen/Qwen2.5-0.5B-Instruct") as anno:
35
29
  ds = anno.annotate_dataset(
36
- "stanfordnlp/imdb",
37
30
  output_dir="outputs/sentiment-imdb-qwen",
31
+ dataset_name="stanfordnlp/imdb",
38
32
  dataset_split="test",
39
33
  new_hub_id=f"{hf_user}/sentiment-imdb",
40
34
  streaming=True,
41
35
  max_num_samples=250,
42
- cache_input_dataset=False, # `True` is generally useful, not for demo purposes
36
+ cache_input_dataset=False, # `True` is generally useful, not for demo purposes
43
37
  prompt_template=prompt_template,
44
38
  output_schema=output_schema,
45
39
  # Backup to HF every 10 samples.
@@ -48,5 +42,6 @@ def main():
48
42
  )
49
43
  print(ds)
50
44
 
45
+
51
46
  if __name__ == "__main__":
52
- main()
47
+ main()
@@ -62,7 +62,7 @@ class Annotator:
62
62
  return self
63
63
 
64
64
  def __exit__(self, exc_type, exc, tb):
65
- self.reset_model()
65
+ self.destroy_model()
66
66
 
67
67
  def _get_skip_idxs(
68
68
  self, *, pdout: Path, idx_column: str, dataset_split: str | None, dataset_config: str | None
@@ -406,29 +406,21 @@ class Annotator:
406
406
  **result,
407
407
  }
408
408
 
409
- def reset_model(self) -> None:
409
+ def destroy_model(self) -> None:
410
410
  """Clean up model resources to free memory.
411
411
 
412
- Destroys the distributed environment, clears GPU cache, and resets internal state..
412
+ Destroys the distributed environment, clears GPU cache, and resets internal state.
413
413
  """
414
- destroy_model_parallel()
415
- destroy_distributed_environment()
414
+ if isinstance(self.pipe, LLM):
415
+ destroy_model_parallel()
416
+ del self.pipe.llm_engine
417
+ del self.pipe
418
+ gc.collect()
419
+ cuda.empty_cache()
420
+ destroy_distributed_environment()
421
+ gc.collect()
416
422
 
417
- try:
418
- # Remove nested attributes if present
419
- if hasattr(self.pipe, "llm_engine") and hasattr(self.pipe.llm_engine, "model_executor"):
420
- del self.pipe.llm_engine.model_executor
421
- except Exception:
422
- pass
423
-
424
- del self.pipe
425
- del self.tokenizer
426
-
427
- self.pipe = None
428
- self.tokenizer = None
429
-
430
- cuda.empty_cache()
431
- gc.collect()
423
+ self.pipe = None
432
424
 
433
425
  def _process_batch(
434
426
  self,
@@ -598,7 +590,14 @@ class Annotator:
598
590
  shutil.rmtree(pdout)
599
591
  pdout.mkdir(exist_ok=True, parents=True)
600
592
 
601
- self._load_tokenizer()
593
+ # We need to clear the model before doing self._load_dataset because the model
594
+ # cannot be be pickled (which is needed for multiprocessing in dataset.map)
595
+ if self.pipe is not None and self.num_proc is not None:
596
+ self.destroy_model()
597
+
598
+ if self.tokenizer is None:
599
+ self._load_tokenizer()
600
+
602
601
  dataset, processed_n_samples = self._load_dataset(
603
602
  prompt_template=prompt_template,
604
603
  idx_column=idx_column,
@@ -1,11 +0,0 @@
1
- quality:
2
- ruff check src/llm_annotator tests/
3
- ruff format --check src/llm_annotator tests/
4
-
5
- style:
6
- ruff check src/llm_annotator tests/ --fix
7
- ruff format src/llm_annotator tests/
8
-
9
- setup:
10
- uv sync --dev
11
- pre-commit install --hook-type pre-push
File without changes
File without changes
File without changes