llm-annotator 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-annotator
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: An easy-to-extend LLM annotator for robust, resumable data annotation.
5
5
  Author-email: Bram Vanroy <2779410+BramVanroy@users.noreply.github.com>
6
6
  License-Expression: Apache-2.0
@@ -47,7 +47,7 @@ select = ["C", "E", "F", "W", "I"]
47
47
 
48
48
  [tool.ruff.lint.isort]
49
49
  lines-after-imports = 2
50
- known-first-party = ["c5"]
50
+ known-first-party = ["llm_annotator"]
51
51
 
52
52
  [tool.ruff.format]
53
53
  # Like Black, use double quotes for strings.
@@ -232,24 +232,18 @@ class Annotator:
232
232
  dataset = self._preprocess_dataset(dataset=dataset)
233
233
 
234
234
  dataset = dataset.map(
235
- lambda sample, idx: {
236
- f"{prefix}prompted": self.tokenizer.apply_chat_template(
237
- [
238
- {
239
- "role": "user",
240
- "content": prompt_template.format(**{fld: sample[fld] for fld in prompt_fields}),
241
- }
242
- ],
243
- tokenize=False,
244
- add_generation_template=True,
245
- enable_thinking=self.enable_thinking,
246
- ),
247
- idx_column: idx,
248
- },
235
+ self.apply_prompt_template,
249
236
  with_indices=True,
250
237
  num_proc=self.num_proc,
238
+ fn_kwargs={
239
+ "prompt_fields": prompt_fields,
240
+ "prompt_template": prompt_template,
241
+ "idx_column": idx_column,
242
+ "prefix": prefix,
243
+ },
251
244
  desc="Applying prompt template",
252
245
  )
246
+
253
247
  if cache_input_dataset:
254
248
  dataset.save_to_disk(p_cached_input_ds)
255
249
 
@@ -273,6 +267,38 @@ class Annotator:
273
267
  dataset = self._postprocess_dataset(dataset=dataset)
274
268
  return dataset, processed_n_samples
275
269
 
270
+ def apply_prompt_template(
271
+ self, sample: dict, idx: int, prompt_fields: Iterable[str], prompt_template: str, idx_column: str, prefix: str
272
+ ) -> dict[str, str | int]:
273
+ """Apply the prompt template to a single dataset sample. Fills in the prompt template with values from the sample,
274
+ based on the prompt_fields.
275
+
276
+ Args:
277
+ sample: The dataset sample to process.
278
+ idx: The index of the sample in the dataset.
279
+ prompt_fields: Fields required by the prompt template.
280
+ prompt_template: The prompt template string with placeholders.
281
+ idx_column: Column name to use as unique identifier.
282
+ prefix: String prefix to use for internal column names.
283
+
284
+ Returns:
285
+ A dictionary with the filled-in prompt and the sample index.
286
+ """
287
+ return {
288
+ f"{prefix}prompted": self.tokenizer.apply_chat_template(
289
+ [
290
+ {
291
+ "role": "user",
292
+ "content": prompt_template.format(**{fld: sample[fld] for fld in prompt_fields}),
293
+ }
294
+ ],
295
+ tokenize=False,
296
+ add_generation_template=True,
297
+ enable_thinking=self.enable_thinking,
298
+ ),
299
+ idx_column: idx,
300
+ }
301
+
276
302
  def _preprocess_dataset(self, *, dataset: Dataset) -> Dataset:
277
303
  """Preprocess the dataset before applying prompt templates.
278
304
 
File without changes
File without changes
File without changes
File without changes