structured-eval 0.1.1__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. {structured_eval-0.1.1 → structured_eval-0.2.0}/PKG-INFO +1 -1
  2. {structured_eval-0.1.1 → structured_eval-0.2.0}/pyproject.toml +1 -1
  3. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_f1.py +5 -1
  4. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_precision.py +5 -1
  5. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_prf1.py +5 -1
  6. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_recall.py +5 -1
  7. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/base.py +19 -2
  8. structured_eval-0.2.0/structured_eval/metrics/character_f1.py +81 -0
  9. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/composite_score.py +2 -1
  10. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/date_distance_score.py +8 -2
  11. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/exponential_numeric_score.py +7 -2
  12. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/fuzzy.py +7 -1
  13. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/levenshtein.py +7 -2
  14. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/numeric.py +8 -0
  15. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/numeric_closeness.py +5 -1
  16. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_accuracy.py +2 -0
  17. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_f1.py +2 -0
  18. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_precision.py +2 -0
  19. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_prf1.py +2 -0
  20. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_recall.py +2 -0
  21. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_type_validity.py +2 -1
  22. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/regex_match.py +7 -1
  23. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/metric.py +2 -1
  24. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/metric.py +4 -1
  25. structured_eval-0.2.0/structured_eval/metrics/token_f1.py +83 -0
  26. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/url_match.py +8 -1
  27. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/__init__.py +2 -0
  28. structured_eval-0.2.0/structured_eval/metrics/utils/null.py +20 -0
  29. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/PKG-INFO +1 -1
  30. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/SOURCES.txt +1 -0
  31. structured_eval-0.1.1/structured_eval/metrics/character_f1.py +0 -50
  32. structured_eval-0.1.1/structured_eval/metrics/token_f1.py +0 -44
  33. {structured_eval-0.1.1 → structured_eval-0.2.0}/LICENSE +0 -0
  34. {structured_eval-0.1.1 → structured_eval-0.2.0}/README.md +0 -0
  35. {structured_eval-0.1.1 → structured_eval-0.2.0}/setup.cfg +0 -0
  36. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/__init__.py +0 -0
  37. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/__init__.py +0 -0
  38. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/base.py +0 -0
  39. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/by_index.py +0 -0
  40. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/by_key.py +0 -0
  41. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/factory.py +0 -0
  42. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/hungarian.py +0 -0
  43. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/api.py +0 -0
  44. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/__init__.py +0 -0
  45. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/aggregator.py +0 -0
  46. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/evaluator.py +0 -0
  47. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/metric_runner.py +0 -0
  48. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/parser.py +0 -0
  49. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/report_builder.py +0 -0
  50. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/tree_builder.py +0 -0
  51. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/__init__.py +0 -0
  52. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/base.py +0 -0
  53. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/json_parser.py +0 -0
  54. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/yaml_parser.py +0 -0
  55. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/__init__.py +0 -0
  56. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/_adapter.py +0 -0
  57. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/deepeval.py +0 -0
  58. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/langsmith.py +0 -0
  59. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/__init__.py +0 -0
  60. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_accuracy.py +0 -0
  61. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_cardinality.py +0 -0
  62. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_exact_match.py +0 -0
  63. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_jaccard_similarity.py +0 -0
  64. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/coverage_leaf_score.py +0 -0
  65. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/exact.py +0 -0
  66. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/field_faithfulness.py +0 -0
  67. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/invoker.py +0 -0
  68. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/mean_score.py +0 -0
  69. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_exact_match.py +0 -0
  70. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/overall_leaf_score.py +0 -0
  71. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/presence.py +0 -0
  72. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/__init__.py +0 -0
  73. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/dsl.py +0 -0
  74. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/engine.py +0 -0
  75. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/__init__.py +0 -0
  76. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/validator.py +0 -0
  77. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/structural_similarity.py +0 -0
  78. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/type_match.py +0 -0
  79. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/array.py +0 -0
  80. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/calculate.py +0 -0
  81. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/number.py +0 -0
  82. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/object_utils.py +0 -0
  83. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/__init__.py +0 -0
  84. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/config.py +0 -0
  85. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/context.py +0 -0
  86. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/metric_result.py +0 -0
  87. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/__init__.py +0 -0
  88. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/array_node.py +0 -0
  89. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/base.py +0 -0
  90. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/object_node.py +0 -0
  91. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/scalar.py +0 -0
  92. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/result.py +0 -0
  93. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/sample.py +0 -0
  94. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/py.typed +0 -0
  95. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/reporting/__init__.py +0 -0
  96. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/reporting/console.py +0 -0
  97. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/__init__.py +0 -0
  98. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/flatten.py +0 -0
  99. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/paths.py +0 -0
  100. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/structured_diff.py +0 -0
  101. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/dependency_links.txt +0 -0
  102. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/requires.txt +0 -0
  103. {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: structured-eval
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Summary: The LLM Structured Output Evaluation Framework
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/kirillpechurin/structured-eval
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "structured-eval"
7
- version = "0.1.1"
7
+ version = "0.2.0"
8
8
  description = "The LLM Structured Output Evaluation Framework"
9
9
  readme = "README.md"
10
10
  license = { text = "Apache-2.0" }
@@ -19,8 +19,12 @@ class ArrayF1(ArrayMetric):
19
19
  name = "array_f1"
20
20
 
21
21
  def __init__(
22
- self, threshold: float = 1.0, mode: stats.GradingMode = stats.GradingMode.HARD
22
+ self,
23
+ threshold: float = 1.0,
24
+ mode: stats.GradingMode = stats.GradingMode.HARD,
25
+ name: str | None = None,
23
26
  ):
27
+ super().__init__(name=name)
24
28
  self.threshold = threshold
25
29
  self.mode = stats.GradingMode(mode)
26
30
 
@@ -21,8 +21,12 @@ class ArrayPrecision(ArrayMetric):
21
21
  name = "array_precision"
22
22
 
23
23
  def __init__(
24
- self, threshold: float = 1.0, mode: stats.GradingMode = stats.GradingMode.HARD
24
+ self,
25
+ threshold: float = 1.0,
26
+ mode: stats.GradingMode = stats.GradingMode.HARD,
27
+ name: str | None = None,
25
28
  ):
29
+ super().__init__(name=name)
26
30
  self.threshold = threshold
27
31
  self.mode = stats.GradingMode(mode)
28
32
 
@@ -21,8 +21,12 @@ class ArrayPRF1(ArrayMetric):
21
21
  name = "array_prf1"
22
22
 
23
23
  def __init__(
24
- self, threshold: float = 1.0, mode: stats.GradingMode = stats.GradingMode.HARD
24
+ self,
25
+ threshold: float = 1.0,
26
+ mode: stats.GradingMode = stats.GradingMode.HARD,
27
+ name: str | None = None,
25
28
  ):
29
+ super().__init__(name=name)
26
30
  self.threshold = threshold
27
31
  self.mode = stats.GradingMode(mode)
28
32
 
@@ -20,8 +20,12 @@ class ArrayRecall(ArrayMetric):
20
20
  name = "array_recall"
21
21
 
22
22
  def __init__(
23
- self, threshold: float = 1.0, mode: stats.GradingMode = stats.GradingMode.HARD
23
+ self,
24
+ threshold: float = 1.0,
25
+ mode: stats.GradingMode = stats.GradingMode.HARD,
26
+ name: str | None = None,
24
27
  ):
28
+ super().__init__(name=name)
25
29
  self.threshold = threshold
26
30
  self.mode = stats.GradingMode(mode)
27
31
 
@@ -30,12 +30,29 @@ class BaseMetric(ABC): # noqa: B024 — registry root; subclasses define the in
30
30
 
31
31
  ``name`` is the key under which a scalar result lands in ``report.metrics``
32
32
  and ``FieldScore.metrics``. A metric that returns a ``dict`` instead writes
33
- each of its keys directly (the ``name`` is then only a registry handle).
34
- Declaring a subclass with a ``name`` registers it automatically.
33
+ each of its keys directly (the ``name`` is then only a registry handle, and
34
+ a per-instance override does not affect those keys). Declaring a subclass
35
+ with a ``name`` registers it automatically.
36
+
37
+ Passing ``name=`` at construction overrides the key **for that instance
38
+ only** — ``Numeric(tolerance=0.01, name="strict")``. Two instances of one
39
+ metric can then sit on the same node and report under distinct keys instead
40
+ of overwriting each other. The class registry, which backs name-string
41
+ resolution (``resolve_metric("numeric")``), is untouched.
42
+
43
+ Every metric defining its own ``__init__`` must accept ``name`` and forward
44
+ it here via ``super().__init__(name=name)``; ``test_metric_contracts.py``
45
+ enforces this across the registry.
35
46
  """
36
47
 
37
48
  name: str = ""
38
49
 
50
+ def __init__(self, name: str | None = None) -> None:
51
+ if name is not None:
52
+ if not name:
53
+ raise ValueError("metric name must be a non-empty string")
54
+ self.name = name
55
+
39
56
  def __init_subclass__(cls, **kwargs: Any) -> None:
40
57
  super().__init_subclass__(**kwargs)
41
58
  if n := getattr(cls, "name", None):
@@ -0,0 +1,81 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import string
5
+ from collections import Counter
6
+ from typing import Any
7
+
8
+ from structured_eval.metrics.base import FieldMetric
9
+ from structured_eval.metrics.utils.null import both_null
10
+
11
+ _IGNORE_PUNCTUATION_CHARS = frozenset(string.punctuation)
12
+ _IGNORE_WHITESPACE_REGEX = re.compile(r"\s+")
13
+
14
+
15
+ class CharacterF1(FieldMetric):
16
+ """Character-overlap F1 for short free-text fields.
17
+
18
+ Characters are matched as a **multiset** (``Counter``), so repeated
19
+ characters contribute only as many times as they appear on both sides.
20
+ Precision and recall are computed over character counts, and their
21
+ harmonic mean is returned. String-only: if either side is not a ``str``
22
+ the score is ``0.0`` (no coercion) — except two ``None``s, which agree
23
+ (``1.0``; see ``metrics.utils.null``).
24
+
25
+ Normalization is applied to both sides before the comparison and each
26
+ step can be turned off independently::
27
+
28
+ CharacterF1(ignore_case=False) # "AB" vs "ab" scores below 1.0
29
+ CharacterF1(ignore_punctuation=False) # "," and "." count as characters
30
+ CharacterF1(ignore_whitespace=False) # spaces count as characters
31
+
32
+ The defaults keep every normalization on. ``ignore_punctuation`` drops the
33
+ ASCII punctuation of ``string.punctuation`` — the same set :class:`TokenF1`
34
+ uses, so ``_`` is dropped and non-ASCII punctuation such as ``«»—`` is kept.
35
+ """
36
+
37
+ name = "character_f1"
38
+
39
+ def __init__(
40
+ self,
41
+ ignore_case: bool = True,
42
+ ignore_whitespace: bool = True,
43
+ ignore_punctuation: bool = True,
44
+ name: str | None = None,
45
+ ):
46
+ super().__init__(name=name)
47
+ self.ignore_case = ignore_case
48
+ self.ignore_whitespace = ignore_whitespace
49
+ self.ignore_punctuation = ignore_punctuation
50
+
51
+ def _characters(self, value: str) -> list[str]:
52
+ if self.ignore_case:
53
+ value = value.lower()
54
+ if self.ignore_punctuation:
55
+ value = "".join(ch for ch in value if ch not in _IGNORE_PUNCTUATION_CHARS)
56
+ if self.ignore_whitespace:
57
+ value = _IGNORE_WHITESPACE_REGEX.sub("", value)
58
+ return list(value)
59
+
60
+ def score(self, actual: Any, expected: Any) -> float:
61
+ if both_null(actual, expected):
62
+ return 1.0
63
+ if not (isinstance(actual, str) and isinstance(expected, str)):
64
+ return 0.0
65
+
66
+ a = self._characters(actual)
67
+ e = self._characters(expected)
68
+
69
+ if not a and not e:
70
+ return 1.0
71
+ if not a or not e:
72
+ return 0.0
73
+
74
+ same = sum((Counter(a) & Counter(e)).values())
75
+ if not same:
76
+ return 0.0
77
+
78
+ precision = same / len(a)
79
+ recall = same / len(e)
80
+
81
+ return 2 * precision * recall / (precision + recall)
@@ -29,7 +29,8 @@ class CompositeScore(AnyNodeMetric):
29
29
 
30
30
  name = "composite_score"
31
31
 
32
- def __init__(self, weights: dict[str, float]) -> None:
32
+ def __init__(self, weights: dict[str, float], name: str | None = None) -> None:
33
+ super().__init__(name=name)
33
34
  if not weights:
34
35
  raise ValueError("CompositeScore requires at least one metric weight")
35
36
  total = sum(weights.values())
@@ -6,6 +6,7 @@ from typing import Any
6
6
  from pydantic import TypeAdapter
7
7
 
8
8
  from structured_eval.metrics.base import FieldMetric
9
+ from structured_eval.metrics.utils.null import both_null
9
10
 
10
11
 
11
12
  def _to_date(value: Any) -> date | None:
@@ -34,17 +35,22 @@ class DateDistanceScore(FieldMetric):
34
35
  compared by their calendar date only (time-of-day is ignored).
35
36
 
36
37
  If either side cannot be read as a date — ``None``, an unparseable string,
37
- or any non-date type — the score is ``0.0``.
38
+ or any non-date type — the score is ``0.0``. Two ``None``s are the exception
39
+ — no date was expected and none was given, so they agree (``1.0``; see
40
+ ``metrics.utils.null``).
38
41
  """
39
42
 
40
43
  name = "date_distance_score"
41
44
 
42
- def __init__(self, max_days: int = 30) -> None:
45
+ def __init__(self, max_days: int = 30, name: str | None = None) -> None:
46
+ super().__init__(name=name)
43
47
  if max_days <= 0:
44
48
  raise ValueError("max_days must be greater than 0")
45
49
  self.max_days = max_days
46
50
 
47
51
  def score(self, actual: Any, expected: Any) -> float:
52
+ if both_null(actual, expected):
53
+ return 1.0
48
54
  if not isinstance(actual, (date, datetime)):
49
55
  actual = _to_date(actual)
50
56
  if not isinstance(expected, (date, datetime)):
@@ -4,6 +4,7 @@ import math
4
4
  from typing import Any
5
5
 
6
6
  from structured_eval.metrics.base import FieldMetric
7
+ from structured_eval.metrics.utils.null import both_null
7
8
  from structured_eval.metrics.utils.number import parse_number
8
9
 
9
10
 
@@ -29,17 +30,21 @@ class ExponentialNumericScore(FieldMetric):
29
30
  :class:`NumericCloseness`, so numeric strings are graded too. The metric
30
31
  applies **only to numbers**: if either side isn't numeric (``None``, a
31
32
  non-numeric string, or a ``bool`` — ``True`` is not ``1``) the score is
32
- ``0.0``.
33
+ ``0.0``. Two ``None``s are the exception — they agree (``1.0``; see
34
+ ``metrics.utils.null``).
33
35
  """
34
36
 
35
37
  name = "exponential_numeric_score"
36
38
 
37
- def __init__(self, scale: float = 1.0) -> None:
39
+ def __init__(self, scale: float = 1.0, name: str | None = None) -> None:
40
+ super().__init__(name=name)
38
41
  if scale <= 0:
39
42
  raise ValueError("scale must be greater than 0")
40
43
  self.scale = scale
41
44
 
42
45
  def score(self, actual: Any, expected: Any) -> float:
46
+ if both_null(actual, expected):
47
+ return 1.0
43
48
  a = parse_number(actual)
44
49
  e = parse_number(expected)
45
50
  if a is None or e is None:
@@ -4,6 +4,7 @@ from enum import StrEnum
4
4
  from typing import Any
5
5
 
6
6
  from structured_eval.metrics.base import FieldMetric
7
+ from structured_eval.metrics.utils.null import both_null
7
8
 
8
9
 
9
10
  class FuzzyMethod(StrEnum):
@@ -27,7 +28,8 @@ class Fuzzy(FieldMetric):
27
28
 
28
29
  ``normalize`` strips surrounding whitespace and lowercases before comparison.
29
30
  String-only: if either side is not a ``str`` the score is 0.0 (no coercion),
30
- consistent with the other text metrics.
31
+ consistent with the other text metrics — except two ``None``s, which agree
32
+ (1.0; see ``metrics.utils.null``).
31
33
  """
32
34
 
33
35
  name = "fuzzy"
@@ -36,11 +38,15 @@ class Fuzzy(FieldMetric):
36
38
  self,
37
39
  method: FuzzyMethod = FuzzyMethod.TOKEN_SORT_RATIO,
38
40
  normalize: bool = True,
41
+ name: str | None = None,
39
42
  ):
43
+ super().__init__(name=name)
40
44
  self.method = FuzzyMethod(method)
41
45
  self.normalize = normalize
42
46
 
43
47
  def score(self, actual: Any, expected: Any) -> float:
48
+ if both_null(actual, expected):
49
+ return 1.0
44
50
  if not (isinstance(actual, str) and isinstance(expected, str)):
45
51
  return 0.0
46
52
  try:
@@ -12,5 +12,10 @@ class Levenshtein(Fuzzy):
12
12
 
13
13
  name = "levenshtein"
14
14
 
15
- def __init__(self, method: FuzzyMethod = FuzzyMethod.RATIO, normalize: bool = True):
16
- super().__init__(method=method, normalize=normalize)
15
+ def __init__(
16
+ self,
17
+ method: FuzzyMethod = FuzzyMethod.RATIO,
18
+ normalize: bool = True,
19
+ name: str | None = None,
20
+ ):
21
+ super().__init__(method=method, normalize=normalize, name=name)
@@ -4,6 +4,7 @@ from enum import StrEnum
4
4
  from typing import Any
5
5
 
6
6
  from structured_eval.metrics.base import FieldMetric
7
+ from structured_eval.metrics.utils.null import both_null
7
8
  from structured_eval.metrics.utils.number import parse_number
8
9
 
9
10
 
@@ -32,6 +33,9 @@ class Numeric(FieldMetric):
32
33
  * ``relative_tolerance`` and/or ``absolute_tolerance`` — explicit bands; a
33
34
  value matches if it falls within *either* band. When either is supplied it
34
35
  takes precedence over ``tolerance``/``mode``.
36
+
37
+ Two ``None``s agree (1.0; see ``metrics.utils.null``); a one-sided ``None``
38
+ is 0.0.
35
39
  """
36
40
 
37
41
  name = "numeric"
@@ -42,13 +46,17 @@ class Numeric(FieldMetric):
42
46
  mode: NumericMode = NumericMode.RELATIVE,
43
47
  relative_tolerance: float | None = None,
44
48
  absolute_tolerance: float | None = None,
49
+ name: str | None = None,
45
50
  ):
51
+ super().__init__(name=name)
46
52
  self.tolerance = tolerance
47
53
  self.mode = NumericMode(mode)
48
54
  self.relative_tolerance = relative_tolerance
49
55
  self.absolute_tolerance = absolute_tolerance
50
56
 
51
57
  def score(self, actual: Any, expected: Any) -> float:
58
+ if both_null(actual, expected):
59
+ return 1.0
52
60
  a = parse_number(actual)
53
61
  e = parse_number(expected)
54
62
  if a is None or e is None:
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
  from typing import Any
4
4
 
5
5
  from structured_eval.metrics.base import FieldMetric
6
+ from structured_eval.metrics.utils.null import both_null
6
7
  from structured_eval.metrics.utils.number import parse_number
7
8
 
8
9
 
@@ -19,12 +20,15 @@ class NumericCloseness(FieldMetric):
19
20
  Values are parsed with the shared lenient numeric parser (same as
20
21
  :class:`Numeric`), so numeric strings are graded too. The metric applies
21
22
  **only to numbers**: if either side isn't numeric (``None``, a non-numeric
22
- string, or a ``bool`` — ``True`` is not ``1``) the score is 0.0.
23
+ string, or a ``bool`` — ``True`` is not ``1``) the score is 0.0. Two
24
+ ``None``s are the exception — they agree (1.0; see ``metrics.utils.null``).
23
25
  """
24
26
 
25
27
  name = "numeric_closeness"
26
28
 
27
29
  def score(self, actual: Any, expected: Any) -> float:
30
+ if both_null(actual, expected):
31
+ return 1.0
28
32
  a = parse_number(actual)
29
33
  e = parse_number(expected)
30
34
  if a is None or e is None:
@@ -31,7 +31,9 @@ class ObjectAccuracy(ObjectMetric):
31
31
  self,
32
32
  score_policy: dict[str, Any] | None = None,
33
33
  weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
34
+ name: str | None = None,
34
35
  ):
36
+ super().__init__(name=name)
35
37
  self.score_policy = score_policy
36
38
  self.weight_mode = stats.WeightMode(weight_mode)
37
39
 
@@ -26,7 +26,9 @@ class ObjectF1(ObjectMetric):
26
26
  threshold: float | None = None,
27
27
  mode: stats.GradingMode = stats.GradingMode.HARD,
28
28
  weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
29
+ name: str | None = None,
29
30
  ):
31
+ super().__init__(name=name)
30
32
  self.score_policy = score_policy
31
33
  self.threshold = threshold
32
34
  self.mode = stats.GradingMode(mode)
@@ -30,7 +30,9 @@ class ObjectPrecision(ObjectMetric):
30
30
  threshold: float | None = None,
31
31
  mode: stats.GradingMode = stats.GradingMode.HARD,
32
32
  weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
33
+ name: str | None = None,
33
34
  ):
35
+ super().__init__(name=name)
34
36
  self.score_policy = score_policy
35
37
  self.threshold = threshold
36
38
  self.mode = stats.GradingMode(mode)
@@ -26,7 +26,9 @@ class ObjectPRF1(ObjectMetric):
26
26
  threshold: float | None = None,
27
27
  mode: stats.GradingMode = stats.GradingMode.HARD,
28
28
  weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
29
+ name: str | None = None,
29
30
  ):
31
+ super().__init__(name=name)
30
32
  self.score_policy = score_policy
31
33
  self.threshold = threshold
32
34
  self.mode = stats.GradingMode(mode)
@@ -25,7 +25,9 @@ class ObjectRecall(ObjectMetric):
25
25
  threshold: float | None = None,
26
26
  mode: stats.GradingMode = stats.GradingMode.HARD,
27
27
  weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
28
+ name: str | None = None,
28
29
  ):
30
+ super().__init__(name=name)
29
31
  self.score_policy = score_policy
30
32
  self.threshold = threshold
31
33
  self.mode = stats.GradingMode(mode)
@@ -23,7 +23,8 @@ class ObjectTypeValidity(ObjectMetric):
23
23
 
24
24
  name = "object_type_validity"
25
25
 
26
- def __init__(self) -> None:
26
+ def __init__(self, name: str | None = None) -> None:
27
+ super().__init__(name=name)
27
28
  self._type_match = MetricInvoker(TypeMatch())
28
29
 
29
30
  def compute(self, node: ObjectNode) -> float:
@@ -4,6 +4,7 @@ import re
4
4
  from typing import Any
5
5
 
6
6
  from structured_eval.metrics.base import FieldMetric
7
+ from structured_eval.metrics.utils.null import both_null
7
8
 
8
9
 
9
10
  class RegexMatch(FieldMetric):
@@ -11,7 +12,8 @@ class RegexMatch(FieldMetric):
11
12
 
12
13
  A **string-only** metric: if either side is not a ``str`` the score is
13
14
  ``0.0`` (use ``Numeric`` for numbers, ``ExactMatch`` for verbatim
14
- equality). For two strings it applies, in order, optional ``lower`` and
15
+ equality) — except two ``None``s, which agree (``1.0``; see
16
+ ``metrics.utils.null``). For two strings it applies, in order, optional ``lower`` and
15
17
  ``strip``, then substitutes every match of ``pattern`` with ``repl``, and
16
18
  compares the results exactly.
17
19
 
@@ -31,7 +33,9 @@ class RegexMatch(FieldMetric):
31
33
  repl: str = " ",
32
34
  lower: bool = True,
33
35
  strip: bool = True,
36
+ name: str | None = None,
34
37
  ):
38
+ super().__init__(name=name)
35
39
  self.pattern = re.compile(pattern) if isinstance(pattern, str) else pattern
36
40
  self.repl = repl
37
41
  self.lower = lower
@@ -46,6 +50,8 @@ class RegexMatch(FieldMetric):
46
50
  return value.strip() if self.strip else value
47
51
 
48
52
  def score(self, actual: Any, expected: Any) -> float:
53
+ if both_null(actual, expected):
54
+ return 1.0
49
55
  if not (isinstance(actual, str) and isinstance(expected, str)):
50
56
  return 0.0
51
57
  return 1.0 if self._normalize(actual) == self._normalize(expected) else 0.0
@@ -21,7 +21,8 @@ class RulePassRate(RootMetric):
21
21
 
22
22
  name = "rule_pass_rate"
23
23
 
24
- def __init__(self, rules: list[Any]):
24
+ def __init__(self, rules: list[Any], name: str | None = None):
25
+ super().__init__(name=name)
25
26
  self.rules = rules
26
27
  self.processor = RuleProcessor()
27
28
 
@@ -21,7 +21,10 @@ class SchemaValidity(RootMetric):
21
21
 
22
22
  name = "schema_validity"
23
23
 
24
- def __init__(self, schema: type[BaseModel] | dict[str, Any]):
24
+ def __init__(
25
+ self, schema: type[BaseModel] | dict[str, Any], name: str | None = None
26
+ ):
27
+ super().__init__(name=name)
25
28
  self.validator = SchemaValidator(schema)
26
29
 
27
30
  def compute(self, node: EvalNode) -> tuple[float, dict[str, Any]]:
@@ -0,0 +1,83 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import string
5
+ from collections import Counter
6
+ from typing import Any
7
+
8
+ from structured_eval.metrics.base import FieldMetric
9
+ from structured_eval.metrics.utils.null import both_null
10
+
11
+ _IGNORE_PUNCTUATION_CHARS = frozenset(string.punctuation)
12
+ _IGNORE_ARTICLES_REGEX = re.compile(r"\b(a|an|the)\b", re.IGNORECASE)
13
+
14
+
15
+ class TokenF1(FieldMetric):
16
+ """SQuAD-style token-overlap F1 — a default for free-text fields.
17
+
18
+ On its defaults this reproduces the ``f1_score`` of the official SQuAD v1.1
19
+ evaluation script: both sides go through the reference ``normalize_answer``
20
+ (lowercase, drop punctuation, drop the articles ``a``/``an``/``the``, collapse
21
+ whitespace), then tokens are matched as a **multiset** (``Counter``) — a
22
+ repeated token only helps as often as it appears on both sides, so
23
+ ``"the the cat"`` vs ``"the cat"`` is 0.8, not 1.0. Precision and recall are
24
+ over the token *counts*; their harmonic mean is the score.
25
+
26
+ Each normalization step can be turned off independently::
27
+
28
+ TokenF1(ignore_case=False) # "AB" vs "ab" scores below 1.0
29
+ TokenF1(ignore_punctuation=False) # "fox." and "fox" are distinct tokens
30
+ TokenF1(ignore_articles=False) # "the" counts as a token like any other
31
+
32
+ Two deliberate departures from the reference script, both because this scores
33
+ fields rather than question answers: two empty strings score 1.0 (the script
34
+ returns 0.0, an empty answer being a failed answer), and a value that is not a
35
+ ``str`` scores 0.0 with no coercion — except two ``None``s, which agree (1.0;
36
+ see ``metrics.utils.null``).
37
+ """
38
+
39
+ name = "token_f1"
40
+
41
+ def __init__(
42
+ self,
43
+ ignore_case: bool = True,
44
+ ignore_punctuation: bool = True,
45
+ ignore_articles: bool = True,
46
+ name: str | None = None,
47
+ ):
48
+ super().__init__(name=name)
49
+ self.ignore_case = ignore_case
50
+ self.ignore_punctuation = ignore_punctuation
51
+ self.ignore_articles = ignore_articles
52
+
53
+ def _tokenize(self, value: str) -> list[str]:
54
+ if self.ignore_case:
55
+ value = value.lower()
56
+ if self.ignore_punctuation:
57
+ value = "".join(ch for ch in value if ch not in _IGNORE_PUNCTUATION_CHARS)
58
+ if self.ignore_articles:
59
+ value = _IGNORE_ARTICLES_REGEX.sub(" ", value)
60
+ return value.split()
61
+
62
+ def score(self, actual: Any, expected: Any) -> float:
63
+ if both_null(actual, expected):
64
+ return 1.0
65
+ if not (isinstance(actual, str) and isinstance(expected, str)):
66
+ return 0.0
67
+
68
+ a = self._tokenize(actual)
69
+ e = self._tokenize(expected)
70
+
71
+ if not a and not e:
72
+ return 1.0
73
+ if not a or not e:
74
+ return 0.0
75
+
76
+ same = sum((Counter(a) & Counter(e)).values())
77
+ if not same:
78
+ return 0.0
79
+
80
+ precision = same / len(a)
81
+ recall = same / len(e)
82
+
83
+ return 2 * precision * recall / (precision + recall)
@@ -4,6 +4,7 @@ from typing import Any
4
4
  from urllib.parse import parse_qsl, unquote, urlsplit, urlunsplit
5
5
 
6
6
  from structured_eval.metrics.base import FieldMetric
7
+ from structured_eval.metrics.utils.null import both_null
7
8
 
8
9
 
9
10
  class UrlMatch(FieldMetric):
@@ -27,7 +28,9 @@ class UrlMatch(FieldMetric):
27
28
 
28
29
  Both sides must be non-empty strings that parse to a URL with a scheme and a
29
30
  host. Anything else — a non-string, an empty string, or a bare path with no
30
- scheme/host — scores ``0.0``.
31
+ scheme/host — scores ``0.0``. Two ``None``s are the exception — no URL was
32
+ expected and none was given, so they agree (``1.0``; see
33
+ ``metrics.utils.null``).
31
34
  """
32
35
 
33
36
  name = "url_match"
@@ -38,7 +41,9 @@ class UrlMatch(FieldMetric):
38
41
  ignore_query: bool = False,
39
42
  ignore_fragment: bool = True,
40
43
  ignore_www: bool = True,
44
+ name: str | None = None,
41
45
  ) -> None:
46
+ super().__init__(name=name)
42
47
  self.ignore_query = ignore_query
43
48
  self.ignore_fragment = ignore_fragment
44
49
  self.ignore_www = ignore_www
@@ -75,6 +80,8 @@ class UrlMatch(FieldMetric):
75
80
  return (urlunsplit((scheme, netloc, path, query, fragment)),)
76
81
 
77
82
  def score(self, actual: Any, expected: Any) -> float:
83
+ if both_null(actual, expected):
84
+ return 1.0
78
85
  norm_actual = self._normalize(actual)
79
86
  norm_expected = self._normalize(expected)
80
87
  if norm_actual is None or norm_expected is None:
@@ -7,4 +7,6 @@ Clearly-scoped modules:
7
7
  * ``object_utils`` — turning an object's matched fields into the
8
8
  ``(score, threshold)`` pairs that ``calculate.prf_counts`` consumes.
9
9
  * ``array`` — the same for an array's aligned items, plus missing/spurious counts.
10
+ * ``number`` — the lenient numeric parsing shared by the numeric field metrics.
11
+ * ``null`` — the ``(None, None) → 1.0`` rule shared by the comparison field metrics.
10
12
  """
@@ -0,0 +1,20 @@
1
+ """The ``(None, None) → 1.0`` rule shared by the comparison field metrics.
2
+
3
+ The schemas under evaluation ask for ``null`` whenever a value is absent, so a
4
+ null expectation met by a null answer is a *correct* answer. Without this gate a
5
+ metric's type check (str / number / date) would reject the pair and score that
6
+ right answer ``0.0``. Only ``None`` counts as null — an empty string, an empty
7
+ list or a missing key are values, and are graded as such.
8
+
9
+ A one-sided ``None`` stays a mismatch: a value was expected and nothing came
10
+ back, or nothing was expected and a value was invented.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Any
16
+
17
+
18
+ def both_null(actual: Any, expected: Any) -> bool:
19
+ """True when neither side has a value — the two agree."""
20
+ return actual is None and expected is None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: structured-eval
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Summary: The LLM Structured Output Evaluation Framework
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/kirillpechurin/structured-eval
@@ -77,6 +77,7 @@ structured_eval/metrics/schema_validity/validator.py
77
77
  structured_eval/metrics/utils/__init__.py
78
78
  structured_eval/metrics/utils/array.py
79
79
  structured_eval/metrics/utils/calculate.py
80
+ structured_eval/metrics/utils/null.py
80
81
  structured_eval/metrics/utils/number.py
81
82
  structured_eval/metrics/utils/object_utils.py
82
83
  structured_eval/models/__init__.py
@@ -1,50 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import re
4
- from collections import Counter
5
- from typing import Any
6
-
7
- from structured_eval.metrics.base import FieldMetric
8
-
9
- _NON_WORD = re.compile(r"[^\w\s]")
10
-
11
-
12
- def _characters(value: Any) -> list[str]:
13
- """Lowercase, drop punctuation and whitespace, split into characters."""
14
- normalized = _NON_WORD.sub("", str(value).lower())
15
- normalized = "".join(normalized.split()) # remove all whitespace
16
- return list(normalized)
17
-
18
-
19
- class CharacterF1(FieldMetric):
20
- """Character-overlap F1 for short free-text fields.
21
-
22
- Characters are matched as a **multiset** (``Counter``), so repeated
23
- characters contribute only as many times as they appear on both sides.
24
- Precision and recall are computed over character counts, and their
25
- harmonic mean is returned. String-only: if either side is not a ``str``
26
- the score is ``0.0`` (no coercion).
27
- """
28
-
29
- name = "character_f1"
30
-
31
- def score(self, actual: Any, expected: Any) -> float:
32
- if not (isinstance(actual, str) and isinstance(expected, str)):
33
- return 0.0
34
-
35
- a = _characters(actual)
36
- e = _characters(expected)
37
-
38
- if not a and not e:
39
- return 1.0
40
- if not a or not e:
41
- return 0.0
42
-
43
- same = sum((Counter(a) & Counter(e)).values())
44
- if not same:
45
- return 0.0
46
-
47
- precision = same / len(a)
48
- recall = same / len(e)
49
-
50
- return 2 * precision * recall / (precision + recall)
@@ -1,44 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import re
4
- from collections import Counter
5
- from typing import Any
6
-
7
- from structured_eval.metrics.base import FieldMetric
8
-
9
- _NON_WORD = re.compile(r"[^\w\s]")
10
-
11
-
12
- def _tokenize(value: Any) -> list[str]:
13
- """Lowercase, drop punctuation, split on whitespace."""
14
- return _NON_WORD.sub(" ", str(value).lower()).split()
15
-
16
-
17
- class TokenF1(FieldMetric):
18
- """SQuAD-style token-overlap F1 — a default for free-text fields.
19
-
20
- Tokens are matched as a **multiset** (``Counter``), counting shared tokens
21
- with multiplicity exactly like the official SQuAD F1 — so a repeated token
22
- only helps as often as it appears on both sides (``"the the cat"`` vs
23
- ``"the cat"`` is 0.8, not 1.0). Precision and recall are over the token
24
- *counts*; their harmonic mean is the score. String-only: if either side is
25
- not a ``str`` the score is 0.0 (no coercion).
26
- """
27
-
28
- name = "token_f1"
29
-
30
- def score(self, actual: Any, expected: Any) -> float:
31
- if not (isinstance(actual, str) and isinstance(expected, str)):
32
- return 0.0
33
- a = _tokenize(actual)
34
- e = _tokenize(expected)
35
- if not a and not e:
36
- return 1.0
37
- if not a or not e:
38
- return 0.0
39
- same = sum((Counter(a) & Counter(e)).values())
40
- if not same:
41
- return 0.0
42
- precision = same / len(a)
43
- recall = same / len(e)
44
- return 2 * precision * recall / (precision + recall)
File without changes