structured-eval 0.1.1__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {structured_eval-0.1.1 → structured_eval-0.2.0}/PKG-INFO +1 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/pyproject.toml +1 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_f1.py +5 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_precision.py +5 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_prf1.py +5 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_recall.py +5 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/base.py +19 -2
- structured_eval-0.2.0/structured_eval/metrics/character_f1.py +81 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/composite_score.py +2 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/date_distance_score.py +8 -2
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/exponential_numeric_score.py +7 -2
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/fuzzy.py +7 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/levenshtein.py +7 -2
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/numeric.py +8 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/numeric_closeness.py +5 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_accuracy.py +2 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_f1.py +2 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_precision.py +2 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_prf1.py +2 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_recall.py +2 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_type_validity.py +2 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/regex_match.py +7 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/metric.py +2 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/metric.py +4 -1
- structured_eval-0.2.0/structured_eval/metrics/token_f1.py +83 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/url_match.py +8 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/__init__.py +2 -0
- structured_eval-0.2.0/structured_eval/metrics/utils/null.py +20 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/PKG-INFO +1 -1
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/SOURCES.txt +1 -0
- structured_eval-0.1.1/structured_eval/metrics/character_f1.py +0 -50
- structured_eval-0.1.1/structured_eval/metrics/token_f1.py +0 -44
- {structured_eval-0.1.1 → structured_eval-0.2.0}/LICENSE +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/README.md +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/setup.cfg +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/base.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/by_index.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/by_key.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/factory.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/alignment/hungarian.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/api.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/aggregator.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/evaluator.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/metric_runner.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/parser.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/report_builder.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/engine/tree_builder.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/base.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/json_parser.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/formats/yaml_parser.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/_adapter.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/deepeval.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/integrations/langsmith.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_accuracy.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_cardinality.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_exact_match.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_jaccard_similarity.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/coverage_leaf_score.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/exact.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/field_faithfulness.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/invoker.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/mean_score.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_exact_match.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/overall_leaf_score.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/presence.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/dsl.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/engine.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/validator.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/structural_similarity.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/type_match.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/array.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/calculate.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/number.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/object_utils.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/config.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/context.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/metric_result.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/array_node.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/base.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/object_node.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/nodes/scalar.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/result.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/models/sample.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/py.typed +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/reporting/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/reporting/console.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/__init__.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/flatten.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/paths.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/utils/structured_diff.py +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/dependency_links.txt +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/requires.txt +0 -0
- {structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/top_level.txt +0 -0
|
@@ -19,8 +19,12 @@ class ArrayF1(ArrayMetric):
|
|
|
19
19
|
name = "array_f1"
|
|
20
20
|
|
|
21
21
|
def __init__(
|
|
22
|
-
self,
|
|
22
|
+
self,
|
|
23
|
+
threshold: float = 1.0,
|
|
24
|
+
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
25
|
+
name: str | None = None,
|
|
23
26
|
):
|
|
27
|
+
super().__init__(name=name)
|
|
24
28
|
self.threshold = threshold
|
|
25
29
|
self.mode = stats.GradingMode(mode)
|
|
26
30
|
|
|
@@ -21,8 +21,12 @@ class ArrayPrecision(ArrayMetric):
|
|
|
21
21
|
name = "array_precision"
|
|
22
22
|
|
|
23
23
|
def __init__(
|
|
24
|
-
self,
|
|
24
|
+
self,
|
|
25
|
+
threshold: float = 1.0,
|
|
26
|
+
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
27
|
+
name: str | None = None,
|
|
25
28
|
):
|
|
29
|
+
super().__init__(name=name)
|
|
26
30
|
self.threshold = threshold
|
|
27
31
|
self.mode = stats.GradingMode(mode)
|
|
28
32
|
|
|
@@ -21,8 +21,12 @@ class ArrayPRF1(ArrayMetric):
|
|
|
21
21
|
name = "array_prf1"
|
|
22
22
|
|
|
23
23
|
def __init__(
|
|
24
|
-
self,
|
|
24
|
+
self,
|
|
25
|
+
threshold: float = 1.0,
|
|
26
|
+
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
27
|
+
name: str | None = None,
|
|
25
28
|
):
|
|
29
|
+
super().__init__(name=name)
|
|
26
30
|
self.threshold = threshold
|
|
27
31
|
self.mode = stats.GradingMode(mode)
|
|
28
32
|
|
|
@@ -20,8 +20,12 @@ class ArrayRecall(ArrayMetric):
|
|
|
20
20
|
name = "array_recall"
|
|
21
21
|
|
|
22
22
|
def __init__(
|
|
23
|
-
self,
|
|
23
|
+
self,
|
|
24
|
+
threshold: float = 1.0,
|
|
25
|
+
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
26
|
+
name: str | None = None,
|
|
24
27
|
):
|
|
28
|
+
super().__init__(name=name)
|
|
25
29
|
self.threshold = threshold
|
|
26
30
|
self.mode = stats.GradingMode(mode)
|
|
27
31
|
|
|
@@ -30,12 +30,29 @@ class BaseMetric(ABC): # noqa: B024 — registry root; subclasses define the in
|
|
|
30
30
|
|
|
31
31
|
``name`` is the key under which a scalar result lands in ``report.metrics``
|
|
32
32
|
and ``FieldScore.metrics``. A metric that returns a ``dict`` instead writes
|
|
33
|
-
each of its keys directly (the ``name`` is then only a registry handle
|
|
34
|
-
|
|
33
|
+
each of its keys directly (the ``name`` is then only a registry handle, and
|
|
34
|
+
a per-instance override does not affect those keys). Declaring a subclass
|
|
35
|
+
with a ``name`` registers it automatically.
|
|
36
|
+
|
|
37
|
+
Passing ``name=`` at construction overrides the key **for that instance
|
|
38
|
+
only** — ``Numeric(tolerance=0.01, name="strict")``. Two instances of one
|
|
39
|
+
metric can then sit on the same node and report under distinct keys instead
|
|
40
|
+
of overwriting each other. The class registry, which backs name-string
|
|
41
|
+
resolution (``resolve_metric("numeric")``), is untouched.
|
|
42
|
+
|
|
43
|
+
Every metric defining its own ``__init__`` must accept ``name`` and forward
|
|
44
|
+
it here via ``super().__init__(name=name)``; ``test_metric_contracts.py``
|
|
45
|
+
enforces this across the registry.
|
|
35
46
|
"""
|
|
36
47
|
|
|
37
48
|
name: str = ""
|
|
38
49
|
|
|
50
|
+
def __init__(self, name: str | None = None) -> None:
|
|
51
|
+
if name is not None:
|
|
52
|
+
if not name:
|
|
53
|
+
raise ValueError("metric name must be a non-empty string")
|
|
54
|
+
self.name = name
|
|
55
|
+
|
|
39
56
|
def __init_subclass__(cls, **kwargs: Any) -> None:
|
|
40
57
|
super().__init_subclass__(**kwargs)
|
|
41
58
|
if n := getattr(cls, "name", None):
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import string
|
|
5
|
+
from collections import Counter
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from structured_eval.metrics.base import FieldMetric
|
|
9
|
+
from structured_eval.metrics.utils.null import both_null
|
|
10
|
+
|
|
11
|
+
_IGNORE_PUNCTUATION_CHARS = frozenset(string.punctuation)
|
|
12
|
+
_IGNORE_WHITESPACE_REGEX = re.compile(r"\s+")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CharacterF1(FieldMetric):
|
|
16
|
+
"""Character-overlap F1 for short free-text fields.
|
|
17
|
+
|
|
18
|
+
Characters are matched as a **multiset** (``Counter``), so repeated
|
|
19
|
+
characters contribute only as many times as they appear on both sides.
|
|
20
|
+
Precision and recall are computed over character counts, and their
|
|
21
|
+
harmonic mean is returned. String-only: if either side is not a ``str``
|
|
22
|
+
the score is ``0.0`` (no coercion) — except two ``None``s, which agree
|
|
23
|
+
(``1.0``; see ``metrics.utils.null``).
|
|
24
|
+
|
|
25
|
+
Normalization is applied to both sides before the comparison and each
|
|
26
|
+
step can be turned off independently::
|
|
27
|
+
|
|
28
|
+
CharacterF1(ignore_case=False) # "AB" vs "ab" scores below 1.0
|
|
29
|
+
CharacterF1(ignore_punctuation=False) # "," and "." count as characters
|
|
30
|
+
CharacterF1(ignore_whitespace=False) # spaces count as characters
|
|
31
|
+
|
|
32
|
+
The defaults keep every normalization on. ``ignore_punctuation`` drops the
|
|
33
|
+
ASCII punctuation of ``string.punctuation`` — the same set :class:`TokenF1`
|
|
34
|
+
uses, so ``_`` is dropped and non-ASCII punctuation such as ``«»—`` is kept.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
name = "character_f1"
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
ignore_case: bool = True,
|
|
42
|
+
ignore_whitespace: bool = True,
|
|
43
|
+
ignore_punctuation: bool = True,
|
|
44
|
+
name: str | None = None,
|
|
45
|
+
):
|
|
46
|
+
super().__init__(name=name)
|
|
47
|
+
self.ignore_case = ignore_case
|
|
48
|
+
self.ignore_whitespace = ignore_whitespace
|
|
49
|
+
self.ignore_punctuation = ignore_punctuation
|
|
50
|
+
|
|
51
|
+
def _characters(self, value: str) -> list[str]:
|
|
52
|
+
if self.ignore_case:
|
|
53
|
+
value = value.lower()
|
|
54
|
+
if self.ignore_punctuation:
|
|
55
|
+
value = "".join(ch for ch in value if ch not in _IGNORE_PUNCTUATION_CHARS)
|
|
56
|
+
if self.ignore_whitespace:
|
|
57
|
+
value = _IGNORE_WHITESPACE_REGEX.sub("", value)
|
|
58
|
+
return list(value)
|
|
59
|
+
|
|
60
|
+
def score(self, actual: Any, expected: Any) -> float:
|
|
61
|
+
if both_null(actual, expected):
|
|
62
|
+
return 1.0
|
|
63
|
+
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
64
|
+
return 0.0
|
|
65
|
+
|
|
66
|
+
a = self._characters(actual)
|
|
67
|
+
e = self._characters(expected)
|
|
68
|
+
|
|
69
|
+
if not a and not e:
|
|
70
|
+
return 1.0
|
|
71
|
+
if not a or not e:
|
|
72
|
+
return 0.0
|
|
73
|
+
|
|
74
|
+
same = sum((Counter(a) & Counter(e)).values())
|
|
75
|
+
if not same:
|
|
76
|
+
return 0.0
|
|
77
|
+
|
|
78
|
+
precision = same / len(a)
|
|
79
|
+
recall = same / len(e)
|
|
80
|
+
|
|
81
|
+
return 2 * precision * recall / (precision + recall)
|
|
@@ -29,7 +29,8 @@ class CompositeScore(AnyNodeMetric):
|
|
|
29
29
|
|
|
30
30
|
name = "composite_score"
|
|
31
31
|
|
|
32
|
-
def __init__(self, weights: dict[str, float]) -> None:
|
|
32
|
+
def __init__(self, weights: dict[str, float], name: str | None = None) -> None:
|
|
33
|
+
super().__init__(name=name)
|
|
33
34
|
if not weights:
|
|
34
35
|
raise ValueError("CompositeScore requires at least one metric weight")
|
|
35
36
|
total = sum(weights.values())
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/date_distance_score.py
RENAMED
|
@@ -6,6 +6,7 @@ from typing import Any
|
|
|
6
6
|
from pydantic import TypeAdapter
|
|
7
7
|
|
|
8
8
|
from structured_eval.metrics.base import FieldMetric
|
|
9
|
+
from structured_eval.metrics.utils.null import both_null
|
|
9
10
|
|
|
10
11
|
|
|
11
12
|
def _to_date(value: Any) -> date | None:
|
|
@@ -34,17 +35,22 @@ class DateDistanceScore(FieldMetric):
|
|
|
34
35
|
compared by their calendar date only (time-of-day is ignored).
|
|
35
36
|
|
|
36
37
|
If either side cannot be read as a date — ``None``, an unparseable string,
|
|
37
|
-
or any non-date type — the score is ``0.0``.
|
|
38
|
+
or any non-date type — the score is ``0.0``. Two ``None``s are the exception
|
|
39
|
+
— no date was expected and none was given, so they agree (``1.0``; see
|
|
40
|
+
``metrics.utils.null``).
|
|
38
41
|
"""
|
|
39
42
|
|
|
40
43
|
name = "date_distance_score"
|
|
41
44
|
|
|
42
|
-
def __init__(self, max_days: int = 30) -> None:
|
|
45
|
+
def __init__(self, max_days: int = 30, name: str | None = None) -> None:
|
|
46
|
+
super().__init__(name=name)
|
|
43
47
|
if max_days <= 0:
|
|
44
48
|
raise ValueError("max_days must be greater than 0")
|
|
45
49
|
self.max_days = max_days
|
|
46
50
|
|
|
47
51
|
def score(self, actual: Any, expected: Any) -> float:
|
|
52
|
+
if both_null(actual, expected):
|
|
53
|
+
return 1.0
|
|
48
54
|
if not isinstance(actual, (date, datetime)):
|
|
49
55
|
actual = _to_date(actual)
|
|
50
56
|
if not isinstance(expected, (date, datetime)):
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/exponential_numeric_score.py
RENAMED
|
@@ -4,6 +4,7 @@ import math
|
|
|
4
4
|
from typing import Any
|
|
5
5
|
|
|
6
6
|
from structured_eval.metrics.base import FieldMetric
|
|
7
|
+
from structured_eval.metrics.utils.null import both_null
|
|
7
8
|
from structured_eval.metrics.utils.number import parse_number
|
|
8
9
|
|
|
9
10
|
|
|
@@ -29,17 +30,21 @@ class ExponentialNumericScore(FieldMetric):
|
|
|
29
30
|
:class:`NumericCloseness`, so numeric strings are graded too. The metric
|
|
30
31
|
applies **only to numbers**: if either side isn't numeric (``None``, a
|
|
31
32
|
non-numeric string, or a ``bool`` — ``True`` is not ``1``) the score is
|
|
32
|
-
``0.0``.
|
|
33
|
+
``0.0``. Two ``None``s are the exception — they agree (``1.0``; see
|
|
34
|
+
``metrics.utils.null``).
|
|
33
35
|
"""
|
|
34
36
|
|
|
35
37
|
name = "exponential_numeric_score"
|
|
36
38
|
|
|
37
|
-
def __init__(self, scale: float = 1.0) -> None:
|
|
39
|
+
def __init__(self, scale: float = 1.0, name: str | None = None) -> None:
|
|
40
|
+
super().__init__(name=name)
|
|
38
41
|
if scale <= 0:
|
|
39
42
|
raise ValueError("scale must be greater than 0")
|
|
40
43
|
self.scale = scale
|
|
41
44
|
|
|
42
45
|
def score(self, actual: Any, expected: Any) -> float:
|
|
46
|
+
if both_null(actual, expected):
|
|
47
|
+
return 1.0
|
|
43
48
|
a = parse_number(actual)
|
|
44
49
|
e = parse_number(expected)
|
|
45
50
|
if a is None or e is None:
|
|
@@ -4,6 +4,7 @@ from enum import StrEnum
|
|
|
4
4
|
from typing import Any
|
|
5
5
|
|
|
6
6
|
from structured_eval.metrics.base import FieldMetric
|
|
7
|
+
from structured_eval.metrics.utils.null import both_null
|
|
7
8
|
|
|
8
9
|
|
|
9
10
|
class FuzzyMethod(StrEnum):
|
|
@@ -27,7 +28,8 @@ class Fuzzy(FieldMetric):
|
|
|
27
28
|
|
|
28
29
|
``normalize`` strips surrounding whitespace and lowercases before comparison.
|
|
29
30
|
String-only: if either side is not a ``str`` the score is 0.0 (no coercion),
|
|
30
|
-
consistent with the other text metrics
|
|
31
|
+
consistent with the other text metrics — except two ``None``s, which agree
|
|
32
|
+
(1.0; see ``metrics.utils.null``).
|
|
31
33
|
"""
|
|
32
34
|
|
|
33
35
|
name = "fuzzy"
|
|
@@ -36,11 +38,15 @@ class Fuzzy(FieldMetric):
|
|
|
36
38
|
self,
|
|
37
39
|
method: FuzzyMethod = FuzzyMethod.TOKEN_SORT_RATIO,
|
|
38
40
|
normalize: bool = True,
|
|
41
|
+
name: str | None = None,
|
|
39
42
|
):
|
|
43
|
+
super().__init__(name=name)
|
|
40
44
|
self.method = FuzzyMethod(method)
|
|
41
45
|
self.normalize = normalize
|
|
42
46
|
|
|
43
47
|
def score(self, actual: Any, expected: Any) -> float:
|
|
48
|
+
if both_null(actual, expected):
|
|
49
|
+
return 1.0
|
|
44
50
|
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
45
51
|
return 0.0
|
|
46
52
|
try:
|
|
@@ -12,5 +12,10 @@ class Levenshtein(Fuzzy):
|
|
|
12
12
|
|
|
13
13
|
name = "levenshtein"
|
|
14
14
|
|
|
15
|
-
def __init__(
|
|
16
|
-
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
method: FuzzyMethod = FuzzyMethod.RATIO,
|
|
18
|
+
normalize: bool = True,
|
|
19
|
+
name: str | None = None,
|
|
20
|
+
):
|
|
21
|
+
super().__init__(method=method, normalize=normalize, name=name)
|
|
@@ -4,6 +4,7 @@ from enum import StrEnum
|
|
|
4
4
|
from typing import Any
|
|
5
5
|
|
|
6
6
|
from structured_eval.metrics.base import FieldMetric
|
|
7
|
+
from structured_eval.metrics.utils.null import both_null
|
|
7
8
|
from structured_eval.metrics.utils.number import parse_number
|
|
8
9
|
|
|
9
10
|
|
|
@@ -32,6 +33,9 @@ class Numeric(FieldMetric):
|
|
|
32
33
|
* ``relative_tolerance`` and/or ``absolute_tolerance`` — explicit bands; a
|
|
33
34
|
value matches if it falls within *either* band. When either is supplied it
|
|
34
35
|
takes precedence over ``tolerance``/``mode``.
|
|
36
|
+
|
|
37
|
+
Two ``None``s agree (1.0; see ``metrics.utils.null``); a one-sided ``None``
|
|
38
|
+
is 0.0.
|
|
35
39
|
"""
|
|
36
40
|
|
|
37
41
|
name = "numeric"
|
|
@@ -42,13 +46,17 @@ class Numeric(FieldMetric):
|
|
|
42
46
|
mode: NumericMode = NumericMode.RELATIVE,
|
|
43
47
|
relative_tolerance: float | None = None,
|
|
44
48
|
absolute_tolerance: float | None = None,
|
|
49
|
+
name: str | None = None,
|
|
45
50
|
):
|
|
51
|
+
super().__init__(name=name)
|
|
46
52
|
self.tolerance = tolerance
|
|
47
53
|
self.mode = NumericMode(mode)
|
|
48
54
|
self.relative_tolerance = relative_tolerance
|
|
49
55
|
self.absolute_tolerance = absolute_tolerance
|
|
50
56
|
|
|
51
57
|
def score(self, actual: Any, expected: Any) -> float:
|
|
58
|
+
if both_null(actual, expected):
|
|
59
|
+
return 1.0
|
|
52
60
|
a = parse_number(actual)
|
|
53
61
|
e = parse_number(expected)
|
|
54
62
|
if a is None or e is None:
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/numeric_closeness.py
RENAMED
|
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
from typing import Any
|
|
4
4
|
|
|
5
5
|
from structured_eval.metrics.base import FieldMetric
|
|
6
|
+
from structured_eval.metrics.utils.null import both_null
|
|
6
7
|
from structured_eval.metrics.utils.number import parse_number
|
|
7
8
|
|
|
8
9
|
|
|
@@ -19,12 +20,15 @@ class NumericCloseness(FieldMetric):
|
|
|
19
20
|
Values are parsed with the shared lenient numeric parser (same as
|
|
20
21
|
:class:`Numeric`), so numeric strings are graded too. The metric applies
|
|
21
22
|
**only to numbers**: if either side isn't numeric (``None``, a non-numeric
|
|
22
|
-
string, or a ``bool`` — ``True`` is not ``1``) the score is 0.0.
|
|
23
|
+
string, or a ``bool`` — ``True`` is not ``1``) the score is 0.0. Two
|
|
24
|
+
``None``s are the exception — they agree (1.0; see ``metrics.utils.null``).
|
|
23
25
|
"""
|
|
24
26
|
|
|
25
27
|
name = "numeric_closeness"
|
|
26
28
|
|
|
27
29
|
def score(self, actual: Any, expected: Any) -> float:
|
|
30
|
+
if both_null(actual, expected):
|
|
31
|
+
return 1.0
|
|
28
32
|
a = parse_number(actual)
|
|
29
33
|
e = parse_number(expected)
|
|
30
34
|
if a is None or e is None:
|
|
@@ -31,7 +31,9 @@ class ObjectAccuracy(ObjectMetric):
|
|
|
31
31
|
self,
|
|
32
32
|
score_policy: dict[str, Any] | None = None,
|
|
33
33
|
weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
|
|
34
|
+
name: str | None = None,
|
|
34
35
|
):
|
|
36
|
+
super().__init__(name=name)
|
|
35
37
|
self.score_policy = score_policy
|
|
36
38
|
self.weight_mode = stats.WeightMode(weight_mode)
|
|
37
39
|
|
|
@@ -26,7 +26,9 @@ class ObjectF1(ObjectMetric):
|
|
|
26
26
|
threshold: float | None = None,
|
|
27
27
|
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
28
28
|
weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
|
|
29
|
+
name: str | None = None,
|
|
29
30
|
):
|
|
31
|
+
super().__init__(name=name)
|
|
30
32
|
self.score_policy = score_policy
|
|
31
33
|
self.threshold = threshold
|
|
32
34
|
self.mode = stats.GradingMode(mode)
|
|
@@ -30,7 +30,9 @@ class ObjectPrecision(ObjectMetric):
|
|
|
30
30
|
threshold: float | None = None,
|
|
31
31
|
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
32
32
|
weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
|
|
33
|
+
name: str | None = None,
|
|
33
34
|
):
|
|
35
|
+
super().__init__(name=name)
|
|
34
36
|
self.score_policy = score_policy
|
|
35
37
|
self.threshold = threshold
|
|
36
38
|
self.mode = stats.GradingMode(mode)
|
|
@@ -26,7 +26,9 @@ class ObjectPRF1(ObjectMetric):
|
|
|
26
26
|
threshold: float | None = None,
|
|
27
27
|
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
28
28
|
weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
|
|
29
|
+
name: str | None = None,
|
|
29
30
|
):
|
|
31
|
+
super().__init__(name=name)
|
|
30
32
|
self.score_policy = score_policy
|
|
31
33
|
self.threshold = threshold
|
|
32
34
|
self.mode = stats.GradingMode(mode)
|
|
@@ -25,7 +25,9 @@ class ObjectRecall(ObjectMetric):
|
|
|
25
25
|
threshold: float | None = None,
|
|
26
26
|
mode: stats.GradingMode = stats.GradingMode.HARD,
|
|
27
27
|
weight_mode: stats.WeightMode = stats.WeightMode.PROPORTIONAL,
|
|
28
|
+
name: str | None = None,
|
|
28
29
|
):
|
|
30
|
+
super().__init__(name=name)
|
|
29
31
|
self.score_policy = score_policy
|
|
30
32
|
self.threshold = threshold
|
|
31
33
|
self.mode = stats.GradingMode(mode)
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_type_validity.py
RENAMED
|
@@ -23,7 +23,8 @@ class ObjectTypeValidity(ObjectMetric):
|
|
|
23
23
|
|
|
24
24
|
name = "object_type_validity"
|
|
25
25
|
|
|
26
|
-
def __init__(self) -> None:
|
|
26
|
+
def __init__(self, name: str | None = None) -> None:
|
|
27
|
+
super().__init__(name=name)
|
|
27
28
|
self._type_match = MetricInvoker(TypeMatch())
|
|
28
29
|
|
|
29
30
|
def compute(self, node: ObjectNode) -> float:
|
|
@@ -4,6 +4,7 @@ import re
|
|
|
4
4
|
from typing import Any
|
|
5
5
|
|
|
6
6
|
from structured_eval.metrics.base import FieldMetric
|
|
7
|
+
from structured_eval.metrics.utils.null import both_null
|
|
7
8
|
|
|
8
9
|
|
|
9
10
|
class RegexMatch(FieldMetric):
|
|
@@ -11,7 +12,8 @@ class RegexMatch(FieldMetric):
|
|
|
11
12
|
|
|
12
13
|
A **string-only** metric: if either side is not a ``str`` the score is
|
|
13
14
|
``0.0`` (use ``Numeric`` for numbers, ``ExactMatch`` for verbatim
|
|
14
|
-
equality)
|
|
15
|
+
equality) — except two ``None``s, which agree (``1.0``; see
|
|
16
|
+
``metrics.utils.null``). For two strings it applies, in order, optional ``lower`` and
|
|
15
17
|
``strip``, then substitutes every match of ``pattern`` with ``repl``, and
|
|
16
18
|
compares the results exactly.
|
|
17
19
|
|
|
@@ -31,7 +33,9 @@ class RegexMatch(FieldMetric):
|
|
|
31
33
|
repl: str = " ",
|
|
32
34
|
lower: bool = True,
|
|
33
35
|
strip: bool = True,
|
|
36
|
+
name: str | None = None,
|
|
34
37
|
):
|
|
38
|
+
super().__init__(name=name)
|
|
35
39
|
self.pattern = re.compile(pattern) if isinstance(pattern, str) else pattern
|
|
36
40
|
self.repl = repl
|
|
37
41
|
self.lower = lower
|
|
@@ -46,6 +50,8 @@ class RegexMatch(FieldMetric):
|
|
|
46
50
|
return value.strip() if self.strip else value
|
|
47
51
|
|
|
48
52
|
def score(self, actual: Any, expected: Any) -> float:
|
|
53
|
+
if both_null(actual, expected):
|
|
54
|
+
return 1.0
|
|
49
55
|
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
50
56
|
return 0.0
|
|
51
57
|
return 1.0 if self._normalize(actual) == self._normalize(expected) else 0.0
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/metric.py
RENAMED
|
@@ -21,7 +21,8 @@ class RulePassRate(RootMetric):
|
|
|
21
21
|
|
|
22
22
|
name = "rule_pass_rate"
|
|
23
23
|
|
|
24
|
-
def __init__(self, rules: list[Any]):
|
|
24
|
+
def __init__(self, rules: list[Any], name: str | None = None):
|
|
25
|
+
super().__init__(name=name)
|
|
25
26
|
self.rules = rules
|
|
26
27
|
self.processor = RuleProcessor()
|
|
27
28
|
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/metric.py
RENAMED
|
@@ -21,7 +21,10 @@ class SchemaValidity(RootMetric):
|
|
|
21
21
|
|
|
22
22
|
name = "schema_validity"
|
|
23
23
|
|
|
24
|
-
def __init__(
|
|
24
|
+
def __init__(
|
|
25
|
+
self, schema: type[BaseModel] | dict[str, Any], name: str | None = None
|
|
26
|
+
):
|
|
27
|
+
super().__init__(name=name)
|
|
25
28
|
self.validator = SchemaValidator(schema)
|
|
26
29
|
|
|
27
30
|
def compute(self, node: EvalNode) -> tuple[float, dict[str, Any]]:
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import string
|
|
5
|
+
from collections import Counter
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from structured_eval.metrics.base import FieldMetric
|
|
9
|
+
from structured_eval.metrics.utils.null import both_null
|
|
10
|
+
|
|
11
|
+
_IGNORE_PUNCTUATION_CHARS = frozenset(string.punctuation)
|
|
12
|
+
_IGNORE_ARTICLES_REGEX = re.compile(r"\b(a|an|the)\b", re.IGNORECASE)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class TokenF1(FieldMetric):
|
|
16
|
+
"""SQuAD-style token-overlap F1 — a default for free-text fields.
|
|
17
|
+
|
|
18
|
+
On its defaults this reproduces the ``f1_score`` of the official SQuAD v1.1
|
|
19
|
+
evaluation script: both sides go through the reference ``normalize_answer``
|
|
20
|
+
(lowercase, drop punctuation, drop the articles ``a``/``an``/``the``, collapse
|
|
21
|
+
whitespace), then tokens are matched as a **multiset** (``Counter``) — a
|
|
22
|
+
repeated token only helps as often as it appears on both sides, so
|
|
23
|
+
``"the the cat"`` vs ``"the cat"`` is 0.8, not 1.0. Precision and recall are
|
|
24
|
+
over the token *counts*; their harmonic mean is the score.
|
|
25
|
+
|
|
26
|
+
Each normalization step can be turned off independently::
|
|
27
|
+
|
|
28
|
+
TokenF1(ignore_case=False) # "AB" vs "ab" scores below 1.0
|
|
29
|
+
TokenF1(ignore_punctuation=False) # "fox." and "fox" are distinct tokens
|
|
30
|
+
TokenF1(ignore_articles=False) # "the" counts as a token like any other
|
|
31
|
+
|
|
32
|
+
Two deliberate departures from the reference script, both because this scores
|
|
33
|
+
fields rather than question answers: two empty strings score 1.0 (the script
|
|
34
|
+
returns 0.0, an empty answer being a failed answer), and a value that is not a
|
|
35
|
+
``str`` scores 0.0 with no coercion — except two ``None``s, which agree (1.0;
|
|
36
|
+
see ``metrics.utils.null``).
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
name = "token_f1"
|
|
40
|
+
|
|
41
|
+
def __init__(
|
|
42
|
+
self,
|
|
43
|
+
ignore_case: bool = True,
|
|
44
|
+
ignore_punctuation: bool = True,
|
|
45
|
+
ignore_articles: bool = True,
|
|
46
|
+
name: str | None = None,
|
|
47
|
+
):
|
|
48
|
+
super().__init__(name=name)
|
|
49
|
+
self.ignore_case = ignore_case
|
|
50
|
+
self.ignore_punctuation = ignore_punctuation
|
|
51
|
+
self.ignore_articles = ignore_articles
|
|
52
|
+
|
|
53
|
+
def _tokenize(self, value: str) -> list[str]:
|
|
54
|
+
if self.ignore_case:
|
|
55
|
+
value = value.lower()
|
|
56
|
+
if self.ignore_punctuation:
|
|
57
|
+
value = "".join(ch for ch in value if ch not in _IGNORE_PUNCTUATION_CHARS)
|
|
58
|
+
if self.ignore_articles:
|
|
59
|
+
value = _IGNORE_ARTICLES_REGEX.sub(" ", value)
|
|
60
|
+
return value.split()
|
|
61
|
+
|
|
62
|
+
def score(self, actual: Any, expected: Any) -> float:
|
|
63
|
+
if both_null(actual, expected):
|
|
64
|
+
return 1.0
|
|
65
|
+
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
66
|
+
return 0.0
|
|
67
|
+
|
|
68
|
+
a = self._tokenize(actual)
|
|
69
|
+
e = self._tokenize(expected)
|
|
70
|
+
|
|
71
|
+
if not a and not e:
|
|
72
|
+
return 1.0
|
|
73
|
+
if not a or not e:
|
|
74
|
+
return 0.0
|
|
75
|
+
|
|
76
|
+
same = sum((Counter(a) & Counter(e)).values())
|
|
77
|
+
if not same:
|
|
78
|
+
return 0.0
|
|
79
|
+
|
|
80
|
+
precision = same / len(a)
|
|
81
|
+
recall = same / len(e)
|
|
82
|
+
|
|
83
|
+
return 2 * precision * recall / (precision + recall)
|
|
@@ -4,6 +4,7 @@ from typing import Any
|
|
|
4
4
|
from urllib.parse import parse_qsl, unquote, urlsplit, urlunsplit
|
|
5
5
|
|
|
6
6
|
from structured_eval.metrics.base import FieldMetric
|
|
7
|
+
from structured_eval.metrics.utils.null import both_null
|
|
7
8
|
|
|
8
9
|
|
|
9
10
|
class UrlMatch(FieldMetric):
|
|
@@ -27,7 +28,9 @@ class UrlMatch(FieldMetric):
|
|
|
27
28
|
|
|
28
29
|
Both sides must be non-empty strings that parse to a URL with a scheme and a
|
|
29
30
|
host. Anything else — a non-string, an empty string, or a bare path with no
|
|
30
|
-
scheme/host — scores ``0.0``.
|
|
31
|
+
scheme/host — scores ``0.0``. Two ``None``s are the exception — no URL was
|
|
32
|
+
expected and none was given, so they agree (``1.0``; see
|
|
33
|
+
``metrics.utils.null``).
|
|
31
34
|
"""
|
|
32
35
|
|
|
33
36
|
name = "url_match"
|
|
@@ -38,7 +41,9 @@ class UrlMatch(FieldMetric):
|
|
|
38
41
|
ignore_query: bool = False,
|
|
39
42
|
ignore_fragment: bool = True,
|
|
40
43
|
ignore_www: bool = True,
|
|
44
|
+
name: str | None = None,
|
|
41
45
|
) -> None:
|
|
46
|
+
super().__init__(name=name)
|
|
42
47
|
self.ignore_query = ignore_query
|
|
43
48
|
self.ignore_fragment = ignore_fragment
|
|
44
49
|
self.ignore_www = ignore_www
|
|
@@ -75,6 +80,8 @@ class UrlMatch(FieldMetric):
|
|
|
75
80
|
return (urlunsplit((scheme, netloc, path, query, fragment)),)
|
|
76
81
|
|
|
77
82
|
def score(self, actual: Any, expected: Any) -> float:
|
|
83
|
+
if both_null(actual, expected):
|
|
84
|
+
return 1.0
|
|
78
85
|
norm_actual = self._normalize(actual)
|
|
79
86
|
norm_expected = self._normalize(expected)
|
|
80
87
|
if norm_actual is None or norm_expected is None:
|
|
@@ -7,4 +7,6 @@ Clearly-scoped modules:
|
|
|
7
7
|
* ``object_utils`` — turning an object's matched fields into the
|
|
8
8
|
``(score, threshold)`` pairs that ``calculate.prf_counts`` consumes.
|
|
9
9
|
* ``array`` — the same for an array's aligned items, plus missing/spurious counts.
|
|
10
|
+
* ``number`` — the lenient numeric parsing shared by the numeric field metrics.
|
|
11
|
+
* ``null`` — the ``(None, None) → 1.0`` rule shared by the comparison field metrics.
|
|
10
12
|
"""
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""The ``(None, None) → 1.0`` rule shared by the comparison field metrics.
|
|
2
|
+
|
|
3
|
+
The schemas under evaluation ask for ``null`` whenever a value is absent, so a
|
|
4
|
+
null expectation met by a null answer is a *correct* answer. Without this gate a
|
|
5
|
+
metric's type check (str / number / date) would reject the pair and score that
|
|
6
|
+
right answer ``0.0``. Only ``None`` counts as null — an empty string, an empty
|
|
7
|
+
list or a missing key are values, and are graded as such.
|
|
8
|
+
|
|
9
|
+
A one-sided ``None`` stays a mismatch: a value was expected and nothing came
|
|
10
|
+
back, or nothing was expected and a value was invented.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def both_null(actual: Any, expected: Any) -> bool:
|
|
19
|
+
"""True when neither side has a value — the two agree."""
|
|
20
|
+
return actual is None and expected is None
|
|
@@ -77,6 +77,7 @@ structured_eval/metrics/schema_validity/validator.py
|
|
|
77
77
|
structured_eval/metrics/utils/__init__.py
|
|
78
78
|
structured_eval/metrics/utils/array.py
|
|
79
79
|
structured_eval/metrics/utils/calculate.py
|
|
80
|
+
structured_eval/metrics/utils/null.py
|
|
80
81
|
structured_eval/metrics/utils/number.py
|
|
81
82
|
structured_eval/metrics/utils/object_utils.py
|
|
82
83
|
structured_eval/models/__init__.py
|
|
@@ -1,50 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import re
|
|
4
|
-
from collections import Counter
|
|
5
|
-
from typing import Any
|
|
6
|
-
|
|
7
|
-
from structured_eval.metrics.base import FieldMetric
|
|
8
|
-
|
|
9
|
-
_NON_WORD = re.compile(r"[^\w\s]")
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
def _characters(value: Any) -> list[str]:
|
|
13
|
-
"""Lowercase, drop punctuation and whitespace, split into characters."""
|
|
14
|
-
normalized = _NON_WORD.sub("", str(value).lower())
|
|
15
|
-
normalized = "".join(normalized.split()) # remove all whitespace
|
|
16
|
-
return list(normalized)
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
class CharacterF1(FieldMetric):
|
|
20
|
-
"""Character-overlap F1 for short free-text fields.
|
|
21
|
-
|
|
22
|
-
Characters are matched as a **multiset** (``Counter``), so repeated
|
|
23
|
-
characters contribute only as many times as they appear on both sides.
|
|
24
|
-
Precision and recall are computed over character counts, and their
|
|
25
|
-
harmonic mean is returned. String-only: if either side is not a ``str``
|
|
26
|
-
the score is ``0.0`` (no coercion).
|
|
27
|
-
"""
|
|
28
|
-
|
|
29
|
-
name = "character_f1"
|
|
30
|
-
|
|
31
|
-
def score(self, actual: Any, expected: Any) -> float:
|
|
32
|
-
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
33
|
-
return 0.0
|
|
34
|
-
|
|
35
|
-
a = _characters(actual)
|
|
36
|
-
e = _characters(expected)
|
|
37
|
-
|
|
38
|
-
if not a and not e:
|
|
39
|
-
return 1.0
|
|
40
|
-
if not a or not e:
|
|
41
|
-
return 0.0
|
|
42
|
-
|
|
43
|
-
same = sum((Counter(a) & Counter(e)).values())
|
|
44
|
-
if not same:
|
|
45
|
-
return 0.0
|
|
46
|
-
|
|
47
|
-
precision = same / len(a)
|
|
48
|
-
recall = same / len(e)
|
|
49
|
-
|
|
50
|
-
return 2 * precision * recall / (precision + recall)
|
|
@@ -1,44 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import re
|
|
4
|
-
from collections import Counter
|
|
5
|
-
from typing import Any
|
|
6
|
-
|
|
7
|
-
from structured_eval.metrics.base import FieldMetric
|
|
8
|
-
|
|
9
|
-
_NON_WORD = re.compile(r"[^\w\s]")
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
def _tokenize(value: Any) -> list[str]:
|
|
13
|
-
"""Lowercase, drop punctuation, split on whitespace."""
|
|
14
|
-
return _NON_WORD.sub(" ", str(value).lower()).split()
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
class TokenF1(FieldMetric):
|
|
18
|
-
"""SQuAD-style token-overlap F1 — a default for free-text fields.
|
|
19
|
-
|
|
20
|
-
Tokens are matched as a **multiset** (``Counter``), counting shared tokens
|
|
21
|
-
with multiplicity exactly like the official SQuAD F1 — so a repeated token
|
|
22
|
-
only helps as often as it appears on both sides (``"the the cat"`` vs
|
|
23
|
-
``"the cat"`` is 0.8, not 1.0). Precision and recall are over the token
|
|
24
|
-
*counts*; their harmonic mean is the score. String-only: if either side is
|
|
25
|
-
not a ``str`` the score is 0.0 (no coercion).
|
|
26
|
-
"""
|
|
27
|
-
|
|
28
|
-
name = "token_f1"
|
|
29
|
-
|
|
30
|
-
def score(self, actual: Any, expected: Any) -> float:
|
|
31
|
-
if not (isinstance(actual, str) and isinstance(expected, str)):
|
|
32
|
-
return 0.0
|
|
33
|
-
a = _tokenize(actual)
|
|
34
|
-
e = _tokenize(expected)
|
|
35
|
-
if not a and not e:
|
|
36
|
-
return 1.0
|
|
37
|
-
if not a or not e:
|
|
38
|
-
return 0.0
|
|
39
|
-
same = sum((Counter(a) & Counter(e)).values())
|
|
40
|
-
if not same:
|
|
41
|
-
return 0.0
|
|
42
|
-
precision = same / len(a)
|
|
43
|
-
recall = same / len(e)
|
|
44
|
-
return 2 * precision * recall / (precision + recall)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_cardinality.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_exact_match.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/array_jaccard_similarity.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/coverage_leaf_score.py
RENAMED
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/field_faithfulness.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/object_exact_match.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/overall_leaf_score.py
RENAMED
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/__init__.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/dsl.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/rule_pass_rate/engine.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/__init__.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/schema_validity/validator.py
RENAMED
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/structural_similarity.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval/metrics/utils/object_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structured_eval-0.1.1 → structured_eval-0.2.0}/structured_eval.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|