turkish-rag-eval 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- turkish_rag_eval-0.2.0/LICENSE +21 -0
- turkish_rag_eval-0.2.0/NOTICE.md +30 -0
- turkish_rag_eval-0.2.0/PKG-INFO +487 -0
- turkish_rag_eval-0.2.0/README.md +448 -0
- turkish_rag_eval-0.2.0/pyproject.toml +58 -0
- turkish_rag_eval-0.2.0/setup.cfg +4 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/__init__.py +14 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/abstain.py +122 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/agreement.py +233 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/bootstrap.py +330 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/charts.py +251 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/chunking.py +133 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/cli.py +159 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/corpus.py +216 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/export_hf.py +119 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/fetch_corpus.py +204 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/gold.py +60 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/groundedness.py +538 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/leaderboard.py +257 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/metrics.py +108 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/models.py +32 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/paths.py +40 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/report.py +275 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/retrieval.py +84 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/run_abstain.py +69 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/run_eval.py +270 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/turkish_text.py +54 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval/validate_gold.py +174 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/PKG-INFO +487 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/SOURCES.txt +52 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/dependency_links.txt +1 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/entry_points.txt +2 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/requires.txt +22 -0
- turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/top_level.txt +1 -0
- turkish_rag_eval-0.2.0/tests/test_agreement.py +153 -0
- turkish_rag_eval-0.2.0/tests/test_blog.py +29 -0
- turkish_rag_eval-0.2.0/tests/test_bootstrap.py +221 -0
- turkish_rag_eval-0.2.0/tests/test_charts.py +98 -0
- turkish_rag_eval-0.2.0/tests/test_cli.py +117 -0
- turkish_rag_eval-0.2.0/tests/test_corpus.py +127 -0
- turkish_rag_eval-0.2.0/tests/test_export_hf.py +82 -0
- turkish_rag_eval-0.2.0/tests/test_fetch_corpus.py +103 -0
- turkish_rag_eval-0.2.0/tests/test_groundedness.py +327 -0
- turkish_rag_eval-0.2.0/tests/test_leaderboard.py +196 -0
- turkish_rag_eval-0.2.0/tests/test_metrics.py +65 -0
- turkish_rag_eval-0.2.0/tests/test_models.py +51 -0
- turkish_rag_eval-0.2.0/tests/test_readme_agreement.py +44 -0
- turkish_rag_eval-0.2.0/tests/test_readme_full.py +54 -0
- turkish_rag_eval-0.2.0/tests/test_readme_groundedness.py +59 -0
- turkish_rag_eval-0.2.0/tests/test_readme_leaderboard.py +66 -0
- turkish_rag_eval-0.2.0/tests/test_report.py +139 -0
- turkish_rag_eval-0.2.0/tests/test_run_eval.py +160 -0
- turkish_rag_eval-0.2.0/tests/test_validate_gold.py +151 -0
- turkish_rag_eval-0.2.0/tests/test_version.py +30 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Rizgar Ozan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Data attribution
|
|
2
|
+
|
|
3
|
+
The retrieval corpus in `data/raw/corpus.json` is built from Turkish Wikipedia
|
|
4
|
+
articles fetched through the MediaWiki API by `turkish-rag-eval fetch-corpus`.
|
|
5
|
+
|
|
6
|
+
Wikipedia text is licensed **CC BY-SA 4.0**. Each document keeps its `title`
|
|
7
|
+
and permanent `url`, so every retrieved chunk can be traced to its source
|
|
8
|
+
article. The corpus file is not committed (see `.gitignore`); run the fetcher
|
|
9
|
+
to rebuild it.
|
|
10
|
+
|
|
11
|
+
## The gold set
|
|
12
|
+
|
|
13
|
+
`data/eval/gold.json` **is** committed, and each item carries a short verbatim
|
|
14
|
+
`answer_span` quoted from the Turkish Wikipedia article named by its `doc_id`.
|
|
15
|
+
Those spans are therefore also **CC BY-SA 4.0**, attributed to the article in the
|
|
16
|
+
same record (`doc_id` = MediaWiki pageid, plus the title and URL in the fetched
|
|
17
|
+
corpus). The questions themselves were written for this project - deliberately
|
|
18
|
+
paraphrased rather than copied - and are released under CC BY-SA 4.0 as well, so
|
|
19
|
+
the whole gold set can be reused under one licence. Single annotator (the repo
|
|
20
|
+
author), one pass, no inter-annotator agreement figure.
|
|
21
|
+
|
|
22
|
+
The **code** in this repository is MIT (see `LICENSE`). The licences do not
|
|
23
|
+
conflict: MIT covers the harness, CC BY-SA 4.0 covers the Wikipedia-derived text
|
|
24
|
+
in `data/`.
|
|
25
|
+
|
|
26
|
+
## No health data
|
|
27
|
+
|
|
28
|
+
No patient data, clinical records or any personal health information is used
|
|
29
|
+
anywhere in this project. The corpus is public encyclopaedic text only, and
|
|
30
|
+
nothing here is a medical device or clinical decision support tool.
|
|
@@ -0,0 +1,487 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: turkish-rag-eval
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Measure which parts of a Turkish RAG pipeline pay off: chunking, stemming, embedding model, fusion.
|
|
5
|
+
Author: Rızgar Ozan
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/RizgarOzan/turkish-rag-eval
|
|
8
|
+
Project-URL: Dataset, https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval
|
|
9
|
+
Project-URL: Issues, https://github.com/RizgarOzan/turkish-rag-eval/issues
|
|
10
|
+
Keywords: turkish,retrieval,rag,bm25,benchmark,evaluation
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Natural Language :: Turkish
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
16
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
License-File: NOTICE.md
|
|
21
|
+
Requires-Dist: numpy>=1.24
|
|
22
|
+
Requires-Dist: rank-bm25>=0.2.2
|
|
23
|
+
Requires-Dist: requests>=2.31
|
|
24
|
+
Provides-Extra: dense
|
|
25
|
+
Requires-Dist: sentence-transformers>=3.0; extra == "dense"
|
|
26
|
+
Requires-Dist: torch>=2.0; extra == "dense"
|
|
27
|
+
Provides-Extra: charts
|
|
28
|
+
Requires-Dist: matplotlib>=3.7; extra == "charts"
|
|
29
|
+
Provides-Extra: llm
|
|
30
|
+
Requires-Dist: anthropic>=0.40; extra == "llm"
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=7.4; extra == "dev"
|
|
33
|
+
Provides-Extra: all
|
|
34
|
+
Requires-Dist: turkish-rag-eval[dense]; extra == "all"
|
|
35
|
+
Requires-Dist: turkish-rag-eval[charts]; extra == "all"
|
|
36
|
+
Requires-Dist: turkish-rag-eval[llm]; extra == "all"
|
|
37
|
+
Requires-Dist: turkish-rag-eval[dev]; extra == "all"
|
|
38
|
+
Dynamic: license-file
|
|
39
|
+
|
|
40
|
+
# turkish-rag-eval
|
|
41
|
+
|
|
42
|
+
[](https://github.com/RizgarOzan/turkish-rag-eval/actions/workflows/validate.yml)
|
|
43
|
+
[](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md)
|
|
44
|
+
[](https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval)
|
|
45
|
+
|
|
46
|
+
**Which parts of a RAG pipeline actually earn their cost on an agglutinative
|
|
47
|
+
language?** Three chunking strategies × four retrievers × six embedding
|
|
48
|
+
models, measured on a hand-labelled Turkish gold set, with confidence
|
|
49
|
+
intervals and a cost column — then pointed at your own documents.
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install git+https://github.com/RizgarOzan/turkish-rag-eval # not on PyPI yet
|
|
53
|
+
turkish-rag-eval run --corpus ./belgelerim --gold ./sorular.json
|
|
54
|
+
turkish-rag-eval report
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Four findings
|
|
58
|
+
|
|
59
|
+
**1. The embedding model matters more than any pipeline choice.** Every model
|
|
60
|
+
trained for retrieval beats stemmed BM25 (0.494) on its own. The best,
|
|
61
|
+
[Mursit-Large-TR-Retrieval](https://huggingface.co/newmindai/Mursit-Large-TR-Retrieval),
|
|
62
|
+
reaches **0.781** nDCG@10 on hierarchical chunks. A small E5 the same size as
|
|
63
|
+
the default goes from 0.501 to 0.642 with nothing else changed, at about the
|
|
64
|
+
same speed (22 ms against 23 ms). The popular default model is the weak link, not dense retrieval.
|
|
65
|
+
See the [Leaderboard](#leaderboard).
|
|
66
|
+
|
|
67
|
+
**2. Turkish stemming is the cheapest real win.** Truncating tokens to a
|
|
68
|
+
5-character prefix before BM25 lifts nDCG@10 by 23–29% on every chunking
|
|
69
|
+
strategy, and a paired bootstrap puts every one of those gains clear of zero
|
|
70
|
+
(`hierarchical +0.111, 95% CI [+0.026, +0.204]`). "diyabet", "diyabetin",
|
|
71
|
+
"diyabete", "diyabetli" are four surface forms of one concept; an unstemmed
|
|
72
|
+
index almost never matches the query's form.
|
|
73
|
+
|
|
74
|
+
**3. The top of the table is a tie, and the tie decides on cost.** With 58
|
|
75
|
+
queries, a paired bootstrap cannot separate the best configuration from the
|
|
76
|
+
other two hybrid ones; the remaining nine are measurably worse. So the
|
|
77
|
+
real choice at the top is price: `sentence + hybrid_rrf` gives the same
|
|
78
|
+
quality at 27 ms instead of 37 ms.
|
|
79
|
+
|
|
80
|
+
<picture>
|
|
81
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/ndcg-intervals-dark.png">
|
|
82
|
+
<img alt="nDCG@10 per configuration with 95% bootstrap intervals; the best cannot be told apart from the other two hybrid configurations, the remaining nine are measurably worse" src="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/ndcg-intervals.png">
|
|
83
|
+
</picture>
|
|
84
|
+
|
|
85
|
+
**4. The intuitive confidence signal is the useless one.** For deciding when
|
|
86
|
+
*not* to answer, the obvious measure — how far ahead the top hit is — is worse
|
|
87
|
+
than answering everything (0.25 selective accuracy against a 0.47 baseline).
|
|
88
|
+
RRF fuses ranks as `1/(60+rank)`, so the top-two gap is ~2% on every query,
|
|
89
|
+
confident or not. The dense retriever's raw cosine works; the margin does not.
|
|
90
|
+
|
|
91
|
+
<picture>
|
|
92
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/abstention-dark.png">
|
|
93
|
+
<img alt="Coverage against selective accuracy for two confidence signals; the top-1 margin performs worse than answering everything" src="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/abstention.png">
|
|
94
|
+
</picture>
|
|
95
|
+
|
|
96
|
+
Longer write-up of the first result:
|
|
97
|
+
[English](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/blog/2026-09-19-bm25-turkish-en.md) ·
|
|
98
|
+
[Türkçe](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/blog/2026-09-19-bm25-turkish-tr.md).
|
|
99
|
+
|
|
100
|
+
**Contents:** [Results](#results) · [Leaderboard](#leaderboard) ·
|
|
101
|
+
[Why not an existing benchmark?](#why-not-an-existing-benchmark) ·
|
|
102
|
+
[Your own corpus](#your-own-corpus) · [Running it](#running-it) ·
|
|
103
|
+
[Gold set](#gold-set) · [Status](#status) · [Limits](#limits) · [Contribute](#contribute) ·
|
|
104
|
+
[Design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md)
|
|
105
|
+
|
|
106
|
+
## Results
|
|
107
|
+
|
|
108
|
+
58 queries, default embedding model
|
|
109
|
+
`paraphrase-multilingual-MiniLM-L12-v2`, CPU only. Intervals are 95% bootstrap
|
|
110
|
+
over queries; `turkish-rag-eval report` regenerates this table.
|
|
111
|
+
|
|
112
|
+
| Chunking | Retriever | nDCG@10 | 95% CI | R@5 | MRR | P95 |
|
|
113
|
+
|---|---|---|---|---|---|---|
|
|
114
|
+
| **hierarchical** | **hybrid_rrf** | **0.613** | [0.508, 0.716] | 0.690 | 0.559 | 37 ms |
|
|
115
|
+
| sentence | hybrid_rrf | 0.558 | [0.453, 0.662] | 0.621 | 0.500 | 27 ms |
|
|
116
|
+
| fixed | hybrid_rrf | 0.517 | [0.404, 0.629] | 0.578 | 0.484 | 31 ms |
|
|
117
|
+
| sentence | bm25_stem5 | 0.510 | [0.403, 0.621] | 0.638 | 0.462 | 4 ms |
|
|
118
|
+
| hierarchical | dense | 0.501 | [0.396, 0.606] | 0.569 | 0.441 | 30 ms |
|
|
119
|
+
| hierarchical | bm25_stem5 | 0.494 | [0.386, 0.605] | 0.569 | 0.444 | 8 ms |
|
|
120
|
+
| fixed | bm25_stem5 | 0.476 | [0.371, 0.585] | 0.526 | 0.421 | 4 ms |
|
|
121
|
+
| sentence | dense | 0.461 | [0.356, 0.567] | 0.552 | 0.410 | 21 ms |
|
|
122
|
+
| fixed | dense | 0.446 | [0.341, 0.555] | 0.491 | 0.402 | 25 ms |
|
|
123
|
+
| sentence | bm25_nostem | 0.411 | [0.308, 0.515] | 0.483 | 0.356 | 4 ms |
|
|
124
|
+
| fixed | bm25_nostem | 0.387 | [0.290, 0.487] | 0.414 | 0.319 | 3 ms |
|
|
125
|
+
| hierarchical | bm25_nostem | 0.383 | [0.285, 0.482] | 0.500 | 0.319 | 6 ms |
|
|
126
|
+
|
|
127
|
+
`turkish-rag-eval report` does not stop at the table — it names the cheapest
|
|
128
|
+
configuration the data cannot separate from the best:
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
Recommended: sentence + hybrid_rrf
|
|
132
|
+
0.558 ndcg@10 against 0.613 for hierarchical + hybrid_rrf, a gap of 0.056
|
|
133
|
+
that a paired bootstrap over 58 shared queries cannot distinguish from zero.
|
|
134
|
+
It answers in 27 ms at P95 against 37 ms.
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Comparisons are **paired**: both systems answered the same queries, so
|
|
138
|
+
resampling per-query differences removes the variance of some queries being
|
|
139
|
+
harder than others. It matters, and this data shows it in the sharpest way.
|
|
140
|
+
`hierarchical + hybrid_rrf` beats `fixed + hybrid_rrf` by 0.097 and beats
|
|
141
|
+
`sentence + bm25_stem5` by 0.104 — yet only the **larger** gap clears zero,
|
|
142
|
+
because its per-query differences are so much steadier (sd 0.34 against 0.40).
|
|
143
|
+
Ranking by the gap alone gets this backwards.
|
|
144
|
+
|
|
145
|
+
Hierarchical chunking only helps the dense retriever, and fixed-size chunking
|
|
146
|
+
— the most common default — lost on every retriever. More in the
|
|
147
|
+
[design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#what-else-the-numbers-say).
|
|
148
|
+
|
|
149
|
+
### Generation: groundedness
|
|
150
|
+
|
|
151
|
+
The R above feeds a G: the 58 questions answered from the top 5
|
|
152
|
+
`hierarchical + bm25_stem5` passages by `groq:openai/gpt-oss-120b`, judged by
|
|
153
|
+
a model from another family, `nvidia:nvidia/nemotron-3-super-120b-a12b`, at
|
|
154
|
+
temperature 0 on `2026-09-30`. The gold span was among the passages for
|
|
155
|
+
33 of 58 questions, and the two groups are scored apart:
|
|
156
|
+
|
|
157
|
+
| Span retrieved (33) | |
|
|
158
|
+
|---|---|
|
|
159
|
+
| answered when the span was retrieved | 93.9% |
|
|
160
|
+
| answer rests on the passages | 96.8% |
|
|
161
|
+
| answer conveys the gold span | 96.8% |
|
|
162
|
+
|
|
163
|
+
| Span not retrieved (25) | |
|
|
164
|
+
|---|---|
|
|
165
|
+
| declined when the span was not retrieved | 80.0% |
|
|
166
|
+
| answered anyway | 20.0% |
|
|
167
|
+
| of those, answer is right | 60.0% |
|
|
168
|
+
| hallucinated (not in the passages) | 4.0% |
|
|
169
|
+
|
|
170
|
+
The generator mostly declines when the evidence is missing, and when it does
|
|
171
|
+
answer, the answer is usually stated in some other retrieved passage — one of
|
|
172
|
+
25 is not. The first scoring rule called every answer without the span a
|
|
173
|
+
hallucination and reported 20%; reading the five by hand showed four stated in
|
|
174
|
+
a retrieved passage, so answers without the span now go to the judge too
|
|
175
|
+
([why](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#groundedness)).
|
|
176
|
+
16 verdicts were checked by hand; 14 agreed. Of the other two, the judge was
|
|
177
|
+
too strict once (an answer inverted "not recommended unless below 7 g/dl") and
|
|
178
|
+
too lenient once (it accepted "hyperuricaemia" as the cause of gout).
|
|
179
|
+
|
|
180
|
+
This row depends on hosted APIs and cannot be reproduced offline; the models
|
|
181
|
+
may change behind the same name. Only the rates are committed
|
|
182
|
+
([`results/groundedness.json`](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/results/groundedness.json)) —
|
|
183
|
+
the free API terms do not allow redistributing raw model output.
|
|
184
|
+
|
|
185
|
+
To rerun it with other models, name them `groq:`, `nvidia:` or `openai:`
|
|
186
|
+
(a bare name goes to Anthropic). Keys can sit in a `.env` file (gitignored) in
|
|
187
|
+
the directory you run from; `openai:` follows `OPENAI_BASE_URL`, so any OpenAI-compatible
|
|
188
|
+
endpoint, a local router included, works.
|
|
189
|
+
|
|
190
|
+
## Leaderboard
|
|
191
|
+
|
|
192
|
+
Dense `nDCG@10` per chunking strategy, the hybrid (dense + stemmed BM25, RRF)
|
|
193
|
+
on hierarchical chunks, and the cost of each model on one 16-thread CPU,
|
|
194
|
+
measured 2026-09-18. Stemmed BM25 alone scores 0.476 / 0.510 / 0.494.
|
|
195
|
+
`turkish-rag-eval leaderboard` rebuilds this from `results/models/`.
|
|
196
|
+
|
|
197
|
+
| Model | Params | Turkish-only | Dense fixed | Dense sentence | Dense hierarchical | Hybrid hierarchical | Embed corpus | Query P95 |
|
|
198
|
+
|---|---|---|---|---|---|---|---|---|
|
|
199
|
+
| paraphrase-multilingual-MiniLM-L12-v2 (default) | 118 M | no | 0.446 | 0.461 | 0.501 | 0.607 | 2 min | 23 ms |
|
|
200
|
+
| [emrecan/bert-base-turkish-cased-mean-nli-stsb-tr](https://huggingface.co/emrecan/bert-base-turkish-cased-mean-nli-stsb-tr) | 111 M | yes | 0.408 | 0.431 | 0.497 | 0.654 | 4 min | 43 ms |
|
|
201
|
+
| [intfloat/multilingual-e5-small](https://huggingface.co/intfloat/multilingual-e5-small) | 118 M | no | 0.654 | 0.644 | 0.642 | 0.639 | 3 min | 22 ms |
|
|
202
|
+
| [intfloat/multilingual-e5-base](https://huggingface.co/intfloat/multilingual-e5-base) | 278 M | no | 0.631 | 0.677 | 0.668 | 0.648 | 10 min | 49 ms |
|
|
203
|
+
| [BAAI/bge-m3](https://huggingface.co/BAAI/bge-m3) | 568 M | no | 0.766 | 0.767 | — | — | > 45 min | — |
|
|
204
|
+
| **[newmindai/Mursit-Large-TR-Retrieval](https://huggingface.co/newmindai/Mursit-Large-TR-Retrieval)** | 404 M | yes | **0.746** | **0.740** | **0.781** | 0.673 | 35 min | 214 ms |
|
|
205
|
+
|
|
206
|
+
The two Turkish-only models are the most-downloaded Turkish entries in the
|
|
207
|
+
Hugging Face `sentence-similarity` category. `bge-m3` was stopped after 45
|
|
208
|
+
minutes, before the hierarchical chunks; its two numbers come from that
|
|
209
|
+
partial run. `google/embeddinggemma-300m` is gated behind a licence click and
|
|
210
|
+
was not run. Per-query results for every completed model are in
|
|
211
|
+
`results/models/`.
|
|
212
|
+
|
|
213
|
+
> These five rows were measured together on one machine before v0.1.0, which
|
|
214
|
+
> is why the timing columns are comparable with each other and not with the
|
|
215
|
+
> main table above. Their `dense` columns are unaffected by the stable
|
|
216
|
+
> tie-break, but each `Hybrid hierarchical` figure will move by roughly +0.006
|
|
217
|
+
> when the model is re-run — the default row's went 0.607 → 0.613.
|
|
218
|
+
> `turkish-rag-eval leaderboard --check` reports them as missing provenance
|
|
219
|
+
> until then.
|
|
220
|
+
|
|
221
|
+
**The top of this table is settled; the middle is not.** Dense hierarchical,
|
|
222
|
+
with 95% bootstrap intervals over the 58 queries: MiniLM 0.501 [0.396, 0.606],
|
|
223
|
+
emrecan 0.497 [0.394, 0.601], e5-small 0.642 [0.542, 0.738], e5-base 0.668
|
|
224
|
+
[0.568, 0.765], Mursit 0.781 [0.701, 0.856]. The intervals overlap, but paired
|
|
225
|
+
over the same queries Mursit beats the runner-up e5-base by +0.113
|
|
226
|
+
[+0.029, +0.202]. The two E5 models cannot be told apart (+0.026
|
|
227
|
+
[-0.039, +0.094]), and neither can MiniLM and emrecan (+0.004
|
|
228
|
+
[-0.123, +0.135]).
|
|
229
|
+
|
|
230
|
+
**Hybrid fusion only pays for a weak dense model.** RRF lifts the small
|
|
231
|
+
default by +0.106 but pulls Mursit down from 0.781 to 0.673, a paired loss of
|
|
232
|
+
-0.108 [-0.193, -0.030]; for the two E5 models it makes no measurable
|
|
233
|
+
difference. "Turkish-only" is not enough either: the `emrecan` model was
|
|
234
|
+
trained for sentence similarity and truncates input at 75 tokens.
|
|
235
|
+
|
|
236
|
+
To submit a model, open a pull request adding `results/models/<org>__<name>/`
|
|
237
|
+
(`turkish-rag-eval run --model <org>/<name>`, then `leaderboard --check`).
|
|
238
|
+
CI re-derives every metric from the per-query relevance arrays committed
|
|
239
|
+
beside it, and checks the harness version and corpus fingerprint.
|
|
240
|
+
|
|
241
|
+
## Why not an existing benchmark?
|
|
242
|
+
|
|
243
|
+
MTEB-style retrieval benchmarks, TR-MTEB included, score an embedding model on
|
|
244
|
+
passages that are already split. They answer "which model?", not "which
|
|
245
|
+
chunker, is Turkish stemming worth it, does a hybrid help, and what does each
|
|
246
|
+
cost on a CPU?". This harness keeps the articles whole, lets every chunker cut
|
|
247
|
+
them its own way, and judges each chunk by the answer span, so pipeline
|
|
248
|
+
choices are compared on the same labels. For a model-only comparison the same
|
|
249
|
+
data exports to the BEIR layout MTEB reads (`turkish-rag-eval export-hf`),
|
|
250
|
+
published as
|
|
251
|
+
[RizgarOzan/turkish-rag-eval](https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval)
|
|
252
|
+
(the 58 human questions, plus all 296 under the `full-*` configs, each marked
|
|
253
|
+
`human` or `llm-draft`); adding it to MTEB is proposed in
|
|
254
|
+
[embeddings-benchmark/mteb#5536](https://github.com/embeddings-benchmark/mteb/issues/5536).
|
|
255
|
+
|
|
256
|
+
## Your own corpus
|
|
257
|
+
|
|
258
|
+
The harness is not tied to its own articles. Point it at a folder of `.txt` or
|
|
259
|
+
`.md` files, a BEIR directory, or a JSON corpus:
|
|
260
|
+
|
|
261
|
+
```bash
|
|
262
|
+
turkish-rag-eval run --corpus ./belgelerim --gold ./sorular.json \
|
|
263
|
+
--retrievers bm25_stem5 bm25_nostem
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
A BM25-only run needs no model download and no torch — enough to answer "which
|
|
267
|
+
chunker, and is stemming worth it on my documents" from the base install.
|
|
268
|
+
|
|
269
|
+
**No labelled questions?** That is the real wall, and the reason most
|
|
270
|
+
benchmarks only ever measure themselves. `bootstrap` drafts a starting point (the name means drafting a gold set here,
|
|
271
|
+
not the statistical resampling above):
|
|
272
|
+
|
|
273
|
+
```bash
|
|
274
|
+
export ANTHROPIC_API_KEY=...
|
|
275
|
+
turkish-rag-eval bootstrap ./belgelerim --out gold-draft.json --per-doc 3
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
It runs the same procedure the contributed questions in this repository used.
|
|
279
|
+
One pass writes questions and marks the answer span; a second pass sees only
|
|
280
|
+
the document and the questions and marks the span again. Where the passes
|
|
281
|
+
disagree the item is written as `"review": "needs-human"` and **never loads**
|
|
282
|
+
until a person settles it. Spans that are not verbatim, and questions copied
|
|
283
|
+
out of their own answer, are dropped with a reason.
|
|
284
|
+
|
|
285
|
+
What you get is a draft to review, not a gold set.
|
|
286
|
+
|
|
287
|
+
## Running it
|
|
288
|
+
|
|
289
|
+
Python 3.10+. CPU only — no GPU anywhere in this project.
|
|
290
|
+
|
|
291
|
+
```bash
|
|
292
|
+
pip install git+https://github.com/RizgarOzan/turkish-rag-eval # metrics, BM25, gold-set tooling
|
|
293
|
+
pip install 'turkish-rag-eval[all] @ git+https://github.com/RizgarOzan/turkish-rag-eval' # + dense retrieval, charts, LLM commands
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
| Command | What it does |
|
|
297
|
+
|---|---|
|
|
298
|
+
| `run` | every chunking × retriever combination; writes `results/` |
|
|
299
|
+
| `report` | intervals, paired comparisons, and a recommendation |
|
|
300
|
+
| `charts` | the two figures above, light and dark |
|
|
301
|
+
| `leaderboard` | rebuild the model table; `--check` verifies every entry |
|
|
302
|
+
| `abstain` | coverage / selective-accuracy curve for the best configuration ([details](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#abstention)) |
|
|
303
|
+
| `groundedness` | score the generation half against the gold spans ([details](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#groundedness); [results](#generation-groundedness)) |
|
|
304
|
+
| `bootstrap` | draft a gold set for your own corpus |
|
|
305
|
+
| `agreement` | inter-annotator agreement over the gold set |
|
|
306
|
+
| `fetch-corpus` | download the Wikipedia snapshot; `--verify` checks the lock; `--include-drafts` for [all 300 questions](#scoring-all-300) |
|
|
307
|
+
| `export-hf` | the BEIR layout MTEB reads |
|
|
308
|
+
| `validate` | every gold file's invariants |
|
|
309
|
+
|
|
310
|
+
From a checkout:
|
|
311
|
+
|
|
312
|
+
```bash
|
|
313
|
+
git clone https://github.com/RizgarOzan/turkish-rag-eval
|
|
314
|
+
cd turkish-rag-eval
|
|
315
|
+
pip install -e '.[all]'
|
|
316
|
+
turkish-rag-eval fetch-corpus # rebuilds data/raw/corpus.json
|
|
317
|
+
turkish-rag-eval run # results/summary.json + per-query files
|
|
318
|
+
python -m pytest -q
|
|
319
|
+
```
|
|
320
|
+
|
|
321
|
+
`run --model <name>` swaps the embedding model; any sentence-transformers
|
|
322
|
+
model works. Results for a non-default model go to
|
|
323
|
+
`results/models/<org>__<name>/`, so the main table is never overwritten.
|
|
324
|
+
|
|
325
|
+
## Gold set
|
|
326
|
+
|
|
327
|
+
| Files | Questions | Labelled by | In the results above |
|
|
328
|
+
|---|---|---|---|
|
|
329
|
+
| `data/eval/gold.json` (health) | 58 | one human | yes |
|
|
330
|
+
| `data/eval/contrib/llm-draft-*.json` (history, geography, astronomy, biology, computing) | 242 | two independent LLM passes | not yet |
|
|
331
|
+
|
|
332
|
+
The set reached its planned 300 questions on 2026-09-26. Questions a model
|
|
333
|
+
drafted are marked `"source": "llm-draft"`. A second model then picked its own
|
|
334
|
+
answer span for each one without seeing the first label
|
|
335
|
+
(`second_annotation`). Two spans agree when one contains the other or their
|
|
336
|
+
token F1 is at least 0.5; agreed items get `"review": "agreed"`, the rest get
|
|
337
|
+
`"needs-human"` and are never loaded. Batches 1 and 2 were 30 of 30 agreed,
|
|
338
|
+
batch 3 was 29 of 30: for "what are several ribosomes working on one mRNA
|
|
339
|
+
called?" the passes picked two different sentences that both name polysomes,
|
|
340
|
+
so that question waits for a person. Batch 4 was 27 of 30: the second pass
|
|
341
|
+
named the other claimant to the Hungarian throne, took the sentence beside the
|
|
342
|
+
lysozyme result instead of the result itself, and answered "cross compilers"
|
|
343
|
+
with the bare term where the first took its definition. Batch 5 was 30 of 30,
|
|
344
|
+
and every pair is a containment pair: 9 identical, 15 differing only by a
|
|
345
|
+
trailing full stop. One of its answers, Robert W. Holley's 1968 Nobel Prize,
|
|
346
|
+
was cut to its last clause on 2026-09-27 and labelled again by the second
|
|
347
|
+
pass: the sentence splitter breaks after "W.", so no hierarchical chunk held
|
|
348
|
+
the whole sentence ([#17](https://github.com/RizgarOzan/turkish-rag-eval/issues/17)). Batch 6 was 30 of 30 on reworded questions whose answers
|
|
349
|
+
often run to two sentences; six first-pass spans were cut to one sentence
|
|
350
|
+
before merging, because the two-sentence version fitted inside no chunk and
|
|
351
|
+
so could never be retrieved. Batch 7 was 30 of 30 again, with 25 identical
|
|
352
|
+
spans; one first-pass span was cut to its clause for the same reason. Like
|
|
353
|
+
batch 5 its questions mostly point at a single sentence, so its high
|
|
354
|
+
agreement says little about harder questions. Batch 8, the last 32, was
|
|
355
|
+
written to be harder: 17 first-pass answers ran to two or more sentences.
|
|
356
|
+
Seven of those crossed a sentence- or hierarchical-chunk boundary and were
|
|
357
|
+
cut, leaving 12 multi-sentence answers. It was still 32 of 32, 29 identical:
|
|
358
|
+
two LLMs reading the same article pick the same sentences even when the
|
|
359
|
+
answer is long, which is one more reason these drafts need a human pass.
|
|
360
|
+
|
|
361
|
+
Drafts stay out of every number above until a re-run says otherwise:
|
|
362
|
+
`load_gold()` skips them unless called with `include_drafts=True`. Synthetic
|
|
363
|
+
test questions are common practice as long as they are declared and their
|
|
364
|
+
agreement is measured. This section is that declaration.
|
|
365
|
+
|
|
366
|
+
The corpus is 54 Turkish Wikipedia articles, 1.09 M characters; 27 of them
|
|
367
|
+
answer at least one question and the other 27 are distractors from the same
|
|
368
|
+
domain.
|
|
369
|
+
|
|
370
|
+
| Batch | Questions | Identical | Mean IoU | Mean token F1 | Cohen's κ, fixed / sentence / hierarchical |
|
|
371
|
+
|---|---|---|---|---|---|
|
|
372
|
+
| 1 — 2026-09-18 (Malazgirt, Kapadokya, Mars, Mitokondri, Linux) | 30 | 16 | 0.816 | 0.878 | 1.00 / 1.00 / 1.00 |
|
|
373
|
+
| 2 — 2026-09-20 (İstanbul'un Fethi, Ağrı Dağı, Jüpiter, Fotosentez, İnternet)¹ | 30 | 4 | 0.427 | 0.549 | 0.94 / 0.99 / 0.99 |
|
|
374
|
+
| 3 — 2026-09-21 (Çaldıran Muharebesi, Tuz Gölü, Satürn, Ribozom, Unix) | 30 | 18 | 0.829 | 0.872 | 0.92 / 0.96 / 0.96 |
|
|
375
|
+
| 4 — 2026-09-24 (Mohaç Muharebesi, Kızılırmak, Venüs, Enzim, Derleyici) | 30 | 15 | 0.747 | 0.792 | 0.96 / 0.96 / 0.96 |
|
|
376
|
+
| 5 — 2026-09-25 (Kösedağ Muharebesi, Uludağ, Neptün, RNA, İşletim sistemi) | 30 | 24 | 0.922 | 0.943 | 1.00 / 1.00 / 1.00 |
|
|
377
|
+
| 6 — 2026-09-25 (Preveze Deniz Muharebesi, Erciyes, Uranüs, Hemoglobin, Veritabanı) | 30 | 20 | 0.869 | 0.901 | 0.93 / 0.90 / 0.87 |
|
|
378
|
+
| 7 — 2026-09-26 (Ankara Muharebesi, Van Gölü, Merkür, DNA, World Wide Web) | 30 | 25 | 0.944 | 0.960 | 0.93 / 0.93 / 0.97 |
|
|
379
|
+
| 8 — 2026-09-26 (Niğbolu Muharebesi, Fırat, Ay, Protein, Yapay zekâ) | 32 | 29 | 0.964 | 0.975 | 0.94 / 0.97 / 1.00 |
|
|
380
|
+
| **All drafts** | 242 | 151 | 0.816 | 0.860 | 0.95 / 0.97 / 0.97 |
|
|
381
|
+
|
|
382
|
+
¹ Re-measured 2026-09-26: the Turkish Wikipedia article *Jüpiter* was rewritten that day to correct errors, and five of its spans no longer matched the live text. Three changed only in wording (a comma, "30,003" → "30" seconds, "dört uydu" → "uydular") and were edited in both labels; two changed in substance — the 40,000 km mantle thickness is gone and the Great Red Spot went from "at least 400 years" to "recorded since 1831" — so those two questions were rewritten and labelled again by both passes. The row was 0.419 / 0.540 / 0.96 / 1.00 / 1.00 before.
|
|
383
|
+
|
|
384
|
+
The passes disagree about how much of a sentence to take, rather than where
|
|
385
|
+
the answer is. Two LLMs tend to pick the same sentence, so read this as a
|
|
386
|
+
sanity check rather than human agreement. Full discussion in the
|
|
387
|
+
[design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#agreement-between-the-two-passes).
|
|
388
|
+
|
|
389
|
+
### Scoring all 300
|
|
390
|
+
|
|
391
|
+
The drafted questions point at 40 articles outside the health snapshot, so
|
|
392
|
+
they get a corpus of their own: the 54 health articles plus those 40, 1.89 M
|
|
393
|
+
characters, pinned by `data/corpus-full.lock.json`. The published snapshot and
|
|
394
|
+
its lock stay as they are.
|
|
395
|
+
|
|
396
|
+
```bash
|
|
397
|
+
turkish-rag-eval fetch-corpus --include-drafts # data/raw/corpus-full.json
|
|
398
|
+
turkish-rag-eval run --include-drafts # results/full/<model>/
|
|
399
|
+
```
|
|
400
|
+
|
|
401
|
+
Measured 2026-09-27 (re-run after the batch 5 fix; Mursit added 2026-10-01) on 296 questions (the 4 `needs-human` drafts never load),
|
|
402
|
+
hierarchical chunks, nDCG@10:
|
|
403
|
+
|
|
404
|
+
| Retriever | 58 human questions | 296 questions (238 drafted) |
|
|
405
|
+
|---|---|---|
|
|
406
|
+
| bm25_stem5 | 0.494 | 0.558 |
|
|
407
|
+
| MiniLM (default), dense | 0.501 | 0.405 |
|
|
408
|
+
| MiniLM (default), hybrid_rrf | 0.613 | 0.578 |
|
|
409
|
+
| multilingual-e5-small, dense | 0.642 | 0.568 |
|
|
410
|
+
| multilingual-e5-small, hybrid_rrf | 0.639 | 0.642 |
|
|
411
|
+
| multilingual-e5-base, dense | 0.668 | 0.646 |
|
|
412
|
+
| multilingual-e5-base, hybrid_rrf | 0.648 | 0.664 |
|
|
413
|
+
| Mursit-Large-TR-Retrieval, dense | 0.781 | 0.613 |
|
|
414
|
+
| Mursit-Large-TR-Retrieval, hybrid_rrf | 0.673 | 0.661 |
|
|
415
|
+
|
|
416
|
+
On the wider set stemmed BM25 gets stronger and every dense model weaker, so
|
|
417
|
+
fusing the two now helps all four models, where on the 58 health questions it
|
|
418
|
+
cost the E5 models and Mursit. Mursit drops the most, from 0.781 to 0.613, and
|
|
419
|
+
falls below e5-base, so its lead on the human set does not carry over to the
|
|
420
|
+
drafted questions yet. Those questions were written by an LLM from one sentence
|
|
421
|
+
each; whether that style suits the smaller models better is open until people
|
|
422
|
+
have checked them. These rows are not in the tables above: most of the
|
|
423
|
+
questions have not been checked by a person yet. The run also names every
|
|
424
|
+
question no chunk can answer; a span split by a chunk boundary is the usual
|
|
425
|
+
reason. Here that is 17 questions with fixed-size chunks and none with
|
|
426
|
+
sentence or hierarchical chunks (the one hierarchical miss was fixed on
|
|
427
|
+
2026-09-27, see batch 5 above). Results for all three chunkers are in
|
|
428
|
+
[`results/full/`](https://github.com/RizgarOzan/turkish-rag-eval/tree/main/results/full).
|
|
429
|
+
|
|
430
|
+
## Status
|
|
431
|
+
|
|
432
|
+
v0.1.0, tagged but not on PyPI yet — install from GitHub as above.
|
|
433
|
+
|
|
434
|
+
- **Works:** the retrieval harness, [`groundedness`](#generation-groundedness) on the 58 human questions, `report`, `leaderboard --check`, `bootstrap`,
|
|
435
|
+
`agreement`, the Hugging Face export, scoring [all 300 questions](#scoring-all-300)
|
|
436
|
+
on a pinned corpus, and CI that validates every contributed question against
|
|
437
|
+
live Wikipedia.
|
|
438
|
+
- **In progress:** human review of the gold set. All 300 planned questions are
|
|
439
|
+
in (58 human, 242 LLM-drafted); the drafts join the results once people have
|
|
440
|
+
reviewed them.
|
|
441
|
+
- **Not yet:** EmbeddingGemma, and `groundedness` on the 242 drafted questions.
|
|
442
|
+
|
|
443
|
+
## Limits
|
|
444
|
+
|
|
445
|
+
- **58 queries is a small set.** The intervals above are the honest width of
|
|
446
|
+
that; most of the table's ordering is not resolved by this much data. They
|
|
447
|
+
replace the eyeballed rule of thumb this project used to carry — "treat
|
|
448
|
+
differences under roughly 0.05 nDCG as noise" — which was both too strict
|
|
449
|
+
for paired comparisons and too loose for unpaired ones.
|
|
450
|
+
- **One annotator for the human set.** The 58 health questions are
|
|
451
|
+
single-annotated, so they have no agreement figure. The 242 drafted questions
|
|
452
|
+
are double-labelled, but by two LLM passes rather than by two people.
|
|
453
|
+
- **The corpus is Wikipedia, and so is much of the training data.** Every
|
|
454
|
+
embedding model ranked here was almost certainly trained on Turkish
|
|
455
|
+
Wikipedia. Absolute scores are therefore optimistic; the *comparison*
|
|
456
|
+
between pipeline choices on the same corpus is what this measures. A
|
|
457
|
+
non-Wikipedia domain is the most valuable thing a contributor could add.
|
|
458
|
+
- **Six embedding models, one corpus.** `bge-m3` is a partial run and
|
|
459
|
+
EmbeddingGemma is missing. `abstain` uses only the default model.
|
|
460
|
+
- **Encyclopaedic text, not clinical text.** Nothing here transfers to a
|
|
461
|
+
clinical setting without re-measurement. No patient data is used anywhere,
|
|
462
|
+
and nothing here is a medical device.
|
|
463
|
+
- **The generation half is one generator, one judge, 58 questions.**
|
|
464
|
+
[Groundedness](#generation-groundedness) was measured once, through hosted
|
|
465
|
+
APIs, on the human set only; the judge was checked by hand on 16 verdicts.
|
|
466
|
+
- **The embedding model is pinned by name, not by revision.**
|
|
467
|
+
|
|
468
|
+
## Contribute
|
|
469
|
+
|
|
470
|
+
The first two limits shrink with every contributor. Adding questions needs no
|
|
471
|
+
ML background — pick a Turkish Wikipedia article, write 5–10 paraphrased
|
|
472
|
+
questions, and open a pull request with one JSON file. A validator checks each
|
|
473
|
+
file against Wikipedia in CI. See [CONTRIBUTING.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/CONTRIBUTING.md) (Türkçe
|
|
474
|
+
açıklama dahil) and the
|
|
475
|
+
[good first issues](https://github.com/RizgarOzan/turkish-rag-eval/issues?q=is%3Aopen+label%3A%22good+first+issue%22).
|
|
476
|
+
|
|
477
|
+
Submitting an embedding model is one command and a pull request — see
|
|
478
|
+
[Leaderboard](#leaderboard).
|
|
479
|
+
|
|
480
|
+
## Data
|
|
481
|
+
|
|
482
|
+
Turkish Wikipedia, CC BY-SA 4.0. See [NOTICE.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md).
|
|
483
|
+
|
|
484
|
+
## Licence
|
|
485
|
+
|
|
486
|
+
Code MIT ([LICENSE](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/LICENSE)); data under `data/` CC BY-SA 4.0
|
|
487
|
+
([NOTICE.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md)).
|