turkish-rag-eval 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. turkish_rag_eval-0.2.0/LICENSE +21 -0
  2. turkish_rag_eval-0.2.0/NOTICE.md +30 -0
  3. turkish_rag_eval-0.2.0/PKG-INFO +487 -0
  4. turkish_rag_eval-0.2.0/README.md +448 -0
  5. turkish_rag_eval-0.2.0/pyproject.toml +58 -0
  6. turkish_rag_eval-0.2.0/setup.cfg +4 -0
  7. turkish_rag_eval-0.2.0/src/turkish_rag_eval/__init__.py +14 -0
  8. turkish_rag_eval-0.2.0/src/turkish_rag_eval/abstain.py +122 -0
  9. turkish_rag_eval-0.2.0/src/turkish_rag_eval/agreement.py +233 -0
  10. turkish_rag_eval-0.2.0/src/turkish_rag_eval/bootstrap.py +330 -0
  11. turkish_rag_eval-0.2.0/src/turkish_rag_eval/charts.py +251 -0
  12. turkish_rag_eval-0.2.0/src/turkish_rag_eval/chunking.py +133 -0
  13. turkish_rag_eval-0.2.0/src/turkish_rag_eval/cli.py +159 -0
  14. turkish_rag_eval-0.2.0/src/turkish_rag_eval/corpus.py +216 -0
  15. turkish_rag_eval-0.2.0/src/turkish_rag_eval/export_hf.py +119 -0
  16. turkish_rag_eval-0.2.0/src/turkish_rag_eval/fetch_corpus.py +204 -0
  17. turkish_rag_eval-0.2.0/src/turkish_rag_eval/gold.py +60 -0
  18. turkish_rag_eval-0.2.0/src/turkish_rag_eval/groundedness.py +538 -0
  19. turkish_rag_eval-0.2.0/src/turkish_rag_eval/leaderboard.py +257 -0
  20. turkish_rag_eval-0.2.0/src/turkish_rag_eval/metrics.py +108 -0
  21. turkish_rag_eval-0.2.0/src/turkish_rag_eval/models.py +32 -0
  22. turkish_rag_eval-0.2.0/src/turkish_rag_eval/paths.py +40 -0
  23. turkish_rag_eval-0.2.0/src/turkish_rag_eval/report.py +275 -0
  24. turkish_rag_eval-0.2.0/src/turkish_rag_eval/retrieval.py +84 -0
  25. turkish_rag_eval-0.2.0/src/turkish_rag_eval/run_abstain.py +69 -0
  26. turkish_rag_eval-0.2.0/src/turkish_rag_eval/run_eval.py +270 -0
  27. turkish_rag_eval-0.2.0/src/turkish_rag_eval/turkish_text.py +54 -0
  28. turkish_rag_eval-0.2.0/src/turkish_rag_eval/validate_gold.py +174 -0
  29. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/PKG-INFO +487 -0
  30. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/SOURCES.txt +52 -0
  31. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/dependency_links.txt +1 -0
  32. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/entry_points.txt +2 -0
  33. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/requires.txt +22 -0
  34. turkish_rag_eval-0.2.0/src/turkish_rag_eval.egg-info/top_level.txt +1 -0
  35. turkish_rag_eval-0.2.0/tests/test_agreement.py +153 -0
  36. turkish_rag_eval-0.2.0/tests/test_blog.py +29 -0
  37. turkish_rag_eval-0.2.0/tests/test_bootstrap.py +221 -0
  38. turkish_rag_eval-0.2.0/tests/test_charts.py +98 -0
  39. turkish_rag_eval-0.2.0/tests/test_cli.py +117 -0
  40. turkish_rag_eval-0.2.0/tests/test_corpus.py +127 -0
  41. turkish_rag_eval-0.2.0/tests/test_export_hf.py +82 -0
  42. turkish_rag_eval-0.2.0/tests/test_fetch_corpus.py +103 -0
  43. turkish_rag_eval-0.2.0/tests/test_groundedness.py +327 -0
  44. turkish_rag_eval-0.2.0/tests/test_leaderboard.py +196 -0
  45. turkish_rag_eval-0.2.0/tests/test_metrics.py +65 -0
  46. turkish_rag_eval-0.2.0/tests/test_models.py +51 -0
  47. turkish_rag_eval-0.2.0/tests/test_readme_agreement.py +44 -0
  48. turkish_rag_eval-0.2.0/tests/test_readme_full.py +54 -0
  49. turkish_rag_eval-0.2.0/tests/test_readme_groundedness.py +59 -0
  50. turkish_rag_eval-0.2.0/tests/test_readme_leaderboard.py +66 -0
  51. turkish_rag_eval-0.2.0/tests/test_report.py +139 -0
  52. turkish_rag_eval-0.2.0/tests/test_run_eval.py +160 -0
  53. turkish_rag_eval-0.2.0/tests/test_validate_gold.py +151 -0
  54. turkish_rag_eval-0.2.0/tests/test_version.py +30 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Rizgar Ozan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,30 @@
1
+ # Data attribution
2
+
3
+ The retrieval corpus in `data/raw/corpus.json` is built from Turkish Wikipedia
4
+ articles fetched through the MediaWiki API by `turkish-rag-eval fetch-corpus`.
5
+
6
+ Wikipedia text is licensed **CC BY-SA 4.0**. Each document keeps its `title`
7
+ and permanent `url`, so every retrieved chunk can be traced to its source
8
+ article. The corpus file is not committed (see `.gitignore`); run the fetcher
9
+ to rebuild it.
10
+
11
+ ## The gold set
12
+
13
+ `data/eval/gold.json` **is** committed, and each item carries a short verbatim
14
+ `answer_span` quoted from the Turkish Wikipedia article named by its `doc_id`.
15
+ Those spans are therefore also **CC BY-SA 4.0**, attributed to the article in the
16
+ same record (`doc_id` = MediaWiki pageid, plus the title and URL in the fetched
17
+ corpus). The questions themselves were written for this project - deliberately
18
+ paraphrased rather than copied - and are released under CC BY-SA 4.0 as well, so
19
+ the whole gold set can be reused under one licence. Single annotator (the repo
20
+ author), one pass, no inter-annotator agreement figure.
21
+
22
+ The **code** in this repository is MIT (see `LICENSE`). The licences do not
23
+ conflict: MIT covers the harness, CC BY-SA 4.0 covers the Wikipedia-derived text
24
+ in `data/`.
25
+
26
+ ## No health data
27
+
28
+ No patient data, clinical records or any personal health information is used
29
+ anywhere in this project. The corpus is public encyclopaedic text only, and
30
+ nothing here is a medical device or clinical decision support tool.
@@ -0,0 +1,487 @@
1
+ Metadata-Version: 2.4
2
+ Name: turkish-rag-eval
3
+ Version: 0.2.0
4
+ Summary: Measure which parts of a Turkish RAG pipeline pay off: chunking, stemming, embedding model, fusion.
5
+ Author: Rızgar Ozan
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/RizgarOzan/turkish-rag-eval
8
+ Project-URL: Dataset, https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval
9
+ Project-URL: Issues, https://github.com/RizgarOzan/turkish-rag-eval/issues
10
+ Keywords: turkish,retrieval,rag,bm25,benchmark,evaluation
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Natural Language :: Turkish
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
16
+ Classifier: Topic :: Text Processing :: Indexing
17
+ Requires-Python: >=3.10
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ License-File: NOTICE.md
21
+ Requires-Dist: numpy>=1.24
22
+ Requires-Dist: rank-bm25>=0.2.2
23
+ Requires-Dist: requests>=2.31
24
+ Provides-Extra: dense
25
+ Requires-Dist: sentence-transformers>=3.0; extra == "dense"
26
+ Requires-Dist: torch>=2.0; extra == "dense"
27
+ Provides-Extra: charts
28
+ Requires-Dist: matplotlib>=3.7; extra == "charts"
29
+ Provides-Extra: llm
30
+ Requires-Dist: anthropic>=0.40; extra == "llm"
31
+ Provides-Extra: dev
32
+ Requires-Dist: pytest>=7.4; extra == "dev"
33
+ Provides-Extra: all
34
+ Requires-Dist: turkish-rag-eval[dense]; extra == "all"
35
+ Requires-Dist: turkish-rag-eval[charts]; extra == "all"
36
+ Requires-Dist: turkish-rag-eval[llm]; extra == "all"
37
+ Requires-Dist: turkish-rag-eval[dev]; extra == "all"
38
+ Dynamic: license-file
39
+
40
+ # turkish-rag-eval
41
+
42
+ [![validate](https://github.com/RizgarOzan/turkish-rag-eval/actions/workflows/validate.yml/badge.svg)](https://github.com/RizgarOzan/turkish-rag-eval/actions/workflows/validate.yml)
43
+ [![licence: MIT + CC BY-SA 4.0](https://img.shields.io/badge/licence-MIT%20%2B%20CC%20BY--SA%204.0-blue)](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md)
44
+ [![dataset on Hugging Face](https://img.shields.io/badge/%F0%9F%A4%97%20dataset-RizgarOzan%2Fturkish--rag--eval-yellow)](https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval)
45
+
46
+ **Which parts of a RAG pipeline actually earn their cost on an agglutinative
47
+ language?** Three chunking strategies × four retrievers × six embedding
48
+ models, measured on a hand-labelled Turkish gold set, with confidence
49
+ intervals and a cost column — then pointed at your own documents.
50
+
51
+ ```bash
52
+ pip install git+https://github.com/RizgarOzan/turkish-rag-eval # not on PyPI yet
53
+ turkish-rag-eval run --corpus ./belgelerim --gold ./sorular.json
54
+ turkish-rag-eval report
55
+ ```
56
+
57
+ ## Four findings
58
+
59
+ **1. The embedding model matters more than any pipeline choice.** Every model
60
+ trained for retrieval beats stemmed BM25 (0.494) on its own. The best,
61
+ [Mursit-Large-TR-Retrieval](https://huggingface.co/newmindai/Mursit-Large-TR-Retrieval),
62
+ reaches **0.781** nDCG@10 on hierarchical chunks. A small E5 the same size as
63
+ the default goes from 0.501 to 0.642 with nothing else changed, at about the
64
+ same speed (22 ms against 23 ms). The popular default model is the weak link, not dense retrieval.
65
+ See the [Leaderboard](#leaderboard).
66
+
67
+ **2. Turkish stemming is the cheapest real win.** Truncating tokens to a
68
+ 5-character prefix before BM25 lifts nDCG@10 by 23–29% on every chunking
69
+ strategy, and a paired bootstrap puts every one of those gains clear of zero
70
+ (`hierarchical +0.111, 95% CI [+0.026, +0.204]`). "diyabet", "diyabetin",
71
+ "diyabete", "diyabetli" are four surface forms of one concept; an unstemmed
72
+ index almost never matches the query's form.
73
+
74
+ **3. The top of the table is a tie, and the tie decides on cost.** With 58
75
+ queries, a paired bootstrap cannot separate the best configuration from the
76
+ other two hybrid ones; the remaining nine are measurably worse. So the
77
+ real choice at the top is price: `sentence + hybrid_rrf` gives the same
78
+ quality at 27 ms instead of 37 ms.
79
+
80
+ <picture>
81
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/ndcg-intervals-dark.png">
82
+ <img alt="nDCG@10 per configuration with 95% bootstrap intervals; the best cannot be told apart from the other two hybrid configurations, the remaining nine are measurably worse" src="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/ndcg-intervals.png">
83
+ </picture>
84
+
85
+ **4. The intuitive confidence signal is the useless one.** For deciding when
86
+ *not* to answer, the obvious measure — how far ahead the top hit is — is worse
87
+ than answering everything (0.25 selective accuracy against a 0.47 baseline).
88
+ RRF fuses ranks as `1/(60+rank)`, so the top-two gap is ~2% on every query,
89
+ confident or not. The dense retriever's raw cosine works; the margin does not.
90
+
91
+ <picture>
92
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/abstention-dark.png">
93
+ <img alt="Coverage against selective accuracy for two confidence signals; the top-1 margin performs worse than answering everything" src="https://raw.githubusercontent.com/RizgarOzan/turkish-rag-eval/main/docs/charts/abstention.png">
94
+ </picture>
95
+
96
+ Longer write-up of the first result:
97
+ [English](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/blog/2026-09-19-bm25-turkish-en.md) ·
98
+ [Türkçe](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/blog/2026-09-19-bm25-turkish-tr.md).
99
+
100
+ **Contents:** [Results](#results) · [Leaderboard](#leaderboard) ·
101
+ [Why not an existing benchmark?](#why-not-an-existing-benchmark) ·
102
+ [Your own corpus](#your-own-corpus) · [Running it](#running-it) ·
103
+ [Gold set](#gold-set) · [Status](#status) · [Limits](#limits) · [Contribute](#contribute) ·
104
+ [Design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md)
105
+
106
+ ## Results
107
+
108
+ 58 queries, default embedding model
109
+ `paraphrase-multilingual-MiniLM-L12-v2`, CPU only. Intervals are 95% bootstrap
110
+ over queries; `turkish-rag-eval report` regenerates this table.
111
+
112
+ | Chunking | Retriever | nDCG@10 | 95% CI | R@5 | MRR | P95 |
113
+ |---|---|---|---|---|---|---|
114
+ | **hierarchical** | **hybrid_rrf** | **0.613** | [0.508, 0.716] | 0.690 | 0.559 | 37 ms |
115
+ | sentence | hybrid_rrf | 0.558 | [0.453, 0.662] | 0.621 | 0.500 | 27 ms |
116
+ | fixed | hybrid_rrf | 0.517 | [0.404, 0.629] | 0.578 | 0.484 | 31 ms |
117
+ | sentence | bm25_stem5 | 0.510 | [0.403, 0.621] | 0.638 | 0.462 | 4 ms |
118
+ | hierarchical | dense | 0.501 | [0.396, 0.606] | 0.569 | 0.441 | 30 ms |
119
+ | hierarchical | bm25_stem5 | 0.494 | [0.386, 0.605] | 0.569 | 0.444 | 8 ms |
120
+ | fixed | bm25_stem5 | 0.476 | [0.371, 0.585] | 0.526 | 0.421 | 4 ms |
121
+ | sentence | dense | 0.461 | [0.356, 0.567] | 0.552 | 0.410 | 21 ms |
122
+ | fixed | dense | 0.446 | [0.341, 0.555] | 0.491 | 0.402 | 25 ms |
123
+ | sentence | bm25_nostem | 0.411 | [0.308, 0.515] | 0.483 | 0.356 | 4 ms |
124
+ | fixed | bm25_nostem | 0.387 | [0.290, 0.487] | 0.414 | 0.319 | 3 ms |
125
+ | hierarchical | bm25_nostem | 0.383 | [0.285, 0.482] | 0.500 | 0.319 | 6 ms |
126
+
127
+ `turkish-rag-eval report` does not stop at the table — it names the cheapest
128
+ configuration the data cannot separate from the best:
129
+
130
+ ```
131
+ Recommended: sentence + hybrid_rrf
132
+ 0.558 ndcg@10 against 0.613 for hierarchical + hybrid_rrf, a gap of 0.056
133
+ that a paired bootstrap over 58 shared queries cannot distinguish from zero.
134
+ It answers in 27 ms at P95 against 37 ms.
135
+ ```
136
+
137
+ Comparisons are **paired**: both systems answered the same queries, so
138
+ resampling per-query differences removes the variance of some queries being
139
+ harder than others. It matters, and this data shows it in the sharpest way.
140
+ `hierarchical + hybrid_rrf` beats `fixed + hybrid_rrf` by 0.097 and beats
141
+ `sentence + bm25_stem5` by 0.104 — yet only the **larger** gap clears zero,
142
+ because its per-query differences are so much steadier (sd 0.34 against 0.40).
143
+ Ranking by the gap alone gets this backwards.
144
+
145
+ Hierarchical chunking only helps the dense retriever, and fixed-size chunking
146
+ — the most common default — lost on every retriever. More in the
147
+ [design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#what-else-the-numbers-say).
148
+
149
+ ### Generation: groundedness
150
+
151
+ The R above feeds a G: the 58 questions answered from the top 5
152
+ `hierarchical + bm25_stem5` passages by `groq:openai/gpt-oss-120b`, judged by
153
+ a model from another family, `nvidia:nvidia/nemotron-3-super-120b-a12b`, at
154
+ temperature 0 on `2026-09-30`. The gold span was among the passages for
155
+ 33 of 58 questions, and the two groups are scored apart:
156
+
157
+ | Span retrieved (33) | |
158
+ |---|---|
159
+ | answered when the span was retrieved | 93.9% |
160
+ | answer rests on the passages | 96.8% |
161
+ | answer conveys the gold span | 96.8% |
162
+
163
+ | Span not retrieved (25) | |
164
+ |---|---|
165
+ | declined when the span was not retrieved | 80.0% |
166
+ | answered anyway | 20.0% |
167
+ | of those, answer is right | 60.0% |
168
+ | hallucinated (not in the passages) | 4.0% |
169
+
170
+ The generator mostly declines when the evidence is missing, and when it does
171
+ answer, the answer is usually stated in some other retrieved passage — one of
172
+ 25 is not. The first scoring rule called every answer without the span a
173
+ hallucination and reported 20%; reading the five by hand showed four stated in
174
+ a retrieved passage, so answers without the span now go to the judge too
175
+ ([why](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#groundedness)).
176
+ 16 verdicts were checked by hand; 14 agreed. Of the other two, the judge was
177
+ too strict once (an answer inverted "not recommended unless below 7 g/dl") and
178
+ too lenient once (it accepted "hyperuricaemia" as the cause of gout).
179
+
180
+ This row depends on hosted APIs and cannot be reproduced offline; the models
181
+ may change behind the same name. Only the rates are committed
182
+ ([`results/groundedness.json`](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/results/groundedness.json)) —
183
+ the free API terms do not allow redistributing raw model output.
184
+
185
+ To rerun it with other models, name them `groq:`, `nvidia:` or `openai:`
186
+ (a bare name goes to Anthropic). Keys can sit in a `.env` file (gitignored) in
187
+ the directory you run from; `openai:` follows `OPENAI_BASE_URL`, so any OpenAI-compatible
188
+ endpoint, a local router included, works.
189
+
190
+ ## Leaderboard
191
+
192
+ Dense `nDCG@10` per chunking strategy, the hybrid (dense + stemmed BM25, RRF)
193
+ on hierarchical chunks, and the cost of each model on one 16-thread CPU,
194
+ measured 2026-09-18. Stemmed BM25 alone scores 0.476 / 0.510 / 0.494.
195
+ `turkish-rag-eval leaderboard` rebuilds this from `results/models/`.
196
+
197
+ | Model | Params | Turkish-only | Dense fixed | Dense sentence | Dense hierarchical | Hybrid hierarchical | Embed corpus | Query P95 |
198
+ |---|---|---|---|---|---|---|---|---|
199
+ | paraphrase-multilingual-MiniLM-L12-v2 (default) | 118 M | no | 0.446 | 0.461 | 0.501 | 0.607 | 2 min | 23 ms |
200
+ | [emrecan/bert-base-turkish-cased-mean-nli-stsb-tr](https://huggingface.co/emrecan/bert-base-turkish-cased-mean-nli-stsb-tr) | 111 M | yes | 0.408 | 0.431 | 0.497 | 0.654 | 4 min | 43 ms |
201
+ | [intfloat/multilingual-e5-small](https://huggingface.co/intfloat/multilingual-e5-small) | 118 M | no | 0.654 | 0.644 | 0.642 | 0.639 | 3 min | 22 ms |
202
+ | [intfloat/multilingual-e5-base](https://huggingface.co/intfloat/multilingual-e5-base) | 278 M | no | 0.631 | 0.677 | 0.668 | 0.648 | 10 min | 49 ms |
203
+ | [BAAI/bge-m3](https://huggingface.co/BAAI/bge-m3) | 568 M | no | 0.766 | 0.767 | — | — | > 45 min | — |
204
+ | **[newmindai/Mursit-Large-TR-Retrieval](https://huggingface.co/newmindai/Mursit-Large-TR-Retrieval)** | 404 M | yes | **0.746** | **0.740** | **0.781** | 0.673 | 35 min | 214 ms |
205
+
206
+ The two Turkish-only models are the most-downloaded Turkish entries in the
207
+ Hugging Face `sentence-similarity` category. `bge-m3` was stopped after 45
208
+ minutes, before the hierarchical chunks; its two numbers come from that
209
+ partial run. `google/embeddinggemma-300m` is gated behind a licence click and
210
+ was not run. Per-query results for every completed model are in
211
+ `results/models/`.
212
+
213
+ > These five rows were measured together on one machine before v0.1.0, which
214
+ > is why the timing columns are comparable with each other and not with the
215
+ > main table above. Their `dense` columns are unaffected by the stable
216
+ > tie-break, but each `Hybrid hierarchical` figure will move by roughly +0.006
217
+ > when the model is re-run — the default row's went 0.607 → 0.613.
218
+ > `turkish-rag-eval leaderboard --check` reports them as missing provenance
219
+ > until then.
220
+
221
+ **The top of this table is settled; the middle is not.** Dense hierarchical,
222
+ with 95% bootstrap intervals over the 58 queries: MiniLM 0.501 [0.396, 0.606],
223
+ emrecan 0.497 [0.394, 0.601], e5-small 0.642 [0.542, 0.738], e5-base 0.668
224
+ [0.568, 0.765], Mursit 0.781 [0.701, 0.856]. The intervals overlap, but paired
225
+ over the same queries Mursit beats the runner-up e5-base by +0.113
226
+ [+0.029, +0.202]. The two E5 models cannot be told apart (+0.026
227
+ [-0.039, +0.094]), and neither can MiniLM and emrecan (+0.004
228
+ [-0.123, +0.135]).
229
+
230
+ **Hybrid fusion only pays for a weak dense model.** RRF lifts the small
231
+ default by +0.106 but pulls Mursit down from 0.781 to 0.673, a paired loss of
232
+ -0.108 [-0.193, -0.030]; for the two E5 models it makes no measurable
233
+ difference. "Turkish-only" is not enough either: the `emrecan` model was
234
+ trained for sentence similarity and truncates input at 75 tokens.
235
+
236
+ To submit a model, open a pull request adding `results/models/<org>__<name>/`
237
+ (`turkish-rag-eval run --model <org>/<name>`, then `leaderboard --check`).
238
+ CI re-derives every metric from the per-query relevance arrays committed
239
+ beside it, and checks the harness version and corpus fingerprint.
240
+
241
+ ## Why not an existing benchmark?
242
+
243
+ MTEB-style retrieval benchmarks, TR-MTEB included, score an embedding model on
244
+ passages that are already split. They answer "which model?", not "which
245
+ chunker, is Turkish stemming worth it, does a hybrid help, and what does each
246
+ cost on a CPU?". This harness keeps the articles whole, lets every chunker cut
247
+ them its own way, and judges each chunk by the answer span, so pipeline
248
+ choices are compared on the same labels. For a model-only comparison the same
249
+ data exports to the BEIR layout MTEB reads (`turkish-rag-eval export-hf`),
250
+ published as
251
+ [RizgarOzan/turkish-rag-eval](https://huggingface.co/datasets/RizgarOzan/turkish-rag-eval)
252
+ (the 58 human questions, plus all 296 under the `full-*` configs, each marked
253
+ `human` or `llm-draft`); adding it to MTEB is proposed in
254
+ [embeddings-benchmark/mteb#5536](https://github.com/embeddings-benchmark/mteb/issues/5536).
255
+
256
+ ## Your own corpus
257
+
258
+ The harness is not tied to its own articles. Point it at a folder of `.txt` or
259
+ `.md` files, a BEIR directory, or a JSON corpus:
260
+
261
+ ```bash
262
+ turkish-rag-eval run --corpus ./belgelerim --gold ./sorular.json \
263
+ --retrievers bm25_stem5 bm25_nostem
264
+ ```
265
+
266
+ A BM25-only run needs no model download and no torch — enough to answer "which
267
+ chunker, and is stemming worth it on my documents" from the base install.
268
+
269
+ **No labelled questions?** That is the real wall, and the reason most
270
+ benchmarks only ever measure themselves. `bootstrap` drafts a starting point (the name means drafting a gold set here,
271
+ not the statistical resampling above):
272
+
273
+ ```bash
274
+ export ANTHROPIC_API_KEY=...
275
+ turkish-rag-eval bootstrap ./belgelerim --out gold-draft.json --per-doc 3
276
+ ```
277
+
278
+ It runs the same procedure the contributed questions in this repository used.
279
+ One pass writes questions and marks the answer span; a second pass sees only
280
+ the document and the questions and marks the span again. Where the passes
281
+ disagree the item is written as `"review": "needs-human"` and **never loads**
282
+ until a person settles it. Spans that are not verbatim, and questions copied
283
+ out of their own answer, are dropped with a reason.
284
+
285
+ What you get is a draft to review, not a gold set.
286
+
287
+ ## Running it
288
+
289
+ Python 3.10+. CPU only — no GPU anywhere in this project.
290
+
291
+ ```bash
292
+ pip install git+https://github.com/RizgarOzan/turkish-rag-eval # metrics, BM25, gold-set tooling
293
+ pip install 'turkish-rag-eval[all] @ git+https://github.com/RizgarOzan/turkish-rag-eval' # + dense retrieval, charts, LLM commands
294
+ ```
295
+
296
+ | Command | What it does |
297
+ |---|---|
298
+ | `run` | every chunking × retriever combination; writes `results/` |
299
+ | `report` | intervals, paired comparisons, and a recommendation |
300
+ | `charts` | the two figures above, light and dark |
301
+ | `leaderboard` | rebuild the model table; `--check` verifies every entry |
302
+ | `abstain` | coverage / selective-accuracy curve for the best configuration ([details](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#abstention)) |
303
+ | `groundedness` | score the generation half against the gold spans ([details](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#groundedness); [results](#generation-groundedness)) |
304
+ | `bootstrap` | draft a gold set for your own corpus |
305
+ | `agreement` | inter-annotator agreement over the gold set |
306
+ | `fetch-corpus` | download the Wikipedia snapshot; `--verify` checks the lock; `--include-drafts` for [all 300 questions](#scoring-all-300) |
307
+ | `export-hf` | the BEIR layout MTEB reads |
308
+ | `validate` | every gold file's invariants |
309
+
310
+ From a checkout:
311
+
312
+ ```bash
313
+ git clone https://github.com/RizgarOzan/turkish-rag-eval
314
+ cd turkish-rag-eval
315
+ pip install -e '.[all]'
316
+ turkish-rag-eval fetch-corpus # rebuilds data/raw/corpus.json
317
+ turkish-rag-eval run # results/summary.json + per-query files
318
+ python -m pytest -q
319
+ ```
320
+
321
+ `run --model <name>` swaps the embedding model; any sentence-transformers
322
+ model works. Results for a non-default model go to
323
+ `results/models/<org>__<name>/`, so the main table is never overwritten.
324
+
325
+ ## Gold set
326
+
327
+ | Files | Questions | Labelled by | In the results above |
328
+ |---|---|---|---|
329
+ | `data/eval/gold.json` (health) | 58 | one human | yes |
330
+ | `data/eval/contrib/llm-draft-*.json` (history, geography, astronomy, biology, computing) | 242 | two independent LLM passes | not yet |
331
+
332
+ The set reached its planned 300 questions on 2026-09-26. Questions a model
333
+ drafted are marked `"source": "llm-draft"`. A second model then picked its own
334
+ answer span for each one without seeing the first label
335
+ (`second_annotation`). Two spans agree when one contains the other or their
336
+ token F1 is at least 0.5; agreed items get `"review": "agreed"`, the rest get
337
+ `"needs-human"` and are never loaded. Batches 1 and 2 were 30 of 30 agreed,
338
+ batch 3 was 29 of 30: for "what are several ribosomes working on one mRNA
339
+ called?" the passes picked two different sentences that both name polysomes,
340
+ so that question waits for a person. Batch 4 was 27 of 30: the second pass
341
+ named the other claimant to the Hungarian throne, took the sentence beside the
342
+ lysozyme result instead of the result itself, and answered "cross compilers"
343
+ with the bare term where the first took its definition. Batch 5 was 30 of 30,
344
+ and every pair is a containment pair: 9 identical, 15 differing only by a
345
+ trailing full stop. One of its answers, Robert W. Holley's 1968 Nobel Prize,
346
+ was cut to its last clause on 2026-09-27 and labelled again by the second
347
+ pass: the sentence splitter breaks after "W.", so no hierarchical chunk held
348
+ the whole sentence ([#17](https://github.com/RizgarOzan/turkish-rag-eval/issues/17)). Batch 6 was 30 of 30 on reworded questions whose answers
349
+ often run to two sentences; six first-pass spans were cut to one sentence
350
+ before merging, because the two-sentence version fitted inside no chunk and
351
+ so could never be retrieved. Batch 7 was 30 of 30 again, with 25 identical
352
+ spans; one first-pass span was cut to its clause for the same reason. Like
353
+ batch 5 its questions mostly point at a single sentence, so its high
354
+ agreement says little about harder questions. Batch 8, the last 32, was
355
+ written to be harder: 17 first-pass answers ran to two or more sentences.
356
+ Seven of those crossed a sentence- or hierarchical-chunk boundary and were
357
+ cut, leaving 12 multi-sentence answers. It was still 32 of 32, 29 identical:
358
+ two LLMs reading the same article pick the same sentences even when the
359
+ answer is long, which is one more reason these drafts need a human pass.
360
+
361
+ Drafts stay out of every number above until a re-run says otherwise:
362
+ `load_gold()` skips them unless called with `include_drafts=True`. Synthetic
363
+ test questions are common practice as long as they are declared and their
364
+ agreement is measured. This section is that declaration.
365
+
366
+ The corpus is 54 Turkish Wikipedia articles, 1.09 M characters; 27 of them
367
+ answer at least one question and the other 27 are distractors from the same
368
+ domain.
369
+
370
+ | Batch | Questions | Identical | Mean IoU | Mean token F1 | Cohen's κ, fixed / sentence / hierarchical |
371
+ |---|---|---|---|---|---|
372
+ | 1 — 2026-09-18 (Malazgirt, Kapadokya, Mars, Mitokondri, Linux) | 30 | 16 | 0.816 | 0.878 | 1.00 / 1.00 / 1.00 |
373
+ | 2 — 2026-09-20 (İstanbul'un Fethi, Ağrı Dağı, Jüpiter, Fotosentez, İnternet)¹ | 30 | 4 | 0.427 | 0.549 | 0.94 / 0.99 / 0.99 |
374
+ | 3 — 2026-09-21 (Çaldıran Muharebesi, Tuz Gölü, Satürn, Ribozom, Unix) | 30 | 18 | 0.829 | 0.872 | 0.92 / 0.96 / 0.96 |
375
+ | 4 — 2026-09-24 (Mohaç Muharebesi, Kızılırmak, Venüs, Enzim, Derleyici) | 30 | 15 | 0.747 | 0.792 | 0.96 / 0.96 / 0.96 |
376
+ | 5 — 2026-09-25 (Kösedağ Muharebesi, Uludağ, Neptün, RNA, İşletim sistemi) | 30 | 24 | 0.922 | 0.943 | 1.00 / 1.00 / 1.00 |
377
+ | 6 — 2026-09-25 (Preveze Deniz Muharebesi, Erciyes, Uranüs, Hemoglobin, Veritabanı) | 30 | 20 | 0.869 | 0.901 | 0.93 / 0.90 / 0.87 |
378
+ | 7 — 2026-09-26 (Ankara Muharebesi, Van Gölü, Merkür, DNA, World Wide Web) | 30 | 25 | 0.944 | 0.960 | 0.93 / 0.93 / 0.97 |
379
+ | 8 — 2026-09-26 (Niğbolu Muharebesi, Fırat, Ay, Protein, Yapay zekâ) | 32 | 29 | 0.964 | 0.975 | 0.94 / 0.97 / 1.00 |
380
+ | **All drafts** | 242 | 151 | 0.816 | 0.860 | 0.95 / 0.97 / 0.97 |
381
+
382
+ ¹ Re-measured 2026-09-26: the Turkish Wikipedia article *Jüpiter* was rewritten that day to correct errors, and five of its spans no longer matched the live text. Three changed only in wording (a comma, "30,003" → "30" seconds, "dört uydu" → "uydular") and were edited in both labels; two changed in substance — the 40,000 km mantle thickness is gone and the Great Red Spot went from "at least 400 years" to "recorded since 1831" — so those two questions were rewritten and labelled again by both passes. The row was 0.419 / 0.540 / 0.96 / 1.00 / 1.00 before.
383
+
384
+ The passes disagree about how much of a sentence to take, rather than where
385
+ the answer is. Two LLMs tend to pick the same sentence, so read this as a
386
+ sanity check rather than human agreement. Full discussion in the
387
+ [design notes](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/docs/design.md#agreement-between-the-two-passes).
388
+
389
+ ### Scoring all 300
390
+
391
+ The drafted questions point at 40 articles outside the health snapshot, so
392
+ they get a corpus of their own: the 54 health articles plus those 40, 1.89 M
393
+ characters, pinned by `data/corpus-full.lock.json`. The published snapshot and
394
+ its lock stay as they are.
395
+
396
+ ```bash
397
+ turkish-rag-eval fetch-corpus --include-drafts # data/raw/corpus-full.json
398
+ turkish-rag-eval run --include-drafts # results/full/<model>/
399
+ ```
400
+
401
+ Measured 2026-09-27 (re-run after the batch 5 fix; Mursit added 2026-10-01) on 296 questions (the 4 `needs-human` drafts never load),
402
+ hierarchical chunks, nDCG@10:
403
+
404
+ | Retriever | 58 human questions | 296 questions (238 drafted) |
405
+ |---|---|---|
406
+ | bm25_stem5 | 0.494 | 0.558 |
407
+ | MiniLM (default), dense | 0.501 | 0.405 |
408
+ | MiniLM (default), hybrid_rrf | 0.613 | 0.578 |
409
+ | multilingual-e5-small, dense | 0.642 | 0.568 |
410
+ | multilingual-e5-small, hybrid_rrf | 0.639 | 0.642 |
411
+ | multilingual-e5-base, dense | 0.668 | 0.646 |
412
+ | multilingual-e5-base, hybrid_rrf | 0.648 | 0.664 |
413
+ | Mursit-Large-TR-Retrieval, dense | 0.781 | 0.613 |
414
+ | Mursit-Large-TR-Retrieval, hybrid_rrf | 0.673 | 0.661 |
415
+
416
+ On the wider set stemmed BM25 gets stronger and every dense model weaker, so
417
+ fusing the two now helps all four models, where on the 58 health questions it
418
+ cost the E5 models and Mursit. Mursit drops the most, from 0.781 to 0.613, and
419
+ falls below e5-base, so its lead on the human set does not carry over to the
420
+ drafted questions yet. Those questions were written by an LLM from one sentence
421
+ each; whether that style suits the smaller models better is open until people
422
+ have checked them. These rows are not in the tables above: most of the
423
+ questions have not been checked by a person yet. The run also names every
424
+ question no chunk can answer; a span split by a chunk boundary is the usual
425
+ reason. Here that is 17 questions with fixed-size chunks and none with
426
+ sentence or hierarchical chunks (the one hierarchical miss was fixed on
427
+ 2026-09-27, see batch 5 above). Results for all three chunkers are in
428
+ [`results/full/`](https://github.com/RizgarOzan/turkish-rag-eval/tree/main/results/full).
429
+
430
+ ## Status
431
+
432
+ v0.1.0, tagged but not on PyPI yet — install from GitHub as above.
433
+
434
+ - **Works:** the retrieval harness, [`groundedness`](#generation-groundedness) on the 58 human questions, `report`, `leaderboard --check`, `bootstrap`,
435
+ `agreement`, the Hugging Face export, scoring [all 300 questions](#scoring-all-300)
436
+ on a pinned corpus, and CI that validates every contributed question against
437
+ live Wikipedia.
438
+ - **In progress:** human review of the gold set. All 300 planned questions are
439
+ in (58 human, 242 LLM-drafted); the drafts join the results once people have
440
+ reviewed them.
441
+ - **Not yet:** EmbeddingGemma, and `groundedness` on the 242 drafted questions.
442
+
443
+ ## Limits
444
+
445
+ - **58 queries is a small set.** The intervals above are the honest width of
446
+ that; most of the table's ordering is not resolved by this much data. They
447
+ replace the eyeballed rule of thumb this project used to carry — "treat
448
+ differences under roughly 0.05 nDCG as noise" — which was both too strict
449
+ for paired comparisons and too loose for unpaired ones.
450
+ - **One annotator for the human set.** The 58 health questions are
451
+ single-annotated, so they have no agreement figure. The 242 drafted questions
452
+ are double-labelled, but by two LLM passes rather than by two people.
453
+ - **The corpus is Wikipedia, and so is much of the training data.** Every
454
+ embedding model ranked here was almost certainly trained on Turkish
455
+ Wikipedia. Absolute scores are therefore optimistic; the *comparison*
456
+ between pipeline choices on the same corpus is what this measures. A
457
+ non-Wikipedia domain is the most valuable thing a contributor could add.
458
+ - **Six embedding models, one corpus.** `bge-m3` is a partial run and
459
+ EmbeddingGemma is missing. `abstain` uses only the default model.
460
+ - **Encyclopaedic text, not clinical text.** Nothing here transfers to a
461
+ clinical setting without re-measurement. No patient data is used anywhere,
462
+ and nothing here is a medical device.
463
+ - **The generation half is one generator, one judge, 58 questions.**
464
+ [Groundedness](#generation-groundedness) was measured once, through hosted
465
+ APIs, on the human set only; the judge was checked by hand on 16 verdicts.
466
+ - **The embedding model is pinned by name, not by revision.**
467
+
468
+ ## Contribute
469
+
470
+ The first two limits shrink with every contributor. Adding questions needs no
471
+ ML background — pick a Turkish Wikipedia article, write 5–10 paraphrased
472
+ questions, and open a pull request with one JSON file. A validator checks each
473
+ file against Wikipedia in CI. See [CONTRIBUTING.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/CONTRIBUTING.md) (Türkçe
474
+ açıklama dahil) and the
475
+ [good first issues](https://github.com/RizgarOzan/turkish-rag-eval/issues?q=is%3Aopen+label%3A%22good+first+issue%22).
476
+
477
+ Submitting an embedding model is one command and a pull request — see
478
+ [Leaderboard](#leaderboard).
479
+
480
+ ## Data
481
+
482
+ Turkish Wikipedia, CC BY-SA 4.0. See [NOTICE.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md).
483
+
484
+ ## Licence
485
+
486
+ Code MIT ([LICENSE](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/LICENSE)); data under `data/` CC BY-SA 4.0
487
+ ([NOTICE.md](https://github.com/RizgarOzan/turkish-rag-eval/blob/main/NOTICE.md)).