Skip to content

Commit 15dc5c8

Browse files
RizgarOzanclaude
andcommitted
Chart the two findings a table hides
A benchmark repository with no picture asked every reader to reconstruct the result from twelve rows. ndcg-intervals plots each configuration with its 95% interval and highlights the three the paired bootstrap cannot separate from the best. Emphasis - one hue plus grey - rather than a colour per row, because the reader's question is "which of these actually differ", and that is a two-group question. Seeing the intervals overlap does more than the table's ordering ever did. abstention plots coverage against selective accuracy for both confidence signals, which is the least expected result here: thresholding on the top-1 margin is worse than answering everything (0.25 against 0.47), because RRF fuses as 1/(60+rank) and the top-two gap is ~2% on every query. Points resting on fewer than five answered queries are dropped; one question should not draw a cliff. Both render in light and dark so the README can pair them with <picture>; a light PNG on a dark page is a glare rectangle. Palette values come from a validated reference and were re-checked on both surfaces for contrast and colour-vision separation before use. Title and subtitle are offset in points from the axes corner, not in axes fractions: the interval chart's height grows with the number of rows, so a fractional offset drifts and a fixed title pad collides. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EJ4G5AQVGTMMDin5XGvqoU
1 parent 8de3e32 commit 15dc5c8

8 files changed

Lines changed: 354 additions & 4 deletions

File tree

‎.github/workflows/validate.yml‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -14,10 +14,10 @@ jobs:
1414
- uses: actions/setup-python@v5
1515
with:
1616
python-version: "3.12"
17-
# The base install plus dev extras: enough for the metrics, the gold-set
18-
# checks and the Turkish tokeniser. Dense retrieval pulls torch and is
19-
# not exercised here, so it stays out.
20-
- run: pip install -e '.[dev]'
17+
# Base install plus dev and charts: enough for the metrics, the gold-set
18+
# checks, the Turkish tokeniser and chart rendering. Dense retrieval pulls
19+
# torch and is not exercised here, so it stays out.
20+
- run: pip install -e '.[dev,charts]'
2121
- run: python -m pytest -q
2222
- name: Offline checks on every question file
2323
run: turkish-rag-eval validate

‎docs/charts/abstention-dark.png‎

114 KB
Loading

‎docs/charts/abstention.png‎

115 KB
Loading
143 KB
Loading

‎docs/charts/ndcg-intervals.png‎

145 KB
Loading

‎src/turkish_rag_eval/charts.py‎

Lines changed: 251 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,251 @@
1+
"""Render the two charts the README leads with.
2+
3+
A benchmark repository with no picture in it asks every reader to reconstruct
4+
the finding from a twelve-row table. These two carry the findings that a table
5+
hides:
6+
7+
``ndcg-intervals`` every configuration with its 95% bootstrap interval, with
8+
the ones the data cannot separate from the best picked out.
9+
The interval is the point: the ranking looks decisive and
10+
mostly is not.
11+
12+
``abstention`` coverage against selective accuracy for both confidence
13+
signals. This is the repository's least expected result -
14+
the intuitive signal (top-1 margin) is useless on fused
15+
rankings, because RRF scores are 1/(60+rank) and the
16+
top-two gap is about 2% for every query, confident or not.
17+
18+
Both are written in light and dark variants, because a light-background PNG in
19+
a dark README is a glare rectangle; the README pairs them with <picture>.
20+
21+
Colours come from the reference palette and were validated for contrast and
22+
colour-vision separation on both surfaces before use. The emphasis chart is
23+
one hue plus grey rather than a colour per bar: the reader's question is
24+
"which of these are actually different", and that is a two-group question.
25+
"""
26+
27+
import argparse
28+
import json
29+
from pathlib import Path
30+
31+
from .paths import ROOT
32+
from .report import DEFAULT_METRIC, load_configurations, recommend
33+
34+
#: Points resting on a handful of answered queries swing between 0 and 1 on
35+
#: one question. abstain.pick_threshold ignores them for the same reason.
36+
MIN_ANSWERED = 5
37+
38+
THEMES = {
39+
"light": {
40+
"surface": "#fcfcfb",
41+
"ink": "#0b0b0b",
42+
"secondary": "#52514e",
43+
"muted": "#898781",
44+
"grid": "#e1e0d9",
45+
"axis": "#c3c2b7",
46+
"accent": "#2a78d6",
47+
"second": "#eb6834",
48+
"recessive": "#898781",
49+
},
50+
"dark": {
51+
"surface": "#1a1a19",
52+
"ink": "#ffffff",
53+
"secondary": "#c3c2b7",
54+
"muted": "#898781",
55+
"grid": "#2c2c2a",
56+
"axis": "#383835",
57+
"accent": "#3987e5",
58+
"second": "#d95926",
59+
"recessive": "#898781",
60+
},
61+
}
62+
63+
64+
def _figure(theme: dict, size: tuple[float, float]):
65+
import matplotlib.pyplot as plt
66+
67+
figure, axes = plt.subplots(figsize=size)
68+
figure.patch.set_facecolor(theme["surface"])
69+
axes.set_facecolor(theme["surface"])
70+
for side in ("top", "right"):
71+
axes.spines[side].set_visible(False)
72+
for side in ("left", "bottom"):
73+
axes.spines[side].set_color(theme["axis"])
74+
axes.spines[side].set_linewidth(1)
75+
axes.tick_params(colors=theme["muted"], labelsize=9, length=0)
76+
return figure, axes
77+
78+
79+
def _titles(axes, theme: dict, title: str, subtitle: str) -> None:
80+
"""Title and subtitle stacked above the axes.
81+
82+
Offset in *points* from the axes corner rather than in axes fractions: the
83+
interval chart's height grows with the number of configurations, so a
84+
fractional offset would drift away from the title as rows are added, and a
85+
fixed title pad would collide with it.
86+
"""
87+
subtitle_lines = subtitle.count("\n") + 1
88+
axes.annotate(subtitle, xy=(0, 1), xycoords="axes fraction",
89+
xytext=(0, 10), textcoords="offset points",
90+
ha="left", va="bottom", fontsize=9.5,
91+
color=theme["secondary"], linespacing=1.45)
92+
axes.annotate(title, xy=(0, 1), xycoords="axes fraction",
93+
xytext=(0, 10 + 15 * subtitle_lines + 8),
94+
textcoords="offset points", ha="left", va="bottom",
95+
fontsize=13.5, fontweight="bold", color=theme["ink"])
96+
97+
98+
def intervals_chart(results_dir: Path, out: Path, theme_name: str,
99+
metric: str = DEFAULT_METRIC) -> Path:
100+
"""Every configuration, sorted, with its interval; the tied ones picked out."""
101+
theme = THEMES[theme_name]
102+
configurations = load_configurations(results_dir)
103+
recommendation = recommend(configurations, metric)
104+
tied = {id(c) for c in recommendation.tied_with_best}
105+
106+
rows = sorted(configurations, key=lambda c: c.mean(metric))
107+
figure, axes = _figure(theme, (8.4, 0.44 * len(rows) + 2.3))
108+
109+
for position, configuration in enumerate(rows):
110+
low, high = configuration.interval(metric)
111+
mean = configuration.mean(metric)
112+
emphasised = id(configuration) in tied
113+
colour = theme["accent"] if emphasised else theme["recessive"]
114+
115+
axes.plot([low, high], [position, position], color=colour, linewidth=2,
116+
solid_capstyle="round", zorder=2,
117+
alpha=1.0 if emphasised else 0.55)
118+
axes.plot([mean], [position], marker="o", markersize=8, color=colour,
119+
markeredgecolor=theme["surface"], markeredgewidth=2, zorder=3,
120+
alpha=1.0 if emphasised else 0.55)
121+
axes.text(high + 0.012, position, f"{mean:.3f}", va="center",
122+
fontsize=9, color=theme["ink"] if emphasised
123+
else theme["muted"])
124+
125+
axes.set_yticks(range(len(rows)))
126+
axes.set_yticklabels([c.name for c in rows], fontsize=9.5,
127+
color=theme["secondary"])
128+
axes.set_xlabel(metric, color=theme["secondary"], fontsize=9.5, labelpad=8)
129+
axes.set_xlim(0.22, 0.82)
130+
axes.set_ylim(-0.7, len(rows) - 0.3)
131+
axes.xaxis.grid(True, color=theme["grid"], linewidth=1)
132+
axes.set_axisbelow(True)
133+
134+
_titles(axes, theme,
135+
"Most of this ranking is not a ranking",
136+
f"{metric} with 95% bootstrap intervals over "
137+
f"{len(recommendation.best.scored)} queries. Highlighted: the "
138+
f"configurations a paired\nbootstrap cannot separate from the best.")
139+
140+
# A two-group emphasis chart still needs its groups named.
141+
from matplotlib.lines import Line2D
142+
axes.legend(
143+
handles=[
144+
Line2D([], [], color=theme["accent"], linewidth=2, marker="o",
145+
markersize=7, label="within noise of the best"),
146+
Line2D([], [], color=theme["recessive"], alpha=0.55, linewidth=2,
147+
marker="o", markersize=7, label="measurably worse"),
148+
],
149+
loc="lower right", frameon=False, fontsize=9,
150+
labelcolor=theme["secondary"])
151+
152+
return _save(figure, out)
153+
154+
155+
def abstention_chart(results_dir: Path, out: Path, theme_name: str) -> Path:
156+
"""Coverage against selective accuracy, one line per confidence signal."""
157+
theme = THEMES[theme_name]
158+
figure, axes = _figure(theme, (7.6, 5.0))
159+
160+
series = [
161+
("score", "dense top-1 cosine", theme["accent"]),
162+
("margin", "top-1 margin", theme["second"]),
163+
]
164+
plotted = 0
165+
for signal, label, colour in series:
166+
path = results_dir / f"abstain_curve_{signal}.json"
167+
if not path.exists():
168+
continue
169+
curve = [p for p in json.loads(path.read_text(encoding="utf-8"))
170+
if p["answered"] >= MIN_ANSWERED
171+
and p["selective_accuracy"] is not None]
172+
if not curve:
173+
continue
174+
coverage = [p["coverage"] for p in curve]
175+
accuracy = [p["selective_accuracy"] for p in curve]
176+
177+
axes.plot(coverage, accuracy, color=colour, linewidth=2, marker="o",
178+
markersize=6, markeredgecolor=theme["surface"],
179+
markeredgewidth=1.5, label=label, zorder=3)
180+
# Direct label at the low-coverage end, where the two lines separate.
181+
leftmost = min(range(len(coverage)), key=lambda i: coverage[i])
182+
axes.annotate(label, (coverage[leftmost], accuracy[leftmost]),
183+
textcoords="offset points", xytext=(8, 8),
184+
fontsize=9.5, color=theme["ink"], fontweight="bold")
185+
plotted += 1
186+
187+
axes.set_xlabel("coverage - share of queries answered automatically",
188+
color=theme["secondary"], fontsize=9.5, labelpad=8)
189+
axes.set_ylabel("selective accuracy - correct among those answered",
190+
color=theme["secondary"], fontsize=9.5, labelpad=8)
191+
axes.set_xlim(0, 1.03)
192+
axes.set_ylim(0, 1.03)
193+
axes.grid(True, color=theme["grid"], linewidth=1)
194+
axes.set_axisbelow(True)
195+
if plotted > 1:
196+
axes.legend(loc="lower left", frameon=False, fontsize=9,
197+
labelcolor=theme["secondary"])
198+
199+
_titles(axes, theme,
200+
"The intuitive confidence signal is the useless one",
201+
f"Thresholds keeping at least {MIN_ANSWERED} answered queries. RRF "
202+
f"fuses ranks as 1/(60+rank), so the top-two\ngap is ~2% on every "
203+
f"query and the margin buys almost no coverage to trade.")
204+
return _save(figure, out)
205+
206+
207+
def _save(figure, out: Path) -> Path:
208+
import matplotlib.pyplot as plt
209+
210+
out.parent.mkdir(parents=True, exist_ok=True)
211+
figure.savefig(out, dpi=200, bbox_inches="tight",
212+
facecolor=figure.get_facecolor())
213+
plt.close(figure)
214+
return out
215+
216+
217+
def add_arguments(parser: argparse.ArgumentParser) -> None:
218+
parser.add_argument("--results", type=Path, default=ROOT / "results")
219+
parser.add_argument("--out", type=Path, default=ROOT / "docs" / "charts")
220+
parser.add_argument("--metric", default=DEFAULT_METRIC)
221+
222+
223+
def run(args) -> int:
224+
if not (args.results / "summary.json").exists():
225+
print(f"{args.results} holds no results - run 'turkish-rag-eval run' first")
226+
return 1
227+
228+
written = []
229+
for theme_name in THEMES:
230+
suffix = "" if theme_name == "light" else "-dark"
231+
written.append(intervals_chart(
232+
args.results, args.out / f"ndcg-intervals{suffix}.png",
233+
theme_name, args.metric))
234+
if (args.results / "abstain_curve_score.json").exists():
235+
written.append(abstention_chart(
236+
args.results, args.out / f"abstention{suffix}.png", theme_name))
237+
238+
for path in written:
239+
print(f" {path.relative_to(ROOT) if ROOT in path.parents else path}")
240+
print(f"{len(written)} chart(s) written")
241+
return 0
242+
243+
244+
def main() -> int:
245+
parser = argparse.ArgumentParser(description=__doc__)
246+
add_arguments(parser)
247+
return run(parser.parse_args())
248+
249+
250+
if __name__ == "__main__":
251+
raise SystemExit(main())

‎src/turkish_rag_eval/cli.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -32,6 +32,7 @@
3232
"report": ("report", "turn results into a recommendation"),
3333
"abstain": ("run_abstain", "produce abstain records and the coverage curve"),
3434
"abstain-report": ("abstain", "re-print the coverage curve from saved records"),
35+
"charts": ("charts", "render the result charts the README embeds"),
3536
"agreement": ("agreement", "inter-annotator agreement over the gold set"),
3637
"fetch-corpus": ("fetch_corpus", "download the Wikipedia corpus snapshot"),
3738
"export-hf": ("export_hf", "export the BEIR layout MTEB reads"),

‎tests/test_charts.py‎

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
"""Charts are checked for the things a rendered image cannot tell you.
2+
3+
Whether they look right is settled by opening them. What is worth pinning is
4+
that both themes are produced, that a results directory without the abstention
5+
run still yields the chart that does exist, and that the noisiest end of the
6+
coverage curve is filtered out rather than plotted.
7+
"""
8+
9+
import json
10+
11+
import pytest
12+
13+
pytest.importorskip("matplotlib")
14+
pytest.importorskip("numpy")
15+
16+
from turkish_rag_eval import charts # noqa: E402
17+
18+
19+
def records(hits):
20+
return [{"question": f"q{i}", "relevance": [hit], "total_relevant": 1,
21+
"found": hit} for i, hit in enumerate(hits)]
22+
23+
24+
@pytest.fixture
25+
def results(tmp_path):
26+
directory = tmp_path / "results"
27+
directory.mkdir()
28+
for name, hits in [("sentence_dense", [True, False] * 15),
29+
("fixed_bm25_stem5", [True, True, False] * 10)]:
30+
(directory / f"perquery_{name}.json").write_text(
31+
json.dumps(records(hits)), encoding="utf-8")
32+
(directory / "summary.json").write_text(json.dumps([
33+
{"chunking": "sentence", "retriever": "dense", "latency_ms_p95": 19},
34+
{"chunking": "fixed", "retriever": "bm25_stem5", "latency_ms_p95": 4},
35+
]), encoding="utf-8")
36+
return directory
37+
38+
39+
def curve(points):
40+
return [{"threshold": t, "coverage": c, "selective_accuracy": a,
41+
"answered": n, "escalated": 58 - n} for t, c, a, n in points]
42+
43+
44+
def test_both_themes_are_written(results, tmp_path):
45+
for theme in ("light", "dark"):
46+
out = tmp_path / f"{theme}.png"
47+
charts.intervals_chart(results, out, theme)
48+
assert out.stat().st_size > 5000, "a blank PNG would be much smaller"
49+
50+
51+
def test_unknown_theme_is_rejected(results, tmp_path):
52+
with pytest.raises(KeyError):
53+
charts.intervals_chart(results, tmp_path / "x.png", "solarized")
54+
55+
56+
def test_thin_thresholds_are_left_out(results, tmp_path):
57+
# One answered query puts a point at exactly 0.0 or 1.0 accuracy; plotting
58+
# it would draw a cliff that is one question wide.
59+
(results / "abstain_curve_score.json").write_text(json.dumps(curve([
60+
(0.0, 1.0, 0.47, 58),
61+
(0.5, 0.40, 0.70, 23),
62+
(0.9, 0.02, 1.00, 1),
63+
])), encoding="utf-8")
64+
65+
out = charts.abstention_chart(results, tmp_path / "abstention.png", "light")
66+
assert out.exists()
67+
68+
69+
def test_a_curve_of_only_thin_points_draws_no_line(results, tmp_path):
70+
(results / "abstain_curve_score.json").write_text(
71+
json.dumps(curve([(0.9, 0.02, 1.0, 1)])), encoding="utf-8")
72+
# Still produces a labelled, empty chart rather than raising.
73+
assert charts.abstention_chart(results, tmp_path / "empty.png", "light").exists()
74+
75+
76+
def test_run_without_results_explains_itself(tmp_path, capsys):
77+
import argparse
78+
79+
parser = argparse.ArgumentParser()
80+
charts.add_arguments(parser)
81+
args = parser.parse_args([])
82+
args.results, args.out = tmp_path, tmp_path / "out"
83+
84+
assert charts.run(args) == 1
85+
assert "turkish-rag-eval run" in capsys.readouterr().out
86+
87+
88+
def test_run_writes_charts_for_every_theme(results, tmp_path):
89+
import argparse
90+
91+
parser = argparse.ArgumentParser()
92+
charts.add_arguments(parser)
93+
args = parser.parse_args([])
94+
args.results, args.out = results, tmp_path / "out"
95+
96+
assert charts.run(args) == 0
97+
written = sorted(p.name for p in (tmp_path / "out").glob("*.png"))
98+
assert written == ["ndcg-intervals-dark.png", "ndcg-intervals.png"]

0 commit comments

Comments
 (0)