|
| 1 | +"""Render the two charts the README leads with. |
| 2 | +
|
| 3 | +A benchmark repository with no picture in it asks every reader to reconstruct |
| 4 | +the finding from a twelve-row table. These two carry the findings that a table |
| 5 | +hides: |
| 6 | +
|
| 7 | +``ndcg-intervals`` every configuration with its 95% bootstrap interval, with |
| 8 | + the ones the data cannot separate from the best picked out. |
| 9 | + The interval is the point: the ranking looks decisive and |
| 10 | + mostly is not. |
| 11 | +
|
| 12 | +``abstention`` coverage against selective accuracy for both confidence |
| 13 | + signals. This is the repository's least expected result - |
| 14 | + the intuitive signal (top-1 margin) is useless on fused |
| 15 | + rankings, because RRF scores are 1/(60+rank) and the |
| 16 | + top-two gap is about 2% for every query, confident or not. |
| 17 | +
|
| 18 | +Both are written in light and dark variants, because a light-background PNG in |
| 19 | +a dark README is a glare rectangle; the README pairs them with <picture>. |
| 20 | +
|
| 21 | +Colours come from the reference palette and were validated for contrast and |
| 22 | +colour-vision separation on both surfaces before use. The emphasis chart is |
| 23 | +one hue plus grey rather than a colour per bar: the reader's question is |
| 24 | +"which of these are actually different", and that is a two-group question. |
| 25 | +""" |
| 26 | + |
| 27 | +import argparse |
| 28 | +import json |
| 29 | +from pathlib import Path |
| 30 | + |
| 31 | +from .paths import ROOT |
| 32 | +from .report import DEFAULT_METRIC, load_configurations, recommend |
| 33 | + |
| 34 | +#: Points resting on a handful of answered queries swing between 0 and 1 on |
| 35 | +#: one question. abstain.pick_threshold ignores them for the same reason. |
| 36 | +MIN_ANSWERED = 5 |
| 37 | + |
| 38 | +THEMES = { |
| 39 | + "light": { |
| 40 | + "surface": "#fcfcfb", |
| 41 | + "ink": "#0b0b0b", |
| 42 | + "secondary": "#52514e", |
| 43 | + "muted": "#898781", |
| 44 | + "grid": "#e1e0d9", |
| 45 | + "axis": "#c3c2b7", |
| 46 | + "accent": "#2a78d6", |
| 47 | + "second": "#eb6834", |
| 48 | + "recessive": "#898781", |
| 49 | + }, |
| 50 | + "dark": { |
| 51 | + "surface": "#1a1a19", |
| 52 | + "ink": "#ffffff", |
| 53 | + "secondary": "#c3c2b7", |
| 54 | + "muted": "#898781", |
| 55 | + "grid": "#2c2c2a", |
| 56 | + "axis": "#383835", |
| 57 | + "accent": "#3987e5", |
| 58 | + "second": "#d95926", |
| 59 | + "recessive": "#898781", |
| 60 | + }, |
| 61 | +} |
| 62 | + |
| 63 | + |
| 64 | +def _figure(theme: dict, size: tuple[float, float]): |
| 65 | + import matplotlib.pyplot as plt |
| 66 | + |
| 67 | + figure, axes = plt.subplots(figsize=size) |
| 68 | + figure.patch.set_facecolor(theme["surface"]) |
| 69 | + axes.set_facecolor(theme["surface"]) |
| 70 | + for side in ("top", "right"): |
| 71 | + axes.spines[side].set_visible(False) |
| 72 | + for side in ("left", "bottom"): |
| 73 | + axes.spines[side].set_color(theme["axis"]) |
| 74 | + axes.spines[side].set_linewidth(1) |
| 75 | + axes.tick_params(colors=theme["muted"], labelsize=9, length=0) |
| 76 | + return figure, axes |
| 77 | + |
| 78 | + |
| 79 | +def _titles(axes, theme: dict, title: str, subtitle: str) -> None: |
| 80 | + """Title and subtitle stacked above the axes. |
| 81 | +
|
| 82 | + Offset in *points* from the axes corner rather than in axes fractions: the |
| 83 | + interval chart's height grows with the number of configurations, so a |
| 84 | + fractional offset would drift away from the title as rows are added, and a |
| 85 | + fixed title pad would collide with it. |
| 86 | + """ |
| 87 | + subtitle_lines = subtitle.count("\n") + 1 |
| 88 | + axes.annotate(subtitle, xy=(0, 1), xycoords="axes fraction", |
| 89 | + xytext=(0, 10), textcoords="offset points", |
| 90 | + ha="left", va="bottom", fontsize=9.5, |
| 91 | + color=theme["secondary"], linespacing=1.45) |
| 92 | + axes.annotate(title, xy=(0, 1), xycoords="axes fraction", |
| 93 | + xytext=(0, 10 + 15 * subtitle_lines + 8), |
| 94 | + textcoords="offset points", ha="left", va="bottom", |
| 95 | + fontsize=13.5, fontweight="bold", color=theme["ink"]) |
| 96 | + |
| 97 | + |
| 98 | +def intervals_chart(results_dir: Path, out: Path, theme_name: str, |
| 99 | + metric: str = DEFAULT_METRIC) -> Path: |
| 100 | + """Every configuration, sorted, with its interval; the tied ones picked out.""" |
| 101 | + theme = THEMES[theme_name] |
| 102 | + configurations = load_configurations(results_dir) |
| 103 | + recommendation = recommend(configurations, metric) |
| 104 | + tied = {id(c) for c in recommendation.tied_with_best} |
| 105 | + |
| 106 | + rows = sorted(configurations, key=lambda c: c.mean(metric)) |
| 107 | + figure, axes = _figure(theme, (8.4, 0.44 * len(rows) + 2.3)) |
| 108 | + |
| 109 | + for position, configuration in enumerate(rows): |
| 110 | + low, high = configuration.interval(metric) |
| 111 | + mean = configuration.mean(metric) |
| 112 | + emphasised = id(configuration) in tied |
| 113 | + colour = theme["accent"] if emphasised else theme["recessive"] |
| 114 | + |
| 115 | + axes.plot([low, high], [position, position], color=colour, linewidth=2, |
| 116 | + solid_capstyle="round", zorder=2, |
| 117 | + alpha=1.0 if emphasised else 0.55) |
| 118 | + axes.plot([mean], [position], marker="o", markersize=8, color=colour, |
| 119 | + markeredgecolor=theme["surface"], markeredgewidth=2, zorder=3, |
| 120 | + alpha=1.0 if emphasised else 0.55) |
| 121 | + axes.text(high + 0.012, position, f"{mean:.3f}", va="center", |
| 122 | + fontsize=9, color=theme["ink"] if emphasised |
| 123 | + else theme["muted"]) |
| 124 | + |
| 125 | + axes.set_yticks(range(len(rows))) |
| 126 | + axes.set_yticklabels([c.name for c in rows], fontsize=9.5, |
| 127 | + color=theme["secondary"]) |
| 128 | + axes.set_xlabel(metric, color=theme["secondary"], fontsize=9.5, labelpad=8) |
| 129 | + axes.set_xlim(0.22, 0.82) |
| 130 | + axes.set_ylim(-0.7, len(rows) - 0.3) |
| 131 | + axes.xaxis.grid(True, color=theme["grid"], linewidth=1) |
| 132 | + axes.set_axisbelow(True) |
| 133 | + |
| 134 | + _titles(axes, theme, |
| 135 | + "Most of this ranking is not a ranking", |
| 136 | + f"{metric} with 95% bootstrap intervals over " |
| 137 | + f"{len(recommendation.best.scored)} queries. Highlighted: the " |
| 138 | + f"configurations a paired\nbootstrap cannot separate from the best.") |
| 139 | + |
| 140 | + # A two-group emphasis chart still needs its groups named. |
| 141 | + from matplotlib.lines import Line2D |
| 142 | + axes.legend( |
| 143 | + handles=[ |
| 144 | + Line2D([], [], color=theme["accent"], linewidth=2, marker="o", |
| 145 | + markersize=7, label="within noise of the best"), |
| 146 | + Line2D([], [], color=theme["recessive"], alpha=0.55, linewidth=2, |
| 147 | + marker="o", markersize=7, label="measurably worse"), |
| 148 | + ], |
| 149 | + loc="lower right", frameon=False, fontsize=9, |
| 150 | + labelcolor=theme["secondary"]) |
| 151 | + |
| 152 | + return _save(figure, out) |
| 153 | + |
| 154 | + |
| 155 | +def abstention_chart(results_dir: Path, out: Path, theme_name: str) -> Path: |
| 156 | + """Coverage against selective accuracy, one line per confidence signal.""" |
| 157 | + theme = THEMES[theme_name] |
| 158 | + figure, axes = _figure(theme, (7.6, 5.0)) |
| 159 | + |
| 160 | + series = [ |
| 161 | + ("score", "dense top-1 cosine", theme["accent"]), |
| 162 | + ("margin", "top-1 margin", theme["second"]), |
| 163 | + ] |
| 164 | + plotted = 0 |
| 165 | + for signal, label, colour in series: |
| 166 | + path = results_dir / f"abstain_curve_{signal}.json" |
| 167 | + if not path.exists(): |
| 168 | + continue |
| 169 | + curve = [p for p in json.loads(path.read_text(encoding="utf-8")) |
| 170 | + if p["answered"] >= MIN_ANSWERED |
| 171 | + and p["selective_accuracy"] is not None] |
| 172 | + if not curve: |
| 173 | + continue |
| 174 | + coverage = [p["coverage"] for p in curve] |
| 175 | + accuracy = [p["selective_accuracy"] for p in curve] |
| 176 | + |
| 177 | + axes.plot(coverage, accuracy, color=colour, linewidth=2, marker="o", |
| 178 | + markersize=6, markeredgecolor=theme["surface"], |
| 179 | + markeredgewidth=1.5, label=label, zorder=3) |
| 180 | + # Direct label at the low-coverage end, where the two lines separate. |
| 181 | + leftmost = min(range(len(coverage)), key=lambda i: coverage[i]) |
| 182 | + axes.annotate(label, (coverage[leftmost], accuracy[leftmost]), |
| 183 | + textcoords="offset points", xytext=(8, 8), |
| 184 | + fontsize=9.5, color=theme["ink"], fontweight="bold") |
| 185 | + plotted += 1 |
| 186 | + |
| 187 | + axes.set_xlabel("coverage - share of queries answered automatically", |
| 188 | + color=theme["secondary"], fontsize=9.5, labelpad=8) |
| 189 | + axes.set_ylabel("selective accuracy - correct among those answered", |
| 190 | + color=theme["secondary"], fontsize=9.5, labelpad=8) |
| 191 | + axes.set_xlim(0, 1.03) |
| 192 | + axes.set_ylim(0, 1.03) |
| 193 | + axes.grid(True, color=theme["grid"], linewidth=1) |
| 194 | + axes.set_axisbelow(True) |
| 195 | + if plotted > 1: |
| 196 | + axes.legend(loc="lower left", frameon=False, fontsize=9, |
| 197 | + labelcolor=theme["secondary"]) |
| 198 | + |
| 199 | + _titles(axes, theme, |
| 200 | + "The intuitive confidence signal is the useless one", |
| 201 | + f"Thresholds keeping at least {MIN_ANSWERED} answered queries. RRF " |
| 202 | + f"fuses ranks as 1/(60+rank), so the top-two\ngap is ~2% on every " |
| 203 | + f"query and the margin buys almost no coverage to trade.") |
| 204 | + return _save(figure, out) |
| 205 | + |
| 206 | + |
| 207 | +def _save(figure, out: Path) -> Path: |
| 208 | + import matplotlib.pyplot as plt |
| 209 | + |
| 210 | + out.parent.mkdir(parents=True, exist_ok=True) |
| 211 | + figure.savefig(out, dpi=200, bbox_inches="tight", |
| 212 | + facecolor=figure.get_facecolor()) |
| 213 | + plt.close(figure) |
| 214 | + return out |
| 215 | + |
| 216 | + |
| 217 | +def add_arguments(parser: argparse.ArgumentParser) -> None: |
| 218 | + parser.add_argument("--results", type=Path, default=ROOT / "results") |
| 219 | + parser.add_argument("--out", type=Path, default=ROOT / "docs" / "charts") |
| 220 | + parser.add_argument("--metric", default=DEFAULT_METRIC) |
| 221 | + |
| 222 | + |
| 223 | +def run(args) -> int: |
| 224 | + if not (args.results / "summary.json").exists(): |
| 225 | + print(f"{args.results} holds no results - run 'turkish-rag-eval run' first") |
| 226 | + return 1 |
| 227 | + |
| 228 | + written = [] |
| 229 | + for theme_name in THEMES: |
| 230 | + suffix = "" if theme_name == "light" else "-dark" |
| 231 | + written.append(intervals_chart( |
| 232 | + args.results, args.out / f"ndcg-intervals{suffix}.png", |
| 233 | + theme_name, args.metric)) |
| 234 | + if (args.results / "abstain_curve_score.json").exists(): |
| 235 | + written.append(abstention_chart( |
| 236 | + args.results, args.out / f"abstention{suffix}.png", theme_name)) |
| 237 | + |
| 238 | + for path in written: |
| 239 | + print(f" {path.relative_to(ROOT) if ROOT in path.parents else path}") |
| 240 | + print(f"{len(written)} chart(s) written") |
| 241 | + return 0 |
| 242 | + |
| 243 | + |
| 244 | +def main() -> int: |
| 245 | + parser = argparse.ArgumentParser(description=__doc__) |
| 246 | + add_arguments(parser) |
| 247 | + return run(parser.parse_args()) |
| 248 | + |
| 249 | + |
| 250 | +if __name__ == "__main__": |
| 251 | + raise SystemExit(main()) |
0 commit comments