From 76ca7f3499e2f2d8087468eb0a1f81562b35fdfb Mon Sep 17 00:00:00 2001 From: hannibal-lee <47944281+hannibal-lee@users.noreply.github.com> Date: Sun, 20 Sep 2026 06:51:14 -0400 Subject: [PATCH 1/2] fix(benchmark): preserve leaderboard scores on rate limits --- .../benchmark_sources/open_llm_leaderboard.py | 14 ++ src/whichllm/models/http.py | 32 ++++- tests/test_http.py | 36 +++++ tests/test_open_llm_leaderboard.py | 123 ++++++++++++++++++ 4 files changed, 203 insertions(+), 2 deletions(-) create mode 100644 tests/test_open_llm_leaderboard.py diff --git a/src/whichllm/models/benchmark_sources/open_llm_leaderboard.py b/src/whichllm/models/benchmark_sources/open_llm_leaderboard.py index 1c616ba..8b7ec51 100644 --- a/src/whichllm/models/benchmark_sources/open_llm_leaderboard.py +++ b/src/whichllm/models/benchmark_sources/open_llm_leaderboard.py @@ -1,11 +1,14 @@ from __future__ import annotations import io +import logging import httpx from whichllm.models.http import get_with_retries +logger = logging.getLogger(__name__) + LEADERBOARD_PARQUET_URL = ( "https://huggingface.co/api/datasets/open-llm-leaderboard/contents" "/parquet/default/train/0.parquet" @@ -53,6 +56,9 @@ async def _fetch_leaderboard_api(client: httpx.AsyncClient) -> dict[str, float]: resp = await get_with_retries( client, LEADERBOARD_ROWS_URL, + attempts=5, + base_delay=1.0, + max_delay=30.0, params={ "dataset": LEADERBOARD_DATASET, "config": "default", @@ -61,6 +67,14 @@ async def _fetch_leaderboard_api(client: httpx.AsyncClient) -> dict[str, float]: "length": "100", }, ) + if resp.status_code == 429 and scores: + logger.warning( + "Open LLM Leaderboard rate-limited at offset %d; " + "using %d scores fetched so far", + offset, + len(scores), + ) + return scores resp.raise_for_status() data = resp.json() rows = data.get("rows", []) diff --git a/src/whichllm/models/http.py b/src/whichllm/models/http.py index a71b076..10337b2 100644 --- a/src/whichllm/models/http.py +++ b/src/whichllm/models/http.py @@ -2,6 +2,8 @@ import asyncio import random +from datetime import datetime, timezone +from email.utils import parsedate_to_datetime import httpx @@ -9,6 +11,26 @@ DEFAULT_ACCEPT_ENCODING = "gzip, deflate" +def _retry_after_delay(response: httpx.Response) -> float | None: + """Return the server-requested retry delay, if it is valid.""" + value = response.headers.get("Retry-After") + if not value: + return None + + try: + delay = float(value) + except ValueError: + try: + retry_at = parsedate_to_datetime(value) + except (TypeError, ValueError, OverflowError): + return None + if retry_at.tzinfo is None: + retry_at = retry_at.replace(tzinfo=timezone.utc) + delay = (retry_at - datetime.now(timezone.utc)).total_seconds() + + return max(0.0, delay) + + async def get_with_retries( client: httpx.AsyncClient, url: str, @@ -25,6 +47,7 @@ async def get_with_retries( last_attempt = max(1, attempts) - 1 for attempt in range(last_attempt + 1): + retry_after = None try: response = await client.get(url, **kwargs) except (httpx.TimeoutException, httpx.TransportError): @@ -33,9 +56,14 @@ async def get_with_retries( else: if response.status_code not in retry_codes or attempt >= last_attempt: return response + if response.status_code == 429: + retry_after = _retry_after_delay(response) - delay = min(max_delay, base_delay * (2**attempt)) - if jitter > 0: + if retry_after is not None: + delay = min(max_delay, retry_after) + else: + delay = min(max_delay, base_delay * (2**attempt)) + if jitter > 0 and retry_after is None: delay += random.uniform(0, jitter) if delay > 0: await asyncio.sleep(delay) diff --git a/tests/test_http.py b/tests/test_http.py index 1210621..04ed797 100644 --- a/tests/test_http.py +++ b/tests/test_http.py @@ -40,6 +40,42 @@ async def run() -> httpx.Response: assert sleeps == [0.01, 0.02] +def test_get_with_retries_honors_retry_after(monkeypatch): + calls = 0 + sleeps: list[float] = [] + + async def fake_sleep(delay: float) -> None: + sleeps.append(delay) + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + if calls == 1: + return httpx.Response( + 429, + headers={"Retry-After": "3"}, + request=request, + ) + return httpx.Response(200, request=request) + + async def run() -> httpx.Response: + monkeypatch.setattr("whichllm.models.http.asyncio.sleep", fake_sleep) + transport = httpx.MockTransport(handler) + async with httpx.AsyncClient(transport=transport) as client: + return await get_with_retries( + client, + "https://example.test/models", + max_delay=10.0, + jitter=0, + ) + + response = asyncio.run(run()) + + assert response.status_code == 200 + assert calls == 2 + assert sleeps == [3.0] + + def test_benchmark_source_retries_429_before_final_failure(monkeypatch): calls = 0 diff --git a/tests/test_open_llm_leaderboard.py b/tests/test_open_llm_leaderboard.py new file mode 100644 index 0000000..82aaf7e --- /dev/null +++ b/tests/test_open_llm_leaderboard.py @@ -0,0 +1,123 @@ +import asyncio +import logging + +import httpx +import pytest + +from whichllm.models.benchmark_sources import open_llm_leaderboard +from whichllm.models.benchmark_sources.open_llm_leaderboard import ( + _fetch_leaderboard_api, + fetch_leaderboard_with_fallback, +) + + +def _rows(offset: int, total: int) -> list[dict[str, dict[str, object]]]: + return [ + { + "row": { + "fullname": f"test/model-{index}", + "Average ⬆️": 26.0, + } + } + for index in range(offset, min(offset + 100, total)) + ] + + +def test_leaderboard_api_fetches_all_46_pages(): + offsets: list[int] = [] + total = 4576 + + def handler(request: httpx.Request) -> httpx.Response: + offset = int(request.url.params["offset"]) + offsets.append(offset) + return httpx.Response( + 200, + json={"rows": _rows(offset, total), "num_rows_total": total}, + request=request, + ) + + async def run() -> dict[str, float]: + transport = httpx.MockTransport(handler) + async with httpx.AsyncClient(transport=transport) as client: + return await _fetch_leaderboard_api(client) + + scores = asyncio.run(run()) + + assert len(scores) == total + assert offsets == list(range(0, 4600, 100)) + + +def test_leaderboard_api_returns_partial_scores_after_exhausted_429( + monkeypatch, caplog +): + offsets: list[int] = [] + + async def fake_sleep(delay: float) -> None: + return None + + def handler(request: httpx.Request) -> httpx.Response: + offset = int(request.url.params["offset"]) + offsets.append(offset) + if offset == 3600: + return httpx.Response(429, request=request) + return httpx.Response( + 200, + json={"rows": _rows(offset, 4576), "num_rows_total": 4576}, + request=request, + ) + + async def run() -> dict[str, float]: + monkeypatch.setattr("whichllm.models.http.asyncio.sleep", fake_sleep) + transport = httpx.MockTransport(handler) + async with httpx.AsyncClient(transport=transport) as client: + return await _fetch_leaderboard_api(client) + + with caplog.at_level(logging.WARNING): + scores = asyncio.run(run()) + + assert len(scores) == 3600 + assert offsets[-5:] == [3600] * 5 + assert "rate-limited at offset 3600" in caplog.text + assert "using 3600 scores fetched so far" in caplog.text + + +def test_leaderboard_api_first_page_429_still_fails(monkeypatch): + calls = 0 + + async def fake_sleep(delay: float) -> None: + return None + + def handler(request: httpx.Request) -> httpx.Response: + nonlocal calls + calls += 1 + return httpx.Response(429, request=request) + + async def run() -> None: + monkeypatch.setattr("whichllm.models.http.asyncio.sleep", fake_sleep) + transport = httpx.MockTransport(handler) + async with httpx.AsyncClient(transport=transport) as client: + with pytest.raises(httpx.HTTPStatusError): + await _fetch_leaderboard_api(client) + + asyncio.run(run()) + + assert calls == 5 + + +def test_leaderboard_falls_back_to_rows_when_pyarrow_is_unavailable(monkeypatch): + expected = {"test/model": 42.0} + + async def unavailable(client: httpx.AsyncClient) -> dict[str, float]: + raise ImportError("pyarrow is unavailable") + + async def rows(client: httpx.AsyncClient) -> dict[str, float]: + return expected + + monkeypatch.setattr(open_llm_leaderboard, "_fetch_leaderboard_parquet", unavailable) + monkeypatch.setattr(open_llm_leaderboard, "_fetch_leaderboard_api", rows) + + async def run() -> dict[str, float]: + async with httpx.AsyncClient() as client: + return await fetch_leaderboard_with_fallback(client) + + assert asyncio.run(run()) == expected From 0cf1744121622cb79249e03692a2e6e754d6a22d Mon Sep 17 00:00:00 2001 From: andy Date: Sat, 3 Oct 2026 20:17:42 +0900 Subject: [PATCH 2/2] fix: validate retry delays and document partial benchmark caching --- docs/troubleshooting.md | 12 +++++++ src/whichllm/models/http.py | 3 ++ tests/test_http.py | 52 ++++++++++++++++++++++++++++ tests/test_open_llm_leaderboard.py | 55 ++++++++++++++++++++++++++++++ 4 files changed, 122 insertions(+) diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index f2941be..c928d4a 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -2,6 +2,18 @@ This page lists common issues and the first checks to make. +## Leaderboard rate limits + +Without pyarrow, whichllm reads the archived Open LLM Leaderboard in pages of +100 rows. It retries HTTP 429 responses before stopping. If an earlier page +succeeded, it keeps the scores collected so far and writes a warning to stderr +with the failed offset and retained score count. A failure on the first page +still fails that source; the other benchmark sources can continue. + +Partial results use the same 24-hour benchmark cache as a complete fetch. The +cache does not record which pages were missing, and later cache reads do not +repeat the warning. Run `whichllm --refresh` to fetch the benchmark sources again. + ## No GPU detected Run: diff --git a/src/whichllm/models/http.py b/src/whichllm/models/http.py index 10337b2..50d6c82 100644 --- a/src/whichllm/models/http.py +++ b/src/whichllm/models/http.py @@ -1,6 +1,7 @@ from __future__ import annotations import asyncio +import math import random from datetime import datetime, timezone from email.utils import parsedate_to_datetime @@ -28,6 +29,8 @@ def _retry_after_delay(response: httpx.Response) -> float | None: retry_at = retry_at.replace(tzinfo=timezone.utc) delay = (retry_at - datetime.now(timezone.utc)).total_seconds() + if not math.isfinite(delay): + return None return max(0.0, delay) diff --git a/tests/test_http.py b/tests/test_http.py index 04ed797..8d62173 100644 --- a/tests/test_http.py +++ b/tests/test_http.py @@ -1,8 +1,10 @@ import asyncio +from datetime import datetime, timezone import httpx import pytest +import whichllm.models.http as http_helpers from whichllm.models.benchmark_sources.chatbot_arena import fetch_arena_scores from whichllm.models.http import get_with_retries @@ -76,6 +78,56 @@ async def run() -> httpx.Response: assert sleeps == [3.0] +@pytest.mark.parametrize( + "header,expected_delay", + [ + ("Thu, 01 Jan 2026 00:00:03 GMT", 3.0), + ("Thu, 01 Jan 2026 00:01:00 GMT", 10.0), + ("Wed, 31 Dec 2025 23:59:59 GMT", 0.0), + ("120", 10.0), + ("invalid", 1.0), + ("NaN", 1.0), + ("Infinity", 1.0), + ("-Infinity", 1.0), + ], +) +def test_retry_after_date_invalid_values_and_cap(monkeypatch, header, expected_delay): + class FixedDatetime(datetime): + @classmethod + def now(cls, tz=None): + return datetime(2026, 1, 1, tzinfo=timezone.utc) + + sleeps = [] + calls = 0 + + async def fake_sleep(delay): + sleeps.append(delay) + + def handler(request): + nonlocal calls + calls += 1 + if calls == 1: + return httpx.Response(429, headers={"Retry-After": header}, request=request) + return httpx.Response(200, request=request) + + async def run(): + monkeypatch.setattr(http_helpers, "datetime", FixedDatetime) + monkeypatch.setattr(http_helpers.asyncio, "sleep", fake_sleep) + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client: + return await get_with_retries( + client, + "https://example.test/models", + attempts=2, + base_delay=1.0, + max_delay=10.0, + jitter=0, + ) + + assert asyncio.run(run()).status_code == 200 + assert calls == 2 + assert sleeps == ([expected_delay] if expected_delay else []) + + def test_benchmark_source_retries_429_before_final_failure(monkeypatch): calls = 0 diff --git a/tests/test_open_llm_leaderboard.py b/tests/test_open_llm_leaderboard.py index 82aaf7e..75d214e 100644 --- a/tests/test_open_llm_leaderboard.py +++ b/tests/test_open_llm_leaderboard.py @@ -121,3 +121,58 @@ async def run() -> dict[str, float]: return await fetch_leaderboard_with_fallback(client) assert asyncio.run(run()) == expected + + +def test_partial_leaderboard_survives_aggregation_and_cache( + monkeypatch, tmp_path, caplog +): + from whichllm.models import benchmark_cache, benchmark_sources + from whichllm.models.benchmark_fetch import fetch_benchmark_scores + + async def fake_sleep(delay): + return None + + def handler(request): + offset = int(request.url.params["offset"]) + if offset: + return httpx.Response(429, request=request) + return httpx.Response( + 200, + json={"rows": _rows(0, 200), "num_rows_total": 200}, + request=request, + ) + + async def partial_source(client): + async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as mocked: + return await _fetch_leaderboard_api(mocked) + + async def empty_source(client): + return {} + + monkeypatch.setattr("whichllm.models.http.asyncio.sleep", fake_sleep) + monkeypatch.setattr( + benchmark_sources, "fetch_leaderboard_with_fallback", partial_source + ) + for source in ( + "fetch_arena_scores", + "fetch_aa_index_scores", + "fetch_aider_polyglot_scores", + "fetch_vision_scores", + ): + monkeypatch.setattr(benchmark_sources, source, empty_source) + monkeypatch.setattr( + benchmark_sources, "get_livebench_data", lambda: {"current/Model-7B": 88.0} + ) + monkeypatch.setattr(benchmark_cache, "CACHE_DIR", tmp_path) + monkeypatch.setattr(benchmark_cache, "BENCHMARK_CACHE", tmp_path / "benchmark.json") + + with caplog.at_level(logging.WARNING): + scores = asyncio.run(fetch_benchmark_scores()) + + assert scores["test/model-0"] > 0 + assert scores["test/model-99"] > 0 + assert "test/model-100" not in scores + assert scores["current/Model-7B"] == 88.0 + assert "using 100 scores fetched so far" in caplog.text + benchmark_cache.save_benchmark_cache(scores) + assert benchmark_cache.load_benchmark_cache() == scores