Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions docs/troubleshooting.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,18 @@

This page lists common issues and the first checks to make.

## Leaderboard rate limits

Without pyarrow, whichllm reads the archived Open LLM Leaderboard in pages of
100 rows. It retries HTTP 429 responses before stopping. If an earlier page
succeeded, it keeps the scores collected so far and writes a warning to stderr
with the failed offset and retained score count. A failure on the first page
still fails that source; the other benchmark sources can continue.

Partial results use the same 24-hour benchmark cache as a complete fetch. The
cache does not record which pages were missing, and later cache reads do not
repeat the warning. Run `whichllm --refresh` to fetch the benchmark sources again.

## No GPU detected

Run:
Expand Down
14 changes: 14 additions & 0 deletions src/whichllm/models/benchmark_sources/open_llm_leaderboard.py
Original file line number Diff line number Diff line change
@@ -1,11 +1,14 @@
from __future__ import annotations

import io
import logging

import httpx

from whichllm.models.http import get_with_retries

logger = logging.getLogger(__name__)

LEADERBOARD_PARQUET_URL = (
"https://huggingface.co/api/datasets/open-llm-leaderboard/contents"
"/parquet/default/train/0.parquet"
Expand Down Expand Up @@ -53,6 +56,9 @@ async def _fetch_leaderboard_api(client: httpx.AsyncClient) -> dict[str, float]:
resp = await get_with_retries(
client,
LEADERBOARD_ROWS_URL,
attempts=5,
base_delay=1.0,
max_delay=30.0,
params={
"dataset": LEADERBOARD_DATASET,
"config": "default",
Expand All @@ -61,6 +67,14 @@ async def _fetch_leaderboard_api(client: httpx.AsyncClient) -> dict[str, float]:
"length": "100",
},
)
if resp.status_code == 429 and scores:
logger.warning(
"Open LLM Leaderboard rate-limited at offset %d; "
"using %d scores fetched so far",
offset,
len(scores),
)
return scores
resp.raise_for_status()
data = resp.json()
rows = data.get("rows", [])
Expand Down
35 changes: 33 additions & 2 deletions src/whichllm/models/http.py
Original file line number Diff line number Diff line change
@@ -1,14 +1,39 @@
from __future__ import annotations

import asyncio
import math
import random
from datetime import datetime, timezone
from email.utils import parsedate_to_datetime

import httpx

RETRYABLE_STATUS_CODES = {408, 429, 500, 502, 503, 504}
DEFAULT_ACCEPT_ENCODING = "gzip, deflate"


def _retry_after_delay(response: httpx.Response) -> float | None:
"""Return the server-requested retry delay, if it is valid."""
value = response.headers.get("Retry-After")
if not value:
return None

try:
delay = float(value)
except ValueError:
try:
retry_at = parsedate_to_datetime(value)
except (TypeError, ValueError, OverflowError):
return None
if retry_at.tzinfo is None:
retry_at = retry_at.replace(tzinfo=timezone.utc)
delay = (retry_at - datetime.now(timezone.utc)).total_seconds()

if not math.isfinite(delay):
return None
return max(0.0, delay)


async def get_with_retries(
client: httpx.AsyncClient,
url: str,
Expand All @@ -25,6 +50,7 @@ async def get_with_retries(
last_attempt = max(1, attempts) - 1

for attempt in range(last_attempt + 1):
retry_after = None
try:
response = await client.get(url, **kwargs)
except (httpx.TimeoutException, httpx.TransportError):
Expand All @@ -33,9 +59,14 @@ async def get_with_retries(
else:
if response.status_code not in retry_codes or attempt >= last_attempt:
return response
if response.status_code == 429:
retry_after = _retry_after_delay(response)

delay = min(max_delay, base_delay * (2**attempt))
if jitter > 0:
if retry_after is not None:
delay = min(max_delay, retry_after)
else:
delay = min(max_delay, base_delay * (2**attempt))
if jitter > 0 and retry_after is None:
delay += random.uniform(0, jitter)
if delay > 0:
await asyncio.sleep(delay)
Expand Down
88 changes: 88 additions & 0 deletions tests/test_http.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,10 @@
import asyncio
from datetime import datetime, timezone

import httpx
import pytest

import whichllm.models.http as http_helpers
from whichllm.models.benchmark_sources.chatbot_arena import fetch_arena_scores
from whichllm.models.http import get_with_retries

Expand Down Expand Up @@ -40,6 +42,92 @@ async def run() -> httpx.Response:
assert sleeps == [0.01, 0.02]


def test_get_with_retries_honors_retry_after(monkeypatch):
calls = 0
sleeps: list[float] = []

async def fake_sleep(delay: float) -> None:
sleeps.append(delay)

def handler(request: httpx.Request) -> httpx.Response:
nonlocal calls
calls += 1
if calls == 1:
return httpx.Response(
429,
headers={"Retry-After": "3"},
request=request,
)
return httpx.Response(200, request=request)

async def run() -> httpx.Response:
monkeypatch.setattr("whichllm.models.http.asyncio.sleep", fake_sleep)
transport = httpx.MockTransport(handler)
async with httpx.AsyncClient(transport=transport) as client:
return await get_with_retries(
client,
"https://example.test/models",
max_delay=10.0,
jitter=0,
)

response = asyncio.run(run())

assert response.status_code == 200
assert calls == 2
assert sleeps == [3.0]


@pytest.mark.parametrize(
"header,expected_delay",
[
("Thu, 01 Jan 2026 00:00:03 GMT", 3.0),
("Thu, 01 Jan 2026 00:01:00 GMT", 10.0),
("Wed, 31 Dec 2025 23:59:59 GMT", 0.0),
("120", 10.0),
("invalid", 1.0),
("NaN", 1.0),
("Infinity", 1.0),
("-Infinity", 1.0),
],
)
def test_retry_after_date_invalid_values_and_cap(monkeypatch, header, expected_delay):
class FixedDatetime(datetime):
@classmethod
def now(cls, tz=None):
return datetime(2026, 1, 1, tzinfo=timezone.utc)

sleeps = []
calls = 0

async def fake_sleep(delay):
sleeps.append(delay)

def handler(request):
nonlocal calls
calls += 1
if calls == 1:
return httpx.Response(429, headers={"Retry-After": header}, request=request)
return httpx.Response(200, request=request)

async def run():
monkeypatch.setattr(http_helpers, "datetime", FixedDatetime)
monkeypatch.setattr(http_helpers.asyncio, "sleep", fake_sleep)
async with httpx.AsyncClient(transport=httpx.MockTransport(handler)) as client:
return await get_with_retries(
client,
"https://example.test/models",
attempts=2,
base_delay=1.0,
max_delay=10.0,
jitter=0,
)

assert asyncio.run(run()).status_code == 200
assert calls == 2
assert sleeps == ([expected_delay] if expected_delay else [])


def test_benchmark_source_retries_429_before_final_failure(monkeypatch):
calls = 0

Expand Down
Loading
Loading