Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
16 commits
Select commit Hold shift + click to select a range
ed96ff2
fix: replace deprecated errors='ignore' in pd.to_numeric with try/except
yasumorishima Feb 4, 2026
a2afe99
fix: replace deprecated errors='ignore' in pd.to_datetime with try/ex…
yasumorishima Feb 4, 2026
a6d31c4
Fix FutureWarning in team_results.py (#459)
yasumorishima Feb 4, 2026
8c04cb6
test: add regression test for Unknown attendance to NaN conversion (#…
yasumorishima Jun 15, 2026
8f7dd75
Fix deprecated GitHub authentication in retrosheet.py (#455)
yasumorishima Feb 4, 2026
29de310
test: add regression tests for retrosheet modern GitHub auth (#455)
yasumorishima Jun 15, 2026
699e32f
Add input validation to team_fielding_bref (#462)
yasumorishima Feb 4, 2026
98f142e
test: add regression test for team_fielding_bref invalid season range…
yasumorishima Jun 15, 2026
e474a6b
Fix team_batting_bref and team_pitching_bref for updated Baseball Ref…
yasumorishima Feb 4, 2026
cab2930
test: add regression tests for team_batting/pitching_bref new BR HTML…
yasumorishima Jun 15, 2026
2c73a5b
Fix team_ids returning empty data for seasons after 2021 (#486)
yasumorishima Feb 4, 2026
3382f8b
test: add regression test for team_ids on seasons after bundled data …
yasumorishima Jun 15, 2026
f1f005c
Fix team_game_logs on pandas 3 (errors=ignore was removed)
yasumorishima Oct 6, 2026
d806aad
Parse date columns with the pandas 3 str dtype in try_parse_dataframe
yasumorishima Oct 6, 2026
2d2b502
Use ATH for the Athletics from 2025 in team_ids (teamIDBR, teamIDretro)
yasumorishima Oct 6, 2026
3234650
Require PyGithub 1.59, the first release with github.Auth (used by re…
yasumorishima Oct 6, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 7 additions & 6 deletions pybaseball/datahelpers/postprocessing.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,14 +30,15 @@ def try_parse_dataframe(

if parse_numerics:
data_copy = coalesce_nulls(data_copy, null_replacement)
data_copy = data_copy.apply(
pd.to_numeric,
errors='ignore',
downcast='signed'
).convert_dtypes(convert_string=False)
for col in data_copy.columns:
try:
data_copy[col] = pd.to_numeric(data_copy[col], downcast='signed')
except (ValueError, TypeError):
pass
data_copy = data_copy.convert_dtypes(convert_string=False)

string_columns = [
dtype_tuple[0] for dtype_tuple in data_copy.dtypes.items() if str(dtype_tuple[1]) in ["object", "string"]
dtype_tuple[0] for dtype_tuple in data_copy.dtypes.items() if str(dtype_tuple[1]) in ["object", "string", "str"]
]
for column in string_columns:
# Only check the first value of the column and test that;
Expand Down
12 changes: 6 additions & 6 deletions pybaseball/retrosheet.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@
from pybaseball.utils import get_text_file
from datetime import datetime
from io import StringIO
from github import Github
from github import Auth, Github
import os
from getpass import getuser, getpass
from github.GithubException import RateLimitExceededException
Expand Down Expand Up @@ -131,7 +131,7 @@ def events(season, type='regular', export_dir='.'):
"the valid types are: 'regular', 'post', and 'asg'.")

try:
g = Github(GH_TOKEN)
g = Github(auth=Auth.Token(GH_TOKEN)) if GH_TOKEN else Github()
repo = g.get_repo('chadwickbureau/retrosheet')
season_folder = [f.path[f.path.rfind('/')+1:] for f in repo.get_contents(f'seasons/{season}')]
season_events = [t for t in season_folder if t.endswith(file_extension)]
Expand All @@ -156,7 +156,7 @@ def rosters(season):
GH_TOKEN=os.getenv('GH_TOKEN', '')

try:
g = Github(GH_TOKEN)
g = Github(auth=Auth.Token(GH_TOKEN)) if GH_TOKEN else Github()
repo = g.get_repo('chadwickbureau/retrosheet')
season_folder = [f.path[f.path.rfind('/')+1:] for f in repo.get_contents(f'seasons/{season}')]
rosters = [t for t in season_folder if t.endswith('.ROS')]
Expand All @@ -179,7 +179,7 @@ def _roster(team, season, checked = False):
GH_TOKEN=os.getenv('GH_TOKEN', '')

if not checked:
g = Github(GH_TOKEN)
g = Github(auth=Auth.Token(GH_TOKEN)) if GH_TOKEN else Github()
try:
repo = g.get_repo('chadwickbureau/retrosheet')
season_folder = [f.path[f.path.rfind('/')+1:] for f in repo.get_contents(f'seasons/{season}')]
Expand Down Expand Up @@ -213,7 +213,7 @@ def schedules(season):
"""
GH_TOKEN=os.getenv('GH_TOKEN', '')
# validate input
g = Github(GH_TOKEN)
g = Github(auth=Auth.Token(GH_TOKEN)) if GH_TOKEN else Github()
repo = g.get_repo('chadwickbureau/retrosheet')
season_folder = [f.path[f.path.rfind('/')+1:] for f in repo.get_contents(f'seasons/{season}')]
file_name = f'{season}schedule.csv'
Expand All @@ -231,7 +231,7 @@ def season_game_logs(season):
"""
GH_TOKEN=os.getenv('GH_TOKEN', '')
# validate input
g = Github(GH_TOKEN)
g = Github(auth=Auth.Token(GH_TOKEN)) if GH_TOKEN else Github()
repo = g.get_repo('chadwickbureau/retrosheet')
season_folder = [f.path[f.path.rfind('/')+1:] for f in repo.get_contents(f'seasons/{season}')]
gamelog_file_name = f'GL{season}.TXT'
Expand Down
9 changes: 7 additions & 2 deletions pybaseball/team_batting.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,10 +40,15 @@ def team_batting_bref(team: str, start_season: int, end_season: Optional[int]=No
response = session.get(stats_url)
soup = BeautifulSoup(response.content, 'html.parser')

table = soup.find_all('table', {'class': 'sortable stats_table'})[0]
table = soup.find('table', {'id': 'players_standard_batting'})
thead = table.find('thead') if table is not None else None
if table is None or thead is None:
raise ValueError(
"Could not find batting data for {} {}. The page structure may have changed.".format(team, season)
)

if headings is None:
headings = [row.text.strip() for row in table.find_all('th')[1:28]]
headings = [th.text.strip() for th in thead.find_all('th')[1:]]

rows = table.find_all('tr')
for row in rows:
Expand Down
5 changes: 5 additions & 0 deletions pybaseball/team_fielding.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,12 @@ def team_fielding_bref(team: str, start_season: int, end_season: Optional[int]=N
)
if end_season is None:
end_season = start_season
if end_season < start_season:
raise ValueError(
"end_season must be greater than or equal to start_season."
)

team = team.upper()
url = "https://www.baseball-reference.com/teams/{}".format(team)

raw_data = []
Expand Down
10 changes: 9 additions & 1 deletion pybaseball/team_game_logs.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,14 @@ def get_table(season: int, team: str, log_type: str) -> pd.DataFrame:
return data


def _to_numeric_or_keep(column: pd.Series) -> pd.Series:
# pandas 3 removed errors="ignore" from pd.to_numeric
try:
return pd.to_numeric(column)
except (ValueError, TypeError):
return column


def postprocess(data: pd.DataFrame) -> pd.DataFrame:
#print(data.columns)
data.drop([('Unnamed: 0_level_0', 'Rk')], axis=1, inplace=True) # drop index column
Expand All @@ -36,7 +44,7 @@ def postprocess(data: pd.DataFrame) -> pd.DataFrame:
data = data.rename(columns= repl_dict).copy()
data[('Unnamed: 3_level_0','Home')] = data[('Unnamed: 3_level_0','Home')].isnull() # '@' if away, empty if home
data = data[data[('Unnamed: 1_level_0','Game')] != 'Gtm'].copy() # drop empty month rows
data = data.apply(pd.to_numeric, errors="ignore")
data = data.apply(_to_numeric_or_keep)
data[('Unnamed: 1_level_0','Game')] = data[('Unnamed: 1_level_0','Game')].astype(int)
return data.reset_index(drop=True)

Expand Down
9 changes: 7 additions & 2 deletions pybaseball/team_pitching.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,10 +42,15 @@ def team_pitching_bref(team: str, start_season: int, end_season: Optional[int]=N
response = session.get(stats_url)
soup = BeautifulSoup(response.content, 'html.parser')

table = soup.find_all('table', {'id': 'team_pitching'})[0]
table = soup.find('table', {'id': 'players_standard_pitching'})
thead = table.find('thead') if table is not None else None
if table is None or thead is None:
raise ValueError(
"Could not find pitching data for {} {}. The page structure may have changed.".format(team, season)
)

if headings is None:
headings = [row.text.strip() for row in table.find_all('th')[1:34]]
headings = [th.text.strip() for th in thead.find_all('th')[1:]]

rows = table.find_all('tr')
for row in rows:
Expand Down
2 changes: 1 addition & 1 deletion pybaseball/team_results.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ def get_table(soup: BeautifulSoup, team: str) -> pd.DataFrame:
df = df.rename(columns=df.iloc[0])
df = df.reindex(df.index.drop(0))
df = df.drop('', axis=1) #not a useful column
df['Attendance'].replace(r'^Unknown$', np.nan, regex=True, inplace = True) # make this a NaN so the column can benumeric
df['Attendance'] = df['Attendance'].replace(r'^Unknown$', np.nan, regex=True) # make this a NaN so the column can be numeric
return df

def process_win_streak(data: pd.DataFrame) -> pd.DataFrame:
Expand Down
14 changes: 14 additions & 0 deletions pybaseball/teamid_lookup.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,20 @@ def team_ids(season: Optional[int] = None, league: str = 'ALL') -> pd.DataFrame:

fg_team_data = pd.read_csv(_DATA_FILENAME, index_col=0)

max_year = int(fg_team_data['yearID'].max())

# If the requested season is beyond the data, extrapolate from the last known year.
# The 30 franchises are unchanged since 2021 (when the bundled data ends), but the
# Athletics moved to Sacramento in 2025 and Baseball Reference and Retrosheet
# list them as ATH from that season on.
if season is not None and season > max_year:
last_year_data = fg_team_data[fg_team_data['yearID'] == max_year].copy()
last_year_data['yearID'] = season
if season >= 2025:
athletics = last_year_data['franchID'] == 'OAK'
last_year_data.loc[athletics, ['teamIDBR', 'teamIDretro']] = 'ATH'
fg_team_data = pd.concat([fg_team_data, last_year_data], ignore_index=True)

if season is not None:
fg_team_data = fg_team_data.query(f"yearID == {season}")

Expand Down
2 changes: 1 addition & 1 deletion setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -87,7 +87,7 @@
'requests>=2.18.1',
'lxml>=4.2.1',
'pyarrow>=1.0.1',
'pygithub>=1.51',
'pygithub>=1.59',
'scipy>=1.4.0',
'matplotlib>=2.0.0',
'tqdm>=4.50.0',
Expand Down
11 changes: 11 additions & 0 deletions tests/pybaseball/data/team_batting_bref.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
<html><body>
<table id="players_standard_batting">
<thead>
<tr><th>Rk</th><th>Tm</th><th>BatAge</th><th>G</th><th>PA</th><th>R</th><th>H</th><th>HR</th><th>RBI</th></tr>
</thead>
<tbody>
<tr><th scope="row">1</th><td>New York Yankees</td><td>28.5</td><td>162</td><td>6300</td><td>900</td><td>1400</td><td>250</td><td>860</td></tr>
<tr><th scope="row">2</th><td>Boston Red Sox</td><td>27.9</td><td>162</td><td>6250</td><td>850</td><td>1450</td><td>220</td><td>820</td></tr>
</tbody>
</table>
</body></html>
11 changes: 11 additions & 0 deletions tests/pybaseball/data/team_pitching_bref.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
<html><body>
<table id="players_standard_pitching">
<thead>
<tr><th>Rk</th><th>Tm</th><th>PAge</th><th>G</th><th>IP</th><th>H</th><th>ER</th><th>ERA</th><th>SO</th><th>BB</th></tr>
</thead>
<tbody>
<tr><th scope="row">1</th><td>New York Yankees</td><td>29.1</td><td>162</td><td>1450.0</td><td>1300</td><td>600</td><td>3.72</td><td>1500</td><td>500</td></tr>
<tr><th scope="row">2</th><td>Boston Red Sox</td><td>28.4</td><td>162</td><td>1440.0</td><td>1350</td><td>650</td><td>4.06</td><td>1400</td><td>520</td></tr>
</tbody>
</table>
</body></html>
12 changes: 12 additions & 0 deletions tests/pybaseball/data/team_results.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
<html><body>
<table>
<thead>
<tr><th>Gm#</th><th>Date</th><th>Tm</th><th>Opp</th><th>HA</th><th>WL</th><th>R</th><th>RA</th><th>Inn</th><th>Rank</th><th>GB</th><th>Win</th><th>Loss</th><th>Save</th><th>Time</th><th>DN</th><th>Streak</th><th>Attendance</th><th></th></tr>
</thead>
<tbody>
<tr><td>2019-04-01</td><td>NYY</td><td>BAL</td><td>Home</td><td>W</td><td>5</td><td>3</td><td>9</td><td>1</td><td>--</td><td>SmithA</td><td>JonesB</td><td>None</td><td>3:01</td><td>D</td><td>+1</td><td>Unknown</td><td>x</td></tr>
<tr><td>2019-04-02</td><td>NYY</td><td>BAL</td><td>Home</td><td>L</td><td>2</td><td>4</td><td>9</td><td>2</td><td>1.0</td><td>DoeC</td><td>RoeD</td><td>None</td><td>2:55</td><td>N</td><td>-1</td><td>40,000</td><td>x</td></tr>
<tr><td>Description row to be skipped</td></tr>
</tbody>
</table>
</body></html>
38 changes: 38 additions & 0 deletions tests/pybaseball/test_retrosheet.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,38 @@
from unittest.mock import MagicMock

import pytest

import pybaseball.retrosheet as retrosheet


def test_events_uses_token_auth(monkeypatch: pytest.MonkeyPatch) -> None:
# Regression test for #455: when GH_TOKEN is set, the GitHub client must be
# built with the modern keyword auth API (Auth.Token) rather than the
# deprecated positional-token form.
monkeypatch.setenv('GH_TOKEN', 'dummy_token')
fake_github = MagicMock()
fake_github.return_value.get_repo.side_effect = RuntimeError("stop")
monkeypatch.setattr(retrosheet, 'Github', fake_github)

with pytest.raises(RuntimeError):
retrosheet.events(2019)

args, kwargs = fake_github.call_args
assert args == ()
assert 'auth' in kwargs


def test_events_anonymous_without_token(monkeypatch: pytest.MonkeyPatch) -> None:
# Regression test for #455: with no GH_TOKEN, the client must be created
# anonymously (Github()) instead of passing an empty token string.
monkeypatch.delenv('GH_TOKEN', raising=False)
fake_github = MagicMock()
fake_github.return_value.get_repo.side_effect = RuntimeError("stop")
monkeypatch.setattr(retrosheet, 'Github', fake_github)

with pytest.raises(RuntimeError):
retrosheet.events(2019)

args, kwargs = fake_github.call_args
assert args == ()
assert 'auth' not in kwargs
5 changes: 4 additions & 1 deletion tests/pybaseball/test_statcast.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,10 @@ def _single_game_raw(get_data_file_contents: Callable[[str], str]) -> str:
@pytest.fixture(name="single_game")
def _single_game(get_data_file_dataframe: GetDataFrameCallable) -> pd.DataFrame:
data = get_data_file_dataframe('single_game_request.csv', parse_dates=[2])
data[data.columns[2]].apply(pd.to_datetime, errors='ignore', format=DATE_FORMAT)
try:
data[data.columns[2]] = pd.to_datetime(data[data.columns[2]], format=DATE_FORMAT)
except (ValueError, TypeError):
pass
return data


Expand Down
33 changes: 33 additions & 0 deletions tests/pybaseball/test_team_batting.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,3 +25,36 @@ def test_team_batting(response_get_monkeypatch: Callable, sample_html: str, samp
team_batting_result = team_batting(season).reset_index(drop=True)

pd.testing.assert_frame_equal(team_batting_result, sample_processed_result, check_dtype=False)


@pytest.fixture(name="sample_bref_html")
def _sample_bref_html(get_data_file_contents: Callable[[str], str]) -> str:
return get_data_file_contents('team_batting_bref.html')


def test_team_batting_bref(bref_get_monkeypatch: Callable, sample_bref_html: str) -> None:
# Regression test for #461: Baseball Reference changed the batting table to
# id='players_standard_batting' with a <thead>. team_batting_bref must parse
# the new structure instead of the removed 'sortable stats_table' class.
from pybaseball.team_batting import team_batting_bref

bref_get_monkeypatch(sample_bref_html)

result = team_batting_bref('NYY', 2019)

assert result is not None
assert not result.empty
assert 'Tm' in result.columns
assert 'Year' in result.columns
assert (result['Year'] == 2019).all()


def test_team_batting_bref_missing_table_raises(bref_get_monkeypatch: Callable) -> None:
# Regression test for #461: a page without the expected table should raise a
# clear ValueError instead of an opaque IndexError.
from pybaseball.team_batting import team_batting_bref

bref_get_monkeypatch("<html><body><p>no table here</p></body></html>")

with pytest.raises(ValueError):
team_batting_bref('NYY', 2019)
9 changes: 9 additions & 0 deletions tests/pybaseball/test_team_fielding.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,3 +25,12 @@ def test_team_fielding(response_get_monkeypatch: Callable, sample_html: str, sam
team_fielding_result = team_fielding(season).reset_index(drop=True)

pd.testing.assert_frame_equal(team_fielding_result, sample_processed_result, check_dtype=False)


def test_team_fielding_bref_invalid_season_range() -> None:
# Regression test for #462: an end_season earlier than start_season should
# raise a clear ValueError before any network request is made.
from pybaseball.team_fielding import team_fielding_bref

with pytest.raises(ValueError):
team_fielding_bref('NYY', 2019, 2018)
45 changes: 45 additions & 0 deletions tests/pybaseball/test_team_game_logs.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
import pandas as pd

from pybaseball.team_game_logs import _to_numeric_or_keep, postprocess


def test_postprocess_converts_numeric_columns_and_keeps_text() -> None:
# Regression test: postprocess used DataFrame.apply(pd.to_numeric, errors="ignore"),
# and pandas 3 rejects errors="ignore", so team_game_logs raised ValueError.
columns = pd.MultiIndex.from_tuples([
('Unnamed: 0_level_0', 'Rk'),
('Unnamed: 1_level_0', 'Gtm'),
('Unnamed: 2_level_0', 'Date'),
('Unnamed: 3_level_0', 'Unnamed: 3_level_1'),
('Unnamed: 4_level_0', 'Opp'),
('Batting', 'R'),
])
data = pd.DataFrame([
['1', '1', 'Mar 28', '@', 'NYY', '3'],
['2', '2', 'Mar 29', None, 'NYY', '5'],
['Rk', 'Gtm', 'Date', None, 'Opp', 'R'],
['3', '3', 'Mar 30', None, 'NYY', '0'],
['', '', '', None, '', '8'],
], columns=columns)

result = postprocess(data)

assert len(result) == 3
assert result[('Unnamed: 1_level_0', 'Game')].tolist() == [1, 2, 3]
assert result[('Unnamed: 3_level_0', 'Home')].tolist() == [False, True, True]
assert pd.api.types.is_numeric_dtype(result[('Batting', 'R')])
assert result[('Batting', 'R')].tolist() == [3, 5, 0]
assert result[('Unnamed: 4_level_0', 'Opp')].tolist() == ['NYY', 'NYY', 'NYY']


def test_numeric_conversion_handles_duplicate_column_labels() -> None:
# Selecting a duplicated label returns a DataFrame, so a per-column
# pd.to_numeric loop would silently leave every column as text.
columns = pd.MultiIndex.from_tuples([('Batting', 'R'), ('Batting', 'R'), ('Opp', 'Opp')])
data = pd.DataFrame([['3', '4', 'NYY'], ['5', '6', 'BOS']], columns=columns)

result = data.apply(_to_numeric_or_keep)

assert result[('Batting', 'R')].values.tolist() == [[3, 4], [5, 6]]
assert all(pd.api.types.is_numeric_dtype(dtype) for dtype in result[('Batting', 'R')].dtypes)
assert result.iloc[:, 2].tolist() == ['NYY', 'BOS']
Loading