From 888de9d59c11725d97fb53d52286dab1043a910b Mon Sep 17 00:00:00 2001 From: Michal Kaszubski <67229322+Cashubski@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:52:55 +0200 Subject: [PATCH] fix(returns): use non-excess kurtosis and omit NaNs in deflated Sharpe ratio MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Deflated Sharpe Ratio standard error of Bailey and López de Prado, sqrt((1 - g3 * SR + (g4 - 1) / 4 * SR^2) / (T - 1)), takes g4 as the non-excess kurtosis (3 for Gaussian returns, which recovers the Lo (2002) standard error sqrt((1 + SR^2 / 2) / (T - 1))). scipy.stats.kurtosis returns the excess kurtosis by default, so the term was off by 3 / 4 * SR^2 and could turn the variance negative, yielding NaN. Missing returns were also replaced by zero-return periods before computing the moments and the horizon, diluting skewness and kurtosis and inflating the horizon, while sharpe_ratio ignores them. They are now omitted from the moments and the per-column horizon. Tests use the reference formula directly rather than only stored values. --- tests/test_returns.py | 67 +++++++++++++++++++++++++++++++++-- vectorbt/returns/accessors.py | 14 ++++---- vectorbt/returns/metrics.py | 14 ++++++-- 3 files changed, 83 insertions(+), 12 deletions(-) diff --git a/tests/test_returns.py b/tests/test_returns.py index 1c02e342..a5ed8268 100644 --- a/tests/test_returns.py +++ b/tests/test_returns.py @@ -292,12 +292,73 @@ def test_sharpe_ratio(self): def test_deflated_sharpe_ratio(self): pd.testing.assert_series_equal( rets.vbt.returns.deflated_sharpe_ratio(risk_free=0.01), - pd.Series([np.nan, np.nan, 0.0005355605507117676], index=rets.columns, name="deflated_sharpe_ratio"), + pd.Series( + [0.2374973566617728, 5.910048655681186e-16, 0.0032783041604936523], + index=rets.columns, + name="deflated_sharpe_ratio", + ), ) pd.testing.assert_series_equal( rets.vbt.returns.deflated_sharpe_ratio(risk_free=0.03), - pd.Series([np.nan, np.nan, 0.0003423112350834066], index=rets.columns, name="deflated_sharpe_ratio"), - ) + pd.Series( + [0.15146264148773053, 5.864288083101271e-16, 0.002229547585230213], + index=rets.columns, + name="deflated_sharpe_ratio", + ), + ) + + def test_deflated_sharpe_ratio_reference(self): + """DSR matches Bailey & López de Prado with non-excess kurtosis and NaNs omitted.""" + from scipy.stats import kurtosis, norm, skew + + from vectorbt.returns.metrics import approx_exp_max_sharpe + + np.random.seed(seed) + n = 500 + arr = np.random.standard_t(4, size=(n, 3)) * 0.01 + 0.001 + arr[np.random.uniform(size=(n, 3)) < 0.05] = np.nan + df = pd.DataFrame(arr, index=pd.date_range("2020", periods=n, freq="D"), columns=["a", "b", "c"]) + acc = df.vbt.returns(freq="D", year_freq="365 days") + ann_factor = 365.0 + sharpe = acc.sharpe_ratio().values + sr = sharpe / np.sqrt(ann_factor) + var_sr = np.var(sharpe, ddof=1) / ann_factor + sr0 = approx_exp_max_sharpe(0, var_sr, 3) + horizon = np.sum(~np.isnan(arr), axis=0) + g3 = skew(arr, axis=0, nan_policy="omit") + g4 = kurtosis(arr, axis=0, fisher=False, nan_policy="omit") + expected = norm.cdf((sr - sr0) * np.sqrt(horizon - 1) / np.sqrt(1 - g3 * sr + (g4 - 1) / 4 * sr**2)) + np.testing.assert_allclose(acc.deflated_sharpe_ratio().values, expected) + # Gaussian returns recover the Lo (2002) standard error sqrt((1 + SR^2 / 2) / (T - 1)) + gauss = np.random.normal(0.001, 0.01, size=(n, 3)) + acc = pd.DataFrame(gauss, index=df.index, columns=df.columns).vbt.returns(freq="D", year_freq="365 days") + sharpe = acc.sharpe_ratio().values + sr = sharpe / np.sqrt(ann_factor) + sr0 = approx_exp_max_sharpe(0, np.var(sharpe, ddof=1) / ann_factor, 3) + g3 = skew(gauss, axis=0) + g4 = kurtosis(gauss, axis=0, fisher=False) + lo = norm.cdf((sr - sr0) * np.sqrt(n - 1) / np.sqrt(1 + sr**2 / 2)) + exact = norm.cdf((sr - sr0) * np.sqrt(n - 1) / np.sqrt(1 - g3 * sr + (g4 - 1) / 4 * sr**2)) + np.testing.assert_allclose(acc.deflated_sharpe_ratio().values, exact) + np.testing.assert_allclose(acc.deflated_sharpe_ratio().values, lo, atol=1e-3) + + def test_deflated_sharpe_ratio_nan_is_not_zero_return(self): + """A missing return must not be counted as a zero-return period.""" + np.random.seed(seed) + n = 300 + arr = np.random.standard_t(4, size=(n, 3)) * 0.01 + 0.001 + index = pd.date_range("2020", periods=n, freq="D") + full = pd.DataFrame(arr, index=index, columns=["a", "b", "c"]) + with_nan = full.copy() + with_nan.iloc[10:60, 0] = np.nan + zero_filled = with_nan.fillna(0.0) + dropped = with_nan.iloc[np.r_[0:10, 60:n]] + kwargs = dict(var_sharpe=0.01, nb_trials=5) + dsr_nan = with_nan.vbt.returns(freq="D", year_freq="365 days").deflated_sharpe_ratio(**kwargs) + dsr_zero = zero_filled.vbt.returns(freq="D", year_freq="365 days").deflated_sharpe_ratio(**kwargs) + dsr_dropped = dropped.vbt.returns(freq="D", year_freq="365 days").deflated_sharpe_ratio(**kwargs) + assert not np.isclose(dsr_nan["a"], dsr_zero["a"], rtol=1e-3, atol=0) + assert np.isclose(dsr_nan["a"], dsr_dropped["a"], rtol=1e-12, atol=0) def test_downside_risk(self): assert isclose(rets["a"].vbt.returns.downside_risk(required_return=0.1), 0.0) diff --git a/vectorbt/returns/accessors.py b/vectorbt/returns/accessors.py index 9bea26c0..5af3fc5d 100644 --- a/vectorbt/returns/accessors.py +++ b/vectorbt/returns/accessors.py @@ -602,17 +602,17 @@ def deflated_sharpe_ratio( if nb_trials is None: nb_trials = self.wrapper.shape_2d[1] returns = to_2d_array(self.obj) - nanmask = np.isnan(returns) - if nanmask.any(): - returns = returns.copy() - returns[nanmask] = 0.0 + # Missing returns are excluded from the moments and the horizon rather than + # counted as zero-return periods, consistent with `ReturnsAccessor.sharpe_ratio`. + # The formula requires the (non-excess) kurtosis: it equals 3 for Gaussian + # returns, which recovers the classic Lo (2002) standard error of the Sharpe ratio. result = metrics.deflated_sharpe_ratio( est_sharpe=sharpe_ratio / np.sqrt(self.ann_factor), var_sharpe=var_sharpe / self.ann_factor, nb_trials=nb_trials, - backtest_horizon=self.wrapper.shape_2d[0], - skew=skew(returns, axis=0, bias=bias), - kurtosis=kurtosis(returns, axis=0, bias=bias), + backtest_horizon=np.sum(~np.isnan(returns), axis=0), + skew=skew(returns, axis=0, bias=bias, nan_policy="omit"), + kurtosis=kurtosis(returns, axis=0, bias=bias, fisher=False, nan_policy="omit"), ) wrap_kwargs = merge_dicts(dict(name_or_index="deflated_sharpe_ratio"), wrap_kwargs) return self.wrapper.wrap_reduced(result, group_by=False, **wrap_kwargs) diff --git a/vectorbt/returns/metrics.py b/vectorbt/returns/metrics.py index 941f7094..c383d313 100644 --- a/vectorbt/returns/metrics.py +++ b/vectorbt/returns/metrics.py @@ -21,13 +21,23 @@ def deflated_sharpe_ratio( est_sharpe: tp.Array1d, var_sharpe: float, nb_trials: int, - backtest_horizon: int, + backtest_horizon: tp.Union[int, tp.Array1d], skew: tp.Array1d, kurtosis: tp.Array1d, ) -> tp.Array1d: """Deflated Sharpe Ratio (DSR). - See [Deflated Sharpe Ratio](https://gmarti.gitlab.io/qfin/2018/05/30/deflated-sharpe-ratio.html).""" + Probability that the estimated Sharpe ratio `est_sharpe` exceeds the expected maximum + Sharpe ratio of `nb_trials` unskilled strategies, using the Sharpe ratio standard error + of Bailey and López de Prado (2012), which depends on the skewness and the kurtosis of + the returns. All Sharpe ratios and moments must be expressed per period (not annualized). + + `kurtosis` is the non-excess (Pearson) kurtosis, which equals 3 for Gaussian returns. + `backtest_horizon` is the number of observed returns and can be given per column. + + See [Deflated Sharpe Ratio](https://gmarti.gitlab.io/qfin/2018/05/30/deflated-sharpe-ratio.html) + and Bailey, D. H., & López de Prado, M. (2014). The Deflated Sharpe Ratio: Correcting for + Selection Bias, Backtest Overfitting and Non-Normality. The Journal of Portfolio Management.""" SR0 = approx_exp_max_sharpe(0, var_sharpe, nb_trials) return norm.cdf(