- name
- strategy-validation
- description
- End-to-end strategy validation — backtest tearsheet (Sharpe, Sortino, Calmar, VaR), walk-forward optimization, Monte Carlo stress testing, parameter sensitivity heatmaps, and strategy A/B testing. Use for validate strategy, backtest report, walk-forward, Monte Carlo test, parameter sensitivity, or any strategy validation.
- kind
- reference
- category
- trading/quant
- status
- active
- tags
- ["backtesting","monte-carlo","quant","strategy","trading","validation"]
- related_skills
- ["backtesting-sim","backtest-report-generator","hurst-exponent-dynamics-crisis-prediction","ml-trading","quant-ml-trading"]
# Strategy Validation — Tearsheet, WFO, Monte Carlo, Sensitivity, A/B Testing
## Overview
End-to-end strategy validation pipeline:
1. **Backtest Tearsheet** — Sharpe, Sortino, Calmar, VaR, CVaR, Win Rate, HTML report
2. **Walk-Forward Optimization** — anchored and rolling WFO with OOS validation
3. **Monte Carlo Stress Testing** — bootstrap, parameter perturbation, regime shuffle
4. **Parameter Sensitivity** — 2D grid sweep, robustness heatmaps, one-at-a-time
5. **Strategy A/B Testing** — paired t-test, Wilcoxon, KS test, Jobson-Korkie Sharpe diff
---
## Section 1: Backtest Tearsheet
```python
import pandas as pd
import numpy as np
from scipy import stats
from datetime import datetime
from typing import Optional
def compute_tearsheet(equity_curve: pd.Series, returns: pd.Series,
trades_df: Optional[pd.DataFrame] = None,
benchmark_returns: Optional[pd.Series] = None,
risk_free_rate: float = 0.04) -> dict:
"""Compute comprehensive strategy tearsheet metrics."""
annual_factor = 252
total_return = (equity_curve.iloc[-1] / equity_curve.iloc[0]) - 1
years = len(returns) / annual_factor
cagr = (1 + total_return) ** (1 / max(years, 0.01)) - 1
peak = equity_curve.cummax()
dd = (equity_curve - peak) / peak
max_dd = dd.min()
dd_durations = []
in_dd = False; start = None
for i, d in enumerate(dd):
if d < 0 and not in_dd: in_dd = True; start = i
elif d == 0 and in_dd: in_dd = False; dd_durations.append(i - start)
vol = returns.std() * np.sqrt(annual_factor)
sharpe = (returns.mean() * annual_factor - risk_free_rate) / max(vol, 1e-10)
downside_ret = returns[returns < 0]
sortino = ((returns.mean() * annual_factor - risk_free_rate) /
(downside_ret.std() * np.sqrt(annual_factor)) if len(downside_ret) > 0 else 0)
calmar = cagr / abs(max_dd) if max_dd != 0 else 0
var_95 = returns.quantile(0.05)
cvar_95 = returns[returns <= var_95].mean()
trade_stats = {}
if trades_df is not None and not trades_df.empty:
closed = trades_df[trades_df["pnl_pips"].notna()]
wins = closed[closed["pnl_pips"] > 0]; losses = closed[closed["pnl_pips"] <= 0]
trade_stats = {
"total_trades": len(closed), "win_rate": round(len(wins) / max(len(closed), 1) * 100, 1),
"avg_win": round(wins["pnl_pips"].mean(), 1) if len(wins) > 0 else 0,
"avg_loss": round(losses["pnl_pips"].mean(), 1) if len(losses) > 0 else 0,
"profit_factor": round(wins["pnl_usd"].sum() / abs(losses["pnl_usd"].sum()), 2) if len(losses) > 0 and losses["pnl_usd"].sum() != 0 else float("inf"),
"expectancy_pips": round(closed["pnl_pips"].mean(), 2),
"largest_win": round(wins["pnl_pips"].max(), 1) if len(wins) > 0 else 0,
"largest_loss": round(losses["pnl_pips"].min(), 1) if len(losses) > 0 else 0,
}
return {
"summary": {"total_return": round(total_return * 100, 2), "cagr": round(cagr * 100, 2),
"sharpe": round(sharpe, 3), "sortino": round(sortino, 3), "calmar": round(calmar, 3),
"volatility": round(vol * 100, 2), "max_drawdown": round(max_dd * 100, 2),
"avg_drawdown_duration": round(np.mean(dd_durations), 0) if dd_durations else 0,
"max_drawdown_duration": max(dd_durations) if dd_durations else 0,
"var_95": round(var_95 * 100, 4), "cvar_95": round(cvar_95 * 100, 4)},
"trade_stats": trade_stats,
"period": f"{equity_curve.index[0]} → {equity_curve.index[-1]}",
"bars": len(returns), "years": round(years, 2),
}
def generate_html_report(tearsheet: dict, monte_carlo: dict, strategy_name: str = "Strategy") -> str:
"""Generate a standalone HTML report with Chart.js."""
return f"""<!DOCTYPE html><html><head><meta charset="utf-8"><title>{strategy_name} Backtest Report</title>
<script src="https://cdn.jsdelivr.net/npm/chart.js"></script>
<style>body{{font-family:-apple-system,sans-serif;margin:40px;background:#0f0f1a;color:#e0e0e0}}
.card{{background:#1a1a2e;border-radius:12px;padding:24px;margin:16px 0}}
.metric{{display:inline-block;margin:12px 24px;text-align:center}}
.metric .value{{font-size:28px;font-weight:bold}}.metric .label{{font-size:12px;color:#888}}
.green{{color:#00e676}}.red{{color:#ff5252}}.yellow{{color:#ffd740}}
h1{{color:#7c4dff}}h2{{color:#448aff;border-bottom:1px solid #333;padding-bottom:8px}}
table{{width:100%;border-collapse:collapse}}th,td{{padding:8px 12px;text-align:left;border-bottom:1px solid #333}}
th{{color:#888}}</style></head><body>
<h1>{strategy_name} — Backtest Report</h1>
<p>Generated: {datetime.utcnow().strftime('%Y-%m-%d %H:%M UTC')}</p>
<div class="card"><h2>Performance Summary</h2>
<div class="metric"><div class="value {'green' if tearsheet['summary']['total_return'] > 0 else 'red'}">{tearsheet['summary']['total_return']}%</div><div class="label">Total Return</div></div>
<div class="metric"><div class="value">{tearsheet['summary']['cagr']}%</div><div class="label">CAGR</div></div>
<div class="metric"><div class="value">{tearsheet['summary']['sharpe']}</div><div class="label">Sharpe</div></div>
<div class="metric"><div class="value">{tearsheet['summary']['sortino']}</div><div class="label">Sortino</div></div>
<div class="metric"><div class="value red">{tearsheet['summary']['max_drawdown']}%</div><div class="label">Max DD</div></div>
</div>
<div class="card"><h2>Trade Statistics</h2><table><tr><th>Metric</th><th>Value</th></tr>
{''.join(f"<tr><td>{k}</td><td>{v}</td></tr>" for k, v in tearsheet.get('trade_stats', {}).items())}
</table></div>
<div class="card" style="background:#2a1a1a;border:1px solid #ff5252;">
<strong style="color:#ff5252;">⚠ DISCLAIMER:</strong> Past performance does not guarantee future results.
Walk-forward out-of-sample validation required before live deployment.
</div></body></html>"""
```
---
## Section 2: Walk-Forward Optimization
```python
from typing import Callable
class WalkForwardOptimizer:
@staticmethod
def anchored_wfo(data: pd.DataFrame, strategy_fn: Callable, optimize_fn: Callable,
train_pct: float = 0.7, n_folds: int = 5) -> dict:
"""Anchored WFO: training window grows, test window is fixed."""
total = len(data); test_size = total // (n_folds + 1); results = []
for fold in range(n_folds):
train_end = total - test_size * (n_folds - fold)
test_end = train_end + test_size
train = data.iloc[:train_end]; test = data.iloc[train_end:test_end]
best_params = optimize_fn(train)
oos_returns = strategy_fn(test, best_params)
sharpe = (oos_returns.mean() / oos_returns.std()) * np.sqrt(252) if oos_returns.std() > 0 else 0
results.append({"fold": fold, "train_size": len(train), "test_size": len(test),
"params": best_params, "oos_sharpe": round(sharpe, 3),
"oos_return": round(oos_returns.sum() * 100, 2),
"oos_trades": len(oos_returns[oos_returns != 0])})
avg_oos_sharpe = np.mean([r["oos_sharpe"] for r in results])
stabilities = []
params_list = [r["params"] for r in results if isinstance(r["params"], dict)]
if params_list:
for key in params_list[0]:
vals = [p.get(key) for p in params_list if isinstance(p.get(key), (int, float))]
if vals and np.mean(vals) != 0:
stabilities.append(max(1 - np.std(vals) / abs(np.mean(vals)), 0))
param_stability = round(np.mean(stabilities), 3) if stabilities else 0
return {
"method": "anchored_walk_forward", "n_folds": n_folds, "fold_results": results,
"avg_oos_sharpe": round(avg_oos_sharpe, 3), "param_stability": param_stability,
"verdict": "ROBUST" if avg_oos_sharpe > 0.5 and param_stability > 0.6
else "MARGINAL" if avg_oos_sharpe > 0
else "FAILED — strategy does not generalize",
}
@staticmethod
def rolling_wfo(data: pd.DataFrame, strategy_fn: Callable, optimize_fn: Callable,
train_bars: int = 500, test_bars: int = 100) -> dict:
"""Rolling WFO: fixed-size training window moves forward."""
results = []
for start in range(0, len(data) - train_bars - test_bars, test_bars):
train = data.iloc[start:start + train_bars]
test = data.iloc[start + train_bars:start + train_bars + test_bars]
best_params = optimize_fn(train)
oos_returns = strategy_fn(test, best_params)
sharpe = (oos_returns.mean() / oos_returns.std()) * np.sqrt(252) if oos_returns.std() > 0 else 0
results.append({"fold": len(results), "params": best_params, "oos_sharpe": round(sharpe, 3)})
return {"method": "rolling_walk_forward", "n_folds": len(results),
"avg_oos_sharpe": round(np.mean([r["oos_sharpe"] for r in results]), 3), "fold_results": results}
```
---
## Section 3: Monte Carlo Stress Testing
```python
class MonteCarloStressTester:
@staticmethod
def monte_carlo_simulation(returns: pd.Series, n_simulations: int = 1000,
n_periods: int = 252, initial_capital: float = 10000) -> dict:
np.random.seed(42)
all_paths = np.zeros((n_simulations, n_periods))
for sim in range(n_simulations):
sampled = np.random.choice(returns.values, size=n_periods, replace=True)
all_paths[sim] = initial_capital * np.cumprod(1 + sampled)
final_values = all_paths[:, -1]
max_drawdowns = []
for path in all_paths:
peak = np.maximum.accumulate(path)
max_drawdowns.append((path - peak).min() / peak.max())
return {
"n_simulations": n_simulations,
"median_final": round(np.median(final_values), 2),
"p5_final": round(np.percentile(final_values, 5), 2),
GitHubで見る