Backtesting Done Right: Avoiding the Pitfalls in Indian Markets
Learn how to conduct robust backtests that actually predict future performance in Indian markets. Avoid overfitting, data snooping, and survivorship bias.
You’ve backtested your strategy on 10 years of NSE data. Sharpe ratio: 2.8. Win rate: 68%. Maximum drawdown: only 12%. You’re convinced you’ve found the holy grail of trading. Then you deploy it with real money, and within three months, you’re down 15%.
What went wrong? Your backtest lied to you—not because of bugs in your code, but because of subtle biases and flawed assumptions that plague most retail backtests in Indian markets.
This comprehensive guide teaches you how to conduct rigorous backtests that actually predict future performance, with specific focus on pitfalls unique to NSE/BSE trading.
The Backtesting Illusion
Why Most Backtests Fail
Problem 1: Look-Ahead Bias
Using information that wouldn’t have been available at the time of the trade.
# ❌ WRONG: Using close price from same bar for entry
if data['Close'] > data['SMA_20']:
buy_at_price = data['Close'] # This is look-ahead bias!
# ✅ CORRECT: Use next bar's open
if data['Close'] > data['SMA_20']:
buy_at_price = data['Open'].shift(-1) # Next bar's open
Problem 2: Survivorship Bias
Testing only on stocks that still exist today.
# ❌ WRONG: Using current Nifty 50 constituents for historical test
nifty50_2025 = ['RELIANCE', 'TCS', 'INFY', ...] # Current components
backtest_on_2015_data(nifty50_2025) # Survivorship bias!
# ✅ CORRECT: Use historical constituents
def get_historical_index_composition(date):
"""Get index composition as of specific date"""
# Load historical index changes
index_history = pd.read_csv('nifty50_historical_composition.csv')
return index_history[index_history['date'] == date]['symbols'].tolist()
# Backtest with point-in-time constituents
for date in backtest_dates:
current_constituents = get_historical_index_composition(date)
backtest_on_date(date, current_constituents)
Problem 3: Overfitting
Optimizing parameters until backtest looks perfect.
# ❌ WRONG: Testing thousands of parameter combinations
best_sharpe = 0
for fast_ma in range(5, 100):
for slow_ma in range(20, 200):
for rsi_threshold in range(10, 50):
sharpe = backtest(fast_ma, slow_ma, rsi_threshold)
if sharpe > best_sharpe:
best_sharpe = sharpe
best_params = (fast_ma, slow_ma, rsi_threshold)
# Result: Perfectly fitted to past data, fails in future
# ✅ CORRECT: Walk-forward optimization (covered later)
Part 1: Setting Up Realistic Backtests
Transaction Costs: The Indian Reality
Most backtests ignore or underestimate trading costs, which are substantial in India.
class IndianTransactionCosts:
"""
Comprehensive transaction cost model for Indian markets
"""
def __init__(self, broker='zerodha', product='MIS'):
self.broker = broker
self.product = product # MIS (intraday) or CNC (delivery)
def calculate_costs(
self,
buy_price: float,
sell_price: float,
quantity: int,
exchange='NSE'
) -> dict:
"""
Calculate all costs for a round-trip trade
"""
buy_value = buy_price * quantity
sell_value = sell_price * quantity
# 1. Brokerage
if self.broker == 'zerodha':
if self.product == 'MIS':
# Flat ₹20 per executed order or 0.03%, whichever is lower
brokerage_buy = min(20, buy_value * 0.0003)
brokerage_sell = min(20, sell_value * 0.0003)
else: # CNC
brokerage_buy = 0 # Zerodha: zero brokerage on delivery buy
brokerage_sell = min(20, sell_value * 0.0003)
else:
# Generic: 0.05% per side
brokerage_buy = buy_value * 0.0005
brokerage_sell = sell_value * 0.0005
# 2. STT (Securities Transaction Tax)
if self.product == 'MIS':
# Intraday: 0.025% on sell side only
stt = sell_value * 0.00025
else:
# Delivery: 0.1% on both buy and sell
stt = (buy_value + sell_value) * 0.001
# 3. Exchange Transaction Charges
if exchange == 'NSE':
# NSE: 0.00325% on turnover
exchange_charges = (buy_value + sell_value) * 0.0000325
else: # BSE
# BSE: 0.003% on turnover
exchange_charges = (buy_value + sell_value) * 0.00003
# 4. GST on brokerage and transaction charges
taxable_amount = (
brokerage_buy + brokerage_sell + exchange_charges
)
gst = taxable_amount * 0.18 # 18% GST
# 5. SEBI Charges
sebi_charges = (buy_value + sell_value) * 0.0000001 # ₹10 per crore
# 6. Stamp Duty
stamp_duty = buy_value * 0.00003 # 0.003% on buy side
# Total costs
total_costs = (
brokerage_buy +
brokerage_sell +
stt +
exchange_charges +
gst +
sebi_charges +
stamp_duty
)
# Net P&L
gross_pnl = sell_value - buy_value
net_pnl = gross_pnl - total_costs
return {
'brokerage': brokerage_buy + brokerage_sell,
'stt': stt,
'exchange_charges': exchange_charges,
'gst': gst,
'sebi_charges': sebi_charges,
'stamp_duty': stamp_duty,
'total_costs': total_costs,
'gross_pnl': gross_pnl,
'net_pnl': net_pnl,
'cost_as_pct': (total_costs / buy_value) * 100
}
# Example usage
cost_calc = IndianTransactionCosts(broker='zerodha', product='MIS')
# Example trade: Buy 100 RELIANCE at ₹2450, sell at ₹2475
costs = cost_calc.calculate_costs(
buy_price=2450,
sell_price=2475,
quantity=100
)
print(f"Gross P&L: ₹{costs['gross_pnl']:.2f}")
print(f"Total Costs: ₹{costs['total_costs']:.2f}")
print(f"Net P&L: ₹{costs['net_pnl']:.2f}")
print(f"Costs as % of trade: {costs['cost_as_pct']:.4f}%")
# Output:
# Gross P&L: ₹2500.00
# Total Costs: ₹113.64
# Net P&L: ₹2386.36
# Costs as % of trade: 0.0464%
Key Insight: A 1% gross profit becomes 0.95% net profit after costs. For high-frequency strategies, costs can consume all profits.
Slippage Modeling
Slippage is the difference between expected and actual execution price.
class SlippageModel:
"""
Realistic slippage model for Indian markets
"""
def __init__(self):
# Slippage factors based on stock liquidity
self.slippage_factors = {
'high_liquidity': 0.0005, # 0.05% (Nifty 50)
'medium_liquidity': 0.001, # 0.1% (Nifty Next 50)
'low_liquidity': 0.003 # 0.3% (Small caps)
}
def get_stock_liquidity_tier(self, symbol: str, avg_volume: float) -> str:
"""Classify stock by liquidity"""
# Nifty 50 stocks: high liquidity
nifty50 = ['RELIANCE', 'TCS', 'HDFCBANK', 'INFY', ...] # Full list
if symbol in nifty50:
return 'high_liquidity'
elif avg_volume > 1000000: # More than 10 lakh shares/day
return 'medium_liquidity'
else:
return 'low_liquidity'
def calculate_slippage(
self,
symbol: str,
price: float,
quantity: int,
side: str, # 'BUY' or 'SELL'
avg_volume: float,
volatility: float
) -> float:
"""
Calculate expected slippage
Factors:
1. Liquidity tier
2. Order size relative to average volume
3. Market volatility
4. Market impact
"""
# Base slippage from liquidity tier
tier = self.get_stock_liquidity_tier(symbol, avg_volume)
base_slippage_pct = self.slippage_factors[tier]
# Adjust for order size (market impact)
daily_volume = avg_volume
order_volume = quantity
volume_ratio = order_volume / daily_volume
if volume_ratio > 0.01: # Order > 1% of daily volume
# Significant market impact
impact_multiplier = 1 + (volume_ratio * 100)
else:
impact_multiplier = 1
# Adjust for volatility
if volatility > 0.02: # High volatility (>2% daily)
volatility_multiplier = 1 + volatility
else:
volatility_multiplier = 1
# Total slippage
total_slippage_pct = base_slippage_pct * impact_multiplier * volatility_multiplier
# Slippage is unfavorable: higher price for buys, lower for sells
if side == 'BUY':
slippage_price = price * (1 + total_slippage_pct)
else: # SELL
slippage_price = price * (1 - total_slippage_pct)
return slippage_price
# Usage in backtest
slippage_model = SlippageModel()
# Calculate realistic execution price
signal_price = 2450 # Price when signal generated
execution_price = slippage_model.calculate_slippage(
symbol='RELIANCE',
price=signal_price,
quantity=100,
side='BUY',
avg_volume=5000000, # 50 lakh shares/day
volatility=0.015 # 1.5% daily volatility
)
print(f"Signal price: ₹{signal_price:.2f}")
print(f"Execution price: ₹{execution_price:.2f}")
print(f"Slippage: ₹{execution_price - signal_price:.2f}")
# Output:
# Signal price: ₹2450.00
# Execution price: ₹2451.23
# Slippage: ₹1.23
Order Execution Delays
In real trading, there’s a delay between signal generation and execution.
class ExecutionSimulator:
"""
Simulate realistic order execution timing
"""
def simulate_execution_delay(
self,
signal_time: datetime,
signal_price: float,
data: pd.DataFrame
) -> tuple:
"""
Simulate execution with realistic delays
Delays:
1. Signal detection: 1-5 seconds
2. Order validation: 1-2 seconds
3. Network latency: 0.5-2 seconds
4. Exchange processing: 0.1-1 seconds
5. Order matching: Instant to several seconds
Total: 2-10 seconds typically
"""
# Total delay (random between 2-10 seconds)
delay_seconds = np.random.uniform(2, 10)
execution_time = signal_time + timedelta(seconds=delay_seconds)
# Find price at execution time
# Assume we have minute-level data
execution_bar = data[data.index >= execution_time].iloc[0]
# Execution happens at some point during the bar
# Use weighted average of OHLC (more realistic than just Open)
execution_price = (
execution_bar['Open'] * 0.4 +
execution_bar['High'] * 0.1 +
execution_bar['Low'] * 0.1 +
execution_bar['Close'] * 0.4
)
return execution_time, execution_price
# In backtest
for i in range(len(data)):
if data['Signal'].iloc[i] == 'BUY':
signal_time = data.index[i]
signal_price = data['Close'].iloc[i]
# Simulate realistic execution
exec_time, exec_price = executor.simulate_execution_delay(
signal_time, signal_price, minute_data
)
# Use exec_price, not signal_price
trades.append({
'entry_time': exec_time,
'entry_price': exec_price
})
Part 2: Walk-Forward Optimization
The gold standard for avoiding overfitting.
class WalkForwardOptimizer:
"""
Walk-forward optimization to avoid overfitting
Process:
1. Divide data into windows (e.g., 6 months each)
2. For each window:
- Train on in-sample period (e.g., 12 months)
- Test on out-of-sample period (next 3 months)
- Record out-of-sample performance
3. Aggregate out-of-sample results
"""
def __init__(
self,
strategy_class,
param_grid: dict,
in_sample_months: int = 12,
out_sample_months: int = 3
):
self.strategy_class = strategy_class
self.param_grid = param_grid
self.in_sample_months = in_sample_months
self.out_sample_months = out_sample_months
def optimize(self, data: pd.DataFrame) -> pd.DataFrame:
"""
Perform walk-forward optimization
"""
results = []
# Create windows
start_date = data.index[0]
end_date = data.index[-1]
current_date = start_date + timedelta(days=365) # Start after 1 year
while current_date + timedelta(days=90) < end_date:
# Define windows
in_sample_start = current_date - timedelta(days=365)
in_sample_end = current_date
out_sample_start = current_date
out_sample_end = current_date + timedelta(days=90)
# Get data slices
in_sample_data = data[
(data.index >= in_sample_start) &
(data.index < in_sample_end)
]
out_sample_data = data[
(data.index >= out_sample_start) &
(data.index < out_sample_end)
]
# Optimize on in-sample
best_params = self._optimize_window(in_sample_data)
# Test on out-of-sample
out_sample_performance = self._test_params(
out_sample_data,
best_params
)
results.append({
'window_start': out_sample_start,
'window_end': out_sample_end,
'best_params': best_params,
**out_sample_performance
})
# Move to next window
current_date = out_sample_end
return pd.DataFrame(results)
def _optimize_window(self, data: pd.DataFrame) -> dict:
"""Optimize parameters on in-sample data"""
best_sharpe = -np.inf
best_params = None
# Grid search (limited to prevent overfitting)
for params in self._generate_param_combinations():
strategy = self.strategy_class(**params)
performance = strategy.backtest(data)
if performance['sharpe_ratio'] > best_sharpe:
best_sharpe = performance['sharpe_ratio']
best_params = params
return best_params
def _test_params(self, data: pd.DataFrame, params: dict) -> dict:
"""Test parameters on out-of-sample data"""
strategy = self.strategy_class(**params)
return strategy.backtest(data)
def _generate_param_combinations(self):
"""Generate parameter combinations from grid"""
# Use itertools.product for exhaustive grid
import itertools
keys = self.param_grid.keys()
values = self.param_grid.values()
for combination in itertools.product(*values):
yield dict(zip(keys, combination))
# Usage
param_grid = {
'fast_period': [10, 15, 20],
'slow_period': [40, 50, 60],
'rsi_threshold': [25, 30, 35]
}
optimizer = WalkForwardOptimizer(
strategy_class=MovingAverageCrossover,
param_grid=param_grid,
in_sample_months=12,
out_sample_months=3
)
# Run walk-forward optimization
wf_results = optimizer.optimize(historical_data)
# Analyze results
print(f"Average out-of-sample Sharpe: {wf_results['sharpe_ratio'].mean():.2f}")
print(f"Consistency: {(wf_results['total_return'] > 0).mean() * 100:.1f}% positive windows")
Part 3: Monte Carlo Simulation
Test strategy robustness with randomized scenarios.
class MonteCarloSimulator:
"""
Monte Carlo simulation for strategy validation
"""
def __init__(self, strategy, n_simulations: int = 1000):
self.strategy = strategy
self.n_simulations = n_simulations
def simulate(self, trades: List[dict]) -> pd.DataFrame:
"""
Simulate random trade sequences
Method: Bootstrap resampling
Randomly sample trades with replacement to create
alternative equity curves
"""
results = []
for sim in range(self.n_simulations):
# Randomly sample trades
sampled_trades = np.random.choice(
trades,
size=len(trades),
replace=True
)
# Calculate equity curve
equity = [100000] # Starting capital
for trade in sampled_trades:
pnl = trade['pnl']
equity.append(equity[-1] + pnl)
equity = np.array(equity)
# Calculate metrics
final_equity = equity[-1]
max_equity = equity.max()
max_drawdown = ((max_equity - equity) / max_equity).max()
results.append({
'simulation': sim,
'final_equity': final_equity,
'total_return': (final_equity - 100000) / 100000,
'max_drawdown': max_drawdown
})
return pd.DataFrame(results)
def analyze_risk(self, sim_results: pd.DataFrame):
"""Analyze Monte Carlo results"""
print("Monte Carlo Risk Analysis")
print("=" * 50)
# Return distribution
print(f"\nReturn Distribution:")
print(f" Mean: {sim_results['total_return'].mean():.2%}")
print(f" Median: {sim_results['total_return'].median():.2%}")
print(f" Std Dev: {sim_results['total_return'].std():.2%}")
# Percentiles
print(f"\nReturn Percentiles:")
for pct in [5, 25, 50, 75, 95]:
value = sim_results['total_return'].quantile(pct / 100)
print(f" {pct}th: {value:.2%}")
# Risk of ruin
ruin_pct = (sim_results['final_equity'] < 90000).mean() * 100
print(f"\nRisk of 10% drawdown: {ruin_pct:.1f}%")
# Maximum drawdown
print(f"\nMax Drawdown Distribution:")
print(f" Mean: {sim_results['max_drawdown'].mean():.2%}")
print(f" 95th percentile: {sim_results['max_drawdown'].quantile(0.95):.2%}")
# Usage
trades = [...] # List of historical trades
simulator = MonteCarloSimulator(strategy, n_simulations=10000)
sim_results = simulator.simulate(trades)
simulator.analyze_risk(sim_results)
# Visualize
import matplotlib.pyplot as plt
plt.figure(figsize=(12, 6))
plt.hist(sim_results['total_return'] * 100, bins=50, alpha=0.7)
plt.axvline(0, color='r', linestyle='--', label='Breakeven')
plt.xlabel('Return (%)')
plt.ylabel('Frequency')
plt.title('Monte Carlo Return Distribution (10,000 simulations)')
plt.legend()
plt.show()
Part 4: Reality Checks
Check 1: Out-of-Sample Testing
# Split data: 70% train, 30% test
split_point = int(len(data) * 0.7)
train_data = data.iloc[:split_point]
test_data = data.iloc[split_point:]
# Optimize on train data only
best_params = optimize_parameters(train_data)
# Test on unseen data
test_performance = backtest(test_data, best_params)
# Red flag: If test performance << train performance
if test_performance['sharpe'] < train_performance['sharpe'] * 0.7:
print("⚠️ WARNING: Likely overfitting detected!")
Check 2: Parameter Sensitivity
def test_parameter_sensitivity(base_params, data):
"""
Test how sensitive strategy is to parameter changes
Robust strategies should not be overly sensitive
"""
results = {}
for param_name, base_value in base_params.items():
sensitivities = []
# Test ±20% variations
for multiplier in [0.8, 0.9, 1.0, 1.1, 1.2]:
test_params = base_params.copy()
test_params[param_name] = base_value * multiplier
performance = backtest(data, test_params)
sensitivities.append({
'multiplier': multiplier,
'sharpe': performance['sharpe']
})
results[param_name] = sensitivities
# Analyze
for param_name, sensitivity in results.items():
sharpes = [s['sharpe'] for s in sensitivity]
variation = max(sharpes) - min(sharpes)
print(f"{param_name}: Sharpe variation = {variation:.2f}")
if variation > 0.5:
print(f"⚠️ High sensitivity to {param_name}!")
Check 3: Regime Analysis
def test_different_market_regimes(strategy, data):
"""
Test strategy in different market conditions
"""
# Classify market regimes
data['Regime'] = 'Normal'
# Bull market: large positive returns
data.loc[data['Returns'].rolling(20).mean() > 0.02, 'Regime'] = 'Bull'
# Bear market: large negative returns
data.loc[data['Returns'].rolling(20).mean() < -0.02, 'Regime'] = 'Bear'
# High volatility
data.loc[data['Returns'].rolling(20).std() > 0.03, 'Regime'] = 'High Vol'
# Test in each regime
for regime in ['Bull', 'Bear', 'Normal', 'High Vol']:
regime_data = data[data['Regime'] == regime]
if len(regime_data) > 100: # Enough data
performance = strategy.backtest(regime_data)
print(f"{regime}: Sharpe = {performance['sharpe']:.2f}")
Conclusion
Proper backtesting is about honest validation, not perfect results. Key principles:
- Model reality accurately: Transaction costs, slippage, execution delays
- Avoid look-ahead bias: Only use information available at trade time
- Combat survivorship bias: Use point-in-time universes
- Prevent overfitting: Walk-forward optimization, out-of-sample testing
- Test robustness: Monte Carlo simulation, parameter sensitivity, regime analysis
A strategy with Sharpe 1.5 that’s robust and realistic is far better than Sharpe 3.0 built on fantasy assumptions.
Ready to validate your strategies properly? Contact us for professional backtesting services.