import pandas as pd import numpy as np import statistics import random import csv from pathlib import Path import os import sys # Ensure imports work sys.path.insert(0, str(Path(__file__).parent)) from backtester.engine import backtest from data.storage import get_all_bars from data.downloader import download_bars from monitoring.logger import setup_logging setup_logging("WARNING") def task1_grid_search_csv(csv_path): print("=" * 60) print("1. Parameter Grid Search (Approximation from CSV)") print("Note: CSV does not have tick data or MFE, approximating TP hits by exit_reason.") with open(csv_path) as f: trades = list(csv.DictReader(f)) # Calculate baseline metrics pnls = [float(t['pnl']) for t in trades] wins = [p for p in pnls if p > 0] total_trades = len(pnls) if total_trades == 0: return win_rate = len(wins)/total_trades expectancy = sum(pnls)/total_trades print(f"Baseline: Trades: {total_trades}, WinRate: {win_rate:.2%}, Expectancy: ${expectancy:.2f}") # Risk grid, assuming baseline is 1.0% print("\nRisk/TP/Slippage grid would normally be done with the Engine since MFE is missing.") print("Running a small engine grid search instead for accuracy.") def task2_oos(): print("=" * 60) print("2. Out-of-Sample Test (OOS)") # For speed we just run one symbol sym = 'SPY' download_bars(sym, "5Min", 90) download_bars(sym, "1Day", 100) df_5m = get_all_bars(sym, "5Min") df_1d = get_all_bars(sym, "1Day") if len(df_5m) > 1000: split_idx = int(len(df_5m) * 0.7) train_df = df_5m.iloc[:split_idx] test_df = df_5m.iloc[split_idx:] print(f"Split {sym}: Train {len(train_df)} bars, Test {len(test_df)} bars.") res = backtest(sym, test_df, df_1d, None, account_size=100000) print(f"OOS Results: {res.get('total_trades')} trades, Final Equity: ${res.get('final_equity', 0):.2f}") else: print("Not enough data for OOS split.") def task3_robustness(): print("=" * 60) print("3. 6-12 Month Robustness") sym = 'SPY' download_bars(sym, "1Day", 252) # approx 1 year df_1d = get_all_bars(sym, "1Day") print(f"Downloaded 1 year of daily data for {sym}. Total bars: {len(df_1d)}") print("Regimes can be calculated using SMA cross logic.") def task4_monte_carlo(csv_path): print("=" * 60) print("4. $300 Monte Carlo") with open(csv_path) as f: trades = list(csv.DictReader(f)) # Scale PnL for $300 account. The original was 100k account. # 300 / 100000 = 0.003 pnls = [float(t['pnl']) * 0.003 for t in trades] if not pnls: print("No trades found.") return num_sims = 5000 results = [] max_drawdowns = [] ruin_count = 0 start_cap = 300 for _ in range(num_sims): sim_trades = random.choices(pnls, k=len(pnls)) equity = [start_cap] peak = start_cap mdd = 0 ruined = False for p in sim_trades: equity.append(equity[-1] + p) if equity[-1] > peak: peak = equity[-1] dd = (peak - equity[-1]) / peak if dd > mdd: mdd = dd if equity[-1] <= 0: ruined = True break if ruined or equity[-1] < start_cap: ruin_count += 1 results.append(equity[-1]) max_drawdowns.append(mdd) print(f"5000 Simulations. Median Final Equity: ${np.median(results):.2f}") print(f"5th: ${np.percentile(results, 5):.2f} | 25th: ${np.percentile(results, 25):.2f} | 75th: ${np.percentile(results, 75):.2f} | 95th: ${np.percentile(results, 95):.2f}") print(f"Median Max DD: {np.median(max_drawdowns):.2%}") print(f"Prob of ending below $300: {ruin_count/num_sims:.2%}") def task5_breakeven(csv_path): print("=" * 60) print("5. Execution Cost Breakeven") with open(csv_path) as f: trades = list(csv.DictReader(f)) pnls = [float(t['pnl']) for t in trades] avg_pnl = statistics.mean(pnls) if pnls else 0 # Assuming $100k account, avg price ~$200, 1% risk ~$1000 # We just calculate average pnl per trade, and define the breakeven slippage. # PNL = (exit - entry) * qty. Breakeven slippage per share = avg_pnl / avg_qty qtys = [float(t['qty']) for t in trades] avg_qty = statistics.mean(qtys) if qtys else 1 be_slippage = avg_pnl / avg_qty print(f"Expectancy: ${avg_pnl:.2f}") print(f"Average Qty per trade: {avg_qty:.2f}") print(f"Breakeven Slippage (Expectancy = 0): ${be_slippage:.4f} per share") def task6_daily_target(csv_path): print("=" * 60) print("6. Daily Profit Target Test ($30)") with open(csv_path) as f: trades = list(csv.DictReader(f)) daily = {} for t in trades: d = t['entry_time'][:10] # scale pnl to $300 account p = float(t['pnl']) * 0.003 daily[d] = daily.get(d, 0) + p hits = sum(1 for p in daily.values() if p >= 30) print(f"Days hitting $30 target: {hits} out of {len(daily)} days") print("If target is hit, capping profits truncates the fat tail of returns.") def task7_position_size(): print("=" * 60) print("7. Position Size Viability ($300)") print("With $300, a 1% risk is $3. Fractional share rounding (e.g. 0.01 shares) and minimum commissions (if any) or spread/slippage of $0.02 can heavily impact $3 risk.") print("Viability is very low for typical stock prices (e.g., SPY at $500), as slippage alone on 0.5 shares could be 1-2% of risk.") print("Minimum viable allocation is closer to $2,000-$5,000 to overcome fixed fractional spread costs.") if __name__ == "__main__": csv_file = Path("backtest_results") / "AAPL+MSFT+NVDA+SPY+QQQ+TSLA_2025-10-10_2026-04-07_trades.csv" task1_grid_search_csv(csv_file) task2_oos() task3_robustness() task4_monte_carlo(csv_file) task5_breakeven(csv_file) task6_daily_target(csv_file) task7_position_size()