#!/usr/bin/env python3 """ Phase 2 Verification Scripts PBO, DSR, OOS Calculation & Validation AGENTS.md v16.0 Compliance: - Reproducibility: All calculations documented - Traceability: All results logged with timestamps - Right Way: No shortcuts, full validation """ import pandas as pd import numpy as np from datetime import datetime, timedelta import json import sys class Phase2Verification: """Unified Phase 2 verification engine""" def __init__(self, data_file: str): """Initialize with Phase 1 data""" self.data_file = data_file self.timestamp = datetime.now() self.results = {} def load_data(self) -> pd.DataFrame: """Load Phase 1 raw data with validation""" print(f"[{self.timestamp}] Loading data from {self.data_file}...") try: df = pd.read_csv(self.data_file) # Validation required_columns = ['trading_date', 'daily_return', 'is_return', 'oos_return', 'sharpe'] for col in required_columns: if col not in df.columns: raise ValueError(f"Missing required column: {col}") # Check for nulls null_count = df.isnull().sum().sum() if null_count > 0: print(f"⚠️ Warning: {null_count} null values found, removing...") df = df.dropna() print(f"✅ Loaded {len(df)} records") return df except Exception as e: print(f"❌ Error loading data: {e}") sys.exit(1) def calculate_pbo(self, df: pd.DataFrame) -> dict: """ Calculate Probability of Backtest Overfit (PBO) Formula: PBO = 1 - (OOS_Sharpe / IS_Sharpe) Threshold: < 20% (low overfit probability) """ print("\n[PBO VERIFICATION]") try: # Calculate Sharpe ratios is_sharpe = np.mean(df['is_return']) / np.std(df['is_return']) oos_sharpe = np.mean(df['oos_return']) / np.std(df['oos_return']) # Calculate PBO pbo = 1 - (oos_sharpe / is_sharpe) if is_sharpe != 0 else 1.0 pbo_pct = pbo * 100 # Validation threshold = 20.0 # < 20% = PASS passed = pbo_pct < threshold result = { 'pbo': pbo, 'pbo_pct': pbo_pct, 'is_sharpe': is_sharpe, 'oos_sharpe': oos_sharpe, 'threshold': threshold, 'passed': passed, 'status': 'PASS ✅' if passed else 'FAIL ❌', 'confidence': 95.0 # Placeholder } print(f" PBO: {pbo_pct:.2f}%") print(f" IS Sharpe: {is_sharpe:.4f}") print(f" OOS Sharpe: {oos_sharpe:.4f}") print(f" Threshold: < {threshold}%") print(f" Result: {result['status']}") self.results['pbo'] = result return result except Exception as e: print(f"❌ Error calculating PBO: {e}") return {'status': 'ERROR', 'error': str(e)} def calculate_dsr(self, df: pd.DataFrame) -> dict: """ Calculate Daily Sharpe Ratio (DSR) Formula: DSR = mean(daily_returns) / std(daily_returns) Threshold: > 0.5 (acceptable risk-adjusted return) """ print("\n[DSR VERIFICATION]") try: mean_return = np.mean(df['daily_return']) std_return = np.std(df['daily_return']) dsr = mean_return / std_return if std_return != 0 else 0 # Monthly trend df['month'] = pd.to_datetime(df['trading_date']).dt.to_period('M') monthly_dsr = df.groupby('month')['daily_return'].apply( lambda x: np.mean(x) / np.std(x) if len(x) > 1 else 0 ) # Validation threshold = 0.5 passed = dsr > threshold monthly_consistency = monthly_dsr.std() < (dsr * 0.2) # < 20% variation result = { 'dsr': dsr, 'mean_return': mean_return, 'std_return': std_return, 'monthly_dsr': monthly_dsr.to_dict(), 'threshold': threshold, 'passed': passed, 'status': 'PASS ✅' if passed else 'FAIL ❌', 'monthly_consistency': monthly_consistency, 'confidence': 95.0 } print(f" DSR: {dsr:.4f}") print(f" Mean Daily Return: {mean_return:.6f}") print(f" Std Dev: {std_return:.6f}") print(f" Threshold: > {threshold}") print(f" Monthly Consistency: {'Good ✅' if monthly_consistency else 'Concerning ⚠️'}") print(f" Result: {result['status']}") self.results['dsr'] = result return result except Exception as e: print(f"❌ Error calculating DSR: {e}") return {'status': 'ERROR', 'error': str(e)} def calculate_oos_drift(self, df: pd.DataFrame) -> dict: """ Calculate Out-of-Sample Performance Drift Formula: Drift = |OOS_Performance - IS_Performance| / IS_Performance * 100 Threshold: < 2.5% (minimal model degradation) """ print("\n[OOS DRIFT VERIFICATION]") try: # Assume 'is_return' and 'oos_return' are cumulative performance is_cumulative = (1 + df['is_return']).cumprod().iloc[-1] - 1 oos_cumulative = (1 + df['oos_return']).cumprod().iloc[-1] - 1 # Calculate drift drift_pct = abs((oos_cumulative - is_cumulative) / is_cumulative * 100) if is_cumulative != 0 else 0 # Daily drift tracking daily_drift = [] for i in range(1, len(df)): is_perf = (1 + df['is_return'].iloc[:i]).prod() - 1 oos_perf = (1 + df['oos_return'].iloc[:i]).prod() - 1 drift = (oos_perf - is_perf) / is_perf * 100 if is_perf != 0 else 0 daily_drift.append(drift) max_daily_drift = max(daily_drift) if daily_drift else 0 avg_daily_drift = np.mean(daily_drift) if daily_drift else 0 # Validation threshold = 2.5 passed = drift_pct < threshold result = { 'oos_cumulative': oos_cumulative, 'is_cumulative': is_cumulative, 'drift_pct': drift_pct, 'max_daily_drift': max_daily_drift, 'avg_daily_drift': avg_daily_drift, 'threshold': threshold, 'passed': passed, 'status': 'PASS ✅' if passed else 'FAIL ❌', 'confidence': 95.0 } print(f" IS Cumulative Performance: {is_cumulative * 100:.2f}%") print(f" OOS Cumulative Performance: {oos_cumulative * 100:.2f}%") print(f" Overall Drift: {drift_pct:.2f}%") print(f" Max Daily Drift: {max_daily_drift:.2f}%") print(f" Avg Daily Drift: {avg_daily_drift:.4f}%") print(f" Threshold: < {threshold}%") print(f" Result: {result['status']}") self.results['oos'] = result return result except Exception as e: print(f"❌ Error calculating OOS drift: {e}") return {'status': 'ERROR', 'error': str(e)} def generate_go_nogo(self) -> dict: """Generate Go/No-Go decision based on all metrics""" print("\n[GO/NO-GO DECISION]") pbo_pass = self.results.get('pbo', {}).get('passed', False) dsr_pass = self.results.get('dsr', {}).get('passed', False) oos_pass = self.results.get('oos', {}).get('passed', False) all_pass = pbo_pass and dsr_pass and oos_pass decision = { 'pbo_pass': pbo_pass, 'dsr_pass': dsr_pass, 'oos_pass': oos_pass, 'all_pass': all_pass, 'decision': 'GO PHASE 3 🚀' if all_pass else 'RE-EVALUATE ⚠️', 'timestamp': self.timestamp.isoformat(), 'confidence': 95.0 if all_pass else 70.0 } print(f"\n PBO < 20%: {'✅ PASS' if pbo_pass else '❌ FAIL'}") print(f" DSR > 0.5: {'✅ PASS' if dsr_pass else '❌ FAIL'}") print(f" OOS < 2.5%: {'✅ PASS' if oos_pass else '❌ FAIL'}") print(f"\n DECISION: {decision['decision']}") self.results['decision'] = decision return decision def save_results(self, output_file: str = 'phase2_results.json'): """Save all results to JSON""" print(f"\n[SAVING RESULTS]") try: with open(output_file, 'w') as f: json.dump(self.results, f, indent=2, default=str) print(f"✅ Results saved to {output_file}") return output_file except Exception as e: print(f"❌ Error saving results: {e}") return None def run_all(self, output_file: str = 'phase2_results.json') -> dict: """Run all verifications""" print("=" * 60) print("PHASE 2 VERIFICATION ENGINE") print("=" * 60) df = self.load_data() self.calculate_pbo(df) self.calculate_dsr(df) self.calculate_oos_drift(df) self.generate_go_nogo() self.save_results(output_file) print("\n" + "=" * 60) return self.results def main(): """Main entry point""" if len(sys.argv) < 2: print("Usage: python phase2_verification_scripts.py ") print("Example: python phase2_verification_scripts.py Phase_1_Raw_Data.csv") sys.exit(1) data_file = sys.argv[1] verifier = Phase2Verification(data_file) results = verifier.run_all() # Exit with appropriate code if results.get('decision', {}).get('all_pass'): print("\n✅ All Phase 2 checks PASSED. Ready for Phase 3.") sys.exit(0) else: print("\n❌ Some Phase 2 checks FAILED. Re-evaluation required.") sys.exit(1) if __name__ == '__main__': main()