#!/usr/bin/env python3 """ Download 90 days of real market data from Databento for ML model training. Downloads OHLCV-1m data for 4 major futures symbols: - ES.FUT (E-mini S&P 500) - Stock index - NQ.FUT (E-mini NASDAQ) - Tech index - ZN.FUT (10-Year Treasury Note) - Fixed income - 6E.FUT (Euro FX) - Currency Target: 90 days × 4 symbols = ~180,000 bars Estimated cost: ~$2.00 (based on Databento pricing) Usage: # Set API key export DATABENTO_API_KEY='your-key-here' # Run download python3 download_ml_training_data.py # Or specify custom date range python3 download_ml_training_data.py --start-date 2024-01-01 --days 90 """ import os import sys from datetime import datetime, timedelta, timezone import argparse import databento as db # Configuration API_KEY = os.getenv("DATABENTO_API_KEY") OUTPUT_DIR = "test_data/real/databento/ml_training" SCHEMA = "ohlcv-1m" DATASET = "GLBX.MDP3" # Symbols to download (futures contracts) SYMBOLS = { "ES.FUT": { "name": "E-mini S&P 500", "category": "Stock Index", "description": "Most liquid equity index futures" }, "NQ.FUT": { "name": "E-mini NASDAQ-100", "category": "Stock Index", "description": "Tech-heavy equity index" }, "ZN.FUT": { "name": "10-Year Treasury Note", "category": "Fixed Income", "description": "Interest rate sensitivity" }, "6E.FUT": { "name": "Euro FX", "category": "Currency", "description": "EUR/USD exchange rate" } } def parse_args(): """Parse command line arguments.""" parser = argparse.ArgumentParser(description="Download ML training data from Databento") parser.add_argument( "--start-date", type=str, default="2024-01-02", help="Start date (YYYY-MM-DD, default: 2024-01-02)" ) parser.add_argument( "--days", type=int, default=90, help="Number of days to download (default: 90)" ) parser.add_argument( "--symbols", type=str, nargs="+", default=list(SYMBOLS.keys()), help="Symbols to download (default: all 4 symbols)" ) parser.add_argument( "--dry-run", action="store_true", help="Preview downloads without executing" ) return parser.parse_args() def generate_date_range(start_date_str, num_days): """Generate list of trading days (excluding weekends).""" start_date = datetime.strptime(start_date_str, "%Y-%m-%d") dates = [] current = start_date while len(dates) < num_days: # Skip weekends (Saturday=5, Sunday=6) if current.weekday() < 5: dates.append(current.strftime("%Y-%m-%d")) current += timedelta(days=1) return dates def estimate_cost(num_days, num_symbols): """Estimate download cost based on Databento pricing.""" # Rough estimate: $0.10-0.15 per symbol per day for 1-minute OHLCV cost_per_symbol_day = 0.12 return num_days * num_symbols * cost_per_symbol_day def download_symbol_data(client, symbol, date_str, output_dir): """Download data for a single symbol and date.""" try: # Parse date (full trading day UTC) start_date = datetime.strptime(date_str, "%Y-%m-%d").replace(tzinfo=timezone.utc) end_date = start_date.replace(hour=23, minute=59, second=59) # Build output filename output_file = os.path.join(output_dir, f"{symbol}_{SCHEMA}_{date_str}.dbn") # Skip if file already exists if os.path.exists(output_file): file_size_kb = os.path.getsize(output_file) / 1024 return { "status": "skipped", "reason": "File already exists", "path": output_file, "size_kb": file_size_kb } # Request data data = client.timeseries.get_range( dataset=DATASET, symbols=[symbol], schema=SCHEMA, start=start_date.isoformat(), end=end_date.isoformat(), ) # Write to file data.to_file(output_file) # Verify data file_size = os.path.getsize(output_file) file_size_kb = file_size / 1024 try: store = db.DBNStore.from_file(output_file) df = store.to_df() record_count = len(df) if record_count == 0: return { "status": "warning", "reason": "No records in file (holiday/no trading)", "path": output_file, "size_kb": file_size_kb, "records": 0 } return { "status": "success", "path": output_file, "size_kb": file_size_kb, "records": record_count, "price_min": float(df['close'].min()), "price_max": float(df['close'].max()), "volume": int(df['volume'].sum()) } except Exception as e: return { "status": "success", "path": output_file, "size_kb": file_size_kb, "warning": f"Could not verify: {e}" } except Exception as e: return { "status": "error", "reason": str(e) } def main(): """Main download orchestration.""" args = parse_args() print("=" * 80) print("ML Training Data Download - Databento") print("=" * 80) print() # Validate API key if not API_KEY: print("❌ ERROR: DATABENTO_API_KEY not found in environment!") print() print("Set it with:") print(" export DATABENTO_API_KEY='your-key-here'") print() print("Get your API key from: https://databento.com/") sys.exit(1) # Generate date range dates = generate_date_range(args.start_date, args.days) # Estimate cost estimated_cost = estimate_cost(len(dates), len(args.symbols)) # Preview print(f"📊 Download Configuration:") print(f" Start date: {args.start_date}") print(f" Trading days: {len(dates)}") print(f" Symbols: {len(args.symbols)} ({', '.join(args.symbols)})") print(f" Schema: {SCHEMA}") print(f" Dataset: {DATASET}") print(f" Output: {OUTPUT_DIR}") print() print(f"📦 Total Downloads: {len(dates) * len(args.symbols)} files") print(f"💰 Estimated Cost: ${estimated_cost:.2f}") print() # Symbol details print("📋 Symbols to Download:") for symbol in args.symbols: if symbol in SYMBOLS: info = SYMBOLS[symbol] print(f" • {symbol}: {info['name']} ({info['category']})") print(f" {info['description']}") else: print(f" • {symbol}: Unknown symbol") print() # Dry run exit if args.dry_run: print("🔍 DRY RUN: Preview complete. Add --no-dry-run to execute.") print() print("First 5 dates to download:") for date in dates[:5]: print(f" • {date}") print(f" ... ({len(dates) - 5} more dates)") return # Confirm before proceeding print("⚠️ This will download data and incur costs (~${:.2f})".format(estimated_cost)) response = input("Proceed with download? (yes/no): ") if response.lower() not in ["yes", "y"]: print("Download cancelled.") sys.exit(0) print() # Create output directory os.makedirs(OUTPUT_DIR, exist_ok=True) # Initialize Databento client try: client = db.Historical(API_KEY) print("✅ Databento client initialized") print() except Exception as e: print(f"❌ Failed to initialize Databento client: {e}") sys.exit(1) # Track progress results = { "success": [], "skipped": [], "warning": [], "error": [] } total_records = 0 total_size_kb = 0 # Download all combinations total_downloads = len(args.symbols) * len(dates) current_download = 0 for symbol in args.symbols: print("-" * 80) print(f"📥 Downloading: {symbol}") if symbol in SYMBOLS: print(f" {SYMBOLS[symbol]['name']} - {SYMBOLS[symbol]['category']}") print("-" * 80) print() for date_str in dates: current_download += 1 progress = (current_download / total_downloads) * 100 print(f"[{current_download}/{total_downloads} - {progress:.1f}%] {symbol} @ {date_str}...", end=" ") result = download_symbol_data(client, symbol, date_str, OUTPUT_DIR) status = result["status"] results[status].append((symbol, date_str, result)) if status == "success": records = result.get("records", 0) size_kb = result.get("size_kb", 0) total_records += records total_size_kb += size_kb print(f"✅ {records} bars, {size_kb:.1f} KB") elif status == "skipped": print(f"⏭️ {result['reason']}") elif status == "warning": print(f"⚠️ {result['reason']}") elif status == "error": print(f"❌ {result['reason']}") print() # Summary print() print("=" * 80) print("📊 DOWNLOAD SUMMARY") print("=" * 80) print() print(f"✅ Successful: {len(results['success'])}") print(f"⏭️ Skipped: {len(results['skipped'])}") print(f"⚠️ Warnings: {len(results['warning'])}") print(f"❌ Errors: {len(results['error'])}") print() print(f"📊 Total Records: {total_records:,} bars") print(f"💾 Total Size: {total_size_kb:,.1f} KB ({total_size_kb/1024:.1f} MB)") print(f"💰 Estimated Cost: ${estimated_cost:.2f}") print() if results['error']: print("❌ ERRORS:") for symbol, date, result in results['error'][:10]: print(f" • {symbol} @ {date}: {result['reason']}") if len(results['error']) > 10: print(f" ... and {len(results['error']) - 10} more") print() # Success criteria success_rate = len(results['success']) / total_downloads * 100 if total_downloads > 0 else 0 print("📋 NEXT STEPS:") print("1. Run ML readiness validation with new data:") print(" cargo test -p ml --test ml_readiness_validation_tests") print() print("2. Run training time benchmarks:") print(" python3 benchmark_training_time.py") print() print("3. Calculate realistic training timeline from benchmarks") print() if success_rate >= 80: print(f"✅ SUCCESS: Downloaded {success_rate:.1f}% of requested data!") print(f" Ready for ML training benchmarks on RTX 3050 Ti") elif success_rate >= 50: print(f"⚠️ PARTIAL SUCCESS: Downloaded {success_rate:.1f}% of data") print(f" May be sufficient for benchmarking, but consider re-downloading missing files") else: print(f"❌ ERROR: Only downloaded {success_rate:.1f}% of data") print(f" Check errors above and retry") sys.exit(1) if __name__ == "__main__": main()