chore: Major codebase cleanup - remove deprecated files and organize structure
- Docker: Delete 23 deprecated Dockerfiles, fix CI/CD to use Dockerfile.foxhunt-build - Config: Remove 36 .env files, keep 4 essential, delete config/environments/ - Docs: Archive 614 Wave D files to docs/archive/wave_d/, 95% reduction in root - Scripts: Delete 56 deprecated scripts, keep 58 production-critical (49% reduction) - Python: Organize 37 scripts into scripts/python/ subdirectories, delete ml/python/ - Build: Remove 1GB artifacts, delete old venvs, clean Python cache from git - Migrations: Delete deprecated directory (4,432 lines), remove duplicate database/migrations/ - Infrastructure: Delete deployment/ (61 files), docs/scripts/ (8 files) Total impact: ~2,500 files cleaned, 750MB+ space freed, zero production impact All deleted scripts backed up to archives. runpod/ and tests/runpod/ preserved. data_acquisition_service retained per user request.
This commit is contained in:
366
scripts/python/data/download_ml_training_data.py
Executable file
366
scripts/python/data/download_ml_training_data.py
Executable file
@@ -0,0 +1,366 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Download 90 days of real market data from Databento for ML model training.
|
||||
|
||||
Downloads OHLCV-1m data for 4 major futures symbols:
|
||||
- ES.FUT (E-mini S&P 500) - Stock index
|
||||
- NQ.FUT (E-mini NASDAQ) - Tech index
|
||||
- ZN.FUT (10-Year Treasury Note) - Fixed income
|
||||
- 6E.FUT (Euro FX) - Currency
|
||||
|
||||
Target: 90 days × 4 symbols = ~180,000 bars
|
||||
Estimated cost: ~$2.00 (based on Databento pricing)
|
||||
|
||||
Usage:
|
||||
# Set API key
|
||||
export DATABENTO_API_KEY='your-key-here'
|
||||
|
||||
# Run download
|
||||
python3 download_ml_training_data.py
|
||||
|
||||
# Or specify custom date range
|
||||
python3 download_ml_training_data.py --start-date 2024-01-01 --days 90
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timedelta, timezone
|
||||
import argparse
|
||||
import databento as db
|
||||
|
||||
# Configuration
|
||||
API_KEY = os.getenv("DATABENTO_API_KEY")
|
||||
OUTPUT_DIR = "test_data/real/databento/ml_training"
|
||||
SCHEMA = "ohlcv-1m"
|
||||
DATASET = "GLBX.MDP3"
|
||||
|
||||
# Symbols to download (futures contracts)
|
||||
SYMBOLS = {
|
||||
"ES.FUT": {
|
||||
"name": "E-mini S&P 500",
|
||||
"category": "Stock Index",
|
||||
"description": "Most liquid equity index futures"
|
||||
},
|
||||
"NQ.FUT": {
|
||||
"name": "E-mini NASDAQ-100",
|
||||
"category": "Stock Index",
|
||||
"description": "Tech-heavy equity index"
|
||||
},
|
||||
"ZN.FUT": {
|
||||
"name": "10-Year Treasury Note",
|
||||
"category": "Fixed Income",
|
||||
"description": "Interest rate sensitivity"
|
||||
},
|
||||
"6E.FUT": {
|
||||
"name": "Euro FX",
|
||||
"category": "Currency",
|
||||
"description": "EUR/USD exchange rate"
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def parse_args():
|
||||
"""Parse command line arguments."""
|
||||
parser = argparse.ArgumentParser(description="Download ML training data from Databento")
|
||||
parser.add_argument(
|
||||
"--start-date",
|
||||
type=str,
|
||||
default="2024-01-02",
|
||||
help="Start date (YYYY-MM-DD, default: 2024-01-02)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--days",
|
||||
type=int,
|
||||
default=90,
|
||||
help="Number of days to download (default: 90)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--symbols",
|
||||
type=str,
|
||||
nargs="+",
|
||||
default=list(SYMBOLS.keys()),
|
||||
help="Symbols to download (default: all 4 symbols)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dry-run",
|
||||
action="store_true",
|
||||
help="Preview downloads without executing"
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def generate_date_range(start_date_str, num_days):
|
||||
"""Generate list of trading days (excluding weekends)."""
|
||||
start_date = datetime.strptime(start_date_str, "%Y-%m-%d")
|
||||
dates = []
|
||||
|
||||
current = start_date
|
||||
while len(dates) < num_days:
|
||||
# Skip weekends (Saturday=5, Sunday=6)
|
||||
if current.weekday() < 5:
|
||||
dates.append(current.strftime("%Y-%m-%d"))
|
||||
current += timedelta(days=1)
|
||||
|
||||
return dates
|
||||
|
||||
|
||||
def estimate_cost(num_days, num_symbols):
|
||||
"""Estimate download cost based on Databento pricing."""
|
||||
# Rough estimate: $0.10-0.15 per symbol per day for 1-minute OHLCV
|
||||
cost_per_symbol_day = 0.12
|
||||
return num_days * num_symbols * cost_per_symbol_day
|
||||
|
||||
|
||||
def download_symbol_data(client, symbol, date_str, output_dir):
|
||||
"""Download data for a single symbol and date."""
|
||||
try:
|
||||
# Parse date (full trading day UTC)
|
||||
start_date = datetime.strptime(date_str, "%Y-%m-%d").replace(tzinfo=timezone.utc)
|
||||
end_date = start_date.replace(hour=23, minute=59, second=59)
|
||||
|
||||
# Build output filename
|
||||
output_file = os.path.join(output_dir, f"{symbol}_{SCHEMA}_{date_str}.dbn")
|
||||
|
||||
# Skip if file already exists
|
||||
if os.path.exists(output_file):
|
||||
file_size_kb = os.path.getsize(output_file) / 1024
|
||||
return {
|
||||
"status": "skipped",
|
||||
"reason": "File already exists",
|
||||
"path": output_file,
|
||||
"size_kb": file_size_kb
|
||||
}
|
||||
|
||||
# Request data
|
||||
data = client.timeseries.get_range(
|
||||
dataset=DATASET,
|
||||
symbols=[symbol],
|
||||
schema=SCHEMA,
|
||||
start=start_date.isoformat(),
|
||||
end=end_date.isoformat(),
|
||||
)
|
||||
|
||||
# Write to file
|
||||
data.to_file(output_file)
|
||||
|
||||
# Verify data
|
||||
file_size = os.path.getsize(output_file)
|
||||
file_size_kb = file_size / 1024
|
||||
|
||||
try:
|
||||
store = db.DBNStore.from_file(output_file)
|
||||
df = store.to_df()
|
||||
record_count = len(df)
|
||||
|
||||
if record_count == 0:
|
||||
return {
|
||||
"status": "warning",
|
||||
"reason": "No records in file (holiday/no trading)",
|
||||
"path": output_file,
|
||||
"size_kb": file_size_kb,
|
||||
"records": 0
|
||||
}
|
||||
|
||||
return {
|
||||
"status": "success",
|
||||
"path": output_file,
|
||||
"size_kb": file_size_kb,
|
||||
"records": record_count,
|
||||
"price_min": float(df['close'].min()),
|
||||
"price_max": float(df['close'].max()),
|
||||
"volume": int(df['volume'].sum())
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
return {
|
||||
"status": "success",
|
||||
"path": output_file,
|
||||
"size_kb": file_size_kb,
|
||||
"warning": f"Could not verify: {e}"
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
return {
|
||||
"status": "error",
|
||||
"reason": str(e)
|
||||
}
|
||||
|
||||
|
||||
def main():
|
||||
"""Main download orchestration."""
|
||||
args = parse_args()
|
||||
|
||||
print("=" * 80)
|
||||
print("ML Training Data Download - Databento")
|
||||
print("=" * 80)
|
||||
print()
|
||||
|
||||
# Validate API key
|
||||
if not API_KEY:
|
||||
print("❌ ERROR: DATABENTO_API_KEY not found in environment!")
|
||||
print()
|
||||
print("Set it with:")
|
||||
print(" export DATABENTO_API_KEY='your-key-here'")
|
||||
print()
|
||||
print("Get your API key from: https://databento.com/")
|
||||
sys.exit(1)
|
||||
|
||||
# Generate date range
|
||||
dates = generate_date_range(args.start_date, args.days)
|
||||
|
||||
# Estimate cost
|
||||
estimated_cost = estimate_cost(len(dates), len(args.symbols))
|
||||
|
||||
# Preview
|
||||
print(f"📊 Download Configuration:")
|
||||
print(f" Start date: {args.start_date}")
|
||||
print(f" Trading days: {len(dates)}")
|
||||
print(f" Symbols: {len(args.symbols)} ({', '.join(args.symbols)})")
|
||||
print(f" Schema: {SCHEMA}")
|
||||
print(f" Dataset: {DATASET}")
|
||||
print(f" Output: {OUTPUT_DIR}")
|
||||
print()
|
||||
print(f"📦 Total Downloads: {len(dates) * len(args.symbols)} files")
|
||||
print(f"💰 Estimated Cost: ${estimated_cost:.2f}")
|
||||
print()
|
||||
|
||||
# Symbol details
|
||||
print("📋 Symbols to Download:")
|
||||
for symbol in args.symbols:
|
||||
if symbol in SYMBOLS:
|
||||
info = SYMBOLS[symbol]
|
||||
print(f" • {symbol}: {info['name']} ({info['category']})")
|
||||
print(f" {info['description']}")
|
||||
else:
|
||||
print(f" • {symbol}: Unknown symbol")
|
||||
print()
|
||||
|
||||
# Dry run exit
|
||||
if args.dry_run:
|
||||
print("🔍 DRY RUN: Preview complete. Add --no-dry-run to execute.")
|
||||
print()
|
||||
print("First 5 dates to download:")
|
||||
for date in dates[:5]:
|
||||
print(f" • {date}")
|
||||
print(f" ... ({len(dates) - 5} more dates)")
|
||||
return
|
||||
|
||||
# Confirm before proceeding
|
||||
print("⚠️ This will download data and incur costs (~${:.2f})".format(estimated_cost))
|
||||
response = input("Proceed with download? (yes/no): ")
|
||||
if response.lower() not in ["yes", "y"]:
|
||||
print("Download cancelled.")
|
||||
sys.exit(0)
|
||||
print()
|
||||
|
||||
# Create output directory
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
|
||||
# Initialize Databento client
|
||||
try:
|
||||
client = db.Historical(API_KEY)
|
||||
print("✅ Databento client initialized")
|
||||
print()
|
||||
except Exception as e:
|
||||
print(f"❌ Failed to initialize Databento client: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
# Track progress
|
||||
results = {
|
||||
"success": [],
|
||||
"skipped": [],
|
||||
"warning": [],
|
||||
"error": []
|
||||
}
|
||||
total_records = 0
|
||||
total_size_kb = 0
|
||||
|
||||
# Download all combinations
|
||||
total_downloads = len(args.symbols) * len(dates)
|
||||
current_download = 0
|
||||
|
||||
for symbol in args.symbols:
|
||||
print("-" * 80)
|
||||
print(f"📥 Downloading: {symbol}")
|
||||
if symbol in SYMBOLS:
|
||||
print(f" {SYMBOLS[symbol]['name']} - {SYMBOLS[symbol]['category']}")
|
||||
print("-" * 80)
|
||||
print()
|
||||
|
||||
for date_str in dates:
|
||||
current_download += 1
|
||||
progress = (current_download / total_downloads) * 100
|
||||
|
||||
print(f"[{current_download}/{total_downloads} - {progress:.1f}%] {symbol} @ {date_str}...", end=" ")
|
||||
|
||||
result = download_symbol_data(client, symbol, date_str, OUTPUT_DIR)
|
||||
|
||||
status = result["status"]
|
||||
results[status].append((symbol, date_str, result))
|
||||
|
||||
if status == "success":
|
||||
records = result.get("records", 0)
|
||||
size_kb = result.get("size_kb", 0)
|
||||
total_records += records
|
||||
total_size_kb += size_kb
|
||||
print(f"✅ {records} bars, {size_kb:.1f} KB")
|
||||
elif status == "skipped":
|
||||
print(f"⏭️ {result['reason']}")
|
||||
elif status == "warning":
|
||||
print(f"⚠️ {result['reason']}")
|
||||
elif status == "error":
|
||||
print(f"❌ {result['reason']}")
|
||||
|
||||
print()
|
||||
|
||||
# Summary
|
||||
print()
|
||||
print("=" * 80)
|
||||
print("📊 DOWNLOAD SUMMARY")
|
||||
print("=" * 80)
|
||||
print()
|
||||
print(f"✅ Successful: {len(results['success'])}")
|
||||
print(f"⏭️ Skipped: {len(results['skipped'])}")
|
||||
print(f"⚠️ Warnings: {len(results['warning'])}")
|
||||
print(f"❌ Errors: {len(results['error'])}")
|
||||
print()
|
||||
print(f"📊 Total Records: {total_records:,} bars")
|
||||
print(f"💾 Total Size: {total_size_kb:,.1f} KB ({total_size_kb/1024:.1f} MB)")
|
||||
print(f"💰 Estimated Cost: ${estimated_cost:.2f}")
|
||||
print()
|
||||
|
||||
if results['error']:
|
||||
print("❌ ERRORS:")
|
||||
for symbol, date, result in results['error'][:10]:
|
||||
print(f" • {symbol} @ {date}: {result['reason']}")
|
||||
if len(results['error']) > 10:
|
||||
print(f" ... and {len(results['error']) - 10} more")
|
||||
print()
|
||||
|
||||
# Success criteria
|
||||
success_rate = len(results['success']) / total_downloads * 100 if total_downloads > 0 else 0
|
||||
|
||||
print("📋 NEXT STEPS:")
|
||||
print("1. Run ML readiness validation with new data:")
|
||||
print(" cargo test -p ml --test ml_readiness_validation_tests")
|
||||
print()
|
||||
print("2. Run training time benchmarks:")
|
||||
print(" python3 benchmark_training_time.py")
|
||||
print()
|
||||
print("3. Calculate realistic training timeline from benchmarks")
|
||||
print()
|
||||
|
||||
if success_rate >= 80:
|
||||
print(f"✅ SUCCESS: Downloaded {success_rate:.1f}% of requested data!")
|
||||
print(f" Ready for ML training benchmarks on RTX 3050 Ti")
|
||||
elif success_rate >= 50:
|
||||
print(f"⚠️ PARTIAL SUCCESS: Downloaded {success_rate:.1f}% of data")
|
||||
print(f" May be sufficient for benchmarking, but consider re-downloading missing files")
|
||||
else:
|
||||
print(f"❌ ERROR: Only downloaded {success_rate:.1f}% of data")
|
||||
print(f" Check errors above and retry")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user