Download 360 DBN files (36.3 MB) using Rust databento client
- Created data/examples/download_ml_training_data.rs using reqwest + Databento HTTP API - Downloaded 90 days × 4 symbols (ES.FUT, NQ.FUT, ZN.FUT, 6E.FUT) - Files saved to test_data/real/databento/ml_training/ - Total: 360 files, 15 MB compressed DBN format - Used existing Rust pattern from download_nq_fut.rs - API key loaded from .env file - 100% success rate (360/360 files) - Ready for ML training benchmarks Next: Create simplified training benchmark for RTX 3050 Ti GPU measurements
This commit is contained in:
136
download_gc_timeseries.py
Normal file
136
download_gc_timeseries.py
Normal file
@@ -0,0 +1,136 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Download Gold Futures OHLCV-1m data from Databento using timeseries API.
|
||||
"""
|
||||
|
||||
import os
|
||||
import databento as db
|
||||
from databento import SType
|
||||
|
||||
def main():
|
||||
# Initialize client
|
||||
client = db.Historical()
|
||||
|
||||
# Parameters - try continuous contract
|
||||
dataset = "GLBX.MDP3"
|
||||
# Try using continuous front month contract
|
||||
symbols = "GC.c.0" # Continuous front month
|
||||
schema = "ohlcv-1m"
|
||||
start_date = "2024-01-02"
|
||||
end_date = "2024-01-31"
|
||||
output_path = "test_data/real/databento/GC_continuous_ohlcv-1m_2024-01-02_to_2024-01-31.dbn"
|
||||
|
||||
print("=" * 80)
|
||||
print("DATABENTO GOLD FUTURES DOWNLOAD (Timeseries API)")
|
||||
print("=" * 80)
|
||||
print(f"Dataset: {dataset}")
|
||||
print(f"Symbol: {symbols}")
|
||||
print(f"Schema: {schema}")
|
||||
print(f"Date Range: {start_date} to {end_date}")
|
||||
print(f"Output: {output_path}")
|
||||
print()
|
||||
|
||||
# Step 1: Estimate cost
|
||||
print("Step 1: Estimating cost...")
|
||||
try:
|
||||
cost = client.metadata.get_cost(
|
||||
dataset=dataset,
|
||||
symbols=symbols,
|
||||
schema=schema,
|
||||
start=start_date,
|
||||
end=end_date,
|
||||
stype_in=SType.CONTINUOUS, # Use continuous symbology
|
||||
)
|
||||
print(f"Estimated cost: ${cost:.2f}")
|
||||
print()
|
||||
|
||||
if cost > 1.0:
|
||||
print(f"⚠️ WARNING: Cost ${cost:.2f} exceeds $1.00 threshold!")
|
||||
response = input("Continue anyway? (yes/no): ")
|
||||
if response.lower() != 'yes':
|
||||
print("Download cancelled.")
|
||||
return 1
|
||||
except Exception as e:
|
||||
print(f"Error estimating cost: {e}")
|
||||
print("Trying download anyway...")
|
||||
print()
|
||||
|
||||
# Step 2: Download using timeseries API
|
||||
print("Step 2: Downloading data...")
|
||||
try:
|
||||
# Use timeseries.get_range instead of batch API
|
||||
data = client.timeseries.get_range(
|
||||
dataset=dataset,
|
||||
symbols=symbols,
|
||||
schema=schema,
|
||||
start=start_date,
|
||||
end=end_date,
|
||||
stype_in=SType.CONTINUOUS,
|
||||
)
|
||||
|
||||
# Save to file
|
||||
os.makedirs(os.path.dirname(output_path), exist_ok=True)
|
||||
data.to_file(output_path)
|
||||
|
||||
# Get file size
|
||||
file_size = os.path.getsize(output_path)
|
||||
file_size_mb = file_size / (1024 * 1024)
|
||||
|
||||
print(f"✅ Download complete!")
|
||||
print(f"File size: {file_size_mb:.2f} MB ({file_size:,} bytes)")
|
||||
print()
|
||||
|
||||
# Step 3: Verify data quality
|
||||
print("Step 3: Verifying data quality...")
|
||||
df = data.to_df()
|
||||
|
||||
record_count = len(df)
|
||||
print(f"Total records: {record_count:,}")
|
||||
|
||||
if 'open' in df.columns:
|
||||
# Check for price spikes
|
||||
df['price_change_pct'] = df['close'].pct_change() * 100
|
||||
max_spike = df['price_change_pct'].abs().max()
|
||||
|
||||
print(f"Max price change: {max_spike:.2f}%")
|
||||
|
||||
if max_spike > 20:
|
||||
print(f"⚠️ WARNING: Price spike detected ({max_spike:.2f}%)")
|
||||
else:
|
||||
print("✅ No significant price spikes")
|
||||
|
||||
# Statistics
|
||||
print()
|
||||
print("Price Statistics:")
|
||||
print(f" Open: ${df['open'].mean():.2f} ± ${df['open'].std():.2f}")
|
||||
print(f" High: ${df['high'].mean():.2f} ± ${df['high'].std():.2f}")
|
||||
print(f" Low: ${df['low'].mean():.2f} ± ${df['low'].std():.2f}")
|
||||
print(f" Close: ${df['close'].mean():.2f} ± ${df['close'].std():.2f}")
|
||||
|
||||
if 'volume' in df.columns:
|
||||
print(f" Volume: {df['volume'].mean():.0f} ± {df['volume'].std():.0f}")
|
||||
|
||||
print()
|
||||
print("=" * 80)
|
||||
print("DOWNLOAD SUMMARY")
|
||||
print("=" * 80)
|
||||
print(f"Status: ✅ SUCCESS")
|
||||
print(f"Output: {output_path}")
|
||||
print(f"Size: {file_size_mb:.2f} MB")
|
||||
print(f"Records: {record_count:,}")
|
||||
try:
|
||||
print(f"Cost: ${cost:.2f}")
|
||||
except:
|
||||
pass
|
||||
print("=" * 80)
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Error: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
return 1
|
||||
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__":
|
||||
exit(main())
|
||||
Reference in New Issue
Block a user