#!/usr/bin/env python3 """ Download Gold Futures OHLCV-1m data from Databento using timeseries API. """ import os import databento as db from databento import SType def main(): # Initialize client client = db.Historical() # Parameters - try continuous contract dataset = "GLBX.MDP3" # Try using continuous front month contract symbols = "GC.c.0" # Continuous front month schema = "ohlcv-1m" start_date = "2024-01-02" end_date = "2024-01-31" output_path = "test_data/real/databento/GC_continuous_ohlcv-1m_2024-01-02_to_2024-01-31.dbn" print("=" * 80) print("DATABENTO GOLD FUTURES DOWNLOAD (Timeseries API)") print("=" * 80) print(f"Dataset: {dataset}") print(f"Symbol: {symbols}") print(f"Schema: {schema}") print(f"Date Range: {start_date} to {end_date}") print(f"Output: {output_path}") print() # Step 1: Estimate cost print("Step 1: Estimating cost...") try: cost = client.metadata.get_cost( dataset=dataset, symbols=symbols, schema=schema, start=start_date, end=end_date, stype_in=SType.CONTINUOUS, # Use continuous symbology ) print(f"Estimated cost: ${cost:.2f}") print() if cost > 1.0: print(f"⚠️ WARNING: Cost ${cost:.2f} exceeds $1.00 threshold!") response = input("Continue anyway? (yes/no): ") if response.lower() != 'yes': print("Download cancelled.") return 1 except Exception as e: print(f"Error estimating cost: {e}") print("Trying download anyway...") print() # Step 2: Download using timeseries API print("Step 2: Downloading data...") try: # Use timeseries.get_range instead of batch API data = client.timeseries.get_range( dataset=dataset, symbols=symbols, schema=schema, start=start_date, end=end_date, stype_in=SType.CONTINUOUS, ) # Save to file os.makedirs(os.path.dirname(output_path), exist_ok=True) data.to_file(output_path) # Get file size file_size = os.path.getsize(output_path) file_size_mb = file_size / (1024 * 1024) print(f"✅ Download complete!") print(f"File size: {file_size_mb:.2f} MB ({file_size:,} bytes)") print() # Step 3: Verify data quality print("Step 3: Verifying data quality...") df = data.to_df() record_count = len(df) print(f"Total records: {record_count:,}") if 'open' in df.columns: # Check for price spikes df['price_change_pct'] = df['close'].pct_change() * 100 max_spike = df['price_change_pct'].abs().max() print(f"Max price change: {max_spike:.2f}%") if max_spike > 20: print(f"⚠️ WARNING: Price spike detected ({max_spike:.2f}%)") else: print("✅ No significant price spikes") # Statistics print() print("Price Statistics:") print(f" Open: ${df['open'].mean():.2f} ± ${df['open'].std():.2f}") print(f" High: ${df['high'].mean():.2f} ± ${df['high'].std():.2f}") print(f" Low: ${df['low'].mean():.2f} ± ${df['low'].std():.2f}") print(f" Close: ${df['close'].mean():.2f} ± ${df['close'].std():.2f}") if 'volume' in df.columns: print(f" Volume: {df['volume'].mean():.0f} ± {df['volume'].std():.0f}") print() print("=" * 80) print("DOWNLOAD SUMMARY") print("=" * 80) print(f"Status: ✅ SUCCESS") print(f"Output: {output_path}") print(f"Size: {file_size_mb:.2f} MB") print(f"Records: {record_count:,}") try: print(f"Cost: ${cost:.2f}") except: pass print("=" * 80) except Exception as e: print(f"❌ Error: {e}") import traceback traceback.print_exc() return 1 return 0 if __name__ == "__main__": exit(main())