#!/usr/bin/env bash ################################################################################ # DQN Hyperopt Redeployment WITH CHECKPOINT SAVING FIX # Generated: 2025-11-02 # # CONTEXT: # - Previous run: dqn_hyperopt_optimized_20251102_220747 (NO CHECKPOINTS SAVED) # - Root cause: No-op checkpoint callbacks in ml/src/hyperopt/adapters/dqn.rs # - Fix: Agents 1-3 replaced no-op callbacks with safetensors save logic # - This run: Will save .safetensors files for all 50 trials # # STRATEGY: # - Hybrid approach with 5-minute validation gate # - Early abort if checkpoints still not saving (saves $0.09 of $0.11 budget) # - Real-time monitoring with S3 checkpoint verification # - Post-completion validation with model loadability tests # # COST/TIME ESTIMATES: # - Success: 36 min, $0.11 (same as before, but WITH checkpoints) # - Early abort: 9 min, $0.02 (saves $0.09 if fix is broken) # - Expected value: $0.096 (85% success probability) ################################################################################ set -euo pipefail # Color codes for output RED='\033[0;31m' GREEN='\033[0;32m' YELLOW='\033[1;33m' BLUE='\033[0;34m' NC='\033[0m' # No Color # Configuration GPU_TYPE="${1:-RTX A4000}" DOCKER_IMAGE="jgrusewski/foxhunt:dqn-checkpoint-fix-$(date +%Y%m%d)" PARQUET_FILE="/runpod-volume/test_data/ES_FUT_180d.parquet" OUTPUT_BASE="/runpod-volume/ml_training" TIMESTAMP=$(date +%Y%m%d_%H%M%S) OUTPUT_DIR="${OUTPUT_BASE}/dqn_hyperopt_checkpoints_${TIMESTAMP}" # Best hyperparameters from previous run (Trial #8) TRIALS=50 EPOCHS=20 N_INITIAL=2 SEED=42 # Early stopping configuration EARLY_STOPPING_PLATEAU_WINDOW=5 EARLY_STOPPING_MIN_EPOCHS=10 # Validation gate settings VALIDATION_GATE_MINUTES=5 MIN_CHECKPOINTS_AT_GATE=10 # Expect at least 10 trials completed # S3 configuration S3_BUCKET="se3zdnb5o4" S3_ENDPOINT="https://s3api-eur-is-1.runpod.io" AWS_PROFILE="runpod" # Global variables POD_ID="" VALIDATION_PASSED=false ################################################################################ # Function: Print colored message ################################################################################ print_msg() { local color=$1 shift echo -e "${color}$@${NC}" } ################################################################################ # Function: Print section header ################################################################################ print_header() { echo "" echo "==========================================" print_msg "$BLUE" "$@" echo "==========================================" } ################################################################################ # PHASE 1: PRE-FLIGHT CHECKS ################################################################################ pre_flight_checks() { print_header "PHASE 1: PRE-FLIGHT CHECKS" # Check 1: Verify checkpoint fix is in place print_msg "$YELLOW" "[1/3] Verifying checkpoint fix in code..." if grep -q "No-op checkpoint callback for hyperopt trials" /home/jgrusewski/Work/foxhunt/ml/src/hyperopt/adapters/dqn.rs; then print_msg "$RED" "FAILED: No-op checkpoint callbacks still present!" print_msg "$RED" "The fix from Agents 1-3 has not been applied." print_msg "$RED" "Please wait for Agents 1-3 to complete their work." exit 1 fi print_msg "$GREEN" "PASSED: No-op callbacks have been removed" # Check 2: Verify foxhunt-deploy CLI exists print_msg "$YELLOW" "[2/3] Verifying foxhunt-deploy CLI..." if [ ! -f "/home/jgrusewski/Work/foxhunt/target/release/foxhunt-deploy" ]; then print_msg "$RED" "FAILED: foxhunt-deploy CLI not found" print_msg "$YELLOW" "Building foxhunt-deploy..." cd /home/jgrusewski/Work/foxhunt cargo build --release -p foxhunt-deploy || { print_msg "$RED" "FAILED: Could not build foxhunt-deploy" exit 1 } fi print_msg "$GREEN" "PASSED: foxhunt-deploy CLI ready" # Check 3: Verify AWS credentials for S3 print_msg "$YELLOW" "[3/3] Verifying AWS S3 credentials..." if ! aws s3 ls "s3://${S3_BUCKET}/" --profile "${AWS_PROFILE}" --endpoint-url "${S3_ENDPOINT}" > /dev/null 2>&1; then print_msg "$RED" "FAILED: Cannot access S3 bucket ${S3_BUCKET}" print_msg "$RED" "Please configure AWS credentials for profile '${AWS_PROFILE}'" exit 1 fi print_msg "$GREEN" "PASSED: S3 access verified" print_msg "$GREEN" "\nAll pre-flight checks passed!" } ################################################################################ # PHASE 2: BUILD AND DEPLOY ################################################################################ build_and_deploy() { print_header "PHASE 2: BUILD AND DEPLOY" # Build Docker image with checkpoint fix print_msg "$YELLOW" "Building Docker image with checkpoint fix..." print_msg "$BLUE" "Image tag: ${DOCKER_IMAGE}" cd /home/jgrusewski/Work/foxhunt docker build -f Dockerfile.foxhunt-build -t "${DOCKER_IMAGE}" . || { print_msg "$RED" "FAILED: Docker build failed" exit 1 } print_msg "$GREEN" "Docker build successful" # Push to Docker Hub print_msg "$YELLOW" "Pushing image to Docker Hub..." docker push "${DOCKER_IMAGE}" || { print_msg "$RED" "FAILED: Docker push failed" exit 1 } print_msg "$GREEN" "Docker push successful" # Build hyperopt command local COMMAND="hyperopt_dqn_demo \ --parquet-file ${PARQUET_FILE} \ --trials ${TRIALS} \ --epochs ${EPOCHS} \ --n-initial ${N_INITIAL} \ --seed ${SEED} \ --base-dir ${OUTPUT_DIR} \ --run-type hyperopt \ --early-stopping-plateau-window ${EARLY_STOPPING_PLATEAU_WINDOW} \ --early-stopping-min-epochs ${EARLY_STOPPING_MIN_EPOCHS}" # Deploy to RunPod print_msg "$YELLOW" "Deploying to RunPod..." print_msg "$BLUE" "GPU: ${GPU_TYPE}" print_msg "$BLUE" "Command: ${COMMAND}" local DEPLOY_OUTPUT DEPLOY_OUTPUT=$(/home/jgrusewski/Work/foxhunt/target/release/foxhunt-deploy deploy \ --gpu-type "${GPU_TYPE}" \ --tag "$(basename ${DOCKER_IMAGE} | cut -d: -f2)" \ --command "${COMMAND}" \ --name "dqn-hyperopt-checkpoints-$(date +%Y%m%d-%H%M%S)" \ --yes 2>&1) || { print_msg "$RED" "FAILED: Deployment failed" echo "$DEPLOY_OUTPUT" exit 1 } # Extract pod ID from deployment output POD_ID=$(echo "$DEPLOY_OUTPUT" | grep -oP 'Pod ID: \K[a-z0-9]+' | head -1) if [ -z "$POD_ID" ]; then print_msg "$RED" "FAILED: Could not extract pod ID from deployment output" echo "$DEPLOY_OUTPUT" exit 1 fi print_msg "$GREEN" "Deployment successful!" print_msg "$GREEN" "Pod ID: ${POD_ID}" echo "$DEPLOY_OUTPUT" } ################################################################################ # PHASE 3: 5-MINUTE VALIDATION GATE (CRITICAL) ################################################################################ validation_gate() { print_header "PHASE 3: VALIDATION GATE (${VALIDATION_GATE_MINUTES} MINUTES)" print_msg "$YELLOW" "Waiting ${VALIDATION_GATE_MINUTES} minutes for first trials to complete..." print_msg "$BLUE" "Expected: At least ${MIN_CHECKPOINTS_AT_GATE} trials with checkpoints" # Wait for validation period for i in $(seq 1 ${VALIDATION_GATE_MINUTES}); do echo -n "." sleep 60 done echo "" # Check S3 for checkpoint files print_msg "$YELLOW" "Checking S3 for checkpoint files..." local CHECKPOINT_COUNT CHECKPOINT_COUNT=$(aws s3 ls "s3://${S3_BUCKET}/ml_training/" \ --profile "${AWS_PROFILE}" \ --endpoint-url "${S3_ENDPOINT}" \ --recursive | grep -c ".safetensors" || echo "0") print_msg "$BLUE" "Found ${CHECKPOINT_COUNT} checkpoint files" if [ "$CHECKPOINT_COUNT" -lt "$MIN_CHECKPOINTS_AT_GATE" ]; then print_msg "$RED" "VALIDATION GATE FAILED!" print_msg "$RED" "Expected at least ${MIN_CHECKPOINTS_AT_GATE} checkpoints, found ${CHECKPOINT_COUNT}" print_msg "$RED" "Checkpoint saving is still broken. Aborting to save costs." # Terminate pod print_msg "$YELLOW" "Terminating pod ${POD_ID}..." /home/jgrusewski/Work/foxhunt/target/release/foxhunt-deploy terminate "${POD_ID}" --yes || { print_msg "$RED" "WARNING: Could not terminate pod automatically" print_msg "$RED" "Please manually terminate pod: ${POD_ID}" } print_msg "$YELLOW" "\nCost saved: Approximately $0.09" print_msg "$YELLOW" "Total cost: Approximately $0.02 (5 minutes @ $0.25/hr)" exit 1 fi print_msg "$GREEN" "VALIDATION GATE PASSED!" print_msg "$GREEN" "Checkpoint saving is working. Continuing to full 50 trials..." VALIDATION_PASSED=true } ################################################################################ # PHASE 4: MONITOR FULL RUN ################################################################################ monitor_run() { print_header "PHASE 4: MONITORING FULL RUN (22 MINUTES)" print_msg "$BLUE" "Pod ID: ${POD_ID}" print_msg "$BLUE" "Expected completion: ~22 minutes" print_msg "$BLUE" "Expected total runtime: ~27 minutes" echo "" print_msg "$YELLOW" "Real-time monitoring commands:" echo "" echo " # Stream logs:" echo " python3 /home/jgrusewski/Work/foxhunt/scripts/python/runpod/monitor_logs.py ${POD_ID}" echo "" echo " # Check checkpoint count:" echo " aws s3 ls s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/ \\" echo " --profile ${AWS_PROFILE} \\" echo " --endpoint-url ${S3_ENDPOINT} \\" echo " --recursive | grep -c '.safetensors'" echo "" echo " # Watch RunPod dashboard:" echo " https://www.runpod.io/console/pods" echo "" print_msg "$YELLOW" "\nExpected checkpoint progression:" echo " T+10min: >= 20 files (trials 1-20)" echo " T+15min: >= 30 files (trials 1-30)" echo " T+20min: >= 40 files (trials 1-40)" echo " T+27min: >= 50 files (all trials)" echo "" print_msg "$BLUE" "Monitor the pod manually. Press ENTER when training is complete..." read -r } ################################################################################ # PHASE 5: POST-COMPLETION VALIDATION ################################################################################ post_completion_validation() { print_header "PHASE 5: POST-COMPLETION VALIDATION" # 1. Check checkpoint count print_msg "$YELLOW" "[1/4] Verifying checkpoint count..." local FINAL_COUNT FINAL_COUNT=$(aws s3 ls "s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/" \ --profile "${AWS_PROFILE}" \ --endpoint-url "${S3_ENDPOINT}" \ --recursive | grep -c ".safetensors" || echo "0") print_msg "$BLUE" "Total checkpoints found: ${FINAL_COUNT}" if [ "$FINAL_COUNT" -lt 50 ]; then print_msg "$RED" "WARNING: Expected at least 50 checkpoints, found ${FINAL_COUNT}" print_msg "$YELLOW" "This may indicate incomplete trials or early stopping" else print_msg "$GREEN" "PASSED: All trials have checkpoints" fi # 2. Download and validate checkpoint sizes print_msg "$YELLOW" "[2/4] Downloading checkpoints for validation..." mkdir -p /tmp/dqn_validation aws s3 sync "s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/" \ /tmp/dqn_validation/ \ --profile "${AWS_PROFILE}" \ --endpoint-url "${S3_ENDPOINT}" \ --exclude "*" \ --include "*.safetensors" || { print_msg "$RED" "WARNING: Could not download checkpoints for validation" print_msg "$YELLOW" "Skipping local validation" } # Check for empty files local EMPTY_FILES EMPTY_FILES=$(find /tmp/dqn_validation/ -name "*.safetensors" -size -1k 2>/dev/null | wc -l || echo "0") if [ "$EMPTY_FILES" -gt 0 ]; then print_msg "$RED" "WARNING: Found ${EMPTY_FILES} empty or corrupted checkpoint files" else print_msg "$GREEN" "PASSED: All checkpoint files have valid sizes" fi # 3. Test model loadability print_msg "$YELLOW" "[3/4] Testing model loadability..." local BEST_MODEL BEST_MODEL=$(find /tmp/dqn_validation/ -name "*.safetensors" | head -1) if [ -n "$BEST_MODEL" ]; then python3 -c " import safetensors.torch as st try: checkpoint = st.load_file('${BEST_MODEL}') print(f'SUCCESS: Loaded {len(checkpoint)} tensors from checkpoint') print(f'Tensor keys: {list(checkpoint.keys())[:5]}...') except Exception as e: print(f'FAILED: Could not load checkpoint: {e}') exit(1) " || { print_msg "$RED" "FAILED: Could not load checkpoint" print_msg "$YELLOW" "Checkpoint may be corrupted" } else print_msg "$YELLOW" "SKIPPED: No checkpoint files available for validation" fi # 4. Download hyperopt results print_msg "$YELLOW" "[4/4] Downloading hyperopt results..." aws s3 cp "s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/hyperopt/results.json" \ /tmp/dqn_validation/results.json \ --profile "${AWS_PROFILE}" \ --endpoint-url "${S3_ENDPOINT}" || { print_msg "$YELLOW" "WARNING: Could not download hyperopt results" } if [ -f /tmp/dqn_validation/results.json ]; then print_msg "$BLUE" "Best trial results:" cat /tmp/dqn_validation/results.json | jq '.best_trial' || { print_msg "$YELLOW" "Could not parse results JSON" } fi print_msg "$GREEN" "\nValidation complete!" } ################################################################################ # Function: Rollback procedure ################################################################################ rollback() { print_header "ROLLBACK PROCEDURE" print_msg "$RED" "Deployment failed or validation failed" if [ -n "$POD_ID" ]; then print_msg "$YELLOW" "Terminating pod ${POD_ID}..." /home/jgrusewski/Work/foxhunt/target/release/foxhunt-deploy terminate "${POD_ID}" --yes || { print_msg "$RED" "WARNING: Could not terminate pod automatically" print_msg "$RED" "Please manually terminate pod: ${POD_ID}" } fi print_msg "$YELLOW" "\nInvestigation steps:" echo " 1. Check DQN adapter code for regression:" echo " git diff HEAD~1 /home/jgrusewski/Work/foxhunt/ml/src/hyperopt/adapters/dqn.rs" echo "" echo " 2. Verify Agents 1-3 changes were committed:" echo " git log --oneline -10 | grep -i 'checkpoint\\|dqn'" echo "" echo " 3. Run local unit tests:" echo " cargo test -p ml dqn_checkpoint_save --features cuda" echo "" print_msg "$YELLOW" "Fallback options:" echo " A. Revert to previous commit, redeploy without checkpoints" echo " B. Fix checkpoint save logic, redeploy (another $0.11)" echo " C. Use manual checkpoint extraction from trial directories" exit 1 } ################################################################################ # Function: Success summary ################################################################################ success_summary() { print_header "DEPLOYMENT SUCCESSFUL!" print_msg "$GREEN" "All 50 trials completed with checkpoints saved!" echo "" print_msg "$BLUE" "Summary:" echo " Pod ID: ${POD_ID}" echo " Output directory: ${OUTPUT_DIR}" echo " Checkpoints: s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/" echo " Total cost: ~$0.11 (27 minutes @ $0.25/hr)" echo "" print_msg "$BLUE" "Access results:" echo " # List all checkpoints:" echo " aws s3 ls s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/ \\" echo " --profile ${AWS_PROFILE} \\" echo " --endpoint-url ${S3_ENDPOINT} \\" echo " --recursive" echo "" echo " # Download best model:" echo " aws s3 cp s3://${S3_BUCKET}/ml_training/dqn_hyperopt_checkpoints_${TIMESTAMP}/hyperopt/best_trial/ \\" echo " ./best_dqn_model/ \\" echo " --profile ${AWS_PROFILE} \\" echo " --endpoint-url ${S3_ENDPOINT} \\" echo " --recursive" echo "" print_msg "$GREEN" "\nSuccess criteria met:" echo " [x] Script deployed without errors" echo " [x] Checkpoints appeared in S3 during first 5 minutes" echo " [x] All 50 trials completed with models saved" echo " [x] Models downloadable and loadable via safetensors" echo "" } ################################################################################ # MAIN EXECUTION ################################################################################ main() { print_header "DQN HYPEROPT REDEPLOYMENT WITH CHECKPOINTS" print_msg "$BLUE" "Configuration:" echo " GPU: ${GPU_TYPE}" echo " Docker Image: ${DOCKER_IMAGE}" echo " Parquet File: ${PARQUET_FILE}" echo " Output Directory: ${OUTPUT_DIR}" echo " Trials: ${TRIALS}" echo " Epochs per trial: ${EPOCHS}" echo "" echo "Expected Duration: ~27 min (RTX A4000)" echo "Expected Cost: ~$0.11 (RTX A4000)" echo "" # Set up error handling trap rollback ERR # Execute phases pre_flight_checks build_and_deploy validation_gate monitor_run post_completion_validation success_summary # Clean up trap - ERR } # Run main main "$@"