#!/bin/bash # # Runpod FP32 TFT-225 Training Test Deployment # # This script deploys a test FP32 TFT-225 model training job to Runpod using spot instances # for cost optimization. It's designed for a single 1-5 minute training run to validate # the deployment before scaling to production. # # COST TARGET: <$0.10 total (spot pricing with auto-termination) # # Requirements: # - runpodctl v1.14.6+ installed # - RUNPOD_API_KEY environment variable set # - test_data/ES_FUT_180d.parquet file exists (2.9MB) # # Usage: # ./scripts/deploy_fp32_runpod_test.sh [--dry-run] # set -euo pipefail # ==================== CONFIGURATION ==================== # Runpod Configuration GPU_TYPE="NVIDIA GeForce RTX 3090" # Cheapest GPU with 24GB VRAM (>4GB required) GPU_COUNT=1 COST_CEILING=0.25 # Max $0.25/hr (spot RTX 3090 ~$0.14/hr) CONTAINER_IMAGE="runpod/pytorch:2.1.0-py3.10-cuda11.8.0-devel-ubuntu22.04" POD_NAME="foxhunt-fp32-tft-test-$(date +%s)" # Training Configuration PARQUET_FILE="test_data/ES_FUT_180d.parquet" EPOCHS=10 # Reduced for quick test (vs 50 production) TRAINING_TIMEOUT=600 # 10 minutes max (safety timeout) # Directories REMOTE_WORKSPACE="/workspace/foxhunt" REMOTE_DATA_DIR="${REMOTE_WORKSPACE}/test_data" REMOTE_OUTPUT_DIR="${REMOTE_WORKSPACE}/models" # Colors for output RED='\033[0;31m' GREEN='\033[0;32m' YELLOW='\033[1;33m' BLUE='\033[0;34m' NC='\033[0m' # No Color # ==================== FUNCTIONS ==================== log_info() { echo -e "${BLUE}[INFO]${NC} $*" } log_success() { echo -e "${GREEN}[SUCCESS]${NC} $*" } log_warning() { echo -e "${YELLOW}[WARNING]${NC} $*" } log_error() { echo -e "${RED}[ERROR]${NC} $*" } check_prerequisites() { log_info "Checking prerequisites..." # Check runpodctl if ! command -v runpodctl &> /dev/null; then log_error "runpodctl not found. Install: brew install runpod/runpodctl/runpodctl" exit 1 fi local version=$(runpodctl --version | grep -oP 'v\K[0-9.]+' || echo "0.0.0") log_info "runpodctl version: v${version}" # Check API key if [[ -z "${RUNPOD_API_KEY:-}" ]]; then log_error "RUNPOD_API_KEY environment variable not set" log_info "Get your API key from: https://www.runpod.io/console/user/settings" log_info "Then run: export RUNPOD_API_KEY='your-key-here'" exit 1 fi # Configure runpodctl log_info "Configuring runpodctl..." runpodctl config --apiKey="${RUNPOD_API_KEY}" 2>/dev/null || { log_error "Failed to configure runpodctl" exit 1 } # Check training data if [[ ! -f "${PARQUET_FILE}" ]]; then log_error "Training data not found: ${PARQUET_FILE}" exit 1 fi local file_size=$(du -h "${PARQUET_FILE}" | cut -f1) log_info "Training data: ${PARQUET_FILE} (${file_size})" log_success "Prerequisites check passed" } estimate_cost() { log_info "Cost Estimation:" echo " GPU Type: ${GPU_TYPE}" echo " Spot Price Ceiling: \$${COST_CEILING}/hr" echo " Expected Spot Rate: ~\$0.14/hr (typical RTX 3090 spot)" echo " Training Duration: ~1-5 minutes" echo " Storage (1GB): ~\$0.0002/hr" echo "" echo " ESTIMATED COST: \$0.02 - \$0.10 per run" echo " MAX COST (10 min): \$0.04 (with auto-termination)" echo "" log_warning "Spot instances can be interrupted. Save checkpoints frequently." } create_pod() { log_info "Creating Runpod spot instance..." # Note: runpodctl doesn't have a direct spot flag in the create command # Spot pricing is automatically used when --cost is specified and available local pod_output=$(runpodctl create pod \ --name "${POD_NAME}" \ --gpuType "${GPU_TYPE}" \ --gpuCount ${GPU_COUNT} \ --cost ${COST_CEILING} \ --communityCloud \ --imageName "${CONTAINER_IMAGE}" \ --containerDiskSize 20 \ --volumeSize 5 \ --volumePath "/workspace" \ --env "CUDA_VISIBLE_DEVICES=0" \ --env "PYTHONUNBUFFERED=1" \ --ports "8888/http" \ --startSSH 2>&1) || { log_error "Failed to create pod" echo "${pod_output}" exit 1 } # Extract pod ID from output POD_ID=$(echo "${pod_output}" | grep -oP '(?<=id: )[a-z0-9]+' || echo "") if [[ -z "${POD_ID}" ]]; then log_error "Failed to extract pod ID from output:" echo "${pod_output}" exit 1 fi log_success "Pod created: ${POD_ID}" echo "${POD_ID}" > /tmp/foxhunt_runpod_test_id.txt # Wait for pod to be ready log_info "Waiting for pod to be ready (max 120s)..." local wait_count=0 while [[ $wait_count -lt 24 ]]; do local status=$(runpodctl get pod "${POD_ID}" 2>/dev/null | grep -i "status" || echo "UNKNOWN") if echo "${status}" | grep -qi "running"; then log_success "Pod is running" sleep 5 # Additional wait for SSH/filesystem return 0 fi sleep 5 wait_count=$((wait_count + 1)) done log_error "Pod failed to become ready within 120s" cleanup_pod exit 1 } upload_data() { log_info "Uploading training data and code..." # Create a temporary directory with all necessary files local tmp_dir=$(mktemp -d) mkdir -p "${tmp_dir}/foxhunt/test_data" # Copy training data cp "${PARQUET_FILE}" "${tmp_dir}/foxhunt/test_data/" # Create a simplified training script cat > "${tmp_dir}/foxhunt/train_remote.sh" << 'EOF' #!/bin/bash set -euo pipefail cd /workspace/foxhunt echo "[INFO] Installing Rust and Cargo..." curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y source $HOME/.cargo/env echo "[INFO] Cloning Foxhunt repository..." git clone https://github.com/yourusername/foxhunt.git repo || { echo "[ERROR] Failed to clone repository. Using local files." } # If git clone failed, we'll use the uploaded data if [[ -d "repo" ]]; then cd repo else cd /workspace/foxhunt fi echo "[INFO] Starting FP32 TFT-225 training..." echo "[INFO] Training data: test_data/ES_FUT_180d.parquet" echo "[INFO] Epochs: ${EPOCHS:-10}" echo "[INFO] GPU: $(nvidia-smi --query-gpu=name --format=csv,noheader)" # Run training cargo run -p ml --example train_tft_parquet --release --features cuda -- \ --parquet-file test_data/ES_FUT_180d.parquet \ --epochs ${EPOCHS:-10} \ 2>&1 | tee /workspace/foxhunt/training.log echo "[INFO] Training complete!" echo "[INFO] Logs saved to: /workspace/foxhunt/training.log" EOF chmod +x "${tmp_dir}/foxhunt/train_remote.sh" # Upload via runpodctl send log_info "Initiating file transfer..." local send_output=$(runpodctl send "${tmp_dir}/foxhunt" 2>&1) local transfer_code=$(echo "${send_output}" | grep -oP '[0-9]{4}-[a-z]+-[a-z]+-[a-z]+' || echo "") if [[ -z "${transfer_code}" ]]; then log_error "Failed to get transfer code" rm -rf "${tmp_dir}" cleanup_pod exit 1 fi log_info "Transfer code: ${transfer_code}" log_info "Run this command on the pod to receive files:" echo " runpodctl receive ${transfer_code}" # Attempt to SSH and receive files log_info "Attempting to receive files on pod..." # Note: This requires SSH access to the pod, which may not be immediate # In practice, you'd SSH into the pod manually and run the receive command rm -rf "${tmp_dir}" log_warning "Manual step required:" log_warning " 1. SSH into pod: runpodctl ssh ${POD_ID}" log_warning " 2. Run: runpodctl receive ${transfer_code}" log_warning " 3. Run: bash /workspace/foxhunt/train_remote.sh" } run_training() { log_info "Training must be started manually via SSH" log_info "SSH command: runpodctl ssh ${POD_ID}" log_info "Then run: bash /workspace/foxhunt/train_remote.sh" log_info "" log_info "Training will:" log_info " - Install Rust/Cargo" log_info " - Clone Foxhunt repo (or use uploaded files)" log_info " - Run FP32 TFT-225 training for ${EPOCHS} epochs" log_info " - Save logs to /workspace/foxhunt/training.log" } download_results() { log_info "To download results after training:" echo " 1. On pod: runpodctl send /workspace/foxhunt/training.log" echo " 2. Locally: runpodctl receive " echo " 3. On pod: runpodctl send /workspace/foxhunt/models/" echo " 4. Locally: runpodctl receive " } cleanup_pod() { if [[ -n "${POD_ID:-}" ]]; then log_info "Terminating pod: ${POD_ID}" runpodctl remove pod "${POD_ID}" 2>/dev/null || { log_warning "Failed to terminate pod. Please terminate manually:" echo " runpodctl remove pod ${POD_ID}" } log_success "Pod terminated" rm -f /tmp/foxhunt_runpod_test_id.txt fi } # ==================== MAIN ==================== main() { local dry_run=false # Parse arguments for arg in "$@"; do case $arg in --dry-run) dry_run=true shift ;; --help) echo "Usage: $0 [--dry-run]" echo "" echo "Options:" echo " --dry-run Show cost estimate and exit without creating pod" echo " --help Show this help message" exit 0 ;; esac done log_info "Foxhunt FP32 TFT-225 Runpod Test Deployment" echo "" check_prerequisites echo "" estimate_cost if [[ "$dry_run" == "true" ]]; then log_info "Dry run complete. No pod created." exit 0 fi echo "" read -p "Proceed with deployment? (yes/no): " confirm if [[ "$confirm" != "yes" ]]; then log_info "Deployment cancelled" exit 0 fi echo "" create_pod echo "" upload_data echo "" run_training echo "" download_results echo "" log_warning "IMPORTANT: Remember to terminate the pod when done to avoid charges!" echo " Terminate now: runpodctl remove pod ${POD_ID}" echo " Or save pod ID for later: ${POD_ID}" echo "" log_info "Deployment complete!" } # Trap to cleanup on exit trap cleanup_pod EXIT INT TERM main "$@"