#!/bin/bash #============================================================================ # FOXHUNT HFT TRADING SYSTEM - COMPREHENSIVE HEALTH CHECK #============================================================================ # Validates system health, performance metrics, and service connectivity # # Usage: ./health-check.sh [OPTIONS] # # Options: # --detailed Show detailed health information # --performance Include performance metrics # --json Output results in JSON format # --continuous Run continuous monitoring (Ctrl+C to stop) # --help Show this help message #============================================================================ set -euo pipefail # Configuration SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" DOCKER_COMPOSE_FILE="docker-compose.production.yml" # Colors for output RED='\033[0;31m' GREEN='\033[0;32m' YELLOW='\033[1;33m' BLUE='\033[0;34m' NC='\033[0m' # No Color # Health check results declare -A health_results declare -A performance_metrics # Logging functions log_info() { echo -e "${BLUE}[INFO]${NC} $1" } log_success() { echo -e "${GREEN}[✓]${NC} $1" } log_warning() { echo -e "${YELLOW}[⚠]${NC} $1" } log_error() { echo -e "${RED}[✗]${NC} $1" } # Show help show_help() { cat << EOF Foxhunt HFT Trading System - Health Check Usage: $0 [OPTIONS] OPTIONS: --detailed Show detailed health information --performance Include performance metrics --json Output results in JSON format --continuous Run continuous monitoring (Ctrl+C to stop) --help Show this help message EXAMPLES: $0 # Basic health check $0 --detailed # Detailed health check $0 --performance # Include performance metrics $0 --continuous # Continuous monitoring EOF } # Check if service is running check_service_status() { local service_name="$1" local container_name="foxhunt-${service_name}-prod" if docker ps --format "table {{.Names}}" | grep -q "$container_name"; then health_results["${service_name}_status"]="running" return 0 else health_results["${service_name}_status"]="stopped" return 1 fi } # Check service health endpoint check_service_health() { local service_name="$1" local health_url="$2" local timeout="${3:-5}" if curl -sf --max-time "$timeout" "$health_url" &>/dev/null; then health_results["${service_name}_health"]="healthy" return 0 else health_results["${service_name}_health"]="unhealthy" return 1 fi } # Get service performance metrics get_service_metrics() { local service_name="$1" local metrics_url="$2" local response if response=$(curl -sf --max-time 5 "$metrics_url" 2>/dev/null); then # Extract key metrics (simplified for demo) local cpu_usage=$(echo "$response" | grep "cpu_usage" | awk '{print $2}' || echo "0") local memory_usage=$(echo "$response" | grep "memory_usage" | awk '{print $2}' || echo "0") local request_rate=$(echo "$response" | grep "request_rate" | awk '{print $2}' || echo "0") performance_metrics["${service_name}_cpu"]="$cpu_usage" performance_metrics["${service_name}_memory"]="$memory_usage" performance_metrics["${service_name}_requests"]="$request_rate" fi } # Check database connectivity check_database() { local db_type="$1" local connection_string="$2" case "$db_type" in "postgresql") if docker exec foxhunt-postgres-prod pg_isready &>/dev/null; then health_results["postgresql_health"]="healthy" log_success "PostgreSQL is accessible" else health_results["postgresql_health"]="unhealthy" log_error "PostgreSQL is not accessible" fi ;; "redis") if docker exec foxhunt-redis-prod redis-cli ping | grep -q PONG; then health_results["redis_health"]="healthy" log_success "Redis is accessible" else health_results["redis_health"]="unhealthy" log_error "Redis is not accessible" fi ;; "influxdb") if check_service_health "influxdb" "http://localhost:8086/ping" 3; then log_success "InfluxDB is accessible" else log_error "InfluxDB is not accessible" fi ;; esac } # Check critical trading metrics check_trading_metrics() { log_info "Checking critical trading metrics..." # Check if trading service is responsive if check_service_health "trading" "http://localhost:8080/health"; then log_success "Trading service is healthy" # Check latency metrics local latency_response if latency_response=$(curl -sf "http://localhost:8080/metrics" 2>/dev/null); then local avg_latency=$(echo "$latency_response" | grep "order_latency" | awk '{print $2}' || echo "unknown") if [[ "$avg_latency" != "unknown" && "$avg_latency" != "" ]]; then if (( $(echo "$avg_latency < 10" | bc -l) )); then log_success "Order latency is acceptable: ${avg_latency}ms" health_results["trading_latency"]="good" else log_warning "Order latency is high: ${avg_latency}ms" health_results["trading_latency"]="high" fi fi fi else log_error "Trading service is not healthy" fi } # Check system resources check_system_resources() { log_info "Checking system resources..." # Memory usage local memory_info memory_info=$(free -m | awk 'NR==2{printf "%.1f", $3*100/$2}') if (( $(echo "$memory_info > 90" | bc -l) )); then log_warning "High memory usage: ${memory_info}%" health_results["memory_usage"]="high" else log_success "Memory usage is acceptable: ${memory_info}%" health_results["memory_usage"]="normal" fi # Disk usage local disk_usage disk_usage=$(df / | awk 'NR==2{print $5}' | sed 's/%//') if [[ "$disk_usage" -gt 85 ]]; then log_warning "High disk usage: ${disk_usage}%" health_results["disk_usage"]="high" else log_success "Disk usage is acceptable: ${disk_usage}%" health_results["disk_usage"]="normal" fi # CPU load local cpu_load cpu_load=$(uptime | awk -F'load average:' '{print $2}' | awk '{print $1}' | sed 's/,//') local cpu_cores=$(nproc) if (( $(echo "$cpu_load > $cpu_cores * 0.8" | bc -l) )); then log_warning "High CPU load: $cpu_load (cores: $cpu_cores)" health_results["cpu_load"]="high" else log_success "CPU load is acceptable: $cpu_load (cores: $cpu_cores)" health_results["cpu_load"]="normal" fi } # Check network connectivity check_network() { log_info "Checking network connectivity..." # Test internal service communication local services=("trading:8080" "ml-training:8082" "backtesting:8083" "tli:8081") for service in "${services[@]}"; do local service_name=$(echo "$service" | cut -d':' -f1) local service_port=$(echo "$service" | cut -d':' -f2) if nc -z localhost "$service_port" 2>/dev/null; then log_success "$service_name service is reachable on port $service_port" else log_error "$service_name service is not reachable on port $service_port" fi done } # Comprehensive health check run_health_check() { local detailed="$1" local include_performance="$2" log_info "Starting comprehensive health check..." # Check core services local services=("trading" "ml-training" "backtesting" "tli") for service in "${services[@]}"; do if check_service_status "$service"; then log_success "$service service is running" else log_error "$service service is not running" fi done # Check infrastructure services check_database "postgresql" "" check_database "redis" "" check_database "influxdb" "" # Check Vault if check_service_health "vault" "http://localhost:8200/v1/sys/health" 5; then log_success "Vault is accessible" else log_error "Vault is not accessible" fi # Check monitoring services if check_service_health "prometheus" "http://localhost:9090/-/healthy"; then log_success "Prometheus is healthy" else log_error "Prometheus is not healthy" fi if check_service_health "grafana" "http://localhost:3000/api/health"; then log_success "Grafana is healthy" else log_error "Grafana is not healthy" fi # Check trading-specific metrics check_trading_metrics # Check system resources if [[ "$detailed" == true ]]; then check_system_resources check_network fi # Get performance metrics if [[ "$include_performance" == true ]]; then log_info "Collecting performance metrics..." get_service_metrics "trading" "http://localhost:8080/metrics" get_service_metrics "ml-training" "http://localhost:8082/metrics" fi } # Output results in JSON format output_json() { echo "{" echo " \"timestamp\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\"," echo " \"health_check\": {" local first=true for key in "${!health_results[@]}"; do if [[ "$first" == false ]]; then echo "," fi echo " \"$key\": \"${health_results[$key]}\"" first=false done if [[ ${#performance_metrics[@]} -gt 0 ]]; then echo " }," echo " \"performance_metrics\": {" first=true for key in "${!performance_metrics[@]}"; do if [[ "$first" == false ]]; then echo "," fi echo " \"$key\": \"${performance_metrics[$key]}\"" first=false done fi echo " }" echo "}" } # Display summary show_summary() { local total_checks=0 local passed_checks=0 for result in "${health_results[@]}"; do ((total_checks++)) if [[ "$result" == "running" || "$result" == "healthy" || "$result" == "good" || "$result" == "normal" ]]; then ((passed_checks++)) fi done echo "" echo "===============================================" echo "HEALTH CHECK SUMMARY" echo "===============================================" echo "Total checks: $total_checks" echo "Passed: $passed_checks" echo "Failed: $((total_checks - passed_checks))" if [[ $passed_checks -eq $total_checks ]]; then log_success "All health checks passed! System is operating normally." elif [[ $passed_checks -gt $((total_checks / 2)) ]]; then log_warning "Some health checks failed. System is partially operational." else log_error "Multiple health checks failed. System requires attention." fi } # Continuous monitoring continuous_monitoring() { local detailed="$1" local include_performance="$2" log_info "Starting continuous monitoring (Press Ctrl+C to stop)..." while true; do clear echo "Foxhunt HFT Health Check - $(date)" echo "========================================" # Reset results health_results=() performance_metrics=() run_health_check "$detailed" "$include_performance" show_summary sleep 30 done } # Main function main() { local detailed=false local include_performance=false local json_output=false local continuous=false # Parse arguments while [[ $# -gt 0 ]]; do case $1 in --detailed) detailed=true shift ;; --performance) include_performance=true shift ;; --json) json_output=true shift ;; --continuous) continuous=true shift ;; --help) show_help exit 0 ;; *) log_error "Unknown option: $1" show_help exit 1 ;; esac done if [[ "$continuous" == true ]]; then continuous_monitoring "$detailed" "$include_performance" else run_health_check "$detailed" "$include_performance" if [[ "$json_output" == true ]]; then output_json else show_summary fi fi } # Run main function main "$@"