Files
foxhunt/proto/monitoring.proto
2026-03-04 12:55:21 +01:00

567 lines
16 KiB
Protocol Buffer

syntax = "proto3";
package monitoring;
// Monitoring Service provides comprehensive system health monitoring, performance metrics collection,
// alerting capabilities, and training metrics for the HFT trading system. This service tracks latency,
// throughput, resource utilization, service health, and ML training progress across all components
// with real-time alerting and streaming.
service MonitoringService {
// Health and Status Monitoring
// Get overall system status and individual service health
rpc GetSystemStatus(GetSystemStatusRequest) returns (GetSystemStatusResponse);
// Stream real-time system status changes
rpc StreamSystemStatus(StreamSystemStatusRequest) returns (stream SystemStatusEvent);
// Perform detailed health checks on services
rpc GetHealthCheck(GetHealthCheckRequest) returns (GetHealthCheckResponse);
// Performance Metrics Collection
// Get system and application metrics
rpc GetMetrics(GetMetricsRequest) returns (GetMetricsResponse);
// Stream real-time performance metrics
rpc StreamMetrics(StreamMetricsRequest) returns (stream MetricsEvent);
// Get detailed latency performance metrics
rpc GetLatencyMetrics(GetLatencyMetricsRequest) returns (GetLatencyMetricsResponse);
// Get throughput and capacity metrics
rpc GetThroughputMetrics(GetThroughputMetricsRequest) returns (GetThroughputMetricsResponse);
// Alerting and Notification System
// Stream real-time system alerts and notifications
rpc StreamAlerts(StreamAlertsRequest) returns (stream AlertEvent);
// Acknowledge an active alert
rpc AcknowledgeAlert(AcknowledgeAlertRequest) returns (AcknowledgeAlertResponse);
// Get all currently active alerts
rpc GetActiveAlerts(GetActiveAlertsRequest) returns (GetActiveAlertsResponse);
// Training Metrics (from monitoring_service)
// Single snapshot of all active training sessions
rpc GetLiveTrainingMetrics(GetLiveTrainingMetricsRequest)
returns (GetLiveTrainingMetricsResponse);
// Server-streaming: pushes updates every N seconds
rpc StreamTrainingMetrics(StreamTrainingMetricsRequest)
returns (stream GetLiveTrainingMetricsResponse);
// Epoch history for a specific session (ring buffer, max 50 epochs)
rpc GetEpochHistory(GetEpochHistoryRequest)
returns (GetEpochHistoryResponse);
// Kubernetes pod status — streams every N seconds
rpc SubscribeClusterPods(SubscribeClusterPodsRequest) returns (stream ClusterPodsResponse);
}
// ============================================================================
// Health and Status Messages
// ============================================================================
message GetSystemStatusRequest {
repeated string service_names = 1;
}
message GetSystemStatusResponse {
SystemStatus overall_status = 1;
repeated ServiceStatus service_statuses = 2;
int64 timestamp = 3;
}
message StreamSystemStatusRequest {
repeated string service_names = 1;
optional int32 update_frequency_seconds = 2;
}
message GetHealthCheckRequest {
optional string service_name = 1;
}
message GetHealthCheckResponse {
HealthStatus health_status = 1;
repeated HealthCheck health_checks = 2;
int64 timestamp = 3;
}
// ============================================================================
// Performance Metrics Messages
// ============================================================================
message GetMetricsRequest {
repeated string metric_names = 1;
optional int64 start_time = 2;
optional int64 end_time = 3;
optional MetricAggregation aggregation = 4;
}
message GetMetricsResponse {
repeated Metric metrics = 1;
int64 timestamp = 2;
}
message StreamMetricsRequest {
repeated string metric_names = 1;
optional int32 update_frequency_seconds = 2;
}
message GetLatencyMetricsRequest {
optional string service_name = 1;
optional string operation_name = 2;
optional int64 start_time = 3;
optional int64 end_time = 4;
}
message GetLatencyMetricsResponse {
repeated LatencyMetric latency_metrics = 1;
}
message GetThroughputMetricsRequest {
optional string service_name = 1;
optional string operation_name = 2;
optional int64 start_time = 3;
optional int64 end_time = 4;
}
message GetThroughputMetricsResponse {
repeated ThroughputMetric throughput_metrics = 1;
}
// ============================================================================
// Alert Messages
// ============================================================================
message StreamAlertsRequest {
optional AlertSeverity min_severity = 1;
repeated string service_names = 2;
repeated AlertType alert_types = 3;
}
message AcknowledgeAlertRequest {
string alert_id = 1;
string acknowledged_by = 2;
optional string note = 3;
}
message AcknowledgeAlertResponse {
bool success = 1;
string message = 2;
int64 timestamp = 3;
}
message GetActiveAlertsRequest {
optional AlertSeverity min_severity = 1;
repeated string service_names = 2;
}
message GetActiveAlertsResponse {
repeated Alert active_alerts = 1;
int32 total_count = 2;
}
// ============================================================================
// Core Data Types (System Health)
// ============================================================================
message ServiceStatus {
string service_name = 1;
ServiceHealth health = 2;
ServiceState state = 3;
optional string version = 4;
optional string error_message = 5;
int64 uptime_seconds = 6;
int64 last_health_check = 7;
map<string, string> metadata = 8;
repeated Dependency dependencies = 9;
}
message SystemStatus {
SystemHealth overall_health = 1;
int32 healthy_services = 2;
int32 total_services = 3;
repeated string critical_issues = 4;
int64 system_uptime_seconds = 5;
SystemMetrics system_metrics = 6;
}
message HealthCheck {
string check_name = 1;
HealthStatus status = 2;
optional string message = 3;
optional double response_time_ms = 4;
int64 last_checked = 5;
map<string, string> details = 6;
}
message Dependency {
string name = 1;
DependencyType dependency_type = 2;
HealthStatus status = 3;
optional string endpoint = 4;
optional double response_time_ms = 5;
int64 last_checked = 6;
}
message SystemMetrics {
double cpu_usage_percent = 1;
double memory_usage_percent = 2;
double disk_usage_percent = 3;
double network_io_mbps = 4;
int32 active_connections = 5;
int32 total_requests = 6;
double avg_response_time_ms = 7;
double error_rate_percent = 8;
}
message Metric {
string name = 1;
MetricType metric_type = 2;
double value = 3;
string unit = 4;
map<string, string> labels = 5;
int64 timestamp = 6;
optional MetricStatistics statistics = 7;
}
message MetricStatistics {
double min = 1;
double max = 2;
double avg = 3;
double percentile_95 = 4;
double percentile_99 = 5;
double std_dev = 6;
int32 sample_count = 7;
}
message LatencyMetric {
string service_name = 1;
string operation_name = 2;
double avg_latency_ms = 3;
double p50_latency_ms = 4;
double p95_latency_ms = 5;
double p99_latency_ms = 6;
double max_latency_ms = 7;
int32 request_count = 8;
int64 time_window_start = 9;
int64 time_window_end = 10;
}
message ThroughputMetric {
string service_name = 1;
string operation_name = 2;
double requests_per_second = 3;
double bytes_per_second = 4;
int32 total_requests = 5;
int64 total_bytes = 6;
int64 time_window_start = 7;
int64 time_window_end = 8;
}
message Alert {
string alert_id = 1;
AlertType alert_type = 2;
AlertSeverity severity = 3;
string title = 4;
string description = 5;
string service_name = 6;
map<string, string> labels = 7;
int64 triggered_at = 8;
optional int64 acknowledged_at = 9;
optional string acknowledged_by = 10;
optional int64 resolved_at = 11;
AlertStatus status = 12;
optional string resolution_note = 13;
}
// ============================================================================
// Event Messages (System Health)
// ============================================================================
message SystemStatusEvent {
SystemStatus system_status = 1;
SystemStatusChangeType change_type = 2;
int64 timestamp = 3;
}
message MetricsEvent {
repeated Metric metrics = 1;
int64 timestamp = 2;
}
message AlertEvent {
Alert alert = 1;
AlertEventType event_type = 2;
int64 timestamp = 3;
}
// ============================================================================
// Training Metrics Messages (from monitoring_service)
// ============================================================================
message GetLiveTrainingMetricsRequest {
string model_filter = 1; // optional: "dqn", "ppo", etc.
}
message StreamTrainingMetricsRequest {
string model_filter = 1;
uint32 interval_seconds = 2; // 0 = server default (3s)
}
message GetLiveTrainingMetricsResponse {
repeated TrainingSession sessions = 1;
GpuSnapshot gpu = 2;
uint32 active_k8s_jobs = 3;
int64 timestamp = 4;
float cpu_percent = 5;
float memory_used_mb = 6;
float memory_total_mb = 7;
}
message TrainingSession {
string model = 1;
string fold = 2;
bool is_hyperopt = 3;
// Epoch/progress
float current_epoch = 4;
float epoch_loss = 5;
float validation_loss = 6;
// Throughput
float batches_per_second = 7;
float batches_processed = 8;
float iteration_seconds = 9;
// Eval metrics
float eval_accuracy = 10;
float eval_precision = 11;
float eval_recall = 12;
float eval_f1 = 13;
// Checkpoint
float checkpoint_size_bytes = 14;
uint32 checkpoint_saves = 15;
uint32 checkpoint_failures = 16;
// Health counters
uint32 nan_detected = 17;
uint32 gradient_explosions = 18;
uint32 feature_errors = 19;
// Hyperopt fields (populated when is_hyperopt=true)
uint32 hyperopt_trial_current = 20;
uint32 hyperopt_trial_total = 21;
float hyperopt_best_objective = 22;
uint32 hyperopt_trials_failed = 23;
// RL diagnostics
float q_value_mean = 24;
float q_value_max = 25;
float policy_entropy = 26;
float kl_divergence = 27;
float advantage_mean = 28;
uint32 replay_buffer_size = 29;
// Gradient & training health
float gradient_norm = 30;
float learning_rate = 31;
float epoch_duration_seconds = 32;
// Hyperopt intra-trial
uint32 hyperopt_trial_epoch = 33;
float hyperopt_trial_best_loss = 34;
float hyperopt_elapsed_seconds = 35;
// Epoch-level financial metrics
float epoch_sharpe = 36;
float epoch_sortino = 37;
float epoch_win_rate = 38;
float epoch_max_drawdown = 39;
float epoch_profit_factor = 40;
float epoch_total_return = 41;
float epoch_avg_return = 42;
uint32 epoch_total_trades = 43;
// Action distribution
float action_buy_pct = 44;
float action_sell_pct = 45;
float action_hold_pct = 46;
}
message GpuSnapshot {
float utilization_percent = 1;
float memory_used_mb = 2;
float memory_total_mb = 3;
float temperature_celsius = 4;
float power_watts = 5;
}
message GetEpochHistoryRequest {
string model = 1;
string fold = 2;
uint32 max_epochs = 3; // 0 = all (up to 50)
}
message EpochFinancialSnapshot {
uint32 epoch = 1;
float sharpe = 2;
float sortino = 3;
float win_rate = 4;
float max_drawdown = 5;
float profit_factor = 6;
float total_return = 7;
float avg_return = 8;
uint32 total_trades = 9;
float loss = 10;
float val_loss = 11;
float learning_rate = 12;
float action_buy_pct = 13;
float action_sell_pct = 14;
float action_hold_pct = 15;
}
message GetEpochHistoryResponse {
string model = 1;
string fold = 2;
repeated EpochFinancialSnapshot epochs = 3;
}
// ============================================================================
// Cluster Pod Messages
// ============================================================================
message SubscribeClusterPodsRequest {
uint32 interval_seconds = 1; // 0 = server default (5s)
}
message ClusterPodsResponse {
repeated PodInfo pods = 1;
int64 timestamp = 2;
}
message PodInfo {
string name = 1;
string service = 2; // app.kubernetes.io/name label
string namespace = 3;
string status = 4; // Running/Pending/CrashLoopBackOff/Completed/etc
int32 restarts = 5;
string age = 6; // human-readable: "2d5h", "3m"
string node = 7;
bool ready = 8;
string version = 9; // from FOXHUNT_BUILD_VERSION env or empty
}
// ============================================================================
// Enums
// ============================================================================
// Health status levels for services
enum ServiceHealth {
SERVICE_HEALTH_UNSPECIFIED = 0; // Default/unknown health
SERVICE_HEALTH_HEALTHY = 1; // Service operating normally
SERVICE_HEALTH_DEGRADED = 2; // Service performance degraded
SERVICE_HEALTH_UNHEALTHY = 3; // Service not functioning properly
SERVICE_HEALTH_CRITICAL = 4; // Service in critical failure state
}
// Operational states of services
enum ServiceState {
SERVICE_STATE_UNSPECIFIED = 0; // Default/unknown state
SERVICE_STATE_STARTING = 1; // Service is starting up
SERVICE_STATE_RUNNING = 2; // Service is running normally
SERVICE_STATE_STOPPING = 3; // Service is shutting down
SERVICE_STATE_STOPPED = 4; // Service is stopped
SERVICE_STATE_ERROR = 5; // Service encountered an error
}
enum SystemHealth {
SYSTEM_HEALTH_UNSPECIFIED = 0;
SYSTEM_HEALTH_HEALTHY = 1;
SYSTEM_HEALTH_DEGRADED = 2;
SYSTEM_HEALTH_UNHEALTHY = 3;
SYSTEM_HEALTH_CRITICAL = 4;
}
enum HealthStatus {
HEALTH_STATUS_UNSPECIFIED = 0;
HEALTH_STATUS_HEALTHY = 1;
HEALTH_STATUS_DEGRADED = 2;
HEALTH_STATUS_UNHEALTHY = 3;
HEALTH_STATUS_CRITICAL = 4;
}
enum DependencyType {
DEPENDENCY_TYPE_UNSPECIFIED = 0;
DEPENDENCY_TYPE_DATABASE = 1;
DEPENDENCY_TYPE_MESSAGE_QUEUE = 2;
DEPENDENCY_TYPE_CACHE = 3;
DEPENDENCY_TYPE_EXTERNAL_API = 4;
DEPENDENCY_TYPE_FILE_SYSTEM = 5;
DEPENDENCY_TYPE_NETWORK = 6;
}
enum MetricType {
METRIC_TYPE_UNSPECIFIED = 0;
METRIC_TYPE_COUNTER = 1;
METRIC_TYPE_GAUGE = 2;
METRIC_TYPE_HISTOGRAM = 3;
METRIC_TYPE_TIMER = 4;
}
enum MetricAggregation {
METRIC_AGGREGATION_UNSPECIFIED = 0;
METRIC_AGGREGATION_SUM = 1;
METRIC_AGGREGATION_AVG = 2;
METRIC_AGGREGATION_MIN = 3;
METRIC_AGGREGATION_MAX = 4;
METRIC_AGGREGATION_COUNT = 5;
}
enum AlertType {
ALERT_TYPE_UNSPECIFIED = 0;
ALERT_TYPE_HEALTH_CHECK = 1;
ALERT_TYPE_PERFORMANCE = 2;
ALERT_TYPE_ERROR_RATE = 3;
ALERT_TYPE_LATENCY = 4;
ALERT_TYPE_THROUGHPUT = 5;
ALERT_TYPE_RESOURCE_USAGE = 6;
ALERT_TYPE_DEPENDENCY = 7;
ALERT_TYPE_SECURITY = 8;
}
// Alert severity levels
enum AlertSeverity {
ALERT_SEVERITY_UNSPECIFIED = 0; // Default/unknown severity
ALERT_SEVERITY_INFO = 1; // Informational alert
ALERT_SEVERITY_WARNING = 2; // Warning requiring attention
ALERT_SEVERITY_CRITICAL = 3; // Critical issue requiring immediate action
ALERT_SEVERITY_EMERGENCY = 4; // Emergency requiring immediate response
}
// Current status of alerts
enum AlertStatus {
ALERT_STATUS_UNSPECIFIED = 0; // Default/unknown status
ALERT_STATUS_ACTIVE = 1; // Alert is currently active
ALERT_STATUS_ACKNOWLEDGED = 2; // Alert has been acknowledged
ALERT_STATUS_RESOLVED = 3; // Alert has been resolved
ALERT_STATUS_SUPPRESSED = 4; // Alert is temporarily suppressed
}
enum SystemStatusChangeType {
SYSTEM_STATUS_CHANGE_TYPE_UNSPECIFIED = 0;
SYSTEM_STATUS_CHANGE_TYPE_HEALTH_IMPROVED = 1;
SYSTEM_STATUS_CHANGE_TYPE_HEALTH_DEGRADED = 2;
SYSTEM_STATUS_CHANGE_TYPE_SERVICE_STARTED = 3;
SYSTEM_STATUS_CHANGE_TYPE_SERVICE_STOPPED = 4;
SYSTEM_STATUS_CHANGE_TYPE_SERVICE_ERROR = 5;
}
enum AlertEventType {
ALERT_EVENT_TYPE_UNSPECIFIED = 0;
ALERT_EVENT_TYPE_TRIGGERED = 1;
ALERT_EVENT_TYPE_ACKNOWLEDGED = 2;
ALERT_EVENT_TYPE_RESOLVED = 3;
ALERT_EVENT_TYPE_ESCALATED = 4;
}