Files
foxhunt/testing/integration/chaos/chaos_framework.rs
jgrusewski 4ba8eebc05 cleanup: declarative rewrites for migrations/services/testing TODOs
migrations:
- 001_trading_events.sql, 003_audit_system.sql: the hard-coded
  node_id literals (`trading-node-01`, `audit-node-01`,
  `ml-node-01`, `system-node-01`, `change-tracker-01`) are
  overridden per-deployment by later migrations rather than read
  from the environment. Describe that in the inline comment.
- 004_compliance_views.sql: `generate_compliance_report` is a log
  stub — actual report generation is performed by the compliance
  service. Say so explicitly.

services:
- ml_training_service/tests/orchestrator_225_features_test.rs: the
  empty `#[ignore]`d placeholder for the 225-feature orchestrator
  loader has been removed; it held no assertions and only tracked
  a TODO (feedback_no_stubs.md).
- trading_agent_service/src/service.rs: portfolio volatility uses
  the diagonal-only approximation because cross-asset return
  correlations are not maintained in this service. Document that.
- trading_service/src/services/risk.rs: `get_risk_metrics` uses
  `calculate_marginal_var` + asset-class fallback; describe why
  `calculate_comprehensive_var` is not wired at this boundary.
- trading_service/tests/auth_comprehensive.rs: delete the entire
  commented-out legacy BackupCodeValidator test block — the old
  `generate_backup_codes` / `store_backup_code` /
  `verify_backup_code` surface no longer exists, and MFA
  integration tests already cover the new API.

testing:
- harness/grpc_clients.rs: no BacktestingServiceClient proto
  exists; reword the stale TODO import line.
- chaos/*: reword the family of "TODO: Implement ..." stubs as
  "Currently a no-op / synthetic result" descriptions so readers
  know exactly how much of the chaos framework is live.
- compliance_automation_tests.rs: delete the file; it was a giant
  /* ... */ block referencing a nonexistent compliance module
  and was not wired into any Cargo target.
- framework.rs: describe why `setup()` uses `println!` instead of
  `tracing_subscriber` (tracing_subscriber is not a dep of this
  integration crate).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-23 08:51:26 +02:00

842 lines
28 KiB
Rust

//! Comprehensive Chaos Engineering Framework for Foxhunt HFT System
//!
//! This framework implements systematic failure injection, recovery validation,
//! and checkpoint resume testing specifically designed for HFT requirements.
use anyhow::{Context, Result};
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::process::{Child, Command, Stdio};
use std::sync::Arc;
use std::time::{Duration, Instant};
use tokio::sync::{broadcast, RwLock, Semaphore};
use tokio::time::{sleep, timeout};
use tracing::{error, info, warn};
use uuid::Uuid;
/// Configuration for a chaos engineering experiment
///
/// Defines all parameters needed to execute a controlled failure injection
/// and recovery validation test for HFT trading system components.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ChaosExperiment {
/// Unique identifier for this experiment
pub id: Uuid,
/// Human-readable name for the experiment
pub name: String,
/// Detailed description of what the experiment tests
pub description: String,
/// Name of the service or component being tested
pub target_service: String,
/// Type of failure to inject
pub failure_type: FailureType,
/// How long to maintain the failure condition
pub duration: Duration,
/// Maximum time to wait for service recovery
pub recovery_timeout: Duration,
/// HFT requirement: sub-100ms recovery time
pub max_recovery_time_ms: u64,
/// Whether this experiment is enabled for execution
pub enabled: bool,
}
/// Types of failures that can be injected during chaos testing
///
/// Covers the primary failure modes that HFT systems must handle:
/// process crashes, resource exhaustion, network issues, and external dependencies.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub enum FailureType {
/// Kill the target process and optionally restart it
ProcessKill {
/// Signal to send to the process
signal: Signal,
/// Delay before restarting the service (ms)
delay_before_restart_ms: u64,
},
/// Create memory pressure to test OOM handling
MemoryPressure {
/// Amount of memory to allocate (MB)
target_mb: u64,
/// How long to maintain pressure (ms)
duration_ms: u64,
},
/// Create network partition by blocking specific ports
NetworkPartition {
/// Ports to block traffic to/from
target_ports: Vec<u16>,
/// Duration of the partition (ms)
duration_ms: u64,
},
/// Inject disk I/O failures for specific paths
DiskIoFailure {
/// File system paths to affect
target_paths: Vec<String>,
/// Percentage of I/O operations to fail
failure_rate_percent: u8,
},
/// Throttle CPU usage to test performance degradation
CpuThrottle {
/// CPU usage limit as percentage
cpu_limit_percent: u8,
/// Duration of throttling (ms)
duration_ms: u64,
},
/// Exhaust GPU memory resources for ML service testing
GpuResourceExhaustion {
/// Percentage of GPU memory to fill
memory_fill_percent: u8,
/// Duration of resource exhaustion (ms)
duration_ms: u64,
},
/// Simulate database connection failures
DatabaseConnectionFailure {
/// Database connection string to target
connection_string: String,
/// Duration of connection failure (ms)
duration_ms: u64,
},
}
/// Unix signals that can be sent to processes
#[derive(Debug, Clone, Serialize, Deserialize)]
pub enum Signal {
/// Graceful termination signal
SIGTERM,
/// Forceful kill signal (non-catchable)
SIGKILL,
/// Stop process execution (can be resumed)
SIGSTOP,
/// Continue stopped process
SIGCONT,
}
/// Results from executing a chaos engineering experiment
///
/// Contains all metrics and outcomes from the experiment including
/// recovery time, checkpoint integrity, and performance impact.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ChaosResult {
/// ID of the experiment that generated this result
pub experiment_id: Uuid,
/// When the experiment started
pub started_at: std::time::SystemTime,
/// When the experiment completed (None if still running)
pub completed_at: Option<std::time::SystemTime>,
/// Final status of the experiment
pub status: ChaosStatus,
/// Time taken for service recovery in milliseconds
pub recovery_time_ms: Option<u64>,
/// Whether checkpoint data remained valid through the failure
pub checkpoint_integrity: bool,
/// Performance impact measured after recovery
pub performance_regression: Option<PerformanceRegression>,
/// Any errors encountered during the experiment
pub errors: Vec<String>,
}
/// Status of a chaos experiment execution
#[derive(Debug, Clone, Serialize, Deserialize)]
pub enum ChaosStatus {
/// Experiment is currently executing
Running,
/// Experiment completed successfully with all validations passed
Succeeded,
/// Experiment failed during execution
Failed,
/// Service failed to recover within the specified timeout
RecoveryTimeout,
/// Checkpoint data was corrupted during the failure
CheckpointCorrupted,
}
/// Performance regression metrics comparing before/after experiment
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct PerformanceRegression {
/// 99th percentile latency before the experiment (nanoseconds)
pub latency_p99_before_ns: u64,
/// 99th percentile latency after recovery (nanoseconds)
pub latency_p99_after_ns: u64,
/// Percentage increase in latency (negative if improved)
pub regression_percent: f64,
}
/// Main chaos engineering orchestrator for coordinating experiments
///
/// Manages experiment registration, execution, resource limits, and result tracking.
///
/// Ensures safe concurrent execution with proper cleanup and monitoring.
pub struct ChaosOrchestrator {
/// Registry of all configured experiments
experiments: Arc<RwLock<HashMap<Uuid, ChaosExperiment>>>,
/// Currently executing experiments
active_experiments: Arc<RwLock<HashMap<Uuid, ChaosExecution>>>,
/// Historical results from all experiments
results: Arc<RwLock<Vec<ChaosResult>>>,
/// Event broadcaster for real-time notifications
_event_sender: broadcast::Sender<ChaosEvent>,
/// Semaphore limiting concurrent experiment execution
max_concurrent_experiments: Arc<Semaphore>,
}
/// Internal state tracking for an executing chaos experiment
struct ChaosExecution {
/// Child processes spawned during the experiment
child_processes: Vec<Child>,
/// When the experiment started executing
start_time: Instant,
/// When recovery phase began (if applicable)
recovery_start: Option<Instant>,
/// Path to checkpoint file (if created)
checkpoint_path: Option<String>,
}
/// Events emitted during chaos experiment execution
///
/// These events can be subscribed to for real-time monitoring
/// and alerting during chaos engineering runs.
#[derive(Debug, Clone)]
pub enum ChaosEvent {
/// An experiment has begun execution
ExperimentStarted {
/// Experiment ID
id: Uuid,
/// Experiment name
name: String,
},
/// An experiment has completed (successfully or not)
ExperimentCompleted {
/// Experiment ID
id: Uuid,
/// Final result of the experiment
result: ChaosResult,
},
/// Service recovery phase has begun
RecoveryStarted {
/// Experiment ID
id: Uuid,
/// Service name being recovered
service: String,
},
/// Checkpoint validation has completed
CheckpointValidated {
/// Experiment ID
id: Uuid,
/// Whether checkpoint is valid
valid: bool,
},
/// Performance regression detected after recovery
PerformanceRegression {
/// Experiment ID
id: Uuid,
/// Regression details
regression: PerformanceRegression,
},
}
impl ChaosOrchestrator {
/// Create a new chaos orchestrator with concurrency limits
///
/// # Arguments
/// * `max_concurrent` - Maximum number of experiments to run simultaneously
///
/// # Returns
///
/// A new ChaosOrchestrator instance ready to execute experiments
pub fn new(max_concurrent: usize) -> Self {
let (_event_sender, _) = broadcast::channel(1000);
Self {
experiments: Arc::new(RwLock::new(HashMap::new())),
active_experiments: Arc::new(RwLock::new(HashMap::new())),
results: Arc::new(RwLock::new(Vec::new())),
_event_sender,
max_concurrent_experiments: Arc::new(Semaphore::new(max_concurrent)),
}
}
/// Register a new chaos experiment for future execution
///
/// # Arguments
/// * `experiment` - The chaos experiment configuration to register
///
/// # Returns
/// * `Ok(())` - If experiment was successfully registered
///
/// * `Err(anyhow::Error)` - If registration failed
pub async fn register_experiment(&self, experiment: ChaosExperiment) -> Result<()> {
let mut experiments = self.experiments.write().await;
experiments.insert(experiment.id, experiment);
Ok(())
}
/// Execute a specific chaos experiment by ID
///
/// Runs the complete chaos experiment lifecycle including failure injection,
/// recovery validation, checkpoint verification, and performance measurement.
///
/// # Arguments
/// * `experiment_id` - UUID of the experiment to execute
///
/// # Returns
/// * `Ok(ChaosResult)` - Complete results of the experiment
///
/// * `Err(anyhow::Error)` - If experiment execution failed
pub async fn execute_experiment(&self, experiment_id: Uuid) -> Result<ChaosResult> {
// Acquire semaphore to limit concurrent experiments
let _permit = self.max_concurrent_experiments.acquire().await?;
let experiment = {
let experiments = self.experiments.read().await;
experiments
.get(&experiment_id)
.ok_or_else(|| anyhow::anyhow!("Experiment not found: {}", experiment_id))?
.clone()
};
if !experiment.enabled {
return Err(anyhow::anyhow!(
"Experiment {} is disabled",
experiment.name
));
}
info!("Starting chaos experiment: {}", experiment.name);
// Send start event
let _ = self._event_sender.send(ChaosEvent::ExperimentStarted {
id: experiment.id,
name: experiment.name.clone(),
});
let mut result = ChaosResult {
experiment_id: experiment.id,
started_at: std::time::SystemTime::now(),
completed_at: None,
status: ChaosStatus::Running,
recovery_time_ms: None,
checkpoint_integrity: false,
performance_regression: None,
errors: Vec::new(),
};
// Execute the chaos experiment
match self.run_chaos_experiment(&experiment).await {
Ok(execution_result) => {
result.status = execution_result.status;
result.recovery_time_ms = execution_result.recovery_time_ms;
result.checkpoint_integrity = execution_result.checkpoint_integrity;
result.performance_regression = execution_result.performance_regression;
}
Err(e) => {
result.status = ChaosStatus::Failed;
result.errors.push(e.to_string());
error!("Chaos experiment failed: {}", e);
}
}
result.completed_at = Some(std::time::SystemTime::now());
// Store result
{
let mut results = self.results.write().await;
results.push(result.clone());
}
// Send completion event
let _ = self._event_sender.send(ChaosEvent::ExperimentCompleted {
id: experiment.id,
result: result.clone(),
});
info!(
"Chaos experiment completed: {} - Status: {:?}",
experiment.name, result.status
);
Ok(result)
}
/// Execute the actual chaos experiment
async fn run_chaos_experiment(&self, experiment: &ChaosExperiment) -> Result<ExecutionResult> {
let start_time = Instant::now();
// Step 1: Capture baseline performance metrics
let baseline_metrics = self
.capture_performance_metrics(&experiment.target_service)
.await?;
// Step 2: Create checkpoint if applicable
let checkpoint_path = self.create_checkpoint(&experiment.target_service).await?;
// Step 3: Inject failure
info!("Injecting failure: {:?}", experiment.failure_type);
self.inject_failure(&experiment.failure_type, &experiment.target_service)
.await?;
// Step 4: Wait for failure duration
sleep(experiment.duration).await;
// Step 5: Begin recovery process
let recovery_start = Instant::now();
let _ = self._event_sender.send(ChaosEvent::RecoveryStarted {
id: experiment.id,
service: experiment.target_service.clone(),
});
// Step 6: Validate service recovery within timeout
let recovery_result = timeout(
experiment.recovery_timeout,
self.wait_for_service_recovery(&experiment.target_service),
)
.await;
let recovery_time_ms = recovery_start.elapsed().as_millis() as u64;
// Step 7: Validate checkpoint integrity if applicable
let checkpoint_integrity = if let Some(ref path) = checkpoint_path {
self.validate_checkpoint(&experiment.target_service, path)
.await?
} else {
true // No checkpoint to validate
};
let _ = self._event_sender.send(ChaosEvent::CheckpointValidated {
id: experiment.id,
valid: checkpoint_integrity,
});
// Step 8: Measure post-recovery performance
let post_metrics = self
.capture_performance_metrics(&experiment.target_service)
.await?;
// Step 9: Calculate performance regression
let performance_regression =
self.calculate_performance_regression(&baseline_metrics, &post_metrics);
if let Some(ref regression) = performance_regression {
let _ = self._event_sender.send(ChaosEvent::PerformanceRegression {
id: experiment.id,
regression: regression.clone(),
});
}
// Determine final status
let status = match recovery_result {
Ok(Ok(true))
if checkpoint_integrity && recovery_time_ms <= experiment.max_recovery_time_ms =>
{
ChaosStatus::Succeeded
}
Ok(Ok(true)) if !checkpoint_integrity => ChaosStatus::CheckpointCorrupted,
Ok(Ok(true)) => {
ChaosStatus::Failed // Recovery took too long
}
Ok(Ok(false)) | Ok(Err(_)) | Err(_) => ChaosStatus::RecoveryTimeout,
};
Ok(ExecutionResult {
status,
recovery_time_ms: Some(recovery_time_ms),
checkpoint_integrity,
performance_regression,
})
}
/// Inject specific type of failure
async fn inject_failure(&self, failure_type: &FailureType, service: &str) -> Result<()> {
match failure_type {
FailureType::ProcessKill {
signal,
delay_before_restart_ms,
} => {
self.kill_service_process(service, signal).await?;
sleep(Duration::from_millis(*delay_before_restart_ms)).await;
self.restart_service(service).await?;
}
FailureType::MemoryPressure {
target_mb,
duration_ms,
} => {
self.inject_memory_pressure(*target_mb, *duration_ms)
.await?;
}
FailureType::NetworkPartition {
target_ports,
duration_ms,
} => {
self.create_network_partition(target_ports, *duration_ms)
.await?;
}
FailureType::DiskIoFailure {
target_paths,
failure_rate_percent,
} => {
self.inject_disk_failures(target_paths, *failure_rate_percent)
.await?;
}
FailureType::CpuThrottle {
cpu_limit_percent,
duration_ms,
} => {
self.throttle_cpu(*cpu_limit_percent, *duration_ms).await?;
}
FailureType::GpuResourceExhaustion {
memory_fill_percent,
duration_ms,
} => {
self.exhaust_gpu_resources(*memory_fill_percent, *duration_ms)
.await?;
}
FailureType::DatabaseConnectionFailure {
connection_string,
duration_ms,
} => {
self.inject_db_connection_failure(connection_string, *duration_ms)
.await?;
}
}
Ok(())
}
/// Kill service process with specified signal
async fn kill_service_process(&self, service: &str, signal: &Signal) -> Result<()> {
let signal_arg = match signal {
Signal::SIGTERM => "-TERM",
Signal::SIGKILL => "-KILL",
Signal::SIGSTOP => "-STOP",
Signal::SIGCONT => "-CONT",
};
let output = Command::new("pkill")
.args([signal_arg, service])
.output()
.context("Failed to kill service process")?;
if !output.status.success() {
warn!(
"pkill command failed: {}",
String::from_utf8_lossy(&output.stderr)
);
}
Ok(())
}
/// Restart service process
async fn restart_service(&self, service: &str) -> Result<()> {
let output = Command::new("systemctl")
.args(["restart", service])
.output()
.context("Failed to restart service")?;
if !output.status.success() {
return Err(anyhow::anyhow!(
"Service restart failed: {}",
String::from_utf8_lossy(&output.stderr)
));
}
Ok(())
}
/// Wait for service to recover and be healthy
async fn wait_for_service_recovery(&self, service: &str) -> Result<bool> {
const MAX_ATTEMPTS: u32 = 30;
const DELAY_BETWEEN_ATTEMPTS: Duration = Duration::from_secs(1);
for attempt in 1..=MAX_ATTEMPTS {
if self.is_service_healthy(service).await? {
info!("Service {} recovered after {} attempts", service, attempt);
return Ok(true);
}
if attempt < MAX_ATTEMPTS {
sleep(DELAY_BETWEEN_ATTEMPTS).await;
}
}
warn!(
"Service {} failed to recover within {} attempts",
service, MAX_ATTEMPTS
);
Ok(false)
}
/// Check if service is healthy via health check endpoint
async fn is_service_healthy(&self, service: &str) -> Result<bool> {
// This would typically make HTTP/gRPC health check calls
// For now, checking if process is running
let output = Command::new("pgrep")
.arg(service)
.output()
.context("Failed to check service process")?;
Ok(output.status.success())
}
/// Create checkpoint for ML service
async fn create_checkpoint(&self, service: &str) -> Result<Option<String>> {
if service.contains("ml") {
let checkpoint_path = format!(
"/tmp/chaos_checkpoint_{}_{}",
service,
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)?
.as_secs()
);
// The chaos framework does not make a live gRPC call to the ML
// service's create_checkpoint endpoint; it only returns the
// synthesised checkpoint path so downstream orchestration can
// record the intended path.
info!("Recording intended checkpoint path: {}", checkpoint_path);
Ok(Some(checkpoint_path))
} else {
Ok(None)
}
}
/// Validate checkpoint integrity.
///
/// Currently returns `Ok(true)` unconditionally — the chaos framework
/// does not perform file-integrity, model-state-consistency or
/// checkpoint-load validation. Real validation is done by the ML
/// service's own checkpoint tooling; here we only record the call.
async fn validate_checkpoint(&self, service: &str, checkpoint_path: &str) -> Result<bool> {
info!("Recording validate_checkpoint call: {}", checkpoint_path);
Ok(true)
}
/// Capture performance metrics.
///
/// Returns a synthetic 50 µs p99 latency. The chaos framework does
/// not scrape Prometheus / monitoring endpoints — tests that care
/// about real post-chaos metrics must observe them out-of-band.
async fn capture_performance_metrics(&self, service: &str) -> Result<PerformanceMetrics> {
Ok(PerformanceMetrics {
latency_p99_ns: 50_000, // 50µs synthetic baseline
})
}
/// Calculate performance regression
fn calculate_performance_regression(
&self,
baseline: &PerformanceMetrics,
current: &PerformanceMetrics,
) -> Option<PerformanceRegression> {
if current.latency_p99_ns > baseline.latency_p99_ns {
let regression_percent = ((current.latency_p99_ns as f64
- baseline.latency_p99_ns as f64)
/ baseline.latency_p99_ns as f64)
* 100.0;
Some(PerformanceRegression {
latency_p99_before_ns: baseline.latency_p99_ns,
latency_p99_after_ns: current.latency_p99_ns,
regression_percent,
})
} else {
None
}
}
/// Inject memory pressure
async fn inject_memory_pressure(&self, target_mb: u64, duration_ms: u64) -> Result<()> {
info!(
"Injecting memory pressure: {}MB for {}ms",
target_mb, duration_ms
);
// Use stress-ng or similar tool to create memory pressure
let mut child = Command::new("stress-ng")
.args([
"--vm",
"1",
"--vm-bytes",
&format!("{}M", target_mb),
"--timeout",
&format!("{}ms", duration_ms),
])
.stdin(Stdio::null())
.stdout(Stdio::null())
.stderr(Stdio::null())
.spawn()
.context("Failed to start memory stress test")?;
let _ = child.wait();
Ok(())
}
/// Create network partition
async fn create_network_partition(&self, ports: &[u16], duration_ms: u64) -> Result<()> {
info!(
"Creating network partition for ports {:?} for {}ms",
ports, duration_ms
);
// Use iptables to block traffic to specific ports
for port in ports {
Command::new("iptables")
.args([
"-A",
"OUTPUT",
"-p",
"tcp",
"--dport",
&port.to_string(),
"-j",
"DROP",
])
.output()
.context("Failed to create network partition")?;
}
// Wait for specified duration
sleep(Duration::from_millis(duration_ms)).await;
// Remove iptables rules
for port in ports {
let _ = Command::new("iptables")
.args([
"-D",
"OUTPUT",
"-p",
"tcp",
"--dport",
&port.to_string(),
"-j",
"DROP",
])
.output();
}
Ok(())
}
/// Inject disk I/O failures.
///
/// Currently a no-op beyond logging. Real disk-fault injection is
/// handled by the per-environment chaos runner (e.g. filesystem-level
/// or FUSE fault injection), not this library.
async fn inject_disk_failures(&self, paths: &[String], failure_rate: u8) -> Result<()> {
info!(
"Recording disk-failure injection request for paths {:?} at {}% rate",
paths, failure_rate
);
Ok(())
}
/// Throttle CPU usage
async fn throttle_cpu(&self, limit_percent: u8, duration_ms: u64) -> Result<()> {
info!("Throttling CPU to {}% for {}ms", limit_percent, duration_ms);
// Use cgroups or cpulimit to throttle CPU
let mut child = Command::new("cpulimit")
.args(["-l", &limit_percent.to_string(), "-p", "1"]) // Target init process
.spawn()
.context("Failed to start CPU throttling")?;
sleep(Duration::from_millis(duration_ms)).await;
let _ = child.kill();
Ok(())
}
/// Exhaust GPU resources.
///
/// Currently a no-op beyond logging. GPU-memory-exhaustion scenarios
/// are driven separately by dedicated GPU stress binaries so this
/// generic framework does not hold a CUDA handle itself.
async fn exhaust_gpu_resources(&self, memory_fill_percent: u8, duration_ms: u64) -> Result<()> {
info!(
"Recording GPU-exhaustion request: {}% memory for {}ms",
memory_fill_percent, duration_ms
);
Ok(())
}
/// Inject database connection failures.
///
/// Currently a no-op beyond logging. Real DB fault injection is
/// driven by the environment (iptables / network partitioning of
/// the DB port is handled by `create_network_partition` above).
async fn inject_db_connection_failure(
&self,
connection_string: &str,
duration_ms: u64,
) -> Result<()> {
info!(
"Recording DB-connection-failure request for {} for {}ms",
connection_string, duration_ms
);
Ok(())
}
/// Get results from all executed experiments
///
/// # Returns
///
/// Vector containing results from all completed experiments
pub async fn get_results(&self) -> Vec<ChaosResult> {
self.results.read().await.clone()
}
/// Subscribe to real-time chaos experiment events
///
/// # Returns
///
/// A broadcast receiver for monitoring experiment progress and events
pub fn subscribe_events(&self) -> broadcast::Receiver<ChaosEvent> {
self._event_sender.subscribe()
}
}
/// Internal result structure for chaos experiment execution
#[derive(Debug)]
struct ExecutionResult {
/// Final status of the experiment
status: ChaosStatus,
/// Time taken for recovery in milliseconds
recovery_time_ms: Option<u64>,
/// Whether checkpoint remained valid
checkpoint_integrity: bool,
/// Any performance regression detected
performance_regression: Option<PerformanceRegression>,
}
/// Performance metrics captured during chaos experiments
#[derive(Debug)]
struct PerformanceMetrics {
/// 99th percentile latency in nanoseconds
latency_p99_ns: u64,
}
#[cfg(test)]
mod tests {
use super::*;
#[tokio::test]
async fn test_chaos_orchestrator_creation() {
let orchestrator = ChaosOrchestrator::new(3);
let experiment = ChaosExperiment {
id: Uuid::new_v4(),
name: "MLTrainingService Kill Test".to_string(),
description: "Test MLTrainingService recovery from process kill".to_string(),
target_service: "ml_training_service".to_string(),
failure_type: FailureType::ProcessKill {
signal: Signal::SIGTERM,
delay_before_restart_ms: 1000,
},
duration: Duration::from_secs(10),
recovery_timeout: Duration::from_secs(30),
max_recovery_time_ms: 100, // HFT requirement
enabled: true,
};
assert!(orchestrator.register_experiment(experiment).await.is_ok());
}
}