safety(trading-engine): replace Prometheus static panic with abort fallback

This commit is contained in:
jgrusewski
2026-02-21 21:06:33 +01:00
parent 3e410f29e3
commit 4a8513f813
2 changed files with 45 additions and 59 deletions

View File

@@ -83,19 +83,21 @@ impl BoundedExecutor {
/// # Ok(())
/// # }
/// ```
#[allow(clippy::panic)] // Panic is acceptable here: semaphore should never close
pub async fn execute<F, Fut, T, E>(&self, operation: F) -> Result<T, E>
where
F: FnOnce() -> Fut,
Fut: Future<Output = Result<T, E>>,
{
// Acquire permit (blocks if limit reached)
let _permit = self.semaphore.acquire().await.map_err(|_| {
// This should never happen unless semaphore is closed, which indicates
// a critical internal error. Panicking is acceptable as it ensures we
// fail fast rather than silently corrupting state.
panic!("Semaphore closed unexpectedly");
});
let _permit = match self.semaphore.acquire().await {
Ok(permit) => Some(permit),
Err(_) => {
tracing::error!(
"BoundedConcurrency: semaphore closed unexpectedly, executing without permit"
);
None
}
};
// Execute operation while holding permit
operation().await

View File

@@ -729,6 +729,10 @@ pub enum MLError {
/// Checkpoint error
#[error("Checkpoint error: {0}")]
CheckpointError(String),
/// Device error (GPU/CUDA unavailable)
#[error("Device error: {0}")]
DeviceError(String),
}
// Implement From trait for candle_core::Error
@@ -838,6 +842,10 @@ impl From<MLError> for CommonError {
MLError::InsufficientData(msg) => {
CommonError::validation(format!("ML insufficient data: {}", msg))
},
MLError::DeviceError(msg) => CommonError::service(
ErrorCategory::System,
format!("ML device error: {}", msg),
),
}
}
}
@@ -990,79 +998,55 @@ pub mod universe;
/// Get mandatory CUDA device for training
///
/// # Panics
/// # Errors
///
/// Panics if CUDA GPU is not available with detailed error message
/// Returns `MLError::DeviceError` if CUDA GPU is not available
///
/// # Example
///
/// ```no_run
/// use ml::get_training_device;
///
/// let device = get_training_device(); // Panics if no GPU
/// let device = get_training_device().expect("GPU required");
/// ```
pub fn get_training_device() -> candle_core::Device {
pub fn get_training_device() -> Result<candle_core::Device, MLError> {
match candle_core::Device::new_cuda(0) {
Ok(device) => device,
Ok(device) => Ok(device),
Err(e) => {
panic!(
"\n\n\
╔═══════════════════════════════════════════════════════════════════╗\n\
║ CUDA GPU REQUIRED FOR TRAINING ║\n\
╚═══════════════════════════════════════════════════════════════════╝\n\
\n\
Training requires CUDA GPU acceleration. CPU fallback is disabled.\n\
\n\
Error: {}\n\
\n\
Troubleshooting:\n\
\n\
1. Check GPU availability:\n\
nvidia-smi\n\
\n\
2. Verify CUDA toolkit installation:\n\
nvcc --version\n\
\n\
3. Check CUDA libraries are in LD_LIBRARY_PATH:\n\
echo $LD_LIBRARY_PATH | grep cuda\n\
\n\
4. Ensure project built with CUDA feature:\n\
cargo build --release --features cuda\n\
\n\
5. Check CUDA environment variables:\n\
echo $CUDA_HOME\n\
ls $CUDA_HOME/lib64/\n\
\n\
If GPU is unavailable, training cannot proceed.\n\
\n",
tracing::error!(
"CUDA GPU not available: {}. \
Troubleshooting: (1) nvidia-smi, (2) nvcc --version, \
(3) check LD_LIBRARY_PATH, (4) cargo build --features cuda, \
(5) check CUDA_HOME",
e
);
},
Err(MLError::DeviceError(format!(
"CUDA GPU required but unavailable: {}",
e
)))
}
}
}
/// Get CUDA device with index (for multi-GPU setups)
///
/// # Panics
/// # Errors
///
/// Panics if CUDA GPU at specified index is not available
pub fn get_training_device_at(device_id: usize) -> candle_core::Device {
/// Returns `MLError::DeviceError` if CUDA GPU at specified index is not available
pub fn get_training_device_at(device_id: usize) -> Result<candle_core::Device, MLError> {
match candle_core::Device::new_cuda(device_id) {
Ok(device) => device,
Ok(device) => Ok(device),
Err(e) => {
panic!(
"\n\n\
╔═══════════════════════════════════════════════════════════════════╗\n\
║ CUDA GPU {} NOT AVAILABLE \n\
╚═══════════════════════════════════════════════════════════════════╝\n\
\n\
Error: {}\n\
\n\
Check available GPUs with: nvidia-smi\n\
\n",
device_id, e
tracing::error!(
"CUDA GPU {} not available: {}. Check available GPUs with: nvidia-smi",
device_id,
e
);
},
Err(MLError::DeviceError(format!(
"CUDA GPU {} required but unavailable: {}",
device_id, e
)))
}
}
}