From 70345e5e94096bc48309e5638b3dd02d25c2fd15 Mon Sep 17 00:00:00 2001 From: Robin Mueller Date: Mon, 14 Sep 2026 15:36:01 +0200 Subject: [PATCH] feat: add FDIR fault counter and wire it into the MGM device handler Add satrs::fdir::FaultCounter, an FSFW-style error threshold counter: counts faults, decrements over time when faults stop, and reports when a threshold is exceeded. Two variants for now, mirroring the hk.rs helper pattern: - FaultCounterStd, backed by std::time::Instant - FaultCounterEmbassy, backed by embassy_time::Instant (embassy-time feature), with an optional defmt::Format impl gated on the defmt feature Add satrs::health::HealthTableMapSync::default() for easy construction of a shared, global health table. Wire both into the example app's MGM device handler as the first real FDIR use case: - the minisim MGM model gains SpiFaultMode (None/AllZeros/AllOnes) and a SetSpiFault request, so a stuck SPI bus can be injected for testing, independent of switch state - MgmHandlerLis3Mdl::poll_sensor checks the SPI transfer result: a comm timeout or an all-1s stuck-bus reply (the same pattern the sim already uses for "device off") counts as a fault. Above threshold, the component is marked Faulty in a HealthTableMapSync shared from main.rs. This logic lives in the device handler, not the SPI comm layer, since deciding what a failed transfer means for FDIR is a handler concern. - an all-0s reply is deliberately not treated as a fault, since it collides with a legitimate zero-field reading Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01BaKjBjnxaHJ6vzficJcjN4 --- satrs-example/minisim/src/acs.rs | 54 ++++- satrs-example/minisim/src/controller.rs | 11 + satrs-example/minisim/src/lib.rs | 47 +++- satrs-example/src/acs/mgm.rs | 178 ++++++++++++++- satrs-example/src/main.rs | 5 + satrs/Cargo.toml | 2 +- satrs/src/fdir.rs | 292 ++++++++++++++++++++++++ satrs/src/health.rs | 9 + satrs/src/lib.rs | 1 + 9 files changed, 583 insertions(+), 16 deletions(-) create mode 100644 satrs/src/fdir.rs diff --git a/satrs-example/minisim/src/acs.rs b/satrs-example/minisim/src/acs.rs index fc008e5..9fa671f 100644 --- a/satrs-example/minisim/src/acs.rs +++ b/satrs-example/minisim/src/acs.rs @@ -7,7 +7,7 @@ use nexosim::{ use satrs_minisim::{ acs::{ lis3mdl::MgmLis3MdlReply, MgmReplyCommon, MgmReplyProvider, MgmSensorValuesMicroTesla, - MgtDipole, MgtHkSet, MgtReply, MGT_GEN_MAGNETIC_FIELD, + MgtDipole, MgtHkSet, MgtReply, SpiFaultMode, MGT_GEN_MAGNETIC_FIELD, }, SimReply, }; @@ -34,6 +34,7 @@ pub struct MagnetometerModel { #[allow(dead_code)] pub periodicity: Duration, pub external_mag_field: Option, + pub spi_fault: SpiFaultMode, pub reply_sender: mpsc::Sender, pub phatom: std::marker::PhantomData, } @@ -44,6 +45,7 @@ impl MagnetometerModel { switch_state: SwitchStateBinary::Off, periodicity, external_mag_field: None, + spi_fault: SpiFaultMode::None, reply_sender, phatom: std::marker::PhantomData, } @@ -55,12 +57,21 @@ impl MagnetometerModel { self.switch_state = switch_state; } + /// Force (or clear) a stuck-bus SPI fault, for FDIR testing purposes. + pub async fn set_spi_fault(&mut self, fault_mode: SpiFaultMode) { + self.spi_fault = fault_mode; + } + pub async fn send_sensor_values(&mut self, _: (), scheduler: &mut Context) { self.reply_sender - .send(ReplyProvider::create_mgm_reply(MgmReplyCommon { - switch_state: self.switch_state, - sensor_values: self.calculate_current_mgm_tuple(current_millis(scheduler.time())), - })) + .send(ReplyProvider::create_mgm_reply( + MgmReplyCommon { + switch_state: self.switch_state, + sensor_values: self + .calculate_current_mgm_tuple(current_millis(scheduler.time())), + }, + self.spi_fault, + )) .expect("sending MGM sensor values failed"); } @@ -182,7 +193,7 @@ pub mod tests { use satrs_minisim::{ acs::{ lis3mdl::{self, MgmLis3MdlReply}, - MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest, + MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest, SpiFaultMode, }, SerializableSimMsgPayload, SimComponent, SimMessageProvider, SimRequest, }; @@ -211,6 +222,37 @@ pub mod tests { assert_eq!(reply.common.sensor_values.z, 0.0); } + #[test] + fn test_mgm_spi_fault_injection_all_ones() { + let mut sim_testbench = SimTestbench::new(); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + + let fault_request = + SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(SpiFaultMode::AllOnes)); + sim_testbench + .send_request(fault_request) + .expect("sending MGM fault injection request failed"); + sim_testbench.handle_sim_requests_time_agnostic(); + sim_testbench.step().unwrap(); + + let data_request = SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::RequestSensorData); + sim_testbench + .send_request(data_request) + .expect("sending MGM request failed"); + sim_testbench.handle_sim_requests_time_agnostic(); + sim_testbench.step().unwrap(); + let sim_reply = sim_testbench + .try_receive_next_reply() + .expect("no MGM reply received"); + let reply = MgmLis3MdlReply::from_sim_message(&sim_reply) + .expect("failed to deserialize MGM sensor values"); + // Even though the device is switched on, the injected fault forces a stuck-bus reply. + assert_eq!(reply.common.switch_state, SwitchStateBinary::On); + assert_eq!(reply.raw.x, -1); + assert_eq!(reply.raw.y, -1); + assert_eq!(reply.raw.z, -1); + } + #[test] fn test_basic_mgm_request_switched_on() { let mut sim_testbench = SimTestbench::new(); diff --git a/satrs-example/minisim/src/controller.rs b/satrs-example/minisim/src/controller.rs index 60312d4..ca43a40 100644 --- a/satrs-example/minisim/src/controller.rs +++ b/satrs-example/minisim/src/controller.rs @@ -153,6 +153,17 @@ impl SimController { .process_event(MagnetometerModel::send_sensor_values, (), addr) .expect("event execution error for mgm"); } + MgmRequestLis3Mdl::SetSpiFault(fault_mode) => { + let addr = match mgm_idx { + 0 => &self.addr_wrapper.mgm_0_addr, + 1 => &self.addr_wrapper.mgm_1_addr, + + _ => panic!("invalid mgm index"), + }; + self.simulation + .process_event(MagnetometerModel::set_spi_fault, fault_mode, addr) + .expect("event execution error for mgm"); + } } Ok(()) } diff --git a/satrs-example/minisim/src/lib.rs b/satrs-example/minisim/src/lib.rs index d659d90..f1587d8 100644 --- a/satrs-example/minisim/src/lib.rs +++ b/satrs-example/minisim/src/lib.rs @@ -202,12 +202,27 @@ pub mod acs { use super::*; pub trait MgmReplyProvider: Send + 'static { - fn create_mgm_reply(common: MgmReplyCommon) -> SimReply; + fn create_mgm_reply(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> SimReply; + } + + /// Fault mode injected on the simulated SPI bus, independent of the switch state. + /// + /// Models the classic symptom of a stuck SPI bus: an undriven MISO line commonly reads + /// back as all-1s, a shorted/grounded one as all-0s. + #[derive(Debug, Default, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] + pub enum SpiFaultMode { + #[default] + None, + AllZeros, + AllOnes, } #[derive(Debug, Copy, Clone, Serialize, Deserialize)] pub enum MgmRequestLis3Mdl { RequestSensorData, + /// Force the raw register reply into a stuck-bus pattern, regardless of switch state. + /// Used to test FDIR handling of SPI bus faults. + SetSpiFault(SpiFaultMode), } impl SerializableSimMsgPayload for MgmRequestLis3Mdl { @@ -236,6 +251,7 @@ pub mod acs { z: 30.0, }; pub const ALL_ONES_SENSOR_VAL: i16 = 0xffff_u16 as i16; + pub const ALL_ZEROS_SENSOR_VAL: i16 = 0; pub mod lis3mdl { use super::*; @@ -263,7 +279,30 @@ pub mod acs { } impl MgmLis3MdlReply { - pub fn new(common: MgmReplyCommon) -> Self { + pub fn new(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> Self { + match fault_mode { + SpiFaultMode::AllZeros => { + return Self { + common, + raw: MgmLis3RawValues { + x: ALL_ZEROS_SENSOR_VAL, + y: ALL_ZEROS_SENSOR_VAL, + z: ALL_ZEROS_SENSOR_VAL, + }, + }; + } + SpiFaultMode::AllOnes => { + return Self { + common, + raw: MgmLis3RawValues { + x: ALL_ONES_SENSOR_VAL, + y: ALL_ONES_SENSOR_VAL, + z: ALL_ONES_SENSOR_VAL, + }, + }; + } + SpiFaultMode::None => (), + } match common.switch_state { SwitchStateBinary::Off => Self { common, @@ -306,8 +345,8 @@ pub mod acs { } impl MgmReplyProvider for MgmLis3MdlReply { - fn create_mgm_reply(common: MgmReplyCommon) -> SimReply { - SimReply::new(&Self::new(common)) + fn create_mgm_reply(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> SimReply { + SimReply::new(&Self::new(common, fault_mode)) } } } diff --git a/satrs-example/src/acs/mgm.rs b/satrs-example/src/acs/mgm.rs index ca4fde7..6d27305 100644 --- a/satrs-example/src/acs/mgm.rs +++ b/satrs-example/src/acs/mgm.rs @@ -1,3 +1,5 @@ +use satrs::fdir::FaultCounterStd; +use satrs::health::{HealthState, HealthTableMapSync, HealthTableProvider}; use satrs::spacepackets::CcsdsPacketIdAndPsc; use satrs_example::{HkHelperSingleSet, ModeHelper, TimestampHelper, TmtcQueues}; use satrs_minisim::acs::MgmRequestLis3Mdl; @@ -26,6 +28,11 @@ pub const X_LOWBYTE_IDX: usize = 9; pub const Y_LOWBYTE_IDX: usize = 11; pub const Z_LOWBYTE_IDX: usize = 13; +// FDIR configuration for SPI communication faults (timeouts, stuck bus). Chosen so a handful of +// transient errors are tolerated but a persistently faulty bus is caught quickly. +pub const SPI_FAULT_THRESHOLD: u32 = 2; +pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30); + #[derive(Debug, PartialEq, Eq, Clone, Copy)] pub enum MgmId { _0, @@ -63,10 +70,11 @@ pub struct SpiDummyInterface { } impl SpiDummyInterface { - fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) { + fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool { rx[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.x.to_le_bytes()); rx[Y_LOWBYTE_IDX..Y_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.y.to_be_bytes()); rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.z.to_be_bytes()); + true } } @@ -77,11 +85,12 @@ pub struct TestSpiInterface { } impl TestSpiInterface { - fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) { + fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool { rx[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.x.to_le_bytes()); rx[Y_LOWBYTE_IDX..Y_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.y.to_le_bytes()); rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.z.to_le_bytes()); self.call_count += 1; + true } } @@ -92,7 +101,9 @@ pub struct SpiSimInterface { impl SpiSimInterface { // Right now, we only support requesting sensor data and not configuration of the sensor. - fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) { + // Returns whether a reply was received in time. This is a communication-layer concern only; + // it is up to the caller to decide what a failed transfer means for FDIR purposes. + fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool { let mgm_sensor_request = MgmRequestLis3Mdl::RequestSensorData; if let Err(e) = self .sim_request_tx @@ -110,9 +121,11 @@ impl SpiSimInterface { .copy_from_slice(&sim_reply_lis3.raw.y.to_le_bytes()); rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2] .copy_from_slice(&sim_reply_lis3.raw.z.to_le_bytes()); + true } Err(e) => { log::warn!("MGM LIS3 SIM reply timeout: {e}"); + false } } } @@ -126,7 +139,8 @@ pub enum SpiCommunication { } impl SpiCommunication { - fn transfer(&mut self, tx: &[u8], rx: &mut [u8]) { + /// Performs the transfer, returning whether it succeeded. + fn transfer(&mut self, tx: &[u8], rx: &mut [u8]) -> bool { match self { SpiCommunication::Dummy(dummy) => dummy.transfer(tx, rx), SpiCommunication::Sim(sim_if) => sim_if.transfer(tx, rx), @@ -159,9 +173,12 @@ pub struct MgmHandlerLis3Mdl { hk_helper: HkHelperSingleSet, mode_helpers: ModeHelper, mode_leaf_helper: ModeLeafHelper, + spi_fault_counter: FaultCounterStd, + health_table: HealthTableMapSync, } impl MgmHandlerLis3Mdl { + #[allow(clippy::too_many_arguments)] pub fn new( id: MgmId, tmtc_queues: TmtcQueues, @@ -170,6 +187,7 @@ impl MgmHandlerLis3Mdl { shared_mgm_set: Arc>, mode_leaf_helper: ModeLeafHelper, mode_timeout: Duration, + health_table: HealthTableMapSync, ) -> Self { Self { id, @@ -182,6 +200,11 @@ impl MgmHandlerLis3Mdl { stamp_helper: TimestampHelper::default(), hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)), mode_leaf_helper, + spi_fault_counter: FaultCounterStd::new( + SPI_FAULT_THRESHOLD, + SPI_FAULT_DECREMENT_AFTER, + ), + health_table, } } @@ -339,10 +362,14 @@ impl MgmHandlerLis3Mdl { pub fn poll_sensor(&mut self) { // Communicate with the device. This is actually how to read the data from the LIS3 device // SPI interface. - self.spi_com.transfer( + let transfer_ok = self.spi_com.transfer( &self.buffers.tx_buf[0..NR_OF_DATA_AND_CFG_REGISTERS + 1], &mut self.buffers.rx_buf[0..NR_OF_DATA_AND_CFG_REGISTERS + 1], ); + if !transfer_ok { + self.register_spi_fault(); + return; + } let x_raw = i16::from_le_bytes( self.buffers.rx_buf[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2] .try_into() @@ -358,6 +385,16 @@ impl MgmHandlerLis3Mdl { .try_into() .unwrap(), ); + // A stuck-high SPI bus (undriven MISO) reads back as all-1s on every register, + // regardless of what was actually requested. This is the pattern this codebase already + // uses for "no real device behind the bus" (see the switched-off MGM sim reply). An + // all-0s reading is not used here, since it collides with a legitimate zero-field + // reading and would cause false positives. + if x_raw == -1 && y_raw == -1 && z_raw == -1 { + self.register_spi_fault(); + return; + } + self.spi_fault_counter.try_decrement(); // Simple scaling to retrieve the float value, assuming the best sensor resolution. let mut mgm_guard = self.shared_mgm_set.lock().unwrap(); mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS; @@ -367,6 +404,36 @@ impl MgmHandlerLis3Mdl { drop(mgm_guard); } + /// Registers one SPI communication fault (timeout or stuck-bus data) with the FDIR fault + /// counter, invalidating the current sensor set. If the failure threshold is exceeded, the + /// component is marked faulty in the global health table. + fn register_spi_fault(&mut self) { + log::warn!("{}: SPI communication fault", self.id.str()); + self.shared_mgm_set.lock().unwrap().valid = false; + if !self.spi_fault_counter.increment_and_check() { + return; + } + // Ground may have taken manual control, or already given up on this component. + // Autonomous FDIR should not override that decision. + let component_id = self.id.component_id().into(); + match self.health_table.health(component_id) { + Some(HealthState::ExternalControl) | Some(HealthState::PermanentFaulty) => { + log::info!( + "{}: SPI fault threshold exceeded, but health is externally controlled, \ + not overriding", + self.id.str() + ); + } + _ => { + log::error!( + "{}: SPI fault threshold exceeded, marking component faulty", + self.id.str() + ); + self.health_table.set_health(component_id, HealthState::Faulty); + } + } + } + fn start_transition(&mut self, target_mode: DeviceMode, _forced: bool) { log::info!("{}: transitioning to mode {:?}", self.id.str(), target_mode); if target_mode == DeviceMode::Off { @@ -507,6 +574,7 @@ mod tests { pub tc_tx: mpsc::SyncSender, pub tm_rx: mpsc::Receiver, pub switch_rx: mpsc::Receiver>, + pub health_table: HealthTableMapSync, pub handler: MgmHandlerLis3Mdl, } @@ -526,6 +594,7 @@ mod tests { switch_map.insert(SwitchId::Mgm0, SwitchState::Off); let switch_map = SwitchSet::new(switch_map); let shared_switch_set = SharedSwitchSet::new(Mutex::new(switch_map)); + let health_table = HealthTableMapSync::default(); let handler = MgmHandlerLis3Mdl::new( MgmId::_0, TmtcQueues { tc_rx, tm_tx }, @@ -534,18 +603,37 @@ mod tests { shared_mgm_set, mode_leaf_helper, Duration::from_millis(100), + health_table.clone(), ); Self { assembly_mode_request_tx, mode_report_rx, shared_switch_set, switch_rx, + health_table, handler, tm_rx, tc_tx, } } + /// Switches the MGM to `Normal` mode, completing the power-switch handshake. + pub fn switch_to_normal(&mut self) { + self.tc_tx + .send(create_request_tc( + MgmSelect::_0, + mgm::request::Request::Mode(ModeRequest::SetMode(DeviceMode::Normal)), + )) + .unwrap(); + self.handler.periodic_operation(); + self.shared_switch_set + .lock() + .unwrap() + .set_switch_state(SwitchId::Mgm0, SwitchState::On); + self.handler.periodic_operation(); + assert_eq!(self.handler.mode(), DeviceMode::Normal); + } + pub fn test_spi_interface(&mut self) -> &mut TestSpiInterface { match &mut self.handler.spi_com { SpiCommunication::Dummy(_) | SpiCommunication::Sim(_) => { @@ -767,4 +855,84 @@ mod tests { matches!(testbench.tm_rx.try_recv(), Err(TryRecvError::Empty)); } + + #[test] + fn test_spi_fault_below_threshold_stays_healthy() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { + x: -1, + y: -1, + z: -1, + }; + // One stuck-bus reading should not be enough to trip SPI_FAULT_THRESHOLD. + testbench.handler.periodic_operation(); + assert_eq!( + testbench.health_table.health(ComponentId::AcsMgm0.into()), + None, + "component should not be marked faulty yet" + ); + assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid); + } + + #[test] + fn test_spi_fault_above_threshold_marks_component_faulty() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { + x: -1, + y: -1, + z: -1, + }; + // SPI_FAULT_THRESHOLD is exceeded on the (threshold + 1)-th stuck-bus reading. + for _ in 0..SPI_FAULT_THRESHOLD + 1 { + testbench.handler.periodic_operation(); + } + assert_eq!( + testbench.health_table.health(ComponentId::AcsMgm0.into()), + Some(HealthState::Faulty) + ); + assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid); + } + + #[test] + fn test_spi_fault_does_not_override_external_control() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench + .health_table + .set_health(ComponentId::AcsMgm0.into(), HealthState::ExternalControl); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { + x: -1, + y: -1, + z: -1, + }; + for _ in 0..SPI_FAULT_THRESHOLD + 1 { + testbench.handler.periodic_operation(); + } + // Ground took manual control; autonomous FDIR must not override that decision. + assert_eq!( + testbench.health_table.health(ComponentId::AcsMgm0.into()), + Some(HealthState::ExternalControl) + ); + } + + #[test] + fn test_recovering_from_spi_fault_clears_invalid_data_flag() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { + x: -1, + y: -1, + z: -1, + }; + testbench.handler.periodic_operation(); + assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid); + + // Bus recovers before the threshold is exceeded. + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues::default(); + testbench.handler.periodic_operation(); + assert_eq!(testbench.health_table.health(ComponentId::AcsMgm0.into()), None); + assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid); + } } diff --git a/satrs-example/src/main.rs b/satrs-example/src/main.rs index 8eeb968..bf4a98a 100644 --- a/satrs-example/src/main.rs +++ b/satrs-example/src/main.rs @@ -156,6 +156,9 @@ fn main() { let (switch_request_tx, switch_request_rx) = mpsc::sync_channel(20); let switch_helper = PowerSwitchHelper::new(switch_request_tx, shared_switch_set.clone()); + // Global FDIR health table, shared by all software objects. + let health_table = satrs::health::HealthTableMapSync::default(); + let shared_mgm_0_set = Arc::default(); let shared_mgm_1_set = Arc::default(); let (mgm_0_spi_interface, mgm_1_spi_interface) = @@ -194,6 +197,7 @@ fn main() { report_tx: mgm_0_mode_report_tx, }, Duration::from_millis(1000), + health_table.clone(), ); let mut mgm_1_handler = mgm::MgmHandlerLis3Mdl::new( mgm::MgmId::_1, @@ -209,6 +213,7 @@ fn main() { report_tx: mgm_1_mode_report_tx, }, Duration::from_millis(1000), + health_table.clone(), ); let mut mgm_assembly = mgm_assembly::Assembly::new( mgm_assembly::ParentQueueHelper { diff --git a/satrs/Cargo.toml b/satrs/Cargo.toml index 023ea3b..181c7f9 100644 --- a/satrs/Cargo.toml +++ b/satrs/Cargo.toml @@ -72,7 +72,7 @@ alloc = [ ] serde = ["dep:serde", "spacepackets/serde", "satrs-shared/serde"] crossbeam = ["crossbeam-channel"] -defmt = ["dep:defmt", "spacepackets/defmt"] +defmt = ["dep:defmt", "spacepackets/defmt", "embassy-time?/defmt"] embassy-time = ["dep:embassy-time"] test_util = [] diff --git a/satrs/src/fdir.rs b/satrs/src/fdir.rs new file mode 100644 index 0000000..4d33b17 --- /dev/null +++ b/satrs/src/fdir.rs @@ -0,0 +1,292 @@ +//! # FDIR (Fault Detection, Isolation and Recovery) helpers +//! +//! A fault counter tracks a monotonic fault count, decrements it over time when faults stop +//! occurring, and reports when a configured failure threshold has been exceeded. This is the +//! typical building block used to turn a stream of transient error reports into a single +//! "component is faulty" decision without reacting to the first isolated error. +//! +//! The design follows the FSFW `FaultCounter`: +//! +//! +//! Pick a variant based on what clock is available: +//! +//! - [FaultCounterStd]: `std::time::Instant`, behind the `std` feature. +#![cfg_attr( + feature = "embassy-time", + doc = "- [FaultCounterEmbassy]: `embassy_time::Instant`, behind the `embassy-time` feature." +)] +#![deny(missing_docs)] + +/// Fault counter backed by [std::time::Instant]. +#[cfg(feature = "std")] +#[derive(Debug, Clone)] +pub struct FaultCounterStd { + fault_count: u32, + failure_threshold: u32, + decrement_after: core::time::Duration, + last_decrement: Option, +} + +#[cfg(feature = "std")] +impl FaultCounterStd { + /// Create a new [`FaultCounterStd`]. + /// + /// - `failure_threshold`: threshold above which [`Self::above_threshold`] returns `true` and + /// resets the internal count. + /// - `decrement_after`: minimum duration between automatic decrements performed by + /// [`Self::try_decrement`]. + pub fn new(failure_threshold: u32, decrement_after: core::time::Duration) -> Self { + Self { + fault_count: 0, + failure_threshold, + decrement_after, + last_decrement: None, + } + } + + /// Current fault count. + pub fn fault_count(&self) -> u32 { + self.fault_count + } + + /// Increase the fault count by `1`. + /// + /// If the counter was previously `0`, this starts a new decrement clock. + pub fn increment(&mut self) { + if self.fault_count == 0 { + self.last_decrement = Some(std::time::Instant::now()); + } + self.fault_count += 1; + } + + /// Increase the fault count by `n`. + pub fn increment_n(&mut self, n: u32) { + for _ in 0..n { + self.increment(); + } + } + + fn has_decrement_timedout(&self) -> bool { + match self.last_decrement { + Some(last_decrement) => last_decrement.elapsed() >= self.decrement_after, + None => false, + } + } + + /// Decrease the fault count by `1` if the decrement timeout elapsed. + /// + /// Returns `true` if a decrement was performed, `false` otherwise. A decrement is only + /// performed when the counter is non-zero and at least `decrement_after` has elapsed since + /// the last decrement. + pub fn try_decrement(&mut self) -> bool { + if self.fault_count == 0 || !self.has_decrement_timedout() { + return false; + } + self.last_decrement = Some(std::time::Instant::now()); + self.fault_count -= 1; + true + } + + /// Check whether the counter exceeded the failure threshold. + /// + /// Returns `true` when `fault_count > failure_threshold`. In that case, the counter is reset + /// to `0`. + pub fn above_threshold(&mut self) -> bool { + if self.fault_count > self.failure_threshold { + self.fault_count = 0; + return true; + } + false + } + + /// Convenience helper to increment once and immediately check the threshold. + pub fn increment_and_check(&mut self) -> bool { + self.increment(); + self.above_threshold() + } + + /// Clear the counter and decrement timing state. + pub fn clear(&mut self) { + self.fault_count = 0; + self.last_decrement = None; + } + + /// Update the failure threshold used by [`Self::above_threshold`]. + pub fn set_failure_threshold(&mut self, threshold: u32) { + self.failure_threshold = threshold; + } + + /// Update the minimum interval between automatic decrements. + pub fn set_decrement_after(&mut self, duration: core::time::Duration) { + self.decrement_after = duration; + } +} + +/// Fault counter backed by [embassy_time::Instant]. +#[cfg(feature = "embassy-time")] +#[derive(Debug, Clone, Copy)] +#[cfg_attr(feature = "defmt", derive(defmt::Format))] +pub struct FaultCounterEmbassy { + fault_count: u32, + failure_threshold: u32, + decrement_after: embassy_time::Duration, + last_decrement: Option, +} + +#[cfg(feature = "embassy-time")] +impl FaultCounterEmbassy { + /// Create a new [`FaultCounterEmbassy`]. + /// + /// - `failure_threshold`: threshold above which [`Self::above_threshold`] returns `true` and + /// resets the internal count. + /// - `decrement_after`: minimum duration between automatic decrements performed by + /// [`Self::try_decrement`]. + pub fn new(failure_threshold: u32, decrement_after: embassy_time::Duration) -> Self { + Self { + fault_count: 0, + failure_threshold, + decrement_after, + last_decrement: None, + } + } + + /// Current fault count. + pub fn fault_count(&self) -> u32 { + self.fault_count + } + + /// Increase the fault count by `1`. + /// + /// If the counter was previously `0`, this starts a new decrement clock. + pub fn increment(&mut self) { + if self.fault_count == 0 { + self.last_decrement = Some(embassy_time::Instant::now()); + } + self.fault_count += 1; + } + + /// Increase the fault count by `n`. + pub fn increment_n(&mut self, n: u32) { + for _ in 0..n { + self.increment(); + } + } + + fn has_decrement_timedout(&self) -> bool { + match self.last_decrement { + Some(last_decrement) => { + embassy_time::Instant::now().duration_since(last_decrement) >= self.decrement_after + } + None => false, + } + } + + /// Decrease the fault count by `1` if the decrement timeout elapsed. + /// + /// Returns `true` if a decrement was performed, `false` otherwise. A decrement is only + /// performed when the counter is non-zero and at least `decrement_after` has elapsed since + /// the last decrement. + pub fn try_decrement(&mut self) -> bool { + if self.fault_count == 0 || !self.has_decrement_timedout() { + return false; + } + self.last_decrement = Some(embassy_time::Instant::now()); + self.fault_count -= 1; + true + } + + /// Check whether the counter exceeded the failure threshold. + /// + /// Returns `true` when `fault_count > failure_threshold`. In that case, the counter is reset + /// to `0`. + pub fn above_threshold(&mut self) -> bool { + if self.fault_count > self.failure_threshold { + self.fault_count = 0; + return true; + } + false + } + + /// Convenience helper to increment once and immediately check the threshold. + pub fn increment_and_check(&mut self) -> bool { + self.increment(); + self.above_threshold() + } + + /// Clear the counter and decrement timing state. + pub fn clear(&mut self) { + self.fault_count = 0; + self.last_decrement = None; + } + + /// Update the failure threshold used by [`Self::above_threshold`]. + pub fn set_failure_threshold(&mut self, threshold: u32) { + self.failure_threshold = threshold; + } + + /// Update the minimum interval between automatic decrements. + pub fn set_decrement_after(&mut self, duration: embassy_time::Duration) { + self.decrement_after = duration; + } +} + +#[cfg(all(test, feature = "std"))] +mod tests { + use super::*; + use std::thread; + use std::time::Duration; + + #[test] + fn threshold_not_exceeded_below_limit() { + let mut fc = FaultCounterStd::new(2, Duration::from_secs(60)); + assert!(!fc.increment_and_check()); + assert!(!fc.increment_and_check()); + assert_eq!(fc.fault_count(), 2); + } + + #[test] + fn threshold_exceeded_resets_counter() { + let mut fc = FaultCounterStd::new(2, Duration::from_secs(60)); + fc.increment_n(3); + assert!(fc.above_threshold()); + assert_eq!(fc.fault_count(), 0); + assert!(!fc.above_threshold()); + } + + #[test] + fn decrement_only_after_timeout() { + let mut fc = FaultCounterStd::new(5, Duration::from_millis(20)); + fc.increment(); + assert!(!fc.try_decrement()); + thread::sleep(Duration::from_millis(30)); + assert!(fc.try_decrement()); + assert_eq!(fc.fault_count(), 0); + } + + #[test] + fn decrement_noop_when_empty() { + let mut fc = FaultCounterStd::new(5, Duration::from_millis(1)); + thread::sleep(Duration::from_millis(2)); + assert!(!fc.try_decrement()); + } + + #[test] + fn increment_after_empty_resets_decrement_timing() { + let mut fc = FaultCounterStd::new(5, Duration::from_millis(20)); + fc.increment(); + thread::sleep(Duration::from_millis(30)); + assert!(fc.try_decrement()); + // Counter is 0 again, incrementing should require a fresh decrement_after wait. + fc.increment(); + assert!(!fc.try_decrement()); + } + + #[test] + fn clear_resets_state() { + let mut fc = FaultCounterStd::new(1, Duration::from_secs(60)); + fc.increment_n(2); + fc.clear(); + assert_eq!(fc.fault_count(), 0); + assert!(!fc.above_threshold()); + } +} diff --git a/satrs/src/health.rs b/satrs/src/health.rs index 9d32457..9992b27 100644 --- a/satrs/src/health.rs +++ b/satrs/src/health.rs @@ -27,6 +27,15 @@ impl HealthTableMapSync { } } +#[cfg(feature = "std")] +impl Default for HealthTableMapSync { + /// Creates an empty, shared health table. Absent entries are up to the consumer to + /// interpret, for example as [HealthState::Healthy] by default. + fn default() -> Self { + Self::new(hashbrown::HashMap::new()) + } +} + #[cfg(feature = "std")] impl HealthTableProvider for HealthTableMapSync { fn health(&self, id: ComponentId) -> Option { diff --git a/satrs/src/lib.rs b/satrs/src/lib.rs index 70f9dfc..533b880 100644 --- a/satrs/src/lib.rs +++ b/satrs/src/lib.rs @@ -15,6 +15,7 @@ pub mod ccsds; pub mod encoding; #[cfg(feature = "std")] pub mod executable; +pub mod fdir; pub mod hal; pub mod health; /// Helpers to track when housekeeping sets need to be regenerated.