From a8e07252c38ff444cedc427b04d1ba462faaeade Mon Sep 17 00:00:00 2001 From: Robin Mueller Date: Thu, 24 Sep 2026 18:34:35 +0200 Subject: [PATCH] MGT fdir --- satrs-example/src/acs/mgm.rs | 176 +++++++-------------------- satrs-example/src/device_fdir.rs | 200 +++++++++++++++++++++++++++++++ satrs-example/src/main.rs | 1 + 3 files changed, 246 insertions(+), 131 deletions(-) create mode 100644 satrs-example/src/device_fdir.rs diff --git a/satrs-example/src/acs/mgm.rs b/satrs-example/src/acs/mgm.rs index 913a0ad..c60c752 100644 --- a/satrs-example/src/acs/mgm.rs +++ b/satrs-example/src/acs/mgm.rs @@ -1,4 +1,4 @@ -use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir}; +use satrs::fdir::{FaultCounterStd, RecoveryEvent}; use satrs::health::HealthTableMapSync; use satrs::spacepackets::CcsdsPacketIdAndPsc; use satrs_example::{HkHelperSingleSet, TimestampHelper, TmtcQueues}; @@ -15,6 +15,7 @@ use types::pcdu::SwitchId; use types::{ComponentId, DeviceMode, HkRequestType, acs::mgm}; use crate::ccsds::pack_ccsds_tm_packet_for_now; +use crate::device_fdir::{DeviceFdir, FdirEvent}; use crate::device_mode::{ModeTransitionEvent, SwitchAndModeHelper}; use crate::eps::PowerSwitchHelper; @@ -34,13 +35,6 @@ pub const Z_LOWBYTE_IDX: usize = 13; pub const SPI_FAULT_THRESHOLD: u32 = 2; pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30); -// FDIR configuration for power cycle recoveries. The component is marked faulty if it would be -// recovered more than RECOVERY_THRESHOLD times before the counter is decremented again. -pub const RECOVERY_THRESHOLD: u32 = 2; -pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60); -/// Time the device stays unpowered during a power cycle, so it can fully discharge. -pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500); - #[derive(Debug, PartialEq, Eq, Clone, Copy)] pub enum MgmId { _0, @@ -48,7 +42,7 @@ pub enum MgmId { } impl MgmId { - pub const fn str(&self) -> &str { + pub const fn str(&self) -> &'static str { match self { MgmId::_0 => "MGM 0", MgmId::_1 => "MGM 1", @@ -191,9 +185,7 @@ pub struct MgmHandlerLis3Mdl { hk_helper: HkHelperSingleSet, switch_and_mode_helper: SwitchAndModeHelper, mode_leaf_helper: ModeLeafHelper, - spi_fault_counter: FaultCounterStd, - fdir: RecoveryFdir, - recovery_off_duration: Duration, + fdir: DeviceFdir, event_tx: mpsc::SyncSender<(ComponentId, mgm::Event)>, } @@ -225,14 +217,12 @@ impl MgmHandlerLis3Mdl { stamp_helper: TimestampHelper::default(), hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)), mode_leaf_helper, - spi_fault_counter: FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER), - fdir: RecoveryFdir::new( - id.component_id().into(), + fdir: DeviceFdir::new( + id.str(), + id.component_id(), health_table, - RECOVERY_THRESHOLD, - RECOVERY_DECREMENT_AFTER, + FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER), ), - recovery_off_duration: RECOVERY_OFF_DURATION, event_tx, } } @@ -253,8 +243,9 @@ impl MgmHandlerLis3Mdl { // Handle assembly related messages. self.handle_mode_leaf_handling(); - self.fdir.periodic_operation(); - self.check_needs_recovery(); + self.fdir + .periodic_operation(&mut self.switch_and_mode_helper); + self.handle_fdir_events(); // Handle mode transitions first. This also takes care of recoveries required by FDIR. if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() { @@ -266,9 +257,14 @@ impl MgmHandlerLis3Mdl { self.handle_mode_transition_failure(tc_commander) } // The mode did not change for other components, so there is nothing to report. - ModeTransitionEvent::PowerCycleDone => self.handle_recovery_done(), + ModeTransitionEvent::PowerCycleDone => { + self.fdir.handle_power_cycle_done(); + self.handle_fdir_events(); + } ModeTransitionEvent::PowerCycleFailed { restore_mode } => { - self.handle_recovery_failure(restore_mode) + self.fdir + .handle_power_cycle_failed(&mut self.switch_and_mode_helper, restore_mode); + self.handle_fdir_events(); } } } @@ -453,7 +449,7 @@ impl MgmHandlerLis3Mdl { return; } // Successfull readout, so we can decrement the counter. - self.spi_fault_counter.try_decrement(); + self.fdir.register_success(); // Simple scaling to retrieve the float value, assuming the best sensor resolution. let mut mgm_guard = self.shared_mgm_set.lock().unwrap(); mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS; @@ -463,122 +459,39 @@ impl MgmHandlerLis3Mdl { drop(mgm_guard); } - /// Registers one SPI fault with the FDIR fault counter, invalidating the current - /// sensor set. If the failure threshold is exceeded, the device is power cycled. If it was - /// power cycled too often, the component is marked faulty and commanded off instead. + /// Registers one SPI fault with the FDIR, invalidating the current sensor set. fn register_spi_fault(&mut self) { log::warn!("{}: stuck-bus SPI fault", self.id.str()); self.shared_mgm_set.lock().unwrap().valid = false; - if !self.spi_fault_counter.increment_and_check() { - return; - } - match self.fdir.handle_fault() { - FaultResponse::Ignored => { - log::info!( - "{}: SPI fault threshold exceeded, but component is already faulty, \ - recovering or externally controlled", - self.id.str() - ); - } - FaultResponse::Recover => { - log::warn!( - "{}: SPI fault threshold exceeded, power cycling device", - self.id.str() - ); - self.send_event(mgm::Event::SpiFaultThresholdExceeded); - self.check_needs_recovery(); - } - FaultResponse::SetFaulty => { - log::error!( - "{}: SPI fault threshold exceeded after too many recoveries, marking \ - component faulty", - self.id.str() - ); - self.send_event(mgm::Event::SpiFaultThresholdExceeded); - self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); - self.switch_off_faulty_device(); - } + self.fdir.register_fault(&mut self.switch_and_mode_helper); + self.handle_fdir_events(); + } + + fn handle_fdir_events(&mut self) { + while let Some(event) = self.fdir.next_event() { + let event = match event { + FdirEvent::FaultThresholdExceeded => mgm::Event::SpiFaultThresholdExceeded, + FdirEvent::Recovery(recovery_event) => { + // The device is power cycled or switched off. + if matches!( + recovery_event, + RecoveryEvent::Started | RecoveryEvent::ThresholdExceeded + ) { + self.shared_mgm_set.lock().unwrap().valid = false; + } + mgm::Event::Recovery(recovery_event) + } + }; + self.send_event(event); } } - fn switch_off_faulty_device(&mut self) { - // Do not restart an already pending Off transition, which would reset the transition - // state machine before it can finish. - if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) { - log::warn!("{}: commanding device off due to fault", self.id.str()); - self.start_transition(DeviceMode::Off, None); - } - } - - /// Starts a power cycle if the health is [satrs::health::HealthState::NeedsRecovery]. The health is set - /// either by the FDIR or by ground. - fn check_needs_recovery(&mut self) { - if self.switch_and_mode_helper.power_cycle_active() - || self.switch_and_mode_helper.target().is_some() - || !self.fdir.needs_recovery() - { - return; - } - if self.mode() == DeviceMode::Off { - // Nothing to power cycle, the next switch-on is a fresh start anyway. - log::info!("{}: device is off, no recovery required", self.id.str()); - self.fdir.recovery_done(); - return; - } - self.start_recovery(self.mode()); - } - - fn start_recovery(&mut self, restore_mode: DeviceMode) { - log::warn!("{}: starting power cycle recovery", self.id.str()); - self.shared_mgm_set.lock().unwrap().valid = false; - self.switch_and_mode_helper - .start_power_cycle(restore_mode, self.recovery_off_duration); - self.send_event(mgm::Event::Recovery(RecoveryEvent::Started)); - } - - fn handle_recovery_done(&mut self) { - log::info!("{}: power cycle recovery done", self.id.str()); - // Faults registered while the device was switched off do not count anymore. - self.spi_fault_counter.clear(); - self.fdir.recovery_done(); - self.send_event(mgm::Event::Recovery(RecoveryEvent::Done)); - } - - /// A failed power cycle costs a recovery attempt like any other fault. - fn handle_recovery_failure(&mut self, restore_mode: DeviceMode) { - self.send_event(mgm::Event::Recovery(RecoveryEvent::Failed)); - match self.fdir.recovery_failed() { - FaultResponse::Recover => { - log::warn!("{}: power cycle recovery failed, retrying", self.id.str()); - self.start_recovery(restore_mode); - } - FaultResponse::SetFaulty => { - log::error!( - "{}: power cycle recovery failed too often, marking component faulty", - self.id.str() - ); - self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); - self.switch_off_faulty_device(); - } - // Ground changed the health during the recovery and is in charge now. - FaultResponse::Ignored => (), - } - } - - /// Mode commands from ground or the parent abort a running recovery. fn handle_mode_command( &mut self, target_mode: DeviceMode, tc_commander: Option, ) { - if self.switch_and_mode_helper.power_cycle_active() { - log::warn!( - "{}: mode command aborts power cycle recovery", - self.id.str() - ); - // Otherwise, the recovery would restart right away. - self.fdir.recovery_done(); - } + self.fdir.handle_mode_command(&self.switch_and_mode_helper); self.start_transition(target_mode, tc_commander); } @@ -662,6 +575,7 @@ mod tests { pcdu::{SwitchRequest, SwitchState, SwitchStateBinary}, }; + use crate::device_fdir::RECOVERY_THRESHOLD; use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet}; use super::*; @@ -734,7 +648,7 @@ mod tests { health_table.clone(), event_tx, ); - handler.recovery_off_duration = Duration::ZERO; + handler.fdir.recovery_off_duration = Duration::ZERO; Self { assembly_mode_request_tx, mode_report_rx, @@ -1132,7 +1046,7 @@ mod tests { mgm::Event::Recovery(RecoveryEvent::Done), ] )); - assert_eq!(testbench.handler.spi_fault_counter.fault_count(), 0); + assert_eq!(testbench.handler.fdir.fault_count(), 0); assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid); } @@ -1343,7 +1257,7 @@ mod tests { testbench.handler.periodic_operation(); testbench.set_switch_state(SwitchState::Off); // Keep the device off until the parent asked for its mode. - testbench.handler.recovery_off_duration = Duration::from_secs(60); + testbench.handler.fdir.recovery_off_duration = Duration::from_secs(60); testbench.handler.periodic_operation(); assert_eq!(testbench.handler.mode(), DeviceMode::Off); testbench.mode_report_rx.try_iter().for_each(drop); diff --git a/satrs-example/src/device_fdir.rs b/satrs-example/src/device_fdir.rs new file mode 100644 index 0000000..25785e6 --- /dev/null +++ b/satrs-example/src/device_fdir.rs @@ -0,0 +1,200 @@ +use std::collections::VecDeque; +use std::time::Duration; + +use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir}; +use satrs::health::{HealthState, HealthTableMapSync}; +use types::{ComponentId, DeviceMode}; + +use crate::device_mode::SwitchAndModeHelper; + +// The component is marked faulty if it would be recovered more than RECOVERY_THRESHOLD times +// before the counter is decremented again. +pub const RECOVERY_THRESHOLD: u32 = 2; +pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60); +/// Time the device stays unpowered during a power cycle, so it can fully discharge. +pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500); + +/// Generic FDIR events. The device handler maps them to its own event type. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub enum FdirEvent { + FaultThresholdExceeded, + Recovery(RecoveryEvent), +} + +/// Fault counting and power cycle recovery for device handlers which own the power switch of +/// their device. +/// +/// The handler detects faults itself and reports them with [Self::register_fault]. This helper +/// then decides whether the device is power cycled, or marked faulty and switched off, and drives +/// the [SwitchAndModeHelper] of the handler accordingly. The handler retrieves the resulting +/// events with [Self::next_event]. +pub struct DeviceFdir { + name: &'static str, + fault_counter: FaultCounterStd, + recovery: RecoveryFdir, + pub recovery_off_duration: Duration, + events: VecDeque, +} + +impl DeviceFdir { + pub fn new( + name: &'static str, + component_id: ComponentId, + health_table: HealthTableMapSync, + fault_counter: FaultCounterStd, + ) -> Self { + Self { + name, + fault_counter, + recovery: RecoveryFdir::new( + component_id.into(), + health_table, + RECOVERY_THRESHOLD, + RECOVERY_DECREMENT_AFTER, + ), + recovery_off_duration: RECOVERY_OFF_DURATION, + events: VecDeque::new(), + } + } + + #[cfg(test)] + pub fn fault_count(&self) -> u32 { + self.fault_counter.fault_count() + } + + pub fn set_health(&mut self, health: HealthState) { + self.recovery.set_health(health); + } + + pub fn next_event(&mut self) -> Option { + self.events.pop_front() + } + + /// Should be called once per cycle, before the mode transition is handled. Starts a power + /// cycle if the health was set to [HealthState::NeedsRecovery] by the FDIR or by ground. + pub fn periodic_operation(&mut self, modes: &mut SwitchAndModeHelper) { + self.recovery.periodic_operation(); + self.check_needs_recovery(modes); + } + + pub fn register_success(&mut self) { + self.fault_counter.try_decrement(); + } + + /// If the fault threshold is exceeded, the device is power cycled. If it was power cycled + /// too often, the component is marked faulty and commanded off instead. + pub fn register_fault(&mut self, modes: &mut SwitchAndModeHelper) { + if !self.fault_counter.increment_and_check() { + return; + } + match self.recovery.handle_fault() { + FaultResponse::Ignored => { + log::info!( + "{}: fault threshold exceeded, but component is already faulty, \ + recovering or externally controlled", + self.name + ); + } + FaultResponse::Recover => { + log::warn!( + "{}: fault threshold exceeded, power cycling device", + self.name + ); + self.events.push_back(FdirEvent::FaultThresholdExceeded); + self.check_needs_recovery(modes); + } + FaultResponse::SetFaulty => { + log::error!( + "{}: fault threshold exceeded after too many recoveries, marking \ + component faulty", + self.name + ); + self.events.push_back(FdirEvent::FaultThresholdExceeded); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(modes); + } + } + } + + /// Mode commands from ground or the parent abort a running recovery. Must be called before + /// the commanded transition is started. + pub fn handle_mode_command(&mut self, modes: &SwitchAndModeHelper) { + if modes.power_cycle_active() { + log::warn!("{}: mode command aborts power cycle recovery", self.name); + // Otherwise, the recovery would restart right away. + self.recovery.recovery_done(); + } + } + + pub fn handle_power_cycle_done(&mut self) { + log::info!("{}: power cycle recovery done", self.name); + // Faults registered while the device was switched off do not count anymore. + self.fault_counter.clear(); + self.recovery.recovery_done(); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Done)); + } + + /// A failed power cycle costs a recovery attempt like any other fault. + pub fn handle_power_cycle_failed( + &mut self, + modes: &mut SwitchAndModeHelper, + restore_mode: DeviceMode, + ) { + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Failed)); + match self.recovery.recovery_failed() { + FaultResponse::Recover => { + log::warn!("{}: power cycle recovery failed, retrying", self.name); + self.start_recovery(modes, restore_mode); + } + FaultResponse::SetFaulty => { + log::error!( + "{}: power cycle recovery failed too often, marking component faulty", + self.name + ); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(modes); + } + // Ground changed the health during the recovery and is in charge now. + FaultResponse::Ignored => (), + } + } + + fn check_needs_recovery(&mut self, modes: &mut SwitchAndModeHelper) { + if modes.power_cycle_active() || modes.target().is_some() || !self.recovery.needs_recovery() + { + return; + } + if modes.mode() == DeviceMode::Off { + // Nothing to power cycle, the next switch-on is a fresh start anyway. + log::info!("{}: device is off, no recovery required", self.name); + self.recovery.recovery_done(); + return; + } + let restore_mode = modes.mode(); + self.start_recovery(modes, restore_mode); + } + + fn start_recovery( + &mut self, + modes: &mut SwitchAndModeHelper, + restore_mode: DeviceMode, + ) { + log::warn!("{}: starting power cycle recovery", self.name); + modes.start_power_cycle(restore_mode, self.recovery_off_duration); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Started)); + } + + fn switch_off_faulty_device(&mut self, modes: &mut SwitchAndModeHelper) { + // Do not restart an already pending Off transition, which would reset the transition + // state machine before it can finish. + if modes.target() != Some(DeviceMode::Off) { + log::warn!("{}: commanding device off due to fault", self.name); + modes.start_transition(DeviceMode::Off, None); + } + } +} diff --git a/satrs-example/src/main.rs b/satrs-example/src/main.rs index b00ba09..3b8212e 100644 --- a/satrs-example/src/main.rs +++ b/satrs-example/src/main.rs @@ -48,6 +48,7 @@ use crate::{ mod acs; mod ccsds; mod controller; +mod device_fdir; mod device_mode; mod eps; mod event_manager;