From 7b74bee581b0f767e6eed787429299c0ba103ea9 Mon Sep 17 00:00:00 2001 From: Robin Mueller Date: Wed, 23 Sep 2026 16:20:26 +0200 Subject: [PATCH] FDIR extensions and improvements for MGM device handler - try power cycling now instead of going to faulty immediately - after too many power cycles in a short time frame, go to faulty - new FDIR/recovery helper which is generic - new failure variants for fault injection: transient failures to test that a power cycles could fix the issue --- satrs-example/client/src/main.rs | 83 +++- satrs-example/minisim/src/acs.rs | 82 +++- satrs-example/minisim/src/lib.rs | 10 +- satrs-example/src/acs/mgm.rs | 630 +++++++++++++++++++++++------ satrs-example/src/device_mode.rs | 404 +++++++++++++++++- satrs-example/types/src/acs/mgm.rs | 9 +- satrs-example/types/src/lib.rs | 8 + satrs/CHANGELOG.md | 3 + satrs/src/fdir.rs | 219 ++++++++++ 9 files changed, 1286 insertions(+), 162 deletions(-) diff --git a/satrs-example/client/src/main.rs b/satrs-example/client/src/main.rs index f9c7bec..9cd9bc8 100644 --- a/satrs-example/client/src/main.rs +++ b/satrs-example/client/src/main.rs @@ -4,7 +4,7 @@ use clap::Parser as _; use satrs_example::config::{OBSW_SERVER_ADDR, SERVER_PORT}; use satrs_minisim::{ SerializableSimMsgPayload, SimComponent, SimCtrlReply, SimCtrlRequest, SimMessageProvider, - SimReply, SimRequest, acs::MgmRequestLis3Mdl, acs::SpiFaultMode, udp::SIM_CTRL_PORT, + SimReply, SimRequest, acs, acs::MgmRequestLis3Mdl, acs::SpiFault, udp::SIM_CTRL_PORT, }; use spacepackets::{CcsdsPacketIdAndPsc, SpacePacketHeader}; use std::{ @@ -89,22 +89,33 @@ impl From for types::ComponentId { } #[derive(Debug, PartialEq, Eq, Clone, Copy, clap::ValueEnum)] -enum SpiFaultModeSelect { +enum FaultMode { None, + /// SPI communication is all zeroes, modelling an unconnected sensor. AllZeros, + /// SPI communication is all ones, modelling a broken sensor. AllOnes, } -impl From for SpiFaultMode { - fn from(mode: SpiFaultModeSelect) -> Self { +impl From for acs::SpiFaultMode { + fn from(mode: FaultMode) -> Self { match mode { - SpiFaultModeSelect::None => SpiFaultMode::None, - SpiFaultModeSelect::AllZeros => SpiFaultMode::AllZeros, - SpiFaultModeSelect::AllOnes => SpiFaultMode::AllOnes, + FaultMode::None => acs::SpiFaultMode::None, + FaultMode::AllZeros => acs::SpiFaultMode::AllZeros, + FaultMode::AllOnes => acs::SpiFaultMode::AllOnes, } } } +#[derive(Debug, Default, PartialEq, Eq, Clone, Copy, clap::ValueEnum)] +enum FaultKind { + /// Cleared when the device is switched off, so a power cycle recovers from it. + Transient, + /// Survives power cycles. + #[default] + Permanent, +} + #[derive(Debug, PartialEq, Eq, Clone, Copy, clap::ValueEnum)] enum HealthStateSelect { Healthy, @@ -139,7 +150,10 @@ struct MgmArgs { /// Only takes effect for MGM0: minisim always routes this fault to the MGM0 model /// regardless of which MGM the request names (a pre-existing minisim limitation). #[arg(long, value_enum)] - spi_fault: Option, + fault: Option, + /// Whether a power cycle clears the injected SPI fault. + #[arg(long, value_enum, default_value_t)] + fault_kind: FaultKind, /// Override the device's FDIR health state, for example to clear a `Faulty` state set by /// the handler after the underlying issue has been fixed or worked around. #[arg(long, value_enum)] @@ -187,11 +201,14 @@ fn handle_mgm_command( target_id: types::ComponentId, args: MgmArgs, ) -> anyhow::Result<()> { - if let Some(mode) = args.spi_fault { + if let Some(mode) = args.fault { if target_id != types::ComponentId::AcsMgm0 { bail!("SPI fault injection is only supported for MGM0 right now (minisim limitation)"); } - inject_mgm_failure(mode.into())?; + inject_mgm_failure(SpiFault { + mode: mode.into(), + cleared_by_power_cycle: args.fault_kind == FaultKind::Transient, + })?; } if args.ping { let request = types::ccsds::CcsdsTcPacketOwned::new_with_request( @@ -477,12 +494,12 @@ fn main() -> anyhow::Result<()> { Ok(()) } -/// Injects the given SPI fault mode directly into minisim's MGM0 model, bypassing the OBSW. +/// Injects the given SPI fault directly into minisim's MGM0 model, bypassing the OBSW. /// /// Confirms the simulator is actually reachable first (same ping/pong check the OBSW's own /// internal sim client does, see `SimClientUdp::attempt_connection`), since a fire-and-forget /// UDP send would otherwise silently do nothing if minisim is not running. -fn inject_mgm_failure(mode: SpiFaultMode) -> anyhow::Result<()> { +fn inject_mgm_failure(fault: SpiFault) -> anyhow::Result<()> { let sim_addr = SocketAddr::new(IpAddr::V4(Ipv4Addr::LOCALHOST), SIM_CTRL_PORT); let sim_socket = UdpSocket::bind("127.0.0.1:0")?; sim_socket.set_read_timeout(Some(Duration::from_millis(200)))?; @@ -514,12 +531,43 @@ fn inject_mgm_failure(mode: SpiFaultMode) -> anyhow::Result<()> { Err(e) => return Err(e.into()), } - let request = SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(mode)); + let request = SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(fault)); sim_socket.send_to(&serde_json::to_vec(&request)?, sim_addr)?; - log::info!("injected SPI fault mode {mode:?} into minisim MGM0"); + log::info!("injected SPI fault {fault:?} into minisim MGM0"); Ok(()) } +/// Each component has its own event type, so the sender ID determines how to decode the event. +fn handle_event(sender_id: types::ComponentId, data: &[u8]) { + fn log_event( + sender_id: types::ComponentId, + data: &[u8], + ) { + match postcard::from_bytes::(data) { + Ok(event) => log::info!("Received event from {:?}: {:?}", sender_id, event), + Err(e) => log::warn!("Failed to deserialize event from {:?}: {}", sender_id, e), + } + } + match sender_id { + types::ComponentId::Controller => log_event::(sender_id, data), + types::ComponentId::AcsMgm0 | types::ComponentId::AcsMgm1 => { + log_event::(sender_id, data) + } + types::ComponentId::AcsMgmAssembly => { + log_event::(sender_id, data) + } + types::ComponentId::EpsPcdu => log_event::(sender_id, data), + // TC source events are sent with the ID of the packet source. + types::ComponentId::UdpServer + | types::ComponentId::TcpServer + | types::ComponentId::Ground => log_event::(sender_id, data), + _ => log::warn!( + "Received event from {:?} with unknown event type", + sender_id + ), + } +} + fn handle_raw_tm_packet(data: &[u8]) -> anyhow::Result<()> { match spacepackets::CcsdsPacketReader::new_with_checksum(data) { Ok(packet) => { @@ -543,12 +591,7 @@ fn handle_raw_tm_packet(data: &[u8]) -> anyhow::Result<()> { ); } if tm_header.message_type == MessageType::Event { - let response = postcard::from_bytes::(remainder); - log::info!( - "Received event from {:?}: {:?}", - tm_header.sender_id, - response.unwrap() - ); + handle_event(tm_header.sender_id, remainder); return Ok(()); } match tm_header.sender_id { diff --git a/satrs-example/minisim/src/acs.rs b/satrs-example/minisim/src/acs.rs index 9fa671f..69e2620 100644 --- a/satrs-example/minisim/src/acs.rs +++ b/satrs-example/minisim/src/acs.rs @@ -7,7 +7,7 @@ use nexosim::{ use satrs_minisim::{ acs::{ lis3mdl::MgmLis3MdlReply, MgmReplyCommon, MgmReplyProvider, MgmSensorValuesMicroTesla, - MgtDipole, MgtHkSet, MgtReply, SpiFaultMode, MGT_GEN_MAGNETIC_FIELD, + MgtDipole, MgtHkSet, MgtReply, SpiFault, MGT_GEN_MAGNETIC_FIELD, }, SimReply, }; @@ -34,7 +34,7 @@ pub struct MagnetometerModel { #[allow(dead_code)] pub periodicity: Duration, pub external_mag_field: Option, - pub spi_fault: SpiFaultMode, + pub spi_fault: SpiFault, pub reply_sender: mpsc::Sender, pub phatom: std::marker::PhantomData, } @@ -45,7 +45,7 @@ impl MagnetometerModel { switch_state: SwitchStateBinary::Off, periodicity, external_mag_field: None, - spi_fault: SpiFaultMode::None, + spi_fault: SpiFault::default(), reply_sender, phatom: std::marker::PhantomData, } @@ -55,11 +55,14 @@ impl MagnetometerModel { impl MagnetometerModel { pub async fn switch_device(&mut self, switch_state: SwitchStateBinary) { self.switch_state = switch_state; + if switch_state == SwitchStateBinary::Off && self.spi_fault.cleared_by_power_cycle { + self.spi_fault = SpiFault::default(); + } } /// Force (or clear) a stuck-bus SPI fault, for FDIR testing purposes. - pub async fn set_spi_fault(&mut self, fault_mode: SpiFaultMode) { - self.spi_fault = fault_mode; + pub async fn set_spi_fault(&mut self, fault: SpiFault) { + self.spi_fault = fault; } pub async fn send_sensor_values(&mut self, _: (), scheduler: &mut Context) { @@ -70,7 +73,7 @@ impl MagnetometerModel { sensor_values: self .calculate_current_mgm_tuple(current_millis(scheduler.time())), }, - self.spi_fault, + self.spi_fault.mode, )) .expect("sending MGM sensor values failed"); } @@ -193,13 +196,16 @@ pub mod tests { use satrs_minisim::{ acs::{ lis3mdl::{self, MgmLis3MdlReply}, - MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest, SpiFaultMode, + MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest, SpiFault, SpiFaultMode, }, SerializableSimMsgPayload, SimComponent, SimMessageProvider, SimRequest, }; use types::pcdu::{SwitchId, SwitchStateBinary}; - use crate::{eps::tests::switch_device_on, test_helpers::SimTestbench}; + use crate::{ + eps::tests::{switch_device_off, switch_device_on}, + test_helpers::SimTestbench, + }; #[test] fn test_basic_mgm_request() { @@ -222,19 +228,20 @@ pub mod tests { assert_eq!(reply.common.sensor_values.z, 0.0); } - #[test] - fn test_mgm_spi_fault_injection_all_ones() { - let mut sim_testbench = SimTestbench::new(); - switch_device_on(&mut sim_testbench, SwitchId::Mgm0); - + fn inject_spi_fault(sim_testbench: &mut SimTestbench, cleared_by_power_cycle: bool) { let fault_request = - SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(SpiFaultMode::AllOnes)); + SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(SpiFault { + mode: SpiFaultMode::AllOnes, + cleared_by_power_cycle, + })); sim_testbench .send_request(fault_request) .expect("sending MGM fault injection request failed"); sim_testbench.handle_sim_requests_time_agnostic(); sim_testbench.step().unwrap(); + } + fn request_mgm_reply(sim_testbench: &mut SimTestbench) -> MgmLis3MdlReply { let data_request = SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::RequestSensorData); sim_testbench .send_request(data_request) @@ -244,13 +251,50 @@ pub mod tests { let sim_reply = sim_testbench .try_receive_next_reply() .expect("no MGM reply received"); - let reply = MgmLis3MdlReply::from_sim_message(&sim_reply) - .expect("failed to deserialize MGM sensor values"); + MgmLis3MdlReply::from_sim_message(&sim_reply) + .expect("failed to deserialize MGM sensor values") + } + + fn is_stuck_bus_reply(reply: &MgmLis3MdlReply) -> bool { + reply.raw.x == -1 && reply.raw.y == -1 && reply.raw.z == -1 + } + + #[test] + fn test_mgm_spi_fault_injection_all_ones() { + let mut sim_testbench = SimTestbench::new(); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + inject_spi_fault(&mut sim_testbench, false); + + let reply = request_mgm_reply(&mut sim_testbench); // Even though the device is switched on, the injected fault forces a stuck-bus reply. assert_eq!(reply.common.switch_state, SwitchStateBinary::On); - assert_eq!(reply.raw.x, -1); - assert_eq!(reply.raw.y, -1); - assert_eq!(reply.raw.z, -1); + assert!(is_stuck_bus_reply(&reply)); + } + + #[test] + fn test_mgm_spi_fault_cleared_by_power_cycle() { + let mut sim_testbench = SimTestbench::new(); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + inject_spi_fault(&mut sim_testbench, true); + assert!(is_stuck_bus_reply(&request_mgm_reply(&mut sim_testbench))); + + switch_device_off(&mut sim_testbench, SwitchId::Mgm0); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + sim_testbench.step_until(Duration::from_millis(50)).unwrap(); + assert!(!is_stuck_bus_reply(&request_mgm_reply(&mut sim_testbench))); + } + + #[test] + fn test_mgm_spi_fault_persists_after_power_cycle() { + let mut sim_testbench = SimTestbench::new(); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + inject_spi_fault(&mut sim_testbench, false); + + switch_device_off(&mut sim_testbench, SwitchId::Mgm0); + switch_device_on(&mut sim_testbench, SwitchId::Mgm0); + let reply = request_mgm_reply(&mut sim_testbench); + assert_eq!(reply.common.switch_state, SwitchStateBinary::On); + assert!(is_stuck_bus_reply(&reply)); } #[test] diff --git a/satrs-example/minisim/src/lib.rs b/satrs-example/minisim/src/lib.rs index f1587d8..33e21fb 100644 --- a/satrs-example/minisim/src/lib.rs +++ b/satrs-example/minisim/src/lib.rs @@ -217,12 +217,20 @@ pub mod acs { AllOnes, } + #[derive(Debug, Default, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] + pub struct SpiFault { + pub mode: SpiFaultMode, + /// The fault is cleared when the device is switched off, so a power cycle recovers + /// from it. + pub cleared_by_power_cycle: bool, + } + #[derive(Debug, Copy, Clone, Serialize, Deserialize)] pub enum MgmRequestLis3Mdl { RequestSensorData, /// Force the raw register reply into a stuck-bus pattern, regardless of switch state. /// Used to test FDIR handling of SPI bus faults. - SetSpiFault(SpiFaultMode), + SetSpiFault(SpiFault), } impl SerializableSimMsgPayload for MgmRequestLis3Mdl { diff --git a/satrs-example/src/acs/mgm.rs b/satrs-example/src/acs/mgm.rs index ae714f7..565bc0b 100644 --- a/satrs-example/src/acs/mgm.rs +++ b/satrs-example/src/acs/mgm.rs @@ -1,5 +1,5 @@ -use satrs::fdir::FaultCounterStd; -use satrs::health::{HealthState, HealthTableMapSync, HealthTableProvider}; +use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir}; +use satrs::health::HealthTableMapSync; use satrs::spacepackets::CcsdsPacketIdAndPsc; use satrs_example::{HkHelperSingleSet, TimestampHelper, TmtcQueues}; use satrs_minisim::acs::MgmRequestLis3Mdl; @@ -36,6 +36,13 @@ pub const Z_LOWBYTE_IDX: usize = 13; pub const SPI_FAULT_THRESHOLD: u32 = 2; pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30); +// FDIR configuration for power cycle recoveries. The component is marked faulty if it would be +// recovered more than RECOVERY_THRESHOLD times before the counter is decremented again. +pub const RECOVERY_THRESHOLD: u32 = 2; +pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60); +/// Time the device stays unpowered during a power cycle, so it can fully discharge. +pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500); + #[derive(Debug, PartialEq, Eq, Clone, Copy)] pub enum MgmId { _0, @@ -158,6 +165,14 @@ pub struct ModeLeafHelper { } /// Example MGM device handler strongly based on the LIS3MDL MEMS device. +/// +/// This device handler includes several components beyond the scope of reading sensor values: +/// +/// - FDIR handling on communication issues. +/// - FDIR helper for power cycling the device on communication issues. +/// - Event generation for certain events like communication issues. +/// - HK helper for periodic data generation. +/// - Mode leaf helper to allow integration into a full ACS mode tree pub struct MgmHandlerLis3Mdl { id: MgmId, tmtc_queues: TmtcQueues, @@ -169,7 +184,8 @@ pub struct MgmHandlerLis3Mdl { switch_and_mode_helper: SwitchAndModeHelper, mode_leaf_helper: ModeLeafHelper, spi_fault_counter: FaultCounterStd, - health_table: HealthTableMapSync, + fdir: RecoveryFdir, + recovery_off_duration: Duration, event_tx: mpsc::SyncSender<(ComponentId, mgm::Event)>, } @@ -202,7 +218,13 @@ impl MgmHandlerLis3Mdl { hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)), mode_leaf_helper, spi_fault_counter: FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER), - health_table, + fdir: RecoveryFdir::new( + id.component_id().into(), + health_table, + RECOVERY_THRESHOLD, + RECOVERY_DECREMENT_AFTER, + ), + recovery_off_duration: RECOVERY_OFF_DURATION, event_tx, } } @@ -212,6 +234,7 @@ impl MgmHandlerLis3Mdl { self.switch_and_mode_helper.mode() } + /// Core function called periodically to drive the handler. pub fn periodic_operation(&mut self) { // Update current time. self.stamp_helper.update_from_now(); @@ -222,7 +245,10 @@ impl MgmHandlerLis3Mdl { // Handle assembly related messages. self.handle_mode_leaf_handling(); - // Handle mode transitions first. + self.fdir.periodic_operation(); + self.check_needs_recovery(); + + // Handle mode transitions first. This also takes care of recoveries required by FDIR. if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() { match event { ModeTransitionEvent::Reached(tc_commander) => { @@ -231,11 +257,18 @@ impl MgmHandlerLis3Mdl { ModeTransitionEvent::Failed(tc_commander) => { self.handle_mode_transition_failure(tc_commander) } + // The mode did not change for other components, so there is nothing to report. + ModeTransitionEvent::PowerCycleDone => self.handle_recovery_done(), + ModeTransitionEvent::PowerCycleFailed { restore_mode } => { + self.handle_recovery_failure(restore_mode) + } } } - // Poll sensor before checking and generating HK. - if self.mode() == DeviceMode::Normal { + // Poll sensor before checking and generating HK. The device is not polled during mode + // transitions, which includes all FDIR actions like power cycling or switching off a + // faulty device. Faults are expected then, and polling would only add noise. + if self.mode() == DeviceMode::Normal && self.switch_and_mode_helper.target().is_none() { log::trace!("polling LIS3MDL sensor {}", self.id.str()); self.poll_sensor(); } @@ -267,12 +300,12 @@ impl MgmHandlerLis3Mdl { } mgm::request::Request::Mode(device_mode) => match device_mode { ModeRequest::SetMode(device_mode) => { - self.start_transition(device_mode, Some(tc_id)); + self.handle_mode_command(device_mode, Some(tc_id)); } ModeRequest::ReadMode => self.send_telemetry( Some(tc_id), mgm::response::Response::Mode(ModeResponse::Mode( - self.mode(), + self.switch_and_mode_helper.reported_mode(), )), ), }, @@ -284,10 +317,7 @@ impl MgmHandlerLis3Mdl { self.id.str(), health_state ); - self.health_table.set_health( - self.id.component_id().into(), - health_state, - ); + self.fdir.set_health(health_state); self.send_telemetry( Some(tc_id), mgm::response::Response::Ok, @@ -316,7 +346,9 @@ impl MgmHandlerLis3Mdl { loop { match self.mode_leaf_helper.request_rx.try_recv() { Ok(request) => match request { - ModeRequest::SetMode(device_mode) => self.start_transition(device_mode, None), + ModeRequest::SetMode(device_mode) => { + self.handle_mode_command(device_mode, None) + } ModeRequest::ReadMode => self.report_mode_to_parent(), }, Err(e) => match e { @@ -400,14 +432,17 @@ impl MgmHandlerLis3Mdl { .unwrap(), ); // A stuck-high SPI bus (undriven MISO) reads back as all-1s on every register, - // regardless of what was actually requested. This is the pattern this codebase already - // uses for "no real device behind the bus" (see the switched-off MGM sim reply). An - // all-0s reading is not used here, since it collides with a legitimate zero-field - // reading and would cause false positives. + // regardless of what was actually requested. + // If our sensor was broken, this is what we would probably see. + // An all zeroes reading is ignored for now. the sensor could theoretically return this. + // In a production app, we also need to check whether the sensor data never varies, which is + // also a fault. We ignore this in this example because the handler is already complex + // enough. if x_raw == -1 && y_raw == -1 && z_raw == -1 { self.register_spi_fault(); return; } + // Successfull readout, so we can decrement the counter. self.spi_fault_counter.try_decrement(); // Simple scaling to retrieve the float value, assuming the best sensor resolution. let mut mgm_guard = self.shared_mgm_set.lock().unwrap(); @@ -418,51 +453,131 @@ impl MgmHandlerLis3Mdl { drop(mgm_guard); } - /// Registers one stuck-bus SPI fault with the FDIR fault counter, invalidating the current - /// sensor set. If the failure threshold is exceeded, the component is marked faulty in the - /// global health table. + /// Registers one SPI fault with the FDIR fault counter, invalidating the current + /// sensor set. If the failure threshold is exceeded, the device is power cycled. If it was + /// power cycled too often, the component is marked faulty and commanded off instead. fn register_spi_fault(&mut self) { log::warn!("{}: stuck-bus SPI fault", self.id.str()); self.shared_mgm_set.lock().unwrap().valid = false; if !self.spi_fault_counter.increment_and_check() { return; } - // Ground may have taken manual control, or already given up on this component. - // Autonomous FDIR should not override that decision. - let component_id = self.id.component_id().into(); - match self.health_table.health(component_id) { - Some(HealthState::ExternalControl) | Some(HealthState::PermanentFaulty) => { + match self.fdir.handle_fault() { + FaultResponse::Ignored => { log::info!( - "{}: SPI fault threshold exceeded, but health is externally controlled, \ - not overriding", + "{}: SPI fault threshold exceeded, but component is already faulty, \ + recovering or externally controlled", self.id.str() ); } - _ => { + FaultResponse::Recover => { + log::warn!( + "{}: SPI fault threshold exceeded, power cycling device", + self.id.str() + ); + self.send_event(mgm::Event::SpiFaultThresholdExceeded); + self.check_needs_recovery(); + } + FaultResponse::SetFaulty => { log::error!( - "{}: SPI fault threshold exceeded, marking component faulty", + "{}: SPI fault threshold exceeded after too many recoveries, marking \ + component faulty", self.id.str() ); - self.health_table - .set_health(component_id, HealthState::Faulty); - if let Err(e) = self.event_tx.send(( - self.id.component_id(), - mgm::Event::SpiFaultThresholdExceeded, - )) { - log::warn!("{}: failed to send fault event: {}", self.id.str(), e); - } - // Do not restart an already pending Off transition: poll_sensor still calls - // this every cycle the fault persists, and current stays Normal until the - // transition completes, so re-triggering here would keep resetting the - // transition state machine before it can ever finish. - if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) { - log::warn!("{}: commanding device off due to fault", self.id.str()); - self.start_transition(DeviceMode::Off, None); - } + self.send_event(mgm::Event::SpiFaultThresholdExceeded); + self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(); } } } + fn switch_off_faulty_device(&mut self) { + // Do not restart an already pending Off transition, which would reset the transition + // state machine before it can finish. + if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) { + log::warn!("{}: commanding device off due to fault", self.id.str()); + self.start_transition(DeviceMode::Off, None); + } + } + + /// Starts a power cycle if the health is [satrs::health::HealthState::NeedsRecovery]. The health is set + /// either by the FDIR or by ground. + fn check_needs_recovery(&mut self) { + if self.switch_and_mode_helper.power_cycle_active() + || self.switch_and_mode_helper.target().is_some() + || !self.fdir.needs_recovery() + { + return; + } + if self.mode() == DeviceMode::Off { + // Nothing to power cycle, the next switch-on is a fresh start anyway. + log::info!("{}: device is off, no recovery required", self.id.str()); + self.fdir.recovery_done(); + return; + } + self.start_recovery(self.mode()); + } + + fn start_recovery(&mut self, restore_mode: DeviceMode) { + log::warn!("{}: starting power cycle recovery", self.id.str()); + self.shared_mgm_set.lock().unwrap().valid = false; + self.switch_and_mode_helper + .start_power_cycle(restore_mode, self.recovery_off_duration); + self.send_event(mgm::Event::Recovery(RecoveryEvent::Started)); + } + + fn handle_recovery_done(&mut self) { + log::info!("{}: power cycle recovery done", self.id.str()); + // Faults registered while the device was switched off do not count anymore. + self.spi_fault_counter.clear(); + self.fdir.recovery_done(); + self.send_event(mgm::Event::Recovery(RecoveryEvent::Done)); + } + + /// A failed power cycle costs a recovery attempt like any other fault. + fn handle_recovery_failure(&mut self, restore_mode: DeviceMode) { + self.send_event(mgm::Event::Recovery(RecoveryEvent::Failed)); + match self.fdir.recovery_failed() { + FaultResponse::Recover => { + log::warn!("{}: power cycle recovery failed, retrying", self.id.str()); + self.start_recovery(restore_mode); + } + FaultResponse::SetFaulty => { + log::error!( + "{}: power cycle recovery failed too often, marking component faulty", + self.id.str() + ); + self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(); + } + // Ground changed the health during the recovery and is in charge now. + FaultResponse::Ignored => (), + } + } + + /// Mode commands from ground or the parent abort a running recovery. + fn handle_mode_command( + &mut self, + target_mode: DeviceMode, + tc_commander: Option, + ) { + if self.switch_and_mode_helper.power_cycle_active() { + log::warn!( + "{}: mode command aborts power cycle recovery", + self.id.str() + ); + // Otherwise, the recovery would restart right away. + self.fdir.recovery_done(); + } + self.start_transition(target_mode, tc_commander); + } + + fn send_event(&self, event: mgm::Event) { + if let Err(e) = self.event_tx.send((self.id.component_id(), event)) { + log::warn!("{}: failed to send event {:?}: {}", self.id.str(), event, e); + } + } + fn start_transition( &mut self, target_mode: DeviceMode, @@ -502,22 +617,15 @@ impl MgmHandlerLis3Mdl { fn announce_mode(&self) { log::info!("{} announcing mode: {:?}", self.id.str(), self.mode()); - if let Err(e) = self - .event_tx - .send((self.id.component_id(), mgm::Event::ModeChanged(self.mode()))) - { - log::warn!( - "{}: failed to send mode changed event: {}", - self.id.str(), - e - ); - } + self.send_event(mgm::Event::ModeChanged(self.mode())); } fn report_mode_to_parent(&self) { self.mode_leaf_helper .report_tx - .send(ModeResponse::Mode(self.mode())) + .send(ModeResponse::Mode( + self.switch_and_mode_helper.reported_mode(), + )) .unwrap(); } @@ -534,6 +642,7 @@ mod tests { }; use arbitrary_int::u11; + use satrs::health::{HealthState, HealthTableProvider}; use satrs::spacepackets::SpacePacketHeader; use satrs_minisim::acs::lis3mdl::MgmLis3RawValues; use types::{ @@ -589,7 +698,7 @@ mod tests { impl MgmTestbench { pub fn new() -> Self { let (assembly_mode_request_tx, assembly_mode_request_rx) = mpsc::sync_channel(5); - let (mode_report_tx, mode_report_rx) = mpsc::sync_channel(5); + let (mode_report_tx, mode_report_rx) = mpsc::sync_channel(10); let mode_leaf_helper = ModeLeafHelper { request_rx: assembly_mode_request_rx, report_tx: mode_report_tx, @@ -603,8 +712,8 @@ mod tests { let switch_map = SwitchSet::new(switch_map); let shared_switch_set = SharedSwitchSet::new(Mutex::new(switch_map)); let health_table = HealthTableMapSync::default(); - let (event_tx, event_rx) = mpsc::sync_channel(5); - let handler = MgmHandlerLis3Mdl::new( + let (event_tx, event_rx) = mpsc::sync_channel(20); + let mut handler = MgmHandlerLis3Mdl::new( MgmId::_0, TmtcQueues { tc_rx, tm_tx }, PowerSwitchHelper::new(switcher_tx, shared_switch_set.clone()), @@ -615,6 +724,7 @@ mod tests { health_table.clone(), event_tx, ); + handler.recovery_off_duration = Duration::ZERO; Self { assembly_mode_request_tx, mode_report_rx, @@ -645,6 +755,75 @@ mod tests { assert_eq!(self.handler.mode(), DeviceMode::Normal); } + pub fn set_switch_state(&self, state: SwitchState) { + self.shared_switch_set + .lock() + .unwrap() + .set_switch_state(SwitchId::Mgm0, state); + } + + pub fn inject_stuck_bus(&mut self) { + self.test_spi_interface().next_mgm_data = MgmLis3RawValues { + x: -1, + y: -1, + z: -1, + }; + } + + /// Drives SPI faults until the SPI fault threshold is exceeded once. + pub fn exceed_spi_fault_threshold(&mut self) { + self.inject_stuck_bus(); + for _ in 0..SPI_FAULT_THRESHOLD + 1 { + self.handler.periodic_operation(); + } + } + + /// Drives a started power cycle recovery to completion, completing both power-switch + /// handshakes. + pub fn complete_power_cycle(&mut self) { + self.handler.periodic_operation(); + self.set_switch_state(SwitchState::Off); + self.handler.periodic_operation(); + assert_eq!(self.handler.mode(), DeviceMode::Off); + self.handler.periodic_operation(); + assert_eq!( + self.handler.switch_and_mode_helper.target(), + Some(DeviceMode::Normal) + ); + self.set_switch_state(SwitchState::On); + self.handler.periodic_operation(); + assert_eq!(self.handler.mode(), DeviceMode::Normal); + } + + /// Drives recoveries with a permanently stuck bus until the component is marked faulty. + pub fn recover_until_faulty(&mut self) { + self.exceed_spi_fault_threshold(); + for _ in 0..RECOVERY_THRESHOLD { + assert_eq!(self.health(), Some(HealthState::NeedsRecovery)); + self.complete_power_cycle(); + // The last cycle of the power cycle already polled once. + for _ in 0..SPI_FAULT_THRESHOLD { + self.handler.periodic_operation(); + } + } + assert_eq!(self.health(), Some(HealthState::Faulty)); + } + + pub fn health(&self) -> Option { + self.health_table.health(ComponentId::AcsMgm0.into()) + } + + pub fn drain_events(&self) -> Vec { + self.event_rx.try_iter().map(|(_, event)| event).collect() + } + + pub fn drain_switch_requests(&self) -> Vec { + self.switch_rx + .try_iter() + .map(|req| req.target_state) + .collect() + } + pub fn test_spi_interface(&mut self) -> &mut TestSpiInterface { match &mut self.handler.spi_com { SpiCommunication::Dummy(_) | SpiCommunication::Sim(_) => { @@ -894,73 +1073,298 @@ mod tests { } #[test] - fn test_spi_fault_above_threshold_marks_component_faulty() { + fn test_spi_fault_above_threshold_starts_recovery() { let mut testbench = MgmTestbench::new(); testbench.switch_to_normal(); - // Drain the mode changed event emitted by switch_to_normal(). - testbench - .event_rx - .try_recv() - .expect("expected mode changed event"); - testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { - x: -1, - y: -1, - z: -1, - }; - // SPI_FAULT_THRESHOLD is exceeded on the (threshold + 1)-th stuck-bus reading. - for _ in 0..SPI_FAULT_THRESHOLD + 1 { - testbench.handler.periodic_operation(); - } - assert_eq!( - testbench.health_table.health(ComponentId::AcsMgm0.into()), - Some(HealthState::Faulty) - ); + testbench.drain_events(); + testbench.exceed_spi_fault_threshold(); + assert_eq!(testbench.health(), Some(HealthState::NeedsRecovery)); assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid); - let (sender_id, event) = testbench.event_rx.try_recv().expect("expected fault event"); - assert_eq!(sender_id, ComponentId::AcsMgm0); - assert!(matches!(event, mgm::Event::SpiFaultThresholdExceeded)); + let events = testbench.drain_events(); + assert!(matches!( + events[..], + [ + mgm::Event::SpiFaultThresholdExceeded, + mgm::Event::Recovery(RecoveryEvent::Started) + ] + )); } #[test] - fn test_spi_fault_above_threshold_commands_device_off() { + fn test_recovery_power_cycles_device() { let mut testbench = MgmTestbench::new(); testbench.switch_to_normal(); - // Drain the switch-on request left over from switch_to_normal(). - testbench - .switch_rx - .try_recv() - .expect("no switch-on request sent"); - testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues { - x: -1, - y: -1, - z: -1, - }; + testbench.drain_events(); + testbench.drain_switch_requests(); + testbench.mode_report_rx.try_iter().for_each(drop); + testbench.exceed_spi_fault_threshold(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues::default(); + let call_count = testbench.test_spi_interface().call_count; + + testbench.complete_power_cycle(); + + // The device is only polled again once the power cycle is done. + assert_eq!(testbench.test_spi_interface().call_count, call_count + 1); + // The power cycle is hidden from the parent. + assert!(testbench.mode_report_rx.try_recv().is_err()); + + assert_eq!(testbench.health(), Some(HealthState::Healthy)); + assert_eq!( + testbench.drain_switch_requests(), + [SwitchStateBinary::Off, SwitchStateBinary::On] + ); + let events = testbench.drain_events(); + assert!(matches!( + events[..], + [ + mgm::Event::SpiFaultThresholdExceeded, + mgm::Event::Recovery(RecoveryEvent::Started), + mgm::Event::Recovery(RecoveryEvent::Done), + ] + )); + assert_eq!(testbench.handler.spi_fault_counter.fault_count(), 0); + assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid); + } + + #[test] + fn test_repeated_recovery_marks_component_faulty() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.recover_until_faulty(); + let events = testbench.drain_events(); + assert!(matches!( + events[..], + [ + .., + mgm::Event::SpiFaultThresholdExceeded, + mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded) + ] + )); + testbench.drain_switch_requests(); + + testbench.handler.periodic_operation(); + assert_eq!(testbench.drain_switch_requests(), [SwitchStateBinary::Off]); + testbench.set_switch_state(SwitchState::Off); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + assert_eq!(testbench.health(), Some(HealthState::Faulty)); + } + + #[test] + fn test_faulty_device_is_not_polled_while_switching_off() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.recover_until_faulty(); + let call_count = testbench.test_spi_interface().call_count; + + // The switch-off takes a while. for _ in 0..SPI_FAULT_THRESHOLD + 1 { testbench.handler.periodic_operation(); } - assert_eq!( - testbench.health_table.health(ComponentId::AcsMgm0.into()), - Some(HealthState::Faulty) - ); - // The Off transition was only started on the last iteration above, so it has not sent - // its switch-off request yet: drive one more cycle to let it do so. + assert_eq!(testbench.test_spi_interface().call_count, call_count); + assert_eq!(testbench.health(), Some(HealthState::Faulty)); + testbench.set_switch_state(SwitchState::Off); testbench.handler.periodic_operation(); - let switch_req = testbench - .switch_rx - .try_recv() - .expect("no switch-off request sent after fault"); - assert_eq!(switch_req.switch_id, SwitchId::Mgm0); - assert_eq!(switch_req.target_state, SwitchStateBinary::Off); - - // Simulate the PCDU acting on the switch-off request. - testbench - .shared_switch_set - .lock() - .unwrap() - .set_switch_state(SwitchId::Mgm0, SwitchState::Off); testbench.handler.periodic_operation(); - assert_eq!(testbench.handler.mode(), DeviceMode::Off); + assert_eq!(testbench.health(), Some(HealthState::Faulty)); + } + + #[test] + fn test_ground_needs_recovery_power_cycles_device() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.drain_events(); + testbench + .tc_tx + .send(create_request_tc( + MgmSelect::_0, + mgm::request::Request::Health(mgm::request::HealthRequest::SetHealth( + HealthState::NeedsRecovery, + )), + )) + .unwrap(); + testbench.handler.periodic_operation(); + assert_eq!( + testbench.handler.switch_and_mode_helper.target(), + Some(DeviceMode::Off) + ); + testbench.set_switch_state(SwitchState::Off); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + testbench.handler.periodic_operation(); + testbench.set_switch_state(SwitchState::On); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Normal); + assert_eq!(testbench.health(), Some(HealthState::Healthy)); + let events = testbench.drain_events(); + assert!(matches!( + events[0], + mgm::Event::Recovery(RecoveryEvent::Started) + )); + assert!(matches!( + events.last(), + Some(mgm::Event::Recovery(RecoveryEvent::Done)) + )); + } + + #[test] + fn test_needs_recovery_while_off_sets_healthy() { + let mut testbench = MgmTestbench::new(); + testbench + .health_table + .set_health(ComponentId::AcsMgm0.into(), HealthState::NeedsRecovery); + testbench.handler.periodic_operation(); + assert_eq!(testbench.health(), Some(HealthState::Healthy)); + assert!(testbench.drain_switch_requests().is_empty()); + } + + #[test] + fn test_power_cycle_switch_on_failures_mark_component_faulty() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.drain_events(); + testbench.mode_report_rx.try_iter().for_each(drop); + testbench.exceed_spi_fault_threshold(); + + // The switch never turns on again. Every failed power cycle costs a recovery attempt. + for _ in 0..RECOVERY_THRESHOLD { + assert_eq!(testbench.health(), Some(HealthState::NeedsRecovery)); + testbench.handler.periodic_operation(); + testbench.set_switch_state(SwitchState::Off); + testbench.handler.periodic_operation(); + testbench.handler.periodic_operation(); + std::thread::sleep(Duration::from_millis(110)); + testbench.handler.periodic_operation(); + } + assert_eq!(testbench.health(), Some(HealthState::Faulty)); + let events = testbench.drain_events(); + let started = events + .iter() + .filter(|e| matches!(e, mgm::Event::Recovery(RecoveryEvent::Started))) + .count(); + assert_eq!(started, RECOVERY_THRESHOLD as usize); + assert!(matches!( + events[..], + [ + .., + mgm::Event::Recovery(RecoveryEvent::Failed), + mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded) + ] + )); + // Retries are hidden from the parent. + assert!(testbench.mode_report_rx.try_recv().is_err()); + + // The faulty device is commanded off, which is reported to the parent. + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + assert!(matches!( + testbench.mode_report_rx.try_recv(), + Ok(ModeResponse::Mode(DeviceMode::Off)) + )); + } + + #[test] + fn test_power_cycle_switch_off_failures_mark_component_faulty() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.drain_events(); + testbench.mode_report_rx.try_iter().for_each(drop); + testbench.exceed_spi_fault_threshold(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues::default(); + + // The switch never turns off. Every failed power cycle costs a recovery attempt. + for _ in 0..RECOVERY_THRESHOLD { + assert_eq!(testbench.health(), Some(HealthState::NeedsRecovery)); + testbench.handler.periodic_operation(); + std::thread::sleep(Duration::from_millis(110)); + testbench.handler.periodic_operation(); + } + assert_eq!(testbench.health(), Some(HealthState::Faulty)); + assert_eq!(testbench.handler.mode(), DeviceMode::Normal); + assert_eq!( + testbench.handler.switch_and_mode_helper.target(), + Some(DeviceMode::Off) + ); + // The mode never changed, so it was not announced or reported. + let events = testbench.drain_events(); + assert!( + !events + .iter() + .any(|e| matches!(e, mgm::Event::ModeChanged(_))) + ); + assert!(testbench.mode_report_rx.try_recv().is_err()); + + testbench.set_switch_state(SwitchState::Off); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + assert!(matches!( + testbench.mode_report_rx.try_recv(), + Ok(ModeResponse::Mode(DeviceMode::Off)) + )); + } + + #[test] + fn test_health_command_during_recovery_is_kept() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.exceed_spi_fault_threshold(); + testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues::default(); + testbench + .tc_tx + .send(create_request_tc( + MgmSelect::_0, + mgm::request::Request::Health(mgm::request::HealthRequest::SetHealth( + HealthState::ExternalControl, + )), + )) + .unwrap(); + + // The power cycle is not cancelled, but it does not override the health set by ground. + testbench.complete_power_cycle(); + assert_eq!(testbench.health(), Some(HealthState::ExternalControl)); + } + + #[test] + fn test_read_mode_during_power_cycle_returns_restored_mode() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.exceed_spi_fault_threshold(); + testbench.handler.periodic_operation(); + testbench.set_switch_state(SwitchState::Off); + // Keep the device off until the parent asked for its mode. + testbench.handler.recovery_off_duration = Duration::from_secs(60); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + testbench.mode_report_rx.try_iter().for_each(drop); + + testbench + .assembly_mode_request_tx + .send(ModeRequest::ReadMode) + .unwrap(); + testbench.handler.periodic_operation(); + assert!(matches!( + testbench.mode_report_rx.try_recv(), + Ok(ModeResponse::Mode(DeviceMode::Normal)) + )); + } + + #[test] + fn test_mode_command_aborts_recovery() { + let mut testbench = MgmTestbench::new(); + testbench.switch_to_normal(); + testbench.exceed_spi_fault_threshold(); + testbench.handler.periodic_operation(); + testbench.set_switch_state(SwitchState::Off); + testbench + .assembly_mode_request_tx + .send(ModeRequest::SetMode(DeviceMode::Off)) + .unwrap(); + testbench.handler.periodic_operation(); + testbench.handler.periodic_operation(); + assert_eq!(testbench.handler.mode(), DeviceMode::Off); + assert_eq!(testbench.handler.switch_and_mode_helper.target(), None); + assert_eq!(testbench.health(), Some(HealthState::Healthy)); } #[test] diff --git a/satrs-example/src/device_mode.rs b/satrs-example/src/device_mode.rs index 2a3349c..c6842f4 100644 --- a/satrs-example/src/device_mode.rs +++ b/satrs-example/src/device_mode.rs @@ -1,16 +1,21 @@ -use std::time::Duration; +use std::time::{Duration, Instant}; use types::pcdu::SwitchId; use crate::eps::PowerSwitchHelper; -/// Modes that distinguish a powered-off state from one or more powered-on states, so +/// This is a helper trait required to make [SwitchAndModeHelper] generic. +/// +/// It allows distinguish a powered-off state from one or more powered-on states, so /// [`SwitchAndModeHelper`] knows which way to drive the switch for a given target mode. pub trait PowerSwitchedMode: Copy + PartialEq { + const OFF: Self; fn requires_power(&self) -> bool; } impl PowerSwitchedMode for types::DeviceMode { + const OFF: Self = types::DeviceMode::Off; + fn requires_power(&self) -> bool { *self != types::DeviceMode::Off } @@ -24,13 +29,36 @@ enum SwitchTransitionState { Done, } +/// Outcome of a single power switch transition. +enum SwitchOutcome { + Reached(Option), + Failed(Option), +} + +/// Dedicated states for power cycling a device. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum PowerCycleState { + Idle, + SwitchingOff { restore_mode: Mode }, + WaitingOff { restore_mode: Mode, since: Instant }, + SwitchingOn { restore_mode: Mode }, +} + /// Outcome of a pending mode transition, once [`SwitchAndModeHelper::handle_mode_transition`] /// has driven it to completion. Carries back whichever TC commanded the transition, if any, so /// the caller can reply to it -- what that reply looks like is handler-specific, so this stays /// out of the helper. -pub enum ModeTransitionEvent { +pub enum ModeTransitionEvent { + /// The target mode was reached. Reached(Option), + /// The target mode could not be reached. Failed(Option), + /// The power cycle completed and the mode before the power cycle was restored. + PowerCycleDone, + /// Power switching failed during the power cycle. The power cycle is not hidden anymore, + /// so [SwitchAndModeHelper::reported_mode] returns the actual mode again. `restore_mode` is + /// the mode the power cycle should have restored, which can be used to retry it. + PowerCycleFailed { restore_mode: Mode }, } /// Drives the on/off power-switch commanding state machine (Idle -> PowerSwitching -> Done) @@ -43,6 +71,8 @@ pub struct SwitchAndModeHelper { mode_helper: satrs_example::ModeHelper, switch_helper: PowerSwitchHelper, switch_id: SwitchId, + power_cycle_state: PowerCycleState, + power_cycle_off_duration: Duration, } impl SwitchAndModeHelper { @@ -56,6 +86,8 @@ impl SwitchAndModeHelper { mode_helper: satrs_example::ModeHelper::new(init_mode, timeout), switch_helper, switch_id, + power_cycle_state: PowerCycleState::Idle, + power_cycle_off_duration: Duration::ZERO, } } @@ -69,16 +101,109 @@ impl SwitchAndModeHelper { self.mode_helper.target } + /// Mode which should be reported to other components. A power cycle is hidden from them, + /// so this is the mode which is restored after the power cycle while one is active. + pub fn reported_mode(&self) -> Mode { + match self.power_cycle_state { + PowerCycleState::SwitchingOff { restore_mode } + | PowerCycleState::WaitingOff { restore_mode, .. } + | PowerCycleState::SwitchingOn { restore_mode } => restore_mode, + PowerCycleState::Idle => self.mode(), + } + } + + #[inline] + pub fn power_cycle_active(&self) -> bool { + self.power_cycle_state != PowerCycleState::Idle + } + + /// Starts a new transition, aborting a running power cycle. pub fn start_transition( &mut self, target_mode: Mode, tc_commander: Option, + ) { + self.power_cycle_state = PowerCycleState::Idle; + self.start_transition_internal(target_mode, tc_commander); + } + + /// Switches the device off, keeps it off for `off_duration` and then switches it to + /// `restore_mode`. Reaching the intermediate off mode does not generate an event. + pub fn start_power_cycle(&mut self, restore_mode: Mode, off_duration: Duration) { + self.power_cycle_state = PowerCycleState::SwitchingOff { restore_mode }; + self.power_cycle_off_duration = off_duration; + self.start_transition_internal(Mode::OFF, None); + } + + fn start_transition_internal( + &mut self, + target_mode: Mode, + tc_commander: Option, ) { self.mode_helper.tc_commander = tc_commander; self.mode_helper.start(target_mode); } - pub fn handle_mode_transition(&mut self) -> Option { + /// This is the main API that the periodic handler of a device handler should call. + /// + /// It handles the switch commanding and returns relevant events. + pub fn handle_mode_transition(&mut self) -> Option> { + // The most probable case: Nothing to do. + if self.target().is_none() && !self.power_cycle_active() { + return None; + } + // Handle this as an extra step so the switch transition after this can proceed. + self.handle_waiting_for_off_when_power_cycling(); + // Core logic: Command the switches, check whether target switch state was reached. + // Note the ?: if a switch transition is on-going, we might do an early return. + let outcome = self.handle_switch_transition()?; + // Regular mode transition without power cycling. + if self.power_cycle_state == PowerCycleState::Idle { + return Some(match outcome { + SwitchOutcome::Reached(tc_commander) => ModeTransitionEvent::Reached(tc_commander), + SwitchOutcome::Failed(tc_commander) => ModeTransitionEvent::Failed(tc_commander), + }); + } + // Power cycling, where a bit more logic is required. + // Handle the error case first. + if let SwitchOutcome::Failed(_) = outcome { + let restore_mode = self.reported_mode(); + self.power_cycle_state = PowerCycleState::Idle; + return Some(ModeTransitionEvent::PowerCycleFailed { restore_mode }); + } + // At this point: The switching was succesfull, so we only match on the + // power cycle state. + match self.power_cycle_state { + // No switching going on for thse cases. + PowerCycleState::Idle | PowerCycleState::WaitingOff { .. } => None, + PowerCycleState::SwitchingOff { restore_mode } => { + self.power_cycle_state = PowerCycleState::WaitingOff { + restore_mode, + since: Instant::now(), + }; + None + } + PowerCycleState::SwitchingOn { .. } => { + // Power is back and we are done. + self.power_cycle_state = PowerCycleState::Idle; + Some(ModeTransitionEvent::PowerCycleDone) + } + } + } + + fn handle_waiting_for_off_when_power_cycling(&mut self) { + if let PowerCycleState::WaitingOff { + restore_mode, + since, + } = self.power_cycle_state + && since.elapsed() >= self.power_cycle_off_duration + { + self.power_cycle_state = PowerCycleState::SwitchingOn { restore_mode }; + self.start_transition_internal(restore_mode, None); + } + } + + fn handle_switch_transition(&mut self) -> Option { let target_mode = self.mode_helper.target?; let switch_target_on = target_mode.requires_power(); if self.mode_helper.transition_state == SwitchTransitionState::Idle { @@ -101,12 +226,279 @@ impl SwitchAndModeHelper { log::info!("switch is {}", if switch_target_on { "on" } else { "off" }); self.mode_helper.transition_state = SwitchTransitionState::Done; } else if self.mode_helper.timed_out() { - return Some(ModeTransitionEvent::Failed(self.mode_helper.finish(false))); + return Some(SwitchOutcome::Failed(self.mode_helper.finish(false))); } } if self.mode_helper.transition_state == SwitchTransitionState::Done { - return Some(ModeTransitionEvent::Reached(self.mode_helper.finish(true))); + return Some(SwitchOutcome::Reached(self.mode_helper.finish(true))); } None } } + +#[cfg(test)] +mod tests { + use std::sync::{Arc, Mutex, mpsc}; + + use arbitrary_int::u11; + use satrs::spacepackets::{CcsdsPacketIdAndPsc, SpacePacketHeader}; + use types::{ + DeviceMode, + pcdu::{SwitchRequest, SwitchState, SwitchStateBinary}, + }; + + use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet}; + + use super::*; + + const TIMEOUT: Duration = Duration::from_millis(50); + + struct Testbench { + helper: SwitchAndModeHelper, + switch_rx: mpsc::Receiver, + shared_switch_set: SharedSwitchSet, + } + + impl Testbench { + fn new() -> Self { + let (switch_tx, switch_rx) = mpsc::sync_channel(10); + let mut switch_map = SwitchMap::new(); + switch_map.insert(SwitchId::Mgm0, SwitchState::Off); + let shared_switch_set: SharedSwitchSet = + Arc::new(Mutex::new(SwitchSet::new(switch_map))); + Self { + helper: SwitchAndModeHelper::new( + DeviceMode::Off, + TIMEOUT, + PowerSwitchHelper::new(switch_tx, shared_switch_set.clone()), + SwitchId::Mgm0, + ), + switch_rx, + shared_switch_set, + } + } + + fn set_switch_state(&self, state: SwitchState) { + self.shared_switch_set + .lock() + .unwrap() + .set_switch_state(SwitchId::Mgm0, state); + } + + fn switch_requests(&self) -> Vec { + self.switch_rx + .try_iter() + .map(|req| req.target_state) + .collect() + } + + /// Drives a transition to `Normal` to completion. + fn switch_to_normal(&mut self) { + self.helper.start_transition(DeviceMode::Normal, None); + self.set_switch_state(SwitchState::On); + assert!(matches!( + self.helper.handle_mode_transition(), + Some(ModeTransitionEvent::Reached(None)) + )); + self.switch_requests(); + } + + /// Starts a power cycle from `Normal` and drives it until the device is off. + fn power_cycle_until_off(&mut self, off_duration: Duration) { + self.switch_to_normal(); + self.helper + .start_power_cycle(DeviceMode::Normal, off_duration); + assert!(self.helper.handle_mode_transition().is_none()); + assert_eq!(self.switch_requests(), [SwitchStateBinary::Off]); + self.set_switch_state(SwitchState::Off); + assert!(self.helper.handle_mode_transition().is_none()); + assert_eq!(self.helper.mode(), DeviceMode::Off); + } + } + + fn tc_id() -> CcsdsPacketIdAndPsc { + CcsdsPacketIdAndPsc::new_from_ccsds_packet(&SpacePacketHeader::new_from_apid(u11::new(1))) + } + + #[test] + fn test_no_transition() { + let mut tb = Testbench::new(); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert_eq!(tb.helper.target(), None); + assert!(tb.helper.handle_mode_transition().is_none()); + assert!(tb.switch_requests().is_empty()); + assert!(!tb.helper.power_cycle_active()); + assert_eq!(tb.helper.reported_mode(), DeviceMode::Off); + } + + #[test] + fn test_switch_on() { + let mut tb = Testbench::new(); + tb.helper + .start_transition(DeviceMode::Normal, Some(tc_id())); + assert!(tb.helper.handle_mode_transition().is_none()); + assert_eq!(tb.switch_requests(), [SwitchStateBinary::On]); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert_eq!(tb.helper.target(), Some(DeviceMode::Normal)); + + tb.set_switch_state(SwitchState::On); + match tb.helper.handle_mode_transition() { + Some(ModeTransitionEvent::Reached(Some(id))) => assert_eq!(id, tc_id()), + _ => panic!("expected mode reached event with TC commander"), + } + assert_eq!(tb.helper.mode(), DeviceMode::Normal); + assert_eq!(tb.helper.target(), None); + assert!(tb.switch_requests().is_empty()); + } + + #[test] + fn test_switch_off() { + let mut tb = Testbench::new(); + tb.switch_to_normal(); + tb.helper.start_transition(DeviceMode::Off, None); + assert!(tb.helper.handle_mode_transition().is_none()); + assert_eq!(tb.switch_requests(), [SwitchStateBinary::Off]); + tb.set_switch_state(SwitchState::Off); + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::Reached(None)) + )); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + } + + #[test] + fn test_switch_already_in_target_state() { + let mut tb = Testbench::new(); + tb.set_switch_state(SwitchState::On); + tb.helper.start_transition(DeviceMode::On, None); + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::Reached(None)) + )); + // The switch command is still sent. + assert_eq!(tb.switch_requests(), [SwitchStateBinary::On]); + assert_eq!(tb.helper.mode(), DeviceMode::On); + } + + #[test] + fn test_switch_timeout() { + let mut tb = Testbench::new(); + tb.helper + .start_transition(DeviceMode::Normal, Some(tc_id())); + assert!(tb.helper.handle_mode_transition().is_none()); + std::thread::sleep(TIMEOUT); + match tb.helper.handle_mode_transition() { + Some(ModeTransitionEvent::Failed(Some(id))) => assert_eq!(id, tc_id()), + _ => panic!("expected mode failed event with TC commander"), + } + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert_eq!(tb.helper.target(), None); + } + + #[test] + fn test_power_cycle() { + let mut tb = Testbench::new(); + tb.power_cycle_until_off(Duration::ZERO); + assert!(tb.helper.power_cycle_active()); + + // The off duration elapsed, so switching on starts right away. + assert!(tb.helper.handle_mode_transition().is_none()); + assert_eq!(tb.switch_requests(), [SwitchStateBinary::On]); + assert_eq!(tb.helper.target(), Some(DeviceMode::Normal)); + tb.set_switch_state(SwitchState::On); + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::PowerCycleDone) + )); + assert_eq!(tb.helper.mode(), DeviceMode::Normal); + assert!(!tb.helper.power_cycle_active()); + } + + #[test] + fn test_power_cycle_reports_restored_mode() { + let mut tb = Testbench::new(); + tb.switch_to_normal(); + tb.helper + .start_power_cycle(DeviceMode::Normal, Duration::from_secs(60)); + assert_eq!(tb.helper.reported_mode(), DeviceMode::Normal); + tb.helper.handle_mode_transition(); + tb.set_switch_state(SwitchState::Off); + tb.helper.handle_mode_transition(); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert_eq!(tb.helper.reported_mode(), DeviceMode::Normal); + } + + #[test] + fn test_power_cycle_reports_restored_mode_while_switching_on() { + let mut tb = Testbench::new(); + tb.power_cycle_until_off(Duration::ZERO); + assert!(tb.helper.handle_mode_transition().is_none()); + assert_eq!(tb.helper.target(), Some(DeviceMode::Normal)); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert_eq!(tb.helper.reported_mode(), DeviceMode::Normal); + } + + #[test] + fn test_power_cycle_waits_off_duration() { + let mut tb = Testbench::new(); + tb.power_cycle_until_off(Duration::from_secs(60)); + for _ in 0..3 { + assert!(tb.helper.handle_mode_transition().is_none()); + } + assert!(tb.switch_requests().is_empty()); + assert_eq!(tb.helper.target(), None); + assert!(tb.helper.power_cycle_active()); + } + + #[test] + fn test_power_cycle_switch_off_timeout() { + let mut tb = Testbench::new(); + tb.switch_to_normal(); + tb.helper + .start_power_cycle(DeviceMode::Normal, Duration::ZERO); + assert!(tb.helper.handle_mode_transition().is_none()); + std::thread::sleep(TIMEOUT); + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::PowerCycleFailed { + restore_mode: DeviceMode::Normal + }) + )); + assert_eq!(tb.helper.mode(), DeviceMode::Normal); + assert!(!tb.helper.power_cycle_active()); + } + + #[test] + fn test_power_cycle_switch_on_timeout() { + let mut tb = Testbench::new(); + tb.power_cycle_until_off(Duration::ZERO); + assert!(tb.helper.handle_mode_transition().is_none()); + std::thread::sleep(TIMEOUT); + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::PowerCycleFailed { + restore_mode: DeviceMode::Normal + }) + )); + assert_eq!(tb.helper.mode(), DeviceMode::Off); + assert!(!tb.helper.power_cycle_active()); + // The failed power cycle is not hidden anymore. + assert_eq!(tb.helper.reported_mode(), DeviceMode::Off); + } + + #[test] + fn test_transition_aborts_power_cycle() { + let mut tb = Testbench::new(); + tb.power_cycle_until_off(Duration::from_secs(60)); + tb.helper.start_transition(DeviceMode::On, Some(tc_id())); + assert!(!tb.helper.power_cycle_active()); + assert_eq!(tb.helper.reported_mode(), DeviceMode::Off); + tb.set_switch_state(SwitchState::On); + // A regular transition event instead of a power cycle event. + assert!(matches!( + tb.helper.handle_mode_transition(), + Some(ModeTransitionEvent::Reached(Some(_))) + )); + assert_eq!(tb.helper.mode(), DeviceMode::On); + } +} diff --git a/satrs-example/types/src/acs/mgm.rs b/satrs-example/types/src/acs/mgm.rs index 2f266eb..602220b 100644 --- a/satrs-example/types/src/acs/mgm.rs +++ b/satrs-example/types/src/acs/mgm.rs @@ -62,11 +62,11 @@ pub struct SensorData { #[strum_discriminants(derive(num_enum::IntoPrimitive))] #[repr(u16)] pub enum Event { - /// The SPI fault counter exceeded its threshold, the component was marked faulty and - /// commanded off. + /// The SPI fault counter exceeded its threshold. Followed by a recovery event. SpiFaultThresholdExceeded, /// A commanded or autonomous mode transition completed. ModeChanged(crate::DeviceMode), + Recovery(satrs::fdir::RecoveryEvent), } impl crate::Message for Event { @@ -77,7 +77,10 @@ impl crate::Message for Event { impl crate::EventId for Event { fn event_id(&self) -> u16 { - EventDiscriminants::from(self).into() + match self { + Event::Recovery(event) => crate::recovery_event_id(*event), + _ => EventDiscriminants::from(self).into(), + } } } diff --git a/satrs-example/types/src/lib.rs b/satrs-example/types/src/lib.rs index aedb2d8..c760d7d 100644 --- a/satrs-example/types/src/lib.rs +++ b/satrs-example/types/src/lib.rs @@ -175,6 +175,14 @@ pub trait EventId { fn event_id(&self) -> u16; } +/// Start of the event ID range for generic FDIR events, which are embedded into the event types +/// of the components. These events have the same ID for all components. +pub const FDIR_EVENT_ID_BASE: u16 = 0x100; + +pub const fn recovery_event_id(event: satrs::fdir::RecoveryEvent) -> u16 { + FDIR_EVENT_ID_BASE + event as u16 +} + /// Generic device mode which covers the requirements of most devices. /// /// The states are related both to the physical and the logical state of the device. Some diff --git a/satrs/CHANGELOG.md b/satrs/CHANGELOG.md index 9e41552..c101022 100644 --- a/satrs/CHANGELOG.md +++ b/satrs/CHANGELOG.md @@ -11,6 +11,9 @@ and this project adheres to [Semantic Versioning](http://semver.org/). - Added `hk` module helpers to track whether a single HK set needs regeneration: `SingleSetHkHelperStd` (`std`), `SingleSetHkHelperEmbassy` (new `embassy-time` feature), and `SingleSetHkHelperCountdown`, generic over the existing `Countdown` trait. +- Added `fdir::RecoveryFdir` (`std`), which escalates component faults to a power cycle + recovery first and to a faulty component if it has to be recovered too often. +- Added `fdir::RecoveryEvent` for recovery related events. # [v0.3.0-alpha.3] 2025-11-06 diff --git a/satrs/src/fdir.rs b/satrs/src/fdir.rs index 4d33b17..045e143 100644 --- a/satrs/src/fdir.rs +++ b/satrs/src/fdir.rs @@ -15,8 +15,47 @@ feature = "embassy-time", doc = "- [FaultCounterEmbassy]: `embassy_time::Instant`, behind the `embassy-time` feature." )] +//! +//! [RecoveryFdir] builds on top of that. It decides whether a component is power cycled or +//! marked faulty when one of its fault counters exceeds its threshold, and keeps the health +//! table up to date during the recovery. It follows the FSFW `DeviceHandlerFailureIsolation`. #![deny(missing_docs)] +#[cfg(feature = "std")] +use crate::health::{HealthState, HealthTableProvider}; + +/// Events related to the recovery of a component. Components are expected to embed this into +/// their own event type. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))] +#[cfg_attr(feature = "defmt", derive(defmt::Format))] +pub enum RecoveryEvent { + /// The component health is [crate::health::HealthState::NeedsRecovery] and it is being + /// power cycled. + Started, + /// The power cycle completed and the component is healthy again. + Done, + /// The power cycle failed. This costs a recovery attempt like any other fault, so it is + /// followed by either a new recovery or [RecoveryEvent::ThresholdExceeded]. + Failed, + /// The component was recovered too often, it was marked faulty. + ThresholdExceeded, +} + +/// Outcome of [RecoveryFdir::handle_fault] and [RecoveryFdir::recovery_failed]. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub enum FaultResponse { + /// The component is already faulty, recovering or externally controlled, so nothing was + /// changed. + Ignored, + /// The health was set to [crate::health::HealthState::NeedsRecovery]. The component should + /// be power cycled. + Recover, + /// The component was recovered too often and its health was set to + /// [crate::health::HealthState::Faulty]. The component should be switched off. + SetFaulty, +} + /// Fault counter backed by [std::time::Instant]. #[cfg(feature = "std")] #[derive(Debug, Clone)] @@ -230,6 +269,106 @@ impl FaultCounterEmbassy { } } +/// Escalates faults of a component to a power cycle recovery first, and to a faulty +/// component if it has to be recovered too often. +/// +/// The component itself runs the power cycle while [Self::needs_recovery] returns `true` and +/// reports the outcome with [Self::recovery_done] or [Self::recovery_failed]. Setting +/// [HealthState::NeedsRecovery] from outside, for example by ground, triggers a recovery as well. +#[cfg(feature = "std")] +#[derive(Debug, Clone)] +pub struct RecoveryFdir { + id: crate::ComponentId, + health_table: HealthTable, + recovery_counter: FaultCounterStd, +} + +#[cfg(feature = "std")] +impl RecoveryFdir { + /// Create a new [RecoveryFdir] for component `id`. + /// + /// The component is marked faulty when it would be recovered more than `recovery_threshold` + /// times, with the recovery count being decremented every `recovery_decrement_after`. + pub fn new( + id: crate::ComponentId, + health_table: HealthTable, + recovery_threshold: u32, + recovery_decrement_after: core::time::Duration, + ) -> Self { + Self { + id, + health_table, + recovery_counter: FaultCounterStd::new(recovery_threshold, recovery_decrement_after), + } + } + + /// Health of the component. Absent entries are returned as `None`. + pub fn health(&self) -> Option { + self.health_table.health(self.id) + } + + /// Set the health of the component. + pub fn set_health(&mut self, health: HealthState) { + self.health_table.set_health(self.id, health); + } + + /// Should be called periodically to decrement the recovery counter. + pub fn periodic_operation(&mut self) { + self.recovery_counter.try_decrement(); + } + + /// Should be called when a fault counter of the component exceeded its threshold. + pub fn handle_fault(&mut self) -> FaultResponse { + // Ground may have taken manual control, or already given up on this component. + // Autonomous FDIR should not override that decision. An already faulty or recovering + // component must not be escalated again. For example, this would allow a faulty component + // to become healthy again, because the recovery counter was reset when it became faulty. + if matches!( + self.health(), + Some(HealthState::ExternalControl) + | Some(HealthState::PermanentFaulty) + | Some(HealthState::Faulty) + | Some(HealthState::NeedsRecovery) + ) { + return FaultResponse::Ignored; + } + self.escalate() + } + + /// Every recovery attempt counts. If there were too many attempts, the component is marked + /// faulty. Otherwise, the component should be recovered (again). + fn escalate(&mut self) -> FaultResponse { + if self.recovery_counter.increment_and_check() { + self.set_health(HealthState::Faulty); + return FaultResponse::SetFaulty; + } + self.set_health(HealthState::NeedsRecovery); + FaultResponse::Recover + } + + /// The component should be power cycled. + pub fn needs_recovery(&self) -> bool { + self.health() == Some(HealthState::NeedsRecovery) + } + + /// The power cycle completed, or was not required. Sets the health back to healthy. + pub fn recovery_done(&mut self) { + // The health might have changed during the recovery. + if self.needs_recovery() { + self.set_health(HealthState::Healthy); + } + } + + /// The power cycle failed. This costs a recovery attempt like any other fault. + pub fn recovery_failed(&mut self) -> FaultResponse { + // The health might have changed during the recovery. + if !self.needs_recovery() { + return FaultResponse::Ignored; + } + self.escalate() + } +} + #[cfg(all(test, feature = "std"))] mod tests { use super::*; @@ -281,6 +420,86 @@ mod tests { assert!(!fc.try_decrement()); } + fn recovery_fdir() -> RecoveryFdir { + recovery_fdir_with_threshold(1) + } + + fn recovery_fdir_with_threshold( + recovery_threshold: u32, + ) -> RecoveryFdir { + RecoveryFdir::new( + 1, + crate::health::HealthTableMapSync::default(), + recovery_threshold, + Duration::from_secs(60), + ) + } + + #[test] + fn first_fault_triggers_recovery() { + let mut fdir = recovery_fdir(); + assert_eq!(fdir.handle_fault(), FaultResponse::Recover); + assert!(fdir.needs_recovery()); + fdir.recovery_done(); + assert_eq!(fdir.health(), Some(HealthState::Healthy)); + } + + #[test] + fn repeated_recovery_sets_faulty() { + let mut fdir = recovery_fdir(); + assert_eq!(fdir.handle_fault(), FaultResponse::Recover); + fdir.recovery_done(); + assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty); + assert_eq!(fdir.health(), Some(HealthState::Faulty)); + } + + #[test] + fn faulty_component_stays_faulty() { + let mut fdir = recovery_fdir(); + fdir.handle_fault(); + fdir.recovery_done(); + assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty); + // The recovery counter was reset, but this must not trigger a new recovery. + assert_eq!(fdir.handle_fault(), FaultResponse::Ignored); + assert_eq!(fdir.health(), Some(HealthState::Faulty)); + } + + #[test] + fn failed_recovery_is_retried() { + let mut fdir = recovery_fdir_with_threshold(2); + fdir.handle_fault(); + assert_eq!(fdir.recovery_failed(), FaultResponse::Recover); + assert!(fdir.needs_recovery()); + assert_eq!(fdir.recovery_failed(), FaultResponse::SetFaulty); + assert_eq!(fdir.health(), Some(HealthState::Faulty)); + } + + #[test] + fn health_is_not_overridden() { + let mut fdir = recovery_fdir(); + for health in [ + HealthState::ExternalControl, + HealthState::PermanentFaulty, + HealthState::Faulty, + HealthState::NeedsRecovery, + ] { + fdir.set_health(health); + assert_eq!(fdir.handle_fault(), FaultResponse::Ignored); + assert_eq!(fdir.health(), Some(health)); + } + } + + #[test] + fn health_changed_during_recovery_is_kept() { + let mut fdir = recovery_fdir(); + fdir.handle_fault(); + fdir.set_health(HealthState::ExternalControl); + fdir.recovery_done(); + assert_eq!(fdir.health(), Some(HealthState::ExternalControl)); + assert_eq!(fdir.recovery_failed(), FaultResponse::Ignored); + assert_eq!(fdir.health(), Some(HealthState::ExternalControl)); + } + #[test] fn clear_resets_state() { let mut fc = FaultCounterStd::new(1, Duration::from_secs(60));