diff --git a/satrs-example/minisim/src/acs/mgt.rs b/satrs-example/minisim/src/acs/mgt.rs index d9e98e7..d9cf811 100644 --- a/satrs-example/minisim/src/acs/mgt.rs +++ b/satrs-example/minisim/src/acs/mgt.rs @@ -10,6 +10,9 @@ use serde::{Deserialize, Serialize}; use std::time::Duration; use types::pcdu::SwitchStateBinary; +/// Time the device needs to answer a command. +const REPLY_DELAY: Duration = Duration::from_millis(15); + /// Simple magnetorquer simulation model. #[derive(Serialize, Deserialize)] pub struct MgtModel { @@ -39,6 +42,9 @@ impl MgtModel { duration_and_dipole: (Duration, mgt::Dipole), cx: &Context, ) { + if self.switch_state != SwitchStateBinary::On { + return; + } self.torque_dipole = duration_and_dipole.1; self.torquing = true; if cx @@ -48,6 +54,7 @@ impl MgtModel { log::warn!("torque clearing can only be set for a future time."); } self.generate_magnetic_field(()).await; + self.schedule_reply(mgt::Reply::Ack, cx); } #[nexosim(schedulable)] @@ -69,22 +76,22 @@ impl MgtModel { if self.switch_state != SwitchStateBinary::On { return; } - cx.schedule_event( - Duration::from_millis(15), - schedulable!(Self::send_housekeeping_data), - (), - ) - .expect("requesting housekeeping data failed") + // The HK is sampled when the command is processed, not when the reply is sent. + let hk = mgt::HkSet { + dipole: self.torque_dipole, + torquing: self.torquing, + }; + self.schedule_reply(mgt::Reply::Hk(hk), cx); + } + + fn schedule_reply(&self, reply: mgt::Reply, cx: &Context) { + cx.schedule_event(REPLY_DELAY, schedulable!(Self::send_reply), reply) + .expect("scheduling MGT reply failed") } #[nexosim(schedulable)] - async fn send_housekeeping_data(&mut self) { - self.reply - .send(SimReply::from(mgt::Reply::Hk(mgt::HkSet { - dipole: self.torque_dipole, - torquing: self.torquing, - }))) - .await; + async fn send_reply(&mut self, reply: mgt::Reply) { + self.reply.send(SimReply::from(reply)).await; } fn calc_magnetic_field(&self, _: mgt::Dipole) -> mgm::SensorValuesMicroTesla { @@ -118,10 +125,17 @@ mod tests { use crate::{eps::tests::switch_device_on, test_helpers::SimTestbench}; + fn decode_reply(sim_reply: SimReply) -> mgt::Reply { + let SimReply::Mgt(frame) = sim_reply else { + panic!("unexpected reply {sim_reply:?}"); + }; + mgt::Reply::from_frame(&frame).expect("invalid MGT reply frame") + } + fn request_hk(sim_testbench: &mut SimTestbench) -> Option { let sim_reply = sim_testbench.request_reply(mgt::Request::RequestHk)?; - let SimReply::Mgt(mgt::Reply::Hk(hk)) = sim_reply else { - panic!("unexpected reply {sim_reply:?}"); + let mgt::Reply::Hk(hk) = decode_reply(sim_reply) else { + panic!("unexpected MGT reply"); }; Some(hk) } @@ -162,7 +176,11 @@ mod tests { .send_request(request) .expect("sending MGT request failed"); sim_testbench.handle_sim_requests_time_agnostic(); - sim_testbench.step_until(Duration::from_millis(5)).unwrap(); + sim_testbench.step_until(Duration::from_millis(20)).unwrap(); + let ack = sim_testbench + .try_receive_next_reply() + .expect("no torque command ack"); + assert_eq!(decode_reply(ack), mgt::Reply::Ack); assert_eq!( request_hk(&mut sim_testbench), @@ -183,6 +201,25 @@ mod tests { ); } + #[test] + fn test_torque_command_not_acked_when_off() { + let mut sim_testbench = SimTestbench::new(); + let reply = sim_testbench.request_reply(mgt::Request::ApplyTorque { + duration: Duration::from_millis(100), + dipole: mgt::Dipole { x: 1, y: 2, z: 3 }, + }); + assert!(reply.is_none()); + } + + #[test] + fn test_invalid_frame_is_dropped() { + let mut sim_testbench = SimTestbench::new(); + switch_device_on(&mut sim_testbench, SwitchId::Mgt); + assert!(sim_testbench + .request_reply(SimRequest::Mgt(vec![0x01, 0x00])) + .is_none()); + } + /// Processes the request without stepping, so scheduled events like the torque clearing do /// not fire. fn process_without_step(sim_testbench: &mut SimTestbench, request: impl Into) { @@ -200,13 +237,15 @@ mod tests { request: mgm::Request::RequestSensorData, }, ); - let sim_reply = sim_testbench - .try_receive_next_reply() - .expect("no MGM reply received"); - let SimReply::Mgm { reply, .. } = sim_reply else { - panic!("unexpected reply {sim_reply:?}"); - }; - reply.sensor_values + // Skips pending MGT replies, for example torque command acks. + loop { + let sim_reply = sim_testbench + .try_receive_next_reply() + .expect("no MGM reply received"); + if let SimReply::Mgm { reply, .. } = sim_reply { + return reply.sensor_values; + } + } } fn start_torquing(sim_testbench: &mut SimTestbench, duration: Duration) { diff --git a/satrs-example/minisim/src/controller.rs b/satrs-example/minisim/src/controller.rs index 8346680..b6b6cce 100644 --- a/satrs-example/minisim/src/controller.rs +++ b/satrs-example/minisim/src/controller.rs @@ -268,7 +268,14 @@ impl SimController { } } - fn handle_mgt_request(&mut self, mgt_request: mgt::Request) { + fn handle_mgt_request(&mut self, frame: Vec) { + let mgt_request = match mgt::Request::from_frame(&frame) { + Ok(request) => request, + Err(e) => { + log::warn!("dropping invalid MGT frame {frame:02x?}: {e}"); + return; + } + }; if MGT_REQ_WIRETAPPING { log::info!("received MGT request: {mgt_request:?}"); } diff --git a/satrs-example/minisim/src/lib.rs b/satrs-example/minisim/src/lib.rs index 6007b68..bc68af7 100644 --- a/satrs-example/minisim/src/lib.rs +++ b/satrs-example/minisim/src/lib.rs @@ -19,8 +19,12 @@ pub enum ComponentId { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub enum SimRequest { SimCtrl(SimCtrlRequest), - Mgm { id: mgm::Id, request: mgm::Request }, - Mgt(mgt::Request), + Mgm { + id: mgm::Id, + request: mgm::Request, + }, + /// Raw frame of the MGT serial protocol. + Mgt(Vec), Pcdu(PcduRequest), } @@ -32,7 +36,7 @@ impl From for SimRequest { impl From for SimRequest { fn from(request: mgt::Request) -> Self { - Self::Mgt(request) + Self::Mgt(request.to_frame()) } } @@ -64,8 +68,12 @@ impl SimRequestWithTime { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub enum SimReply { SimCtrl(SimCtrlReply), - Mgm { id: mgm::Id, reply: mgm::Reply }, - Mgt(mgt::Reply), + Mgm { + id: mgm::Id, + reply: mgm::Reply, + }, + /// Raw frame of the MGT serial protocol. + Mgt(Vec), Pcdu(PcduReply), } @@ -88,7 +96,7 @@ impl From for SimReply { impl From for SimReply { fn from(reply: mgt::Reply) -> Self { - Self::Mgt(reply) + Self::Mgt(reply.to_frame()) } } @@ -263,11 +271,40 @@ pub mod acs { } } + /// Simple serial protocol of the magnetorquer. + /// + /// The first byte of each frame is the packet ID. The high bit of the ID is set for replies. + /// All fields are big endian. Every command is answered with exactly one reply, but only + /// if the device is powered. The device drops invalid frames. + /// + /// A data link layer is deliberately skipped for simplicity. A real serial link would need + /// framing and error detection, for example COBS encoding and a CRC. Here, the transport + /// always delivers complete and intact frames. pub mod mgt { use std::time::Duration; use serde::{Deserialize, Serialize}; + pub mod packet_id { + pub const REQUEST_HK: u8 = 0x01; + /// Payload: dipole (3 x i16), duration in milliseconds (u32). + pub const APPLY_TORQUE: u8 = 0x02; + /// Payload: dipole (3 x i16), torquing flag (u8). + pub const HK: u8 = 0x81; + /// Reply to [APPLY_TORQUE]. + pub const ACK: u8 = 0x82; + } + + #[derive(Debug, Copy, Clone, PartialEq, Eq, thiserror::Error)] + pub enum FrameError { + #[error("empty frame")] + Empty, + #[error("unknown packet ID {0:#04x}")] + UnknownPacketId(u8), + #[error("invalid length {len} for packet ID {id:#04x}")] + InvalidLength { id: u8, len: usize }, + } + // Simple model using i16 values. #[derive(Default, Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct Dipole { @@ -276,12 +313,84 @@ pub mod acs { pub z: i16, } + impl Dipole { + const LEN: usize = 6; + + fn write_to(&self, frame: &mut Vec) { + frame.extend_from_slice(&self.x.to_be_bytes()); + frame.extend_from_slice(&self.y.to_be_bytes()); + frame.extend_from_slice(&self.z.to_be_bytes()); + } + + fn read_from(buf: &[u8]) -> Self { + Self { + x: i16::from_be_bytes([buf[0], buf[1]]), + y: i16::from_be_bytes([buf[2], buf[3]]), + z: i16::from_be_bytes([buf[4], buf[5]]), + } + } + } + + /// Checks the frame length and returns the packet ID and the payload. + fn split_frame( + frame: &[u8], + payload_len: impl Fn(u8) -> Option, + ) -> Result<(u8, &[u8]), FrameError> { + let (&id, payload) = frame.split_first().ok_or(FrameError::Empty)?; + let expected_len = payload_len(id).ok_or(FrameError::UnknownPacketId(id))?; + if payload.len() != expected_len { + return Err(FrameError::InvalidLength { + id, + len: frame.len(), + }); + } + Ok((id, payload)) + } + #[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] pub enum Request { - ApplyTorque { duration: Duration, dipole: Dipole }, + /// The duration has millisecond resolution on the wire. + ApplyTorque { + duration: Duration, + dipole: Dipole, + }, RequestHk, } + impl Request { + pub fn to_frame(&self) -> Vec { + match self { + Request::RequestHk => vec![packet_id::REQUEST_HK], + Request::ApplyTorque { duration, dipole } => { + let mut frame = vec![packet_id::APPLY_TORQUE]; + dipole.write_to(&mut frame); + let duration_ms = u32::try_from(duration.as_millis()).unwrap_or(u32::MAX); + frame.extend_from_slice(&duration_ms.to_be_bytes()); + frame + } + } + } + + pub fn from_frame(frame: &[u8]) -> Result { + let (id, payload) = split_frame(frame, |id| match id { + packet_id::REQUEST_HK => Some(0), + packet_id::APPLY_TORQUE => Some(Dipole::LEN + 4), + _ => None, + })?; + Ok(match id { + packet_id::REQUEST_HK => Request::RequestHk, + _ => { + let duration_ms = + u32::from_be_bytes(payload[Dipole::LEN..].try_into().unwrap()); + Request::ApplyTorque { + duration: Duration::from_millis(duration_ms.into()), + dipole: Dipole::read_from(payload), + } + } + }) + } + } + #[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct HkSet { pub dipole: Dipole, @@ -291,6 +400,95 @@ pub mod acs { #[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)] pub enum Reply { Hk(HkSet), + Ack, + } + + impl Reply { + pub fn to_frame(&self) -> Vec { + match self { + Reply::Hk(hk) => { + let mut frame = vec![packet_id::HK]; + hk.dipole.write_to(&mut frame); + frame.push(hk.torquing as u8); + frame + } + Reply::Ack => vec![packet_id::ACK], + } + } + + pub fn from_frame(frame: &[u8]) -> Result { + let (id, payload) = split_frame(frame, |id| match id { + packet_id::HK => Some(Dipole::LEN + 1), + packet_id::ACK => Some(0), + _ => None, + })?; + Ok(match id { + packet_id::ACK => Reply::Ack, + _ => Reply::Hk(HkSet { + dipole: Dipole::read_from(payload), + torquing: payload[Dipole::LEN] != 0, + }), + }) + } + } + + #[cfg(test)] + mod tests { + use super::*; + + #[test] + fn test_apply_torque_frame() { + let request = Request::ApplyTorque { + duration: Duration::from_millis(0x0102_0304), + dipole: Dipole { + x: -2, + y: 0x0506, + z: 0x0708, + }, + }; + let frame = request.to_frame(); + assert_eq!( + frame, + [0x02, 0xff, 0xfe, 0x05, 0x06, 0x07, 0x08, 0x01, 0x02, 0x03, 0x04] + ); + assert_eq!(Request::from_frame(&frame), Ok(request)); + } + + #[test] + fn test_request_hk_frame() { + assert_eq!(Request::RequestHk.to_frame(), [0x01]); + assert_eq!(Request::from_frame(&[0x01]), Ok(Request::RequestHk)); + } + + #[test] + fn test_reply_frames() { + let hk = Reply::Hk(HkSet { + dipole: Dipole { x: 1, y: 2, z: 3 }, + torquing: true, + }); + let frame = hk.to_frame(); + assert_eq!(frame, [0x81, 0, 1, 0, 2, 0, 3, 1]); + assert_eq!(Reply::from_frame(&frame), Ok(hk)); + assert_eq!(Reply::Ack.to_frame(), [0x82]); + assert_eq!(Reply::from_frame(&[0x82]), Ok(Reply::Ack)); + } + + #[test] + fn test_invalid_frames() { + assert_eq!(Request::from_frame(&[]), Err(FrameError::Empty)); + assert_eq!( + Request::from_frame(&[0x81]), + Err(FrameError::UnknownPacketId(0x81)) + ); + assert_eq!( + Request::from_frame(&[0x01, 0x00]), + Err(FrameError::InvalidLength { id: 0x01, len: 2 }) + ); + assert_eq!( + Reply::from_frame(&[0x81, 0, 1]), + Err(FrameError::InvalidLength { id: 0x81, len: 3 }) + ); + } } } } diff --git a/satrs-example/src/acs/mgm.rs b/satrs-example/src/acs/mgm.rs index 913a0ad..c60c752 100644 --- a/satrs-example/src/acs/mgm.rs +++ b/satrs-example/src/acs/mgm.rs @@ -1,4 +1,4 @@ -use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir}; +use satrs::fdir::{FaultCounterStd, RecoveryEvent}; use satrs::health::HealthTableMapSync; use satrs::spacepackets::CcsdsPacketIdAndPsc; use satrs_example::{HkHelperSingleSet, TimestampHelper, TmtcQueues}; @@ -15,6 +15,7 @@ use types::pcdu::SwitchId; use types::{ComponentId, DeviceMode, HkRequestType, acs::mgm}; use crate::ccsds::pack_ccsds_tm_packet_for_now; +use crate::device_fdir::{DeviceFdir, FdirEvent}; use crate::device_mode::{ModeTransitionEvent, SwitchAndModeHelper}; use crate::eps::PowerSwitchHelper; @@ -34,13 +35,6 @@ pub const Z_LOWBYTE_IDX: usize = 13; pub const SPI_FAULT_THRESHOLD: u32 = 2; pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30); -// FDIR configuration for power cycle recoveries. The component is marked faulty if it would be -// recovered more than RECOVERY_THRESHOLD times before the counter is decremented again. -pub const RECOVERY_THRESHOLD: u32 = 2; -pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60); -/// Time the device stays unpowered during a power cycle, so it can fully discharge. -pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500); - #[derive(Debug, PartialEq, Eq, Clone, Copy)] pub enum MgmId { _0, @@ -48,7 +42,7 @@ pub enum MgmId { } impl MgmId { - pub const fn str(&self) -> &str { + pub const fn str(&self) -> &'static str { match self { MgmId::_0 => "MGM 0", MgmId::_1 => "MGM 1", @@ -191,9 +185,7 @@ pub struct MgmHandlerLis3Mdl { hk_helper: HkHelperSingleSet, switch_and_mode_helper: SwitchAndModeHelper, mode_leaf_helper: ModeLeafHelper, - spi_fault_counter: FaultCounterStd, - fdir: RecoveryFdir, - recovery_off_duration: Duration, + fdir: DeviceFdir, event_tx: mpsc::SyncSender<(ComponentId, mgm::Event)>, } @@ -225,14 +217,12 @@ impl MgmHandlerLis3Mdl { stamp_helper: TimestampHelper::default(), hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)), mode_leaf_helper, - spi_fault_counter: FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER), - fdir: RecoveryFdir::new( - id.component_id().into(), + fdir: DeviceFdir::new( + id.str(), + id.component_id(), health_table, - RECOVERY_THRESHOLD, - RECOVERY_DECREMENT_AFTER, + FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER), ), - recovery_off_duration: RECOVERY_OFF_DURATION, event_tx, } } @@ -253,8 +243,9 @@ impl MgmHandlerLis3Mdl { // Handle assembly related messages. self.handle_mode_leaf_handling(); - self.fdir.periodic_operation(); - self.check_needs_recovery(); + self.fdir + .periodic_operation(&mut self.switch_and_mode_helper); + self.handle_fdir_events(); // Handle mode transitions first. This also takes care of recoveries required by FDIR. if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() { @@ -266,9 +257,14 @@ impl MgmHandlerLis3Mdl { self.handle_mode_transition_failure(tc_commander) } // The mode did not change for other components, so there is nothing to report. - ModeTransitionEvent::PowerCycleDone => self.handle_recovery_done(), + ModeTransitionEvent::PowerCycleDone => { + self.fdir.handle_power_cycle_done(); + self.handle_fdir_events(); + } ModeTransitionEvent::PowerCycleFailed { restore_mode } => { - self.handle_recovery_failure(restore_mode) + self.fdir + .handle_power_cycle_failed(&mut self.switch_and_mode_helper, restore_mode); + self.handle_fdir_events(); } } } @@ -453,7 +449,7 @@ impl MgmHandlerLis3Mdl { return; } // Successfull readout, so we can decrement the counter. - self.spi_fault_counter.try_decrement(); + self.fdir.register_success(); // Simple scaling to retrieve the float value, assuming the best sensor resolution. let mut mgm_guard = self.shared_mgm_set.lock().unwrap(); mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS; @@ -463,122 +459,39 @@ impl MgmHandlerLis3Mdl { drop(mgm_guard); } - /// Registers one SPI fault with the FDIR fault counter, invalidating the current - /// sensor set. If the failure threshold is exceeded, the device is power cycled. If it was - /// power cycled too often, the component is marked faulty and commanded off instead. + /// Registers one SPI fault with the FDIR, invalidating the current sensor set. fn register_spi_fault(&mut self) { log::warn!("{}: stuck-bus SPI fault", self.id.str()); self.shared_mgm_set.lock().unwrap().valid = false; - if !self.spi_fault_counter.increment_and_check() { - return; - } - match self.fdir.handle_fault() { - FaultResponse::Ignored => { - log::info!( - "{}: SPI fault threshold exceeded, but component is already faulty, \ - recovering or externally controlled", - self.id.str() - ); - } - FaultResponse::Recover => { - log::warn!( - "{}: SPI fault threshold exceeded, power cycling device", - self.id.str() - ); - self.send_event(mgm::Event::SpiFaultThresholdExceeded); - self.check_needs_recovery(); - } - FaultResponse::SetFaulty => { - log::error!( - "{}: SPI fault threshold exceeded after too many recoveries, marking \ - component faulty", - self.id.str() - ); - self.send_event(mgm::Event::SpiFaultThresholdExceeded); - self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); - self.switch_off_faulty_device(); - } + self.fdir.register_fault(&mut self.switch_and_mode_helper); + self.handle_fdir_events(); + } + + fn handle_fdir_events(&mut self) { + while let Some(event) = self.fdir.next_event() { + let event = match event { + FdirEvent::FaultThresholdExceeded => mgm::Event::SpiFaultThresholdExceeded, + FdirEvent::Recovery(recovery_event) => { + // The device is power cycled or switched off. + if matches!( + recovery_event, + RecoveryEvent::Started | RecoveryEvent::ThresholdExceeded + ) { + self.shared_mgm_set.lock().unwrap().valid = false; + } + mgm::Event::Recovery(recovery_event) + } + }; + self.send_event(event); } } - fn switch_off_faulty_device(&mut self) { - // Do not restart an already pending Off transition, which would reset the transition - // state machine before it can finish. - if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) { - log::warn!("{}: commanding device off due to fault", self.id.str()); - self.start_transition(DeviceMode::Off, None); - } - } - - /// Starts a power cycle if the health is [satrs::health::HealthState::NeedsRecovery]. The health is set - /// either by the FDIR or by ground. - fn check_needs_recovery(&mut self) { - if self.switch_and_mode_helper.power_cycle_active() - || self.switch_and_mode_helper.target().is_some() - || !self.fdir.needs_recovery() - { - return; - } - if self.mode() == DeviceMode::Off { - // Nothing to power cycle, the next switch-on is a fresh start anyway. - log::info!("{}: device is off, no recovery required", self.id.str()); - self.fdir.recovery_done(); - return; - } - self.start_recovery(self.mode()); - } - - fn start_recovery(&mut self, restore_mode: DeviceMode) { - log::warn!("{}: starting power cycle recovery", self.id.str()); - self.shared_mgm_set.lock().unwrap().valid = false; - self.switch_and_mode_helper - .start_power_cycle(restore_mode, self.recovery_off_duration); - self.send_event(mgm::Event::Recovery(RecoveryEvent::Started)); - } - - fn handle_recovery_done(&mut self) { - log::info!("{}: power cycle recovery done", self.id.str()); - // Faults registered while the device was switched off do not count anymore. - self.spi_fault_counter.clear(); - self.fdir.recovery_done(); - self.send_event(mgm::Event::Recovery(RecoveryEvent::Done)); - } - - /// A failed power cycle costs a recovery attempt like any other fault. - fn handle_recovery_failure(&mut self, restore_mode: DeviceMode) { - self.send_event(mgm::Event::Recovery(RecoveryEvent::Failed)); - match self.fdir.recovery_failed() { - FaultResponse::Recover => { - log::warn!("{}: power cycle recovery failed, retrying", self.id.str()); - self.start_recovery(restore_mode); - } - FaultResponse::SetFaulty => { - log::error!( - "{}: power cycle recovery failed too often, marking component faulty", - self.id.str() - ); - self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded)); - self.switch_off_faulty_device(); - } - // Ground changed the health during the recovery and is in charge now. - FaultResponse::Ignored => (), - } - } - - /// Mode commands from ground or the parent abort a running recovery. fn handle_mode_command( &mut self, target_mode: DeviceMode, tc_commander: Option, ) { - if self.switch_and_mode_helper.power_cycle_active() { - log::warn!( - "{}: mode command aborts power cycle recovery", - self.id.str() - ); - // Otherwise, the recovery would restart right away. - self.fdir.recovery_done(); - } + self.fdir.handle_mode_command(&self.switch_and_mode_helper); self.start_transition(target_mode, tc_commander); } @@ -662,6 +575,7 @@ mod tests { pcdu::{SwitchRequest, SwitchState, SwitchStateBinary}, }; + use crate::device_fdir::RECOVERY_THRESHOLD; use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet}; use super::*; @@ -734,7 +648,7 @@ mod tests { health_table.clone(), event_tx, ); - handler.recovery_off_duration = Duration::ZERO; + handler.fdir.recovery_off_duration = Duration::ZERO; Self { assembly_mode_request_tx, mode_report_rx, @@ -1132,7 +1046,7 @@ mod tests { mgm::Event::Recovery(RecoveryEvent::Done), ] )); - assert_eq!(testbench.handler.spi_fault_counter.fault_count(), 0); + assert_eq!(testbench.handler.fdir.fault_count(), 0); assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid); } @@ -1343,7 +1257,7 @@ mod tests { testbench.handler.periodic_operation(); testbench.set_switch_state(SwitchState::Off); // Keep the device off until the parent asked for its mode. - testbench.handler.recovery_off_duration = Duration::from_secs(60); + testbench.handler.fdir.recovery_off_duration = Duration::from_secs(60); testbench.handler.periodic_operation(); assert_eq!(testbench.handler.mode(), DeviceMode::Off); testbench.mode_report_rx.try_iter().for_each(drop); diff --git a/satrs-example/src/acs/mgt.rs b/satrs-example/src/acs/mgt.rs index 2044ef4..4b81de2 100644 --- a/satrs-example/src/acs/mgt.rs +++ b/satrs-example/src/acs/mgt.rs @@ -5,7 +5,7 @@ use std::time::{Duration, Instant}; use satrs::spacepackets::CcsdsPacketIdAndPsc; use satrs_example::{HkHelperSingleSet, TmtcQueues}; use satrs_minisim::acs::mgt as sim_mgt; -use satrs_minisim::{SimReply, SimRequestWithTime}; +use satrs_minisim::{SimReply, SimRequest, SimRequestWithTime}; use types::acs::mgt::{ self, HkSet, request::{ModeRequest, Request}, @@ -23,41 +23,41 @@ use crate::eps::PowerSwitchHelper; pub struct DummyInterface { dipole: sim_mgt::Dipole, torque_end: Option, - hk_requested: bool, + replies: VecDeque>, } impl DummyInterface { - fn send(&mut self, request: sim_mgt::Request) { - match request { + fn send(&mut self, frame: &[u8]) { + let Ok(request) = sim_mgt::Request::from_frame(frame) else { + return; + }; + let reply = match request { sim_mgt::Request::ApplyTorque { duration, dipole } => { self.dipole = dipole; self.torque_end = Some(Instant::now() + duration); + sim_mgt::Reply::Ack } - sim_mgt::Request::RequestHk => self.hk_requested = true, - } - } - - fn try_recv_hk(&mut self) -> Option { - if !std::mem::take(&mut self.hk_requested) { - return None; - } - let torquing = self.torque_end.is_some_and(|end| Instant::now() < end); - Some(sim_mgt::HkSet { - dipole: if torquing { - self.dipole - } else { - sim_mgt::Dipole::default() - }, - torquing, - }) + sim_mgt::Request::RequestHk => { + let torquing = self.torque_end.is_some_and(|end| Instant::now() < end); + sim_mgt::Reply::Hk(sim_mgt::HkSet { + dipole: if torquing { + self.dipole + } else { + sim_mgt::Dipole::default() + }, + torquing, + }) + } + }; + self.replies.push_back(reply.to_frame()); } } -/// Records all requests and returns injected HK replies. +/// Records all sent frames and returns injected reply frames. #[derive(Default)] pub struct TestInterface { - pub sent_requests: Vec, - pub hk_replies: VecDeque, + pub sent_frames: Vec>, + pub replies: VecDeque>, } pub struct SimInterface { @@ -66,19 +66,21 @@ pub struct SimInterface { } impl SimInterface { - fn send(&mut self, request: sim_mgt::Request) { + fn send(&mut self, frame: &[u8]) { if let Err(e) = self .sim_request_tx - .send(SimRequestWithTime::new_with_epoch_time(request)) + .send(SimRequestWithTime::new_with_epoch_time(SimRequest::Mgt( + frame.to_vec(), + ))) { log::error!("failed to send MGT SIM request: {e}"); } } - fn try_recv_hk(&mut self) -> Option { + fn try_recv(&mut self) -> Option> { let sim_reply = self.sim_reply_rx.try_recv().ok()?; match sim_reply { - SimReply::Mgt(sim_mgt::Reply::Hk(hk)) => Some(hk), + SimReply::Mgt(frame) => Some(frame), _ => { log::warn!("unexpected MGT SIM reply: {sim_reply:?}"); None @@ -87,6 +89,7 @@ impl SimInterface { } } +/// Frame based transport to the device. The handler implements the protocol on top of it. pub enum MgtCommunication { Dummy(DummyInterface), Sim(SimInterface), @@ -95,19 +98,19 @@ pub enum MgtCommunication { } impl MgtCommunication { - fn send(&mut self, request: sim_mgt::Request) { + fn send(&mut self, frame: &[u8]) { match self { - MgtCommunication::Dummy(dummy) => dummy.send(request), - MgtCommunication::Sim(sim) => sim.send(request), - MgtCommunication::Test(test) => test.sent_requests.push(request), + MgtCommunication::Dummy(dummy) => dummy.send(frame), + MgtCommunication::Sim(sim) => sim.send(frame), + MgtCommunication::Test(test) => test.sent_frames.push(frame.to_vec()), } } - fn try_recv_hk(&mut self) -> Option { + fn try_recv(&mut self) -> Option> { match self { - MgtCommunication::Dummy(dummy) => dummy.try_recv_hk(), - MgtCommunication::Sim(sim) => sim.try_recv_hk(), - MgtCommunication::Test(test) => test.hk_replies.pop_front(), + MgtCommunication::Dummy(dummy) => dummy.replies.pop_front(), + MgtCommunication::Sim(sim) => sim.try_recv(), + MgtCommunication::Test(test) => test.replies.pop_front(), } } } @@ -183,25 +186,39 @@ impl MgtHandler { } if self.ready_for_commanding() { - self.com.send(sim_mgt::Request::RequestHk); - } - while let Some(hk) = self.com.try_recv_hk() { - self.hk_set = HkSet { - valid: true, - dipole: types::acs::mgt::Dipole { - x: hk.dipole.x, - y: hk.dipole.y, - z: hk.dipole.z, - }, - torquing: hk.torquing, - }; + self.send_request(sim_mgt::Request::RequestHk); } + self.handle_replies(); if self.hk_helper.needs_generation() { self.send_telemetry(None, Response::Hk(self.hk_set)); } } + fn send_request(&mut self, request: sim_mgt::Request) { + self.com.send(&request.to_frame()); + } + + fn handle_replies(&mut self) { + while let Some(frame) = self.com.try_recv() { + match sim_mgt::Reply::from_frame(&frame) { + Ok(sim_mgt::Reply::Hk(hk)) => { + self.hk_set = HkSet { + valid: true, + dipole: types::acs::mgt::Dipole { + x: hk.dipole.x, + y: hk.dipole.y, + z: hk.dipole.z, + }, + torquing: hk.torquing, + }; + } + Ok(sim_mgt::Reply::Ack) => (), + Err(e) => log::warn!("MGT: invalid reply frame {frame:02x?}: {e}"), + } + } + } + fn ready_for_commanding(&self) -> bool { self.mode() == DeviceMode::Normal && self.switch_and_mode_helper.target().is_none() } @@ -271,7 +288,7 @@ impl MgtHandler { self.send_telemetry(Some(tc_id), Response::NotInNormalMode); return; } - self.com.send(sim_mgt::Request::ApplyTorque { + self.send_request(sim_mgt::Request::ApplyTorque { duration, dipole: sim_mgt::Dipole { x: dipole.x, @@ -355,6 +372,19 @@ mod tests { use super::*; + impl TestInterface { + fn sent_requests(&self) -> Vec { + self.sent_frames + .iter() + .map(|frame| sim_mgt::Request::from_frame(frame).unwrap()) + .collect() + } + + fn push_reply(&mut self, reply: sim_mgt::Reply) { + self.replies.push_back(reply.to_frame()); + } + } + struct MgtTestbench { parent_request_tx: mpsc::SyncSender, parent_report_rx: mpsc::Receiver, @@ -442,7 +472,7 @@ mod tests { let mut testbench = MgtTestbench::new(); testbench.handler.periodic_operation(); assert_eq!(testbench.handler.mode(), DeviceMode::Off); - assert!(testbench.test_interface().sent_requests.is_empty()); + assert!(testbench.test_interface().sent_frames.is_empty()); } #[test] @@ -504,14 +534,14 @@ mod tests { }); testbench.handler.periodic_operation(); assert_eq!(testbench.next_response(), Response::NotInNormalMode); - assert!(testbench.test_interface().sent_requests.is_empty()); + assert!(testbench.test_interface().sent_frames.is_empty()); } #[test] fn test_torque_command_forwarded_in_normal_mode() { let mut testbench = MgtTestbench::new(); testbench.switch_to_normal(); - testbench.test_interface().sent_requests.clear(); + testbench.test_interface().sent_frames.clear(); testbench.send_tc(Request::ApplyTorque { dipole: mgt::Dipole { x: 1, y: 2, z: 3 }, @@ -520,7 +550,7 @@ mod tests { testbench.handler.periodic_operation(); assert_eq!(testbench.next_response(), Response::Ok); assert_eq!( - testbench.test_interface().sent_requests.first(), + testbench.test_interface().sent_requests().first(), Some(&sim_mgt::Request::ApplyTorque { duration: Duration::from_millis(100), dipole: sim_mgt::Dipole { x: 1, y: 2, z: 3 }, @@ -535,17 +565,16 @@ mod tests { assert!( testbench .test_interface() - .sent_requests + .sent_requests() .contains(&sim_mgt::Request::RequestHk) ); testbench .test_interface() - .hk_replies - .push_back(sim_mgt::HkSet { + .push_reply(sim_mgt::Reply::Hk(sim_mgt::HkSet { dipole: sim_mgt::Dipole { x: 1, y: 2, z: 3 }, torquing: true, - }); + })); testbench.handler.periodic_operation(); testbench.send_tc(Request::Hk(HkRequestType::OneShot)); testbench.handler.periodic_operation(); @@ -565,11 +594,10 @@ mod tests { testbench.switch_to_normal(); testbench .test_interface() - .hk_replies - .push_back(sim_mgt::HkSet { + .push_reply(sim_mgt::Reply::Hk(sim_mgt::HkSet { dipole: sim_mgt::Dipole::default(), torquing: false, - }); + })); testbench.handler.periodic_operation(); testbench.send_tc(Request::Mode(ModeRequest::SetMode(DeviceMode::Off))); diff --git a/satrs-example/src/device_fdir.rs b/satrs-example/src/device_fdir.rs new file mode 100644 index 0000000..25785e6 --- /dev/null +++ b/satrs-example/src/device_fdir.rs @@ -0,0 +1,200 @@ +use std::collections::VecDeque; +use std::time::Duration; + +use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir}; +use satrs::health::{HealthState, HealthTableMapSync}; +use types::{ComponentId, DeviceMode}; + +use crate::device_mode::SwitchAndModeHelper; + +// The component is marked faulty if it would be recovered more than RECOVERY_THRESHOLD times +// before the counter is decremented again. +pub const RECOVERY_THRESHOLD: u32 = 2; +pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60); +/// Time the device stays unpowered during a power cycle, so it can fully discharge. +pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500); + +/// Generic FDIR events. The device handler maps them to its own event type. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub enum FdirEvent { + FaultThresholdExceeded, + Recovery(RecoveryEvent), +} + +/// Fault counting and power cycle recovery for device handlers which own the power switch of +/// their device. +/// +/// The handler detects faults itself and reports them with [Self::register_fault]. This helper +/// then decides whether the device is power cycled, or marked faulty and switched off, and drives +/// the [SwitchAndModeHelper] of the handler accordingly. The handler retrieves the resulting +/// events with [Self::next_event]. +pub struct DeviceFdir { + name: &'static str, + fault_counter: FaultCounterStd, + recovery: RecoveryFdir, + pub recovery_off_duration: Duration, + events: VecDeque, +} + +impl DeviceFdir { + pub fn new( + name: &'static str, + component_id: ComponentId, + health_table: HealthTableMapSync, + fault_counter: FaultCounterStd, + ) -> Self { + Self { + name, + fault_counter, + recovery: RecoveryFdir::new( + component_id.into(), + health_table, + RECOVERY_THRESHOLD, + RECOVERY_DECREMENT_AFTER, + ), + recovery_off_duration: RECOVERY_OFF_DURATION, + events: VecDeque::new(), + } + } + + #[cfg(test)] + pub fn fault_count(&self) -> u32 { + self.fault_counter.fault_count() + } + + pub fn set_health(&mut self, health: HealthState) { + self.recovery.set_health(health); + } + + pub fn next_event(&mut self) -> Option { + self.events.pop_front() + } + + /// Should be called once per cycle, before the mode transition is handled. Starts a power + /// cycle if the health was set to [HealthState::NeedsRecovery] by the FDIR or by ground. + pub fn periodic_operation(&mut self, modes: &mut SwitchAndModeHelper) { + self.recovery.periodic_operation(); + self.check_needs_recovery(modes); + } + + pub fn register_success(&mut self) { + self.fault_counter.try_decrement(); + } + + /// If the fault threshold is exceeded, the device is power cycled. If it was power cycled + /// too often, the component is marked faulty and commanded off instead. + pub fn register_fault(&mut self, modes: &mut SwitchAndModeHelper) { + if !self.fault_counter.increment_and_check() { + return; + } + match self.recovery.handle_fault() { + FaultResponse::Ignored => { + log::info!( + "{}: fault threshold exceeded, but component is already faulty, \ + recovering or externally controlled", + self.name + ); + } + FaultResponse::Recover => { + log::warn!( + "{}: fault threshold exceeded, power cycling device", + self.name + ); + self.events.push_back(FdirEvent::FaultThresholdExceeded); + self.check_needs_recovery(modes); + } + FaultResponse::SetFaulty => { + log::error!( + "{}: fault threshold exceeded after too many recoveries, marking \ + component faulty", + self.name + ); + self.events.push_back(FdirEvent::FaultThresholdExceeded); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(modes); + } + } + } + + /// Mode commands from ground or the parent abort a running recovery. Must be called before + /// the commanded transition is started. + pub fn handle_mode_command(&mut self, modes: &SwitchAndModeHelper) { + if modes.power_cycle_active() { + log::warn!("{}: mode command aborts power cycle recovery", self.name); + // Otherwise, the recovery would restart right away. + self.recovery.recovery_done(); + } + } + + pub fn handle_power_cycle_done(&mut self) { + log::info!("{}: power cycle recovery done", self.name); + // Faults registered while the device was switched off do not count anymore. + self.fault_counter.clear(); + self.recovery.recovery_done(); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Done)); + } + + /// A failed power cycle costs a recovery attempt like any other fault. + pub fn handle_power_cycle_failed( + &mut self, + modes: &mut SwitchAndModeHelper, + restore_mode: DeviceMode, + ) { + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Failed)); + match self.recovery.recovery_failed() { + FaultResponse::Recover => { + log::warn!("{}: power cycle recovery failed, retrying", self.name); + self.start_recovery(modes, restore_mode); + } + FaultResponse::SetFaulty => { + log::error!( + "{}: power cycle recovery failed too often, marking component faulty", + self.name + ); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded)); + self.switch_off_faulty_device(modes); + } + // Ground changed the health during the recovery and is in charge now. + FaultResponse::Ignored => (), + } + } + + fn check_needs_recovery(&mut self, modes: &mut SwitchAndModeHelper) { + if modes.power_cycle_active() || modes.target().is_some() || !self.recovery.needs_recovery() + { + return; + } + if modes.mode() == DeviceMode::Off { + // Nothing to power cycle, the next switch-on is a fresh start anyway. + log::info!("{}: device is off, no recovery required", self.name); + self.recovery.recovery_done(); + return; + } + let restore_mode = modes.mode(); + self.start_recovery(modes, restore_mode); + } + + fn start_recovery( + &mut self, + modes: &mut SwitchAndModeHelper, + restore_mode: DeviceMode, + ) { + log::warn!("{}: starting power cycle recovery", self.name); + modes.start_power_cycle(restore_mode, self.recovery_off_duration); + self.events + .push_back(FdirEvent::Recovery(RecoveryEvent::Started)); + } + + fn switch_off_faulty_device(&mut self, modes: &mut SwitchAndModeHelper) { + // Do not restart an already pending Off transition, which would reset the transition + // state machine before it can finish. + if modes.target() != Some(DeviceMode::Off) { + log::warn!("{}: commanding device off due to fault", self.name); + modes.start_transition(DeviceMode::Off, None); + } + } +} diff --git a/satrs-example/src/main.rs b/satrs-example/src/main.rs index b00ba09..3b8212e 100644 --- a/satrs-example/src/main.rs +++ b/satrs-example/src/main.rs @@ -48,6 +48,7 @@ use crate::{ mod acs; mod ccsds; mod controller; +mod device_fdir; mod device_mode; mod eps; mod event_manager;