MGT fdir #292

Open
muellerr wants to merge 1 commits from mgt-fdir into main
8 changed files with 954 additions and 231 deletions
+62 -23
View File
@@ -10,6 +10,9 @@ use serde::{Deserialize, Serialize};
use std::time::Duration;
use types::pcdu::SwitchStateBinary;
/// Time the device needs to answer a command.
const REPLY_DELAY: Duration = Duration::from_millis(10);
/// Simple magnetorquer simulation model.
#[derive(Serialize, Deserialize)]
pub struct MgtModel {
@@ -39,6 +42,9 @@ impl MgtModel {
duration_and_dipole: (Duration, mgt::Dipole),
cx: &Context<Self>,
) {
if self.switch_state != SwitchStateBinary::On {
return;
}
self.torque_dipole = duration_and_dipole.1;
self.torquing = true;
if cx
@@ -48,6 +54,7 @@ impl MgtModel {
log::warn!("torque clearing can only be set for a future time.");
}
self.generate_magnetic_field(()).await;
self.schedule_reply(mgt::Reply::Ack, cx);
}
#[nexosim(schedulable)]
@@ -69,22 +76,22 @@ impl MgtModel {
if self.switch_state != SwitchStateBinary::On {
return;
}
cx.schedule_event(
Duration::from_millis(15),
schedulable!(Self::send_housekeeping_data),
(),
)
.expect("requesting housekeeping data failed")
// The HK is sampled when the command is processed, not when the reply is sent.
let hk = mgt::HkSet {
dipole: self.torque_dipole,
torquing: self.torquing,
};
self.schedule_reply(mgt::Reply::Hk(hk), cx);
}
fn schedule_reply(&self, reply: mgt::Reply, cx: &Context<Self>) {
cx.schedule_event(REPLY_DELAY, schedulable!(Self::send_reply), reply)
.expect("scheduling MGT reply failed")
}
#[nexosim(schedulable)]
async fn send_housekeeping_data(&mut self) {
self.reply
.send(SimReply::from(mgt::Reply::Hk(mgt::HkSet {
dipole: self.torque_dipole,
torquing: self.torquing,
})))
.await;
async fn send_reply(&mut self, reply: mgt::Reply) {
self.reply.send(SimReply::from(reply)).await;
}
fn calc_magnetic_field(&self, _: mgt::Dipole) -> mgm::SensorValuesMicroTesla {
@@ -118,10 +125,17 @@ mod tests {
use crate::{eps::tests::switch_device_on, test_helpers::SimTestbench};
fn decode_reply(sim_reply: SimReply) -> mgt::Reply {
let SimReply::Mgt(frame) = sim_reply else {
panic!("unexpected reply {sim_reply:?}");
};
mgt::Reply::from_frame(&frame).expect("invalid MGT reply frame")
}
fn request_hk(sim_testbench: &mut SimTestbench) -> Option<mgt::HkSet> {
let sim_reply = sim_testbench.request_reply(mgt::Request::RequestHk)?;
let SimReply::Mgt(mgt::Reply::Hk(hk)) = sim_reply else {
panic!("unexpected reply {sim_reply:?}");
let mgt::Reply::Hk(hk) = decode_reply(sim_reply) else {
panic!("unexpected MGT reply");
};
Some(hk)
}
@@ -162,7 +176,11 @@ mod tests {
.send_request(request)
.expect("sending MGT request failed");
sim_testbench.handle_sim_requests_time_agnostic();
sim_testbench.step_until(Duration::from_millis(5)).unwrap();
sim_testbench.step_until(Duration::from_millis(20)).unwrap();
let ack = sim_testbench
.try_receive_next_reply()
.expect("no torque command ack");
assert_eq!(decode_reply(ack), mgt::Reply::Ack);
assert_eq!(
request_hk(&mut sim_testbench),
@@ -183,6 +201,25 @@ mod tests {
);
}
#[test]
fn test_torque_command_not_acked_when_off() {
let mut sim_testbench = SimTestbench::new();
let reply = sim_testbench.request_reply(mgt::Request::ApplyTorque {
duration: Duration::from_millis(100),
dipole: mgt::Dipole { x: 1, y: 2, z: 3 },
});
assert!(reply.is_none());
}
#[test]
fn test_invalid_frame_is_dropped() {
let mut sim_testbench = SimTestbench::new();
switch_device_on(&mut sim_testbench, SwitchId::Mgt);
assert!(sim_testbench
.request_reply(SimRequest::Mgt(vec![0x01, 0x00]))
.is_none());
}
/// Processes the request without stepping, so scheduled events like the torque clearing do
/// not fire.
fn process_without_step(sim_testbench: &mut SimTestbench, request: impl Into<SimRequest>) {
@@ -200,13 +237,15 @@ mod tests {
request: mgm::Request::RequestSensorData,
},
);
let sim_reply = sim_testbench
.try_receive_next_reply()
.expect("no MGM reply received");
let SimReply::Mgm { reply, .. } = sim_reply else {
panic!("unexpected reply {sim_reply:?}");
};
reply.sensor_values
// Skips pending MGT replies, for example torque command acks.
loop {
let sim_reply = sim_testbench
.try_receive_next_reply()
.expect("no MGM reply received");
if let SimReply::Mgm { reply, .. } = sim_reply {
return reply.sensor_values;
}
}
}
fn start_torquing(sim_testbench: &mut SimTestbench, duration: Duration) {
+8 -1
View File
@@ -268,7 +268,14 @@ impl SimController {
}
}
fn handle_mgt_request(&mut self, mgt_request: mgt::Request) {
fn handle_mgt_request(&mut self, frame: Vec<u8>) {
let mgt_request = match mgt::Request::from_frame(&frame) {
Ok(request) => request,
Err(e) => {
log::warn!("dropping invalid MGT frame {frame:02x?}: {e}");
return;
}
};
if MGT_REQ_WIRETAPPING {
log::info!("received MGT request: {mgt_request:?}");
}
+205 -7
View File
@@ -19,8 +19,12 @@ pub enum ComponentId {
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
pub enum SimRequest {
SimCtrl(SimCtrlRequest),
Mgm { id: mgm::Id, request: mgm::Request },
Mgt(mgt::Request),
Mgm {
id: mgm::Id,
request: mgm::Request,
},
/// Raw frame of the MGT serial protocol.
Mgt(Vec<u8>),
Pcdu(PcduRequest),
}
@@ -32,7 +36,7 @@ impl From<SimCtrlRequest> for SimRequest {
impl From<mgt::Request> for SimRequest {
fn from(request: mgt::Request) -> Self {
Self::Mgt(request)
Self::Mgt(request.to_frame())
}
}
@@ -64,8 +68,12 @@ impl SimRequestWithTime {
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
pub enum SimReply {
SimCtrl(SimCtrlReply),
Mgm { id: mgm::Id, reply: mgm::Reply },
Mgt(mgt::Reply),
Mgm {
id: mgm::Id,
reply: mgm::Reply,
},
/// Raw frame of the MGT serial protocol.
Mgt(Vec<u8>),
Pcdu(PcduReply),
}
@@ -88,7 +96,7 @@ impl From<SimCtrlReply> for SimReply {
impl From<mgt::Reply> for SimReply {
fn from(reply: mgt::Reply) -> Self {
Self::Mgt(reply)
Self::Mgt(reply.to_frame())
}
}
@@ -263,11 +271,40 @@ pub mod acs {
}
}
/// Simple serial protocol of the magnetorquer.
///
/// The first byte of each frame is the packet ID. The high bit of the ID is set for replies.
/// All fields are big endian. Every command is answered with exactly one reply, but only
/// if the device is powered. The device drops invalid frames.
///
/// A data link layer is deliberately skipped for simplicity. A real serial link would need
/// framing and error detection, for example COBS encoding and a CRC. Here, the transport
/// always delivers complete and intact frames.
pub mod mgt {
use std::time::Duration;
use serde::{Deserialize, Serialize};
pub mod packet_id {
pub const REQUEST_HK: u8 = 0x01;
/// Payload: dipole (3 x i16), duration in milliseconds (u32).
pub const APPLY_TORQUE: u8 = 0x02;
/// Payload: dipole (3 x i16), torquing flag (u8).
pub const HK: u8 = 0x81;
/// Reply to [APPLY_TORQUE].
pub const ACK: u8 = 0x82;
}
#[derive(Debug, Copy, Clone, PartialEq, Eq, thiserror::Error)]
pub enum FrameError {
#[error("empty frame")]
Empty,
#[error("unknown packet ID {0:#04x}")]
UnknownPacketId(u8),
#[error("invalid length {len} for packet ID {id:#04x}")]
InvalidLength { id: u8, len: usize },
}
// Simple model using i16 values.
#[derive(Default, Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct Dipole {
@@ -276,12 +313,84 @@ pub mod acs {
pub z: i16,
}
impl Dipole {
const LEN: usize = 6;
fn write_to(&self, frame: &mut Vec<u8>) {
frame.extend_from_slice(&self.x.to_be_bytes());
frame.extend_from_slice(&self.y.to_be_bytes());
frame.extend_from_slice(&self.z.to_be_bytes());
}
fn read_from(buf: &[u8]) -> Self {
Self {
x: i16::from_be_bytes([buf[0], buf[1]]),
y: i16::from_be_bytes([buf[2], buf[3]]),
z: i16::from_be_bytes([buf[4], buf[5]]),
}
}
}
/// Checks the frame length and returns the packet ID and the payload.
fn split_frame(
frame: &[u8],
payload_len: impl Fn(u8) -> Option<usize>,
) -> Result<(u8, &[u8]), FrameError> {
let (&id, payload) = frame.split_first().ok_or(FrameError::Empty)?;
let expected_len = payload_len(id).ok_or(FrameError::UnknownPacketId(id))?;
if payload.len() != expected_len {
return Err(FrameError::InvalidLength {
id,
len: frame.len(),
});
}
Ok((id, payload))
}
#[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub enum Request {
ApplyTorque { duration: Duration, dipole: Dipole },
/// The duration has millisecond resolution on the wire.
ApplyTorque {
duration: Duration,
dipole: Dipole,
},
RequestHk,
}
impl Request {
pub fn to_frame(&self) -> Vec<u8> {
match self {
Request::RequestHk => vec![packet_id::REQUEST_HK],
Request::ApplyTorque { duration, dipole } => {
let mut frame = vec![packet_id::APPLY_TORQUE];
dipole.write_to(&mut frame);
let duration_ms = u32::try_from(duration.as_millis()).unwrap_or(u32::MAX);
frame.extend_from_slice(&duration_ms.to_be_bytes());
frame
}
}
}
pub fn from_frame(frame: &[u8]) -> Result<Self, FrameError> {
let (id, payload) = split_frame(frame, |id| match id {
packet_id::REQUEST_HK => Some(0),
packet_id::APPLY_TORQUE => Some(Dipole::LEN + 4),
_ => None,
})?;
Ok(match id {
packet_id::REQUEST_HK => Request::RequestHk,
_ => {
let duration_ms =
u32::from_be_bytes(payload[Dipole::LEN..].try_into().unwrap());
Request::ApplyTorque {
duration: Duration::from_millis(duration_ms.into()),
dipole: Dipole::read_from(payload),
}
}
})
}
}
#[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct HkSet {
pub dipole: Dipole,
@@ -291,6 +400,95 @@ pub mod acs {
#[derive(Debug, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub enum Reply {
Hk(HkSet),
Ack,
}
impl Reply {
pub fn to_frame(&self) -> Vec<u8> {
match self {
Reply::Hk(hk) => {
let mut frame = vec![packet_id::HK];
hk.dipole.write_to(&mut frame);
frame.push(hk.torquing as u8);
frame
}
Reply::Ack => vec![packet_id::ACK],
}
}
pub fn from_frame(frame: &[u8]) -> Result<Self, FrameError> {
let (id, payload) = split_frame(frame, |id| match id {
packet_id::HK => Some(Dipole::LEN + 1),
packet_id::ACK => Some(0),
_ => None,
})?;
Ok(match id {
packet_id::ACK => Reply::Ack,
_ => Reply::Hk(HkSet {
dipole: Dipole::read_from(payload),
torquing: payload[Dipole::LEN] != 0,
}),
})
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_apply_torque_frame() {
let request = Request::ApplyTorque {
duration: Duration::from_millis(0x0102_0304),
dipole: Dipole {
x: -2,
y: 0x0506,
z: 0x0708,
},
};
let frame = request.to_frame();
assert_eq!(
frame,
[0x02, 0xff, 0xfe, 0x05, 0x06, 0x07, 0x08, 0x01, 0x02, 0x03, 0x04]
);
assert_eq!(Request::from_frame(&frame), Ok(request));
}
#[test]
fn test_request_hk_frame() {
assert_eq!(Request::RequestHk.to_frame(), [0x01]);
assert_eq!(Request::from_frame(&[0x01]), Ok(Request::RequestHk));
}
#[test]
fn test_reply_frames() {
let hk = Reply::Hk(HkSet {
dipole: Dipole { x: 1, y: 2, z: 3 },
torquing: true,
});
let frame = hk.to_frame();
assert_eq!(frame, [0x81, 0, 1, 0, 2, 0, 3, 1]);
assert_eq!(Reply::from_frame(&frame), Ok(hk));
assert_eq!(Reply::Ack.to_frame(), [0x82]);
assert_eq!(Reply::from_frame(&[0x82]), Ok(Reply::Ack));
}
#[test]
fn test_invalid_frames() {
assert_eq!(Request::from_frame(&[]), Err(FrameError::Empty));
assert_eq!(
Request::from_frame(&[0x81]),
Err(FrameError::UnknownPacketId(0x81))
);
assert_eq!(
Request::from_frame(&[0x01, 0x00]),
Err(FrameError::InvalidLength { id: 0x01, len: 2 })
);
assert_eq!(
Reply::from_frame(&[0x81, 0, 1]),
Err(FrameError::InvalidLength { id: 0x81, len: 3 })
);
}
}
}
}
+45 -131
View File
@@ -1,4 +1,4 @@
use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir};
use satrs::fdir::{FaultCounterStd, RecoveryEvent};
use satrs::health::HealthTableMapSync;
use satrs::spacepackets::CcsdsPacketIdAndPsc;
use satrs_example::{HkHelperSingleSet, TimestampHelper, TmtcQueues};
@@ -15,6 +15,7 @@ use types::pcdu::SwitchId;
use types::{ComponentId, DeviceMode, HkRequestType, acs::mgm};
use crate::ccsds::pack_ccsds_tm_packet_for_now;
use crate::device_fdir::{DeviceFdir, FdirEvent};
use crate::device_mode::{ModeTransitionEvent, SwitchAndModeHelper};
use crate::eps::PowerSwitchHelper;
@@ -34,13 +35,6 @@ pub const Z_LOWBYTE_IDX: usize = 13;
pub const SPI_FAULT_THRESHOLD: u32 = 2;
pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30);
// FDIR configuration for power cycle recoveries. The component is marked faulty if it would be
// recovered more than RECOVERY_THRESHOLD times before the counter is decremented again.
pub const RECOVERY_THRESHOLD: u32 = 2;
pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60);
/// Time the device stays unpowered during a power cycle, so it can fully discharge.
pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500);
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
pub enum MgmId {
_0,
@@ -48,7 +42,7 @@ pub enum MgmId {
}
impl MgmId {
pub const fn str(&self) -> &str {
pub const fn str(&self) -> &'static str {
match self {
MgmId::_0 => "MGM 0",
MgmId::_1 => "MGM 1",
@@ -191,9 +185,7 @@ pub struct MgmHandlerLis3Mdl {
hk_helper: HkHelperSingleSet,
switch_and_mode_helper: SwitchAndModeHelper<DeviceMode>,
mode_leaf_helper: ModeLeafHelper,
spi_fault_counter: FaultCounterStd,
fdir: RecoveryFdir<HealthTableMapSync>,
recovery_off_duration: Duration,
fdir: DeviceFdir,
event_tx: mpsc::SyncSender<(ComponentId, mgm::Event)>,
}
@@ -225,14 +217,12 @@ impl MgmHandlerLis3Mdl {
stamp_helper: TimestampHelper::default(),
hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)),
mode_leaf_helper,
spi_fault_counter: FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER),
fdir: RecoveryFdir::new(
id.component_id().into(),
fdir: DeviceFdir::new(
id.str(),
id.component_id(),
health_table,
RECOVERY_THRESHOLD,
RECOVERY_DECREMENT_AFTER,
FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER),
),
recovery_off_duration: RECOVERY_OFF_DURATION,
event_tx,
}
}
@@ -253,8 +243,9 @@ impl MgmHandlerLis3Mdl {
// Handle assembly related messages.
self.handle_mode_leaf_handling();
self.fdir.periodic_operation();
self.check_needs_recovery();
self.fdir
.periodic_operation(&mut self.switch_and_mode_helper);
self.handle_fdir_events();
// Handle mode transitions first. This also takes care of recoveries required by FDIR.
if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() {
@@ -266,9 +257,14 @@ impl MgmHandlerLis3Mdl {
self.handle_mode_transition_failure(tc_commander)
}
// The mode did not change for other components, so there is nothing to report.
ModeTransitionEvent::PowerCycleDone => self.handle_recovery_done(),
ModeTransitionEvent::PowerCycleDone => {
self.fdir.handle_power_cycle_done();
self.handle_fdir_events();
}
ModeTransitionEvent::PowerCycleFailed { restore_mode } => {
self.handle_recovery_failure(restore_mode)
self.fdir
.handle_power_cycle_failed(&mut self.switch_and_mode_helper, restore_mode);
self.handle_fdir_events();
}
}
}
@@ -453,7 +449,7 @@ impl MgmHandlerLis3Mdl {
return;
}
// Successfull readout, so we can decrement the counter.
self.spi_fault_counter.try_decrement();
self.fdir.register_success();
// Simple scaling to retrieve the float value, assuming the best sensor resolution.
let mut mgm_guard = self.shared_mgm_set.lock().unwrap();
mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS;
@@ -463,122 +459,39 @@ impl MgmHandlerLis3Mdl {
drop(mgm_guard);
}
/// Registers one SPI fault with the FDIR fault counter, invalidating the current
/// sensor set. If the failure threshold is exceeded, the device is power cycled. If it was
/// power cycled too often, the component is marked faulty and commanded off instead.
/// Registers one SPI fault with the FDIR, invalidating the current sensor set.
fn register_spi_fault(&mut self) {
log::warn!("{}: stuck-bus SPI fault", self.id.str());
self.shared_mgm_set.lock().unwrap().valid = false;
if !self.spi_fault_counter.increment_and_check() {
return;
}
match self.fdir.handle_fault() {
FaultResponse::Ignored => {
log::info!(
"{}: SPI fault threshold exceeded, but component is already faulty, \
recovering or externally controlled",
self.id.str()
);
}
FaultResponse::Recover => {
log::warn!(
"{}: SPI fault threshold exceeded, power cycling device",
self.id.str()
);
self.send_event(mgm::Event::SpiFaultThresholdExceeded);
self.check_needs_recovery();
}
FaultResponse::SetFaulty => {
log::error!(
"{}: SPI fault threshold exceeded after too many recoveries, marking \
component faulty",
self.id.str()
);
self.send_event(mgm::Event::SpiFaultThresholdExceeded);
self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded));
self.switch_off_faulty_device();
}
self.fdir.register_fault(&mut self.switch_and_mode_helper);
self.handle_fdir_events();
}
fn handle_fdir_events(&mut self) {
while let Some(event) = self.fdir.next_event() {
let event = match event {
FdirEvent::FaultThresholdExceeded => mgm::Event::SpiFaultThresholdExceeded,
FdirEvent::Recovery(recovery_event) => {
// The device is power cycled or switched off.
if matches!(
recovery_event,
RecoveryEvent::Started | RecoveryEvent::ThresholdExceeded
) {
self.shared_mgm_set.lock().unwrap().valid = false;
}
mgm::Event::Recovery(recovery_event)
}
};
self.send_event(event);
}
}
fn switch_off_faulty_device(&mut self) {
// Do not restart an already pending Off transition, which would reset the transition
// state machine before it can finish.
if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) {
log::warn!("{}: commanding device off due to fault", self.id.str());
self.start_transition(DeviceMode::Off, None);
}
}
/// Starts a power cycle if the health is [satrs::health::HealthState::NeedsRecovery]. The health is set
/// either by the FDIR or by ground.
fn check_needs_recovery(&mut self) {
if self.switch_and_mode_helper.power_cycle_active()
|| self.switch_and_mode_helper.target().is_some()
|| !self.fdir.needs_recovery()
{
return;
}
if self.mode() == DeviceMode::Off {
// Nothing to power cycle, the next switch-on is a fresh start anyway.
log::info!("{}: device is off, no recovery required", self.id.str());
self.fdir.recovery_done();
return;
}
self.start_recovery(self.mode());
}
fn start_recovery(&mut self, restore_mode: DeviceMode) {
log::warn!("{}: starting power cycle recovery", self.id.str());
self.shared_mgm_set.lock().unwrap().valid = false;
self.switch_and_mode_helper
.start_power_cycle(restore_mode, self.recovery_off_duration);
self.send_event(mgm::Event::Recovery(RecoveryEvent::Started));
}
fn handle_recovery_done(&mut self) {
log::info!("{}: power cycle recovery done", self.id.str());
// Faults registered while the device was switched off do not count anymore.
self.spi_fault_counter.clear();
self.fdir.recovery_done();
self.send_event(mgm::Event::Recovery(RecoveryEvent::Done));
}
/// A failed power cycle costs a recovery attempt like any other fault.
fn handle_recovery_failure(&mut self, restore_mode: DeviceMode) {
self.send_event(mgm::Event::Recovery(RecoveryEvent::Failed));
match self.fdir.recovery_failed() {
FaultResponse::Recover => {
log::warn!("{}: power cycle recovery failed, retrying", self.id.str());
self.start_recovery(restore_mode);
}
FaultResponse::SetFaulty => {
log::error!(
"{}: power cycle recovery failed too often, marking component faulty",
self.id.str()
);
self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded));
self.switch_off_faulty_device();
}
// Ground changed the health during the recovery and is in charge now.
FaultResponse::Ignored => (),
}
}
/// Mode commands from ground or the parent abort a running recovery.
fn handle_mode_command(
&mut self,
target_mode: DeviceMode,
tc_commander: Option<CcsdsPacketIdAndPsc>,
) {
if self.switch_and_mode_helper.power_cycle_active() {
log::warn!(
"{}: mode command aborts power cycle recovery",
self.id.str()
);
// Otherwise, the recovery would restart right away.
self.fdir.recovery_done();
}
self.fdir.handle_mode_command(&self.switch_and_mode_helper);
self.start_transition(target_mode, tc_commander);
}
@@ -662,6 +575,7 @@ mod tests {
pcdu::{SwitchRequest, SwitchState, SwitchStateBinary},
};
use crate::device_fdir::RECOVERY_THRESHOLD;
use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet};
use super::*;
@@ -734,7 +648,7 @@ mod tests {
health_table.clone(),
event_tx,
);
handler.recovery_off_duration = Duration::ZERO;
handler.fdir.recovery_off_duration = Duration::ZERO;
Self {
assembly_mode_request_tx,
mode_report_rx,
@@ -1132,7 +1046,7 @@ mod tests {
mgm::Event::Recovery(RecoveryEvent::Done),
]
));
assert_eq!(testbench.handler.spi_fault_counter.fault_count(), 0);
assert_eq!(testbench.handler.fdir.fault_count(), 0);
assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid);
}
@@ -1343,7 +1257,7 @@ mod tests {
testbench.handler.periodic_operation();
testbench.set_switch_state(SwitchState::Off);
// Keep the device off until the parent asked for its mode.
testbench.handler.recovery_off_duration = Duration::from_secs(60);
testbench.handler.fdir.recovery_off_duration = Duration::from_secs(60);
testbench.handler.periodic_operation();
assert_eq!(testbench.handler.mode(), DeviceMode::Off);
testbench.mode_report_rx.try_iter().for_each(drop);
+415 -67
View File
@@ -2,62 +2,71 @@ use std::collections::VecDeque;
use std::sync::mpsc;
use std::time::{Duration, Instant};
use satrs::fdir::{FaultCounterStd, RecoveryEvent};
use satrs::health::HealthTableMapSync;
use satrs::spacepackets::CcsdsPacketIdAndPsc;
use satrs_example::{HkHelperSingleSet, TmtcQueues};
use satrs_minisim::acs::mgt as sim_mgt;
use satrs_minisim::{SimReply, SimRequestWithTime};
use satrs_minisim::{SimReply, SimRequest, SimRequestWithTime};
use types::acs::mgt::{
self, HkSet,
request::{ModeRequest, Request},
request::{HealthRequest, ModeRequest, Request},
response::{ModeResponse, Response},
};
use types::pcdu::SwitchId;
use types::{ComponentId, DeviceMode, HkRequestType};
use crate::ccsds::pack_ccsds_tm_packet_for_now;
use crate::device_fdir::{DeviceFdir, FdirEvent};
use crate::device_mode::{ModeTransitionEvent, SwitchAndModeHelper};
use crate::eps::PowerSwitchHelper;
// FDIR configuration for a stalled communication link. Chosen so a handful of transient
// timeouts or garbled frames are tolerated but a persistently unresponsive device is caught
// quickly.
pub const COMM_FAULT_THRESHOLD: u32 = 2;
pub const COMM_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30);
/// Interface for ideal device which never fails.
#[derive(Default)]
pub struct DummyInterface {
dipole: sim_mgt::Dipole,
torque_end: Option<Instant>,
hk_requested: bool,
replies: VecDeque<Vec<u8>>,
}
impl DummyInterface {
fn send(&mut self, request: sim_mgt::Request) {
match request {
fn send(&mut self, frame: &[u8]) {
let Ok(request) = sim_mgt::Request::from_frame(frame) else {
return;
};
let reply = match request {
sim_mgt::Request::ApplyTorque { duration, dipole } => {
self.dipole = dipole;
self.torque_end = Some(Instant::now() + duration);
sim_mgt::Reply::Ack
}
sim_mgt::Request::RequestHk => self.hk_requested = true,
}
}
fn try_recv_hk(&mut self) -> Option<sim_mgt::HkSet> {
if !std::mem::take(&mut self.hk_requested) {
return None;
}
let torquing = self.torque_end.is_some_and(|end| Instant::now() < end);
Some(sim_mgt::HkSet {
dipole: if torquing {
self.dipole
} else {
sim_mgt::Dipole::default()
},
torquing,
})
sim_mgt::Request::RequestHk => {
let torquing = self.torque_end.is_some_and(|end| Instant::now() < end);
sim_mgt::Reply::Hk(sim_mgt::HkSet {
dipole: if torquing {
self.dipole
} else {
sim_mgt::Dipole::default()
},
torquing,
})
}
};
self.replies.push_back(reply.to_frame());
}
}
/// Records all requests and returns injected HK replies.
/// Records all sent frames and returns injected reply frames.
#[derive(Default)]
pub struct TestInterface {
pub sent_requests: Vec<sim_mgt::Request>,
pub hk_replies: VecDeque<sim_mgt::HkSet>,
pub sent_frames: Vec<Vec<u8>>,
pub replies: VecDeque<Vec<u8>>,
}
pub struct SimInterface {
@@ -66,19 +75,21 @@ pub struct SimInterface {
}
impl SimInterface {
fn send(&mut self, request: sim_mgt::Request) {
fn send(&mut self, frame: &[u8]) {
if let Err(e) = self
.sim_request_tx
.send(SimRequestWithTime::new_with_epoch_time(request))
.send(SimRequestWithTime::new_with_epoch_time(SimRequest::Mgt(
frame.to_vec(),
)))
{
log::error!("failed to send MGT SIM request: {e}");
}
}
fn try_recv_hk(&mut self) -> Option<sim_mgt::HkSet> {
fn try_recv(&mut self) -> Option<Vec<u8>> {
let sim_reply = self.sim_reply_rx.try_recv().ok()?;
match sim_reply {
SimReply::Mgt(sim_mgt::Reply::Hk(hk)) => Some(hk),
SimReply::Mgt(frame) => Some(frame),
_ => {
log::warn!("unexpected MGT SIM reply: {sim_reply:?}");
None
@@ -87,6 +98,7 @@ impl SimInterface {
}
}
/// Frame based transport to the device. The handler implements the protocol on top of it.
pub enum MgtCommunication {
Dummy(DummyInterface),
Sim(SimInterface),
@@ -95,19 +107,19 @@ pub enum MgtCommunication {
}
impl MgtCommunication {
fn send(&mut self, request: sim_mgt::Request) {
fn send(&mut self, frame: &[u8]) {
match self {
MgtCommunication::Dummy(dummy) => dummy.send(request),
MgtCommunication::Sim(sim) => sim.send(request),
MgtCommunication::Test(test) => test.sent_requests.push(request),
MgtCommunication::Dummy(dummy) => dummy.send(frame),
MgtCommunication::Sim(sim) => sim.send(frame),
MgtCommunication::Test(test) => test.sent_frames.push(frame.to_vec()),
}
}
fn try_recv_hk(&mut self) -> Option<sim_mgt::HkSet> {
fn try_recv(&mut self) -> Option<Vec<u8>> {
match self {
MgtCommunication::Dummy(dummy) => dummy.try_recv_hk(),
MgtCommunication::Sim(sim) => sim.try_recv_hk(),
MgtCommunication::Test(test) => test.hk_replies.pop_front(),
MgtCommunication::Dummy(dummy) => dummy.replies.pop_front(),
MgtCommunication::Sim(sim) => sim.try_recv(),
MgtCommunication::Test(test) => test.replies.pop_front(),
}
}
}
@@ -124,6 +136,10 @@ pub struct ModeLeafHelper {
/// The device is powered through the PCDU and only accepts torque commands in normal mode.
/// In normal mode, the device HK is polled every cycle. The replies arrive asynchronously and
/// are cached as the HK set of the handler.
///
/// Communication is monitored for FDIR: a reply timeout or an undecodable frame counts as a
/// fault, which can trigger a power cycle recovery or mark the device faulty, same as the MGM
/// SPI fault handling.
pub struct MgtHandler {
tmtc_queues: TmtcQueues,
pub com: MgtCommunication,
@@ -131,16 +147,22 @@ pub struct MgtHandler {
hk_helper: HkHelperSingleSet,
switch_and_mode_helper: SwitchAndModeHelper<DeviceMode>,
mode_leaf_helper: ModeLeafHelper,
fdir: DeviceFdir,
/// Set once HK is polled and cleared once a reply arrives. Still set by the time the next
/// poll is due means the previous reply never arrived, which counts as a comm fault.
awaiting_reply: bool,
event_tx: mpsc::SyncSender<mgt::Event>,
}
impl MgtHandler {
#[allow(clippy::too_many_arguments)]
pub fn new(
tmtc_queues: TmtcQueues,
switch_helper: PowerSwitchHelper,
com: MgtCommunication,
mode_leaf_helper: ModeLeafHelper,
mode_timeout: Duration,
health_table: HealthTableMapSync,
event_tx: mpsc::SyncSender<mgt::Event>,
) -> Self {
Self {
@@ -155,6 +177,13 @@ impl MgtHandler {
SwitchId::Mgt,
),
mode_leaf_helper,
fdir: DeviceFdir::new(
"MGT",
ComponentId::AcsMgt,
health_table,
FaultCounterStd::new(COMM_FAULT_THRESHOLD, COMM_FAULT_DECREMENT_AFTER),
),
awaiting_reply: false,
event_tx,
}
}
@@ -168,6 +197,10 @@ impl MgtHandler {
self.handle_telecommands();
self.handle_mode_leaf_handling();
self.fdir
.periodic_operation(&mut self.switch_and_mode_helper);
self.handle_fdir_events();
if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() {
match event {
ModeTransitionEvent::Reached(tc_commander) => {
@@ -176,25 +209,28 @@ impl MgtHandler {
ModeTransitionEvent::Failed(tc_commander) => {
self.handle_mode_transition_failure(tc_commander)
}
// No power cycles are started without FDIR.
ModeTransitionEvent::PowerCycleDone
| ModeTransitionEvent::PowerCycleFailed { .. } => (),
// The mode did not change for other components, so there is nothing to report.
ModeTransitionEvent::PowerCycleDone => {
self.fdir.handle_power_cycle_done();
self.handle_fdir_events();
}
ModeTransitionEvent::PowerCycleFailed { restore_mode } => {
self.fdir
.handle_power_cycle_failed(&mut self.switch_and_mode_helper, restore_mode);
self.handle_fdir_events();
}
}
}
// Process replies received since the last call before deciding whether the previous
// poll was answered in time.
self.handle_replies();
if self.ready_for_commanding() {
self.com.send(sim_mgt::Request::RequestHk);
}
while let Some(hk) = self.com.try_recv_hk() {
self.hk_set = HkSet {
valid: true,
dipole: types::acs::mgt::Dipole {
x: hk.dipole.x,
y: hk.dipole.y,
z: hk.dipole.z,
},
torquing: hk.torquing,
};
self.poll_hk();
} else {
// Not polling, so a reply missed while off must not be flagged as a fault later.
self.awaiting_reply = false;
}
if self.hk_helper.needs_generation() {
@@ -202,6 +238,75 @@ impl MgtHandler {
}
}
fn send_request(&mut self, request: sim_mgt::Request) {
self.com.send(&request.to_frame());
}
/// Polls HK. If the reply to the previous poll never arrived, that is a comm fault.
fn poll_hk(&mut self) {
if self.awaiting_reply {
log::warn!("MGT: no reply to previous poll");
self.register_comm_fault();
}
self.awaiting_reply = true;
self.send_request(sim_mgt::Request::RequestHk);
}
fn handle_replies(&mut self) {
while let Some(frame) = self.com.try_recv() {
match sim_mgt::Reply::from_frame(&frame) {
Ok(sim_mgt::Reply::Hk(hk)) => {
self.hk_set = HkSet {
valid: true,
dipole: types::acs::mgt::Dipole {
x: hk.dipole.x,
y: hk.dipole.y,
z: hk.dipole.z,
},
torquing: hk.torquing,
};
self.register_comm_success();
}
Ok(sim_mgt::Reply::Ack) => self.register_comm_success(),
Err(e) => {
log::warn!("MGT: invalid reply frame {frame:02x?}: {e}");
self.awaiting_reply = false;
self.register_comm_fault();
}
}
}
}
fn register_comm_success(&mut self) {
self.awaiting_reply = false;
self.fdir.register_success();
}
fn register_comm_fault(&mut self) {
self.hk_set.valid = false;
self.fdir.register_fault(&mut self.switch_and_mode_helper);
self.handle_fdir_events();
}
fn handle_fdir_events(&mut self) {
while let Some(event) = self.fdir.next_event() {
let event = match event {
FdirEvent::FaultThresholdExceeded => mgt::Event::CommFaultThresholdExceeded,
FdirEvent::Recovery(recovery_event) => {
// The device is power cycled or switched off.
if matches!(
recovery_event,
RecoveryEvent::Started | RecoveryEvent::ThresholdExceeded
) {
self.hk_set.valid = false;
}
mgt::Event::Recovery(recovery_event)
}
};
self.send_event(event);
}
}
fn ready_for_commanding(&self) -> bool {
self.mode() == DeviceMode::Normal && self.switch_and_mode_helper.target().is_none()
}
@@ -225,10 +330,18 @@ impl MgtHandler {
Request::Ping => self.send_telemetry(Some(tc_id), Response::Ok),
Request::Hk(hk_request) => self.handle_hk_request(tc_id, hk_request),
Request::Mode(ModeRequest::SetMode(mode)) => {
self.start_transition(mode, Some(tc_id))
self.handle_mode_command(mode, Some(tc_id))
}
Request::Mode(ModeRequest::ReadMode) => self
.send_telemetry(Some(tc_id), Response::Mode(ModeResponse::Mode(self.mode()))),
Request::Health(HealthRequest::SetHealth(health_state)) => {
log::info!(
"MGT: setting health to {:?} via ground command",
health_state
);
self.fdir.set_health(health_state);
self.send_telemetry(Some(tc_id), Response::Ok);
}
Request::ApplyTorque { dipole, duration } => {
self.handle_torque_command(tc_id, dipole, duration)
}
@@ -239,12 +352,21 @@ impl MgtHandler {
fn handle_mode_leaf_handling(&mut self) {
while let Ok(request) = self.mode_leaf_helper.request_rx.try_recv() {
match request {
ModeRequest::SetMode(mode) => self.start_transition(mode, None),
ModeRequest::SetMode(mode) => self.handle_mode_command(mode, None),
ModeRequest::ReadMode => self.report_mode_to_parent(),
}
}
}
fn handle_mode_command(
&mut self,
target_mode: DeviceMode,
tc_commander: Option<CcsdsPacketIdAndPsc>,
) {
self.fdir.handle_mode_command(&self.switch_and_mode_helper);
self.start_transition(target_mode, tc_commander);
}
fn handle_hk_request(&mut self, tc_id: CcsdsPacketIdAndPsc, hk_request: HkRequestType) {
match hk_request {
HkRequestType::OneShot => self.send_telemetry(Some(tc_id), Response::Hk(self.hk_set)),
@@ -271,7 +393,7 @@ impl MgtHandler {
self.send_telemetry(Some(tc_id), Response::NotInNormalMode);
return;
}
self.com.send(sim_mgt::Request::ApplyTorque {
self.send_request(sim_mgt::Request::ApplyTorque {
duration,
dipole: sim_mgt::Dipole {
x: dipole.x,
@@ -344,6 +466,7 @@ mod tests {
use std::sync::Mutex;
use arbitrary_int::u11;
use satrs::health::{HealthState, HealthTableProvider};
use satrs::spacepackets::SpacePacketHeader;
use types::{
Apid, Message as _, TcHeader,
@@ -351,10 +474,24 @@ mod tests {
pcdu::{SwitchRequest, SwitchState, SwitchStateBinary},
};
use crate::device_fdir::RECOVERY_THRESHOLD;
use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet};
use super::*;
impl TestInterface {
fn sent_requests(&self) -> Vec<sim_mgt::Request> {
self.sent_frames
.iter()
.map(|frame| sim_mgt::Request::from_frame(frame).unwrap())
.collect()
}
fn push_reply(&mut self, reply: sim_mgt::Reply) {
self.replies.push_back(reply.to_frame());
}
}
struct MgtTestbench {
parent_request_tx: mpsc::SyncSender<ModeRequest>,
parent_report_rx: mpsc::Receiver<ModeResponse>,
@@ -363,6 +500,7 @@ mod tests {
tc_tx: mpsc::SyncSender<CcsdsTcPacketOwned>,
tm_rx: mpsc::Receiver<CcsdsTmPacketOwned>,
event_rx: mpsc::Receiver<mgt::Event>,
health_table: HealthTableMapSync,
handler: MgtHandler,
}
@@ -377,7 +515,8 @@ mod tests {
let mut switch_map = SwitchMap::new();
switch_map.insert(SwitchId::Mgt, SwitchState::Off);
let shared_switch_set = SharedSwitchSet::new(Mutex::new(SwitchSet::new(switch_map)));
let handler = MgtHandler::new(
let health_table = HealthTableMapSync::default();
let mut handler = MgtHandler::new(
TmtcQueues { tc_rx, tm_tx },
PowerSwitchHelper::new(switch_tx, shared_switch_set.clone()),
MgtCommunication::Test(TestInterface::default()),
@@ -386,8 +525,10 @@ mod tests {
report_tx,
},
Duration::from_millis(100),
health_table.clone(),
event_tx,
);
handler.fdir.recovery_off_duration = Duration::ZERO;
Self {
parent_request_tx,
parent_report_rx,
@@ -396,6 +537,7 @@ mod tests {
tc_tx,
tm_rx,
event_rx,
health_table,
handler,
}
}
@@ -435,6 +577,70 @@ mod tests {
_ => panic!("unexpected MGT interface"),
}
}
fn set_switch_state(&self, state: SwitchState) {
self.shared_switch_set
.lock()
.unwrap()
.set_switch_state(SwitchId::Mgt, state);
}
fn health(&self) -> Option<HealthState> {
self.health_table.health(ComponentId::AcsMgt.into())
}
fn drain_events(&self) -> Vec<mgt::Event> {
self.event_rx.try_iter().collect()
}
fn drain_switch_requests(&self) -> Vec<SwitchStateBinary> {
self.switch_rx
.try_iter()
.map(|req| req.target_state)
.collect()
}
/// Drives comm timeout faults until the fault threshold is exceeded once. No replies
/// must be queued on the test interface for this to trigger. Assumes one poll is
/// already outstanding, for example right after [Self::switch_to_normal].
fn exceed_comm_fault_threshold(&mut self) {
for _ in 0..COMM_FAULT_THRESHOLD + 1 {
self.handler.periodic_operation();
}
}
/// Drives a started power cycle recovery to completion, completing both power-switch
/// handshakes.
fn complete_power_cycle(&mut self) {
self.handler.periodic_operation();
self.set_switch_state(SwitchState::Off);
self.handler.periodic_operation();
assert_eq!(self.handler.mode(), DeviceMode::Off);
self.handler.periodic_operation();
assert_eq!(
self.handler.switch_and_mode_helper.target(),
Some(DeviceMode::Normal)
);
self.set_switch_state(SwitchState::On);
self.handler.periodic_operation();
assert_eq!(self.handler.mode(), DeviceMode::Normal);
}
/// Drives recoveries with a permanently stalled link until the component is marked
/// faulty.
fn recover_until_faulty(&mut self) {
self.exceed_comm_fault_threshold();
for _ in 0..RECOVERY_THRESHOLD {
assert_eq!(self.health(), Some(HealthState::NeedsRecovery));
self.complete_power_cycle();
// The last cycle of the power cycle only starts a new poll, so the full
// threshold is needed again to trip.
for _ in 0..COMM_FAULT_THRESHOLD + 1 {
self.handler.periodic_operation();
}
}
assert_eq!(self.health(), Some(HealthState::Faulty));
}
}
#[test]
@@ -442,7 +648,7 @@ mod tests {
let mut testbench = MgtTestbench::new();
testbench.handler.periodic_operation();
assert_eq!(testbench.handler.mode(), DeviceMode::Off);
assert!(testbench.test_interface().sent_requests.is_empty());
assert!(testbench.test_interface().sent_frames.is_empty());
}
#[test]
@@ -504,14 +710,14 @@ mod tests {
});
testbench.handler.periodic_operation();
assert_eq!(testbench.next_response(), Response::NotInNormalMode);
assert!(testbench.test_interface().sent_requests.is_empty());
assert!(testbench.test_interface().sent_frames.is_empty());
}
#[test]
fn test_torque_command_forwarded_in_normal_mode() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.test_interface().sent_requests.clear();
testbench.test_interface().sent_frames.clear();
testbench.send_tc(Request::ApplyTorque {
dipole: mgt::Dipole { x: 1, y: 2, z: 3 },
@@ -520,7 +726,7 @@ mod tests {
testbench.handler.periodic_operation();
assert_eq!(testbench.next_response(), Response::Ok);
assert_eq!(
testbench.test_interface().sent_requests.first(),
testbench.test_interface().sent_requests().first(),
Some(&sim_mgt::Request::ApplyTorque {
duration: Duration::from_millis(100),
dipole: sim_mgt::Dipole { x: 1, y: 2, z: 3 },
@@ -535,17 +741,16 @@ mod tests {
assert!(
testbench
.test_interface()
.sent_requests
.sent_requests()
.contains(&sim_mgt::Request::RequestHk)
);
testbench
.test_interface()
.hk_replies
.push_back(sim_mgt::HkSet {
.push_reply(sim_mgt::Reply::Hk(sim_mgt::HkSet {
dipole: sim_mgt::Dipole { x: 1, y: 2, z: 3 },
torquing: true,
});
}));
testbench.handler.periodic_operation();
testbench.send_tc(Request::Hk(HkRequestType::OneShot));
testbench.handler.periodic_operation();
@@ -565,11 +770,10 @@ mod tests {
testbench.switch_to_normal();
testbench
.test_interface()
.hk_replies
.push_back(sim_mgt::HkSet {
.push_reply(sim_mgt::Reply::Hk(sim_mgt::HkSet {
dipole: sim_mgt::Dipole::default(),
torquing: false,
});
}));
testbench.handler.periodic_operation();
testbench.send_tc(Request::Mode(ModeRequest::SetMode(DeviceMode::Off)));
@@ -593,4 +797,148 @@ mod tests {
testbench.handler.periodic_operation();
assert!(testbench.tm_rx.try_recv().is_err());
}
#[test]
fn test_missing_replies_below_threshold_stay_healthy() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
// One missing reply should not be enough to trip COMM_FAULT_THRESHOLD.
testbench.handler.periodic_operation();
assert_eq!(testbench.health(), None);
assert!(!testbench.handler.hk_set.valid);
}
#[test]
fn test_missing_replies_above_threshold_starts_recovery() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.drain_events();
testbench.exceed_comm_fault_threshold();
assert_eq!(testbench.health(), Some(HealthState::NeedsRecovery));
assert!(!testbench.handler.hk_set.valid);
let events = testbench.drain_events();
assert!(matches!(
events[..],
[
mgt::Event::CommFaultThresholdExceeded,
mgt::Event::Recovery(RecoveryEvent::Started)
]
));
}
#[test]
fn test_invalid_frame_counts_as_comm_fault() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.drain_events();
// Consumed by the outstanding HK poll from switch_to_normal, then two more invalid
// frames to exceed the threshold.
for _ in 0..COMM_FAULT_THRESHOLD + 1 {
testbench.test_interface().replies.push_back(vec![0xff]);
testbench.handler.periodic_operation();
}
assert_eq!(testbench.health(), Some(HealthState::NeedsRecovery));
}
#[test]
fn test_recovery_power_cycles_device() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.drain_events();
testbench.drain_switch_requests();
testbench.exceed_comm_fault_threshold();
testbench.complete_power_cycle();
assert_eq!(testbench.health(), Some(HealthState::Healthy));
assert_eq!(
testbench.drain_switch_requests(),
[SwitchStateBinary::Off, SwitchStateBinary::On]
);
let events = testbench.drain_events();
assert!(matches!(
events[..],
[
mgt::Event::CommFaultThresholdExceeded,
mgt::Event::Recovery(RecoveryEvent::Started),
mgt::Event::Recovery(RecoveryEvent::Done),
]
));
}
#[test]
fn test_repeated_recovery_marks_component_faulty() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.recover_until_faulty();
let events = testbench.drain_events();
assert!(matches!(
events[..],
[
..,
mgt::Event::CommFaultThresholdExceeded,
mgt::Event::Recovery(RecoveryEvent::ThresholdExceeded)
]
));
testbench.drain_switch_requests();
testbench.handler.periodic_operation();
assert_eq!(testbench.drain_switch_requests(), [SwitchStateBinary::Off]);
testbench.set_switch_state(SwitchState::Off);
testbench.handler.periodic_operation();
assert_eq!(testbench.handler.mode(), DeviceMode::Off);
assert_eq!(testbench.health(), Some(HealthState::Faulty));
}
#[test]
fn test_ground_health_override_clears_faulty_state() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.recover_until_faulty();
testbench.set_switch_state(SwitchState::Off);
testbench.handler.periodic_operation();
testbench.send_tc(Request::Health(HealthRequest::SetHealth(
HealthState::Healthy,
)));
testbench.handler.periodic_operation();
assert_eq!(testbench.next_response(), Response::Ok);
assert_eq!(testbench.health(), Some(HealthState::Healthy));
}
#[test]
fn test_mode_command_aborts_recovery() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
testbench.exceed_comm_fault_threshold();
testbench.handler.periodic_operation();
testbench.set_switch_state(SwitchState::Off);
testbench
.parent_request_tx
.send(ModeRequest::SetMode(DeviceMode::Off))
.unwrap();
testbench.handler.periodic_operation();
testbench.handler.periodic_operation();
assert_eq!(testbench.handler.mode(), DeviceMode::Off);
assert_eq!(testbench.handler.switch_and_mode_helper.target(), None);
assert_eq!(testbench.health(), Some(HealthState::Healthy));
}
#[test]
fn test_replies_arriving_in_time_stay_healthy() {
let mut testbench = MgtTestbench::new();
testbench.switch_to_normal();
// A reply for the outstanding poll arrives before the next poll is sent, every cycle.
for _ in 0..COMM_FAULT_THRESHOLD + 5 {
testbench
.test_interface()
.push_reply(sim_mgt::Reply::Hk(sim_mgt::HkSet {
dipole: sim_mgt::Dipole::default(),
torquing: false,
}));
testbench.handler.periodic_operation();
}
assert_eq!(testbench.health(), None);
assert!(testbench.handler.hk_set.valid);
}
}
+200
View File
@@ -0,0 +1,200 @@
use std::collections::VecDeque;
use std::time::Duration;
use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir};
use satrs::health::{HealthState, HealthTableMapSync};
use types::{ComponentId, DeviceMode};
use crate::device_mode::SwitchAndModeHelper;
// The component is marked faulty if it would be recovered more than RECOVERY_THRESHOLD times
// before the counter is decremented again.
pub const RECOVERY_THRESHOLD: u32 = 2;
pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60);
/// Time the device stays unpowered during a power cycle, so it can fully discharge.
pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500);
/// Generic FDIR events. The device handler maps them to its own event type.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
pub enum FdirEvent {
FaultThresholdExceeded,
Recovery(RecoveryEvent),
}
/// Fault counting and power cycle recovery for device handlers which own the power switch of
/// their device.
///
/// The handler detects faults itself and reports them with [Self::register_fault]. This helper
/// then decides whether the device is power cycled, or marked faulty and switched off, and drives
/// the [SwitchAndModeHelper] of the handler accordingly. The handler retrieves the resulting
/// events with [Self::next_event].
pub struct DeviceFdir {
name: &'static str,
fault_counter: FaultCounterStd,
recovery: RecoveryFdir<HealthTableMapSync>,
pub recovery_off_duration: Duration,
events: VecDeque<FdirEvent>,
}
impl DeviceFdir {
pub fn new(
name: &'static str,
component_id: ComponentId,
health_table: HealthTableMapSync,
fault_counter: FaultCounterStd,
) -> Self {
Self {
name,
fault_counter,
recovery: RecoveryFdir::new(
component_id.into(),
health_table,
RECOVERY_THRESHOLD,
RECOVERY_DECREMENT_AFTER,
),
recovery_off_duration: RECOVERY_OFF_DURATION,
events: VecDeque::new(),
}
}
#[cfg(test)]
pub fn fault_count(&self) -> u32 {
self.fault_counter.fault_count()
}
pub fn set_health(&mut self, health: HealthState) {
self.recovery.set_health(health);
}
pub fn next_event(&mut self) -> Option<FdirEvent> {
self.events.pop_front()
}
/// Should be called once per cycle, before the mode transition is handled. Starts a power
/// cycle if the health was set to [HealthState::NeedsRecovery] by the FDIR or by ground.
pub fn periodic_operation(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
self.recovery.periodic_operation();
self.check_needs_recovery(modes);
}
pub fn register_success(&mut self) {
self.fault_counter.try_decrement();
}
/// If the fault threshold is exceeded, the device is power cycled. If it was power cycled
/// too often, the component is marked faulty and commanded off instead.
pub fn register_fault(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
if !self.fault_counter.increment_and_check() {
return;
}
match self.recovery.handle_fault() {
FaultResponse::Ignored => {
log::info!(
"{}: fault threshold exceeded, but component is already faulty, \
recovering or externally controlled",
self.name
);
}
FaultResponse::Recover => {
log::warn!(
"{}: fault threshold exceeded, power cycling device",
self.name
);
self.events.push_back(FdirEvent::FaultThresholdExceeded);
self.check_needs_recovery(modes);
}
FaultResponse::SetFaulty => {
log::error!(
"{}: fault threshold exceeded after too many recoveries, marking \
component faulty",
self.name
);
self.events.push_back(FdirEvent::FaultThresholdExceeded);
self.events
.push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded));
self.switch_off_faulty_device(modes);
}
}
}
/// Mode commands from ground or the parent abort a running recovery. Must be called before
/// the commanded transition is started.
pub fn handle_mode_command(&mut self, modes: &SwitchAndModeHelper<DeviceMode>) {
if modes.power_cycle_active() {
log::warn!("{}: mode command aborts power cycle recovery", self.name);
// Otherwise, the recovery would restart right away.
self.recovery.recovery_done();
}
}
pub fn handle_power_cycle_done(&mut self) {
log::info!("{}: power cycle recovery done", self.name);
// Faults registered while the device was switched off do not count anymore.
self.fault_counter.clear();
self.recovery.recovery_done();
self.events
.push_back(FdirEvent::Recovery(RecoveryEvent::Done));
}
/// A failed power cycle costs a recovery attempt like any other fault.
pub fn handle_power_cycle_failed(
&mut self,
modes: &mut SwitchAndModeHelper<DeviceMode>,
restore_mode: DeviceMode,
) {
self.events
.push_back(FdirEvent::Recovery(RecoveryEvent::Failed));
match self.recovery.recovery_failed() {
FaultResponse::Recover => {
log::warn!("{}: power cycle recovery failed, retrying", self.name);
self.start_recovery(modes, restore_mode);
}
FaultResponse::SetFaulty => {
log::error!(
"{}: power cycle recovery failed too often, marking component faulty",
self.name
);
self.events
.push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded));
self.switch_off_faulty_device(modes);
}
// Ground changed the health during the recovery and is in charge now.
FaultResponse::Ignored => (),
}
}
fn check_needs_recovery(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
if modes.power_cycle_active() || modes.target().is_some() || !self.recovery.needs_recovery()
{
return;
}
if modes.mode() == DeviceMode::Off {
// Nothing to power cycle, the next switch-on is a fresh start anyway.
log::info!("{}: device is off, no recovery required", self.name);
self.recovery.recovery_done();
return;
}
let restore_mode = modes.mode();
self.start_recovery(modes, restore_mode);
}
fn start_recovery(
&mut self,
modes: &mut SwitchAndModeHelper<DeviceMode>,
restore_mode: DeviceMode,
) {
log::warn!("{}: starting power cycle recovery", self.name);
modes.start_power_cycle(restore_mode, self.recovery_off_duration);
self.events
.push_back(FdirEvent::Recovery(RecoveryEvent::Started));
}
fn switch_off_faulty_device(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
// Do not restart an already pending Off transition, which would reset the transition
// state machine before it can finish.
if modes.target() != Some(DeviceMode::Off) {
log::warn!("{}: commanding device off due to fault", self.name);
modes.start_transition(DeviceMode::Off, None);
}
}
}
+2
View File
@@ -48,6 +48,7 @@ use crate::{
mod acs;
mod ccsds;
mod controller;
mod device_fdir;
mod device_mode;
mod eps;
mod event_manager;
@@ -278,6 +279,7 @@ fn main() {
report_tx: mgt_report_tx,
},
Duration::from_millis(1000),
health_table.clone(),
mgt_event_tx,
);
+17 -2
View File
@@ -26,11 +26,19 @@ pub mod request {
ReadMode,
}
#[derive(serde::Serialize, serde::Deserialize, Debug, Clone, Copy, PartialEq, Eq)]
pub enum HealthRequest {
/// Overrides the device's autonomous FDIR health state, for example to clear a `Faulty`
/// state set by the handler after ground has fixed or worked around the underlying issue.
SetHealth(satrs::health::HealthState),
}
#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug)]
pub enum Request {
Ping,
Hk(HkRequestType),
Mode(ModeRequest),
Health(HealthRequest),
/// Only accepted in normal mode.
ApplyTorque {
dipole: Dipole,
@@ -44,6 +52,7 @@ pub mod request {
Request::Ping => crate::MessageType::Verification,
Request::Hk(_) => crate::MessageType::Hk,
Request::Mode(_) => crate::MessageType::Mode,
Request::Health(_) => crate::MessageType::Health,
Request::ApplyTorque { .. } => crate::MessageType::Action,
}
}
@@ -54,8 +63,11 @@ pub mod request {
#[strum_discriminants(derive(num_enum::IntoPrimitive))]
#[repr(u16)]
pub enum Event {
/// A commanded mode transition completed.
/// The communication fault counter exceeded its threshold. Followed by a recovery event.
CommFaultThresholdExceeded,
/// A commanded or autonomous mode transition completed.
ModeChanged(DeviceMode),
Recovery(satrs::fdir::RecoveryEvent),
}
impl crate::Message for Event {
@@ -66,7 +78,10 @@ impl crate::Message for Event {
impl crate::EventId for Event {
fn event_id(&self) -> u16 {
EventDiscriminants::from(self).into()
match self {
Event::Recovery(event) => crate::recovery_event_id(*event),
_ => EventDiscriminants::from(self).into(),
}
}
}