MGT fdir
This commit is contained in:
+45
-131
@@ -1,4 +1,4 @@
|
||||
use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir};
|
||||
use satrs::fdir::{FaultCounterStd, RecoveryEvent};
|
||||
use satrs::health::HealthTableMapSync;
|
||||
use satrs::spacepackets::CcsdsPacketIdAndPsc;
|
||||
use satrs_example::{HkHelperSingleSet, TimestampHelper, TmtcQueues};
|
||||
@@ -15,6 +15,7 @@ use types::pcdu::SwitchId;
|
||||
use types::{ComponentId, DeviceMode, HkRequestType, acs::mgm};
|
||||
|
||||
use crate::ccsds::pack_ccsds_tm_packet_for_now;
|
||||
use crate::device_fdir::{DeviceFdir, FdirEvent};
|
||||
use crate::device_mode::{ModeTransitionEvent, SwitchAndModeHelper};
|
||||
use crate::eps::PowerSwitchHelper;
|
||||
|
||||
@@ -34,13 +35,6 @@ pub const Z_LOWBYTE_IDX: usize = 13;
|
||||
pub const SPI_FAULT_THRESHOLD: u32 = 2;
|
||||
pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30);
|
||||
|
||||
// FDIR configuration for power cycle recoveries. The component is marked faulty if it would be
|
||||
// recovered more than RECOVERY_THRESHOLD times before the counter is decremented again.
|
||||
pub const RECOVERY_THRESHOLD: u32 = 2;
|
||||
pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60);
|
||||
/// Time the device stays unpowered during a power cycle, so it can fully discharge.
|
||||
pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500);
|
||||
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub enum MgmId {
|
||||
_0,
|
||||
@@ -48,7 +42,7 @@ pub enum MgmId {
|
||||
}
|
||||
|
||||
impl MgmId {
|
||||
pub const fn str(&self) -> &str {
|
||||
pub const fn str(&self) -> &'static str {
|
||||
match self {
|
||||
MgmId::_0 => "MGM 0",
|
||||
MgmId::_1 => "MGM 1",
|
||||
@@ -191,9 +185,7 @@ pub struct MgmHandlerLis3Mdl {
|
||||
hk_helper: HkHelperSingleSet,
|
||||
switch_and_mode_helper: SwitchAndModeHelper<DeviceMode>,
|
||||
mode_leaf_helper: ModeLeafHelper,
|
||||
spi_fault_counter: FaultCounterStd,
|
||||
fdir: RecoveryFdir<HealthTableMapSync>,
|
||||
recovery_off_duration: Duration,
|
||||
fdir: DeviceFdir,
|
||||
event_tx: mpsc::SyncSender<(ComponentId, mgm::Event)>,
|
||||
}
|
||||
|
||||
@@ -225,14 +217,12 @@ impl MgmHandlerLis3Mdl {
|
||||
stamp_helper: TimestampHelper::default(),
|
||||
hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)),
|
||||
mode_leaf_helper,
|
||||
spi_fault_counter: FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER),
|
||||
fdir: RecoveryFdir::new(
|
||||
id.component_id().into(),
|
||||
fdir: DeviceFdir::new(
|
||||
id.str(),
|
||||
id.component_id(),
|
||||
health_table,
|
||||
RECOVERY_THRESHOLD,
|
||||
RECOVERY_DECREMENT_AFTER,
|
||||
FaultCounterStd::new(SPI_FAULT_THRESHOLD, SPI_FAULT_DECREMENT_AFTER),
|
||||
),
|
||||
recovery_off_duration: RECOVERY_OFF_DURATION,
|
||||
event_tx,
|
||||
}
|
||||
}
|
||||
@@ -253,8 +243,9 @@ impl MgmHandlerLis3Mdl {
|
||||
// Handle assembly related messages.
|
||||
self.handle_mode_leaf_handling();
|
||||
|
||||
self.fdir.periodic_operation();
|
||||
self.check_needs_recovery();
|
||||
self.fdir
|
||||
.periodic_operation(&mut self.switch_and_mode_helper);
|
||||
self.handle_fdir_events();
|
||||
|
||||
// Handle mode transitions first. This also takes care of recoveries required by FDIR.
|
||||
if let Some(event) = self.switch_and_mode_helper.handle_mode_transition() {
|
||||
@@ -266,9 +257,14 @@ impl MgmHandlerLis3Mdl {
|
||||
self.handle_mode_transition_failure(tc_commander)
|
||||
}
|
||||
// The mode did not change for other components, so there is nothing to report.
|
||||
ModeTransitionEvent::PowerCycleDone => self.handle_recovery_done(),
|
||||
ModeTransitionEvent::PowerCycleDone => {
|
||||
self.fdir.handle_power_cycle_done();
|
||||
self.handle_fdir_events();
|
||||
}
|
||||
ModeTransitionEvent::PowerCycleFailed { restore_mode } => {
|
||||
self.handle_recovery_failure(restore_mode)
|
||||
self.fdir
|
||||
.handle_power_cycle_failed(&mut self.switch_and_mode_helper, restore_mode);
|
||||
self.handle_fdir_events();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -453,7 +449,7 @@ impl MgmHandlerLis3Mdl {
|
||||
return;
|
||||
}
|
||||
// Successfull readout, so we can decrement the counter.
|
||||
self.spi_fault_counter.try_decrement();
|
||||
self.fdir.register_success();
|
||||
// Simple scaling to retrieve the float value, assuming the best sensor resolution.
|
||||
let mut mgm_guard = self.shared_mgm_set.lock().unwrap();
|
||||
mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS;
|
||||
@@ -463,122 +459,39 @@ impl MgmHandlerLis3Mdl {
|
||||
drop(mgm_guard);
|
||||
}
|
||||
|
||||
/// Registers one SPI fault with the FDIR fault counter, invalidating the current
|
||||
/// sensor set. If the failure threshold is exceeded, the device is power cycled. If it was
|
||||
/// power cycled too often, the component is marked faulty and commanded off instead.
|
||||
/// Registers one SPI fault with the FDIR, invalidating the current sensor set.
|
||||
fn register_spi_fault(&mut self) {
|
||||
log::warn!("{}: stuck-bus SPI fault", self.id.str());
|
||||
self.shared_mgm_set.lock().unwrap().valid = false;
|
||||
if !self.spi_fault_counter.increment_and_check() {
|
||||
return;
|
||||
}
|
||||
match self.fdir.handle_fault() {
|
||||
FaultResponse::Ignored => {
|
||||
log::info!(
|
||||
"{}: SPI fault threshold exceeded, but component is already faulty, \
|
||||
recovering or externally controlled",
|
||||
self.id.str()
|
||||
);
|
||||
}
|
||||
FaultResponse::Recover => {
|
||||
log::warn!(
|
||||
"{}: SPI fault threshold exceeded, power cycling device",
|
||||
self.id.str()
|
||||
);
|
||||
self.send_event(mgm::Event::SpiFaultThresholdExceeded);
|
||||
self.check_needs_recovery();
|
||||
}
|
||||
FaultResponse::SetFaulty => {
|
||||
log::error!(
|
||||
"{}: SPI fault threshold exceeded after too many recoveries, marking \
|
||||
component faulty",
|
||||
self.id.str()
|
||||
);
|
||||
self.send_event(mgm::Event::SpiFaultThresholdExceeded);
|
||||
self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded));
|
||||
self.switch_off_faulty_device();
|
||||
}
|
||||
self.fdir.register_fault(&mut self.switch_and_mode_helper);
|
||||
self.handle_fdir_events();
|
||||
}
|
||||
|
||||
fn handle_fdir_events(&mut self) {
|
||||
while let Some(event) = self.fdir.next_event() {
|
||||
let event = match event {
|
||||
FdirEvent::FaultThresholdExceeded => mgm::Event::SpiFaultThresholdExceeded,
|
||||
FdirEvent::Recovery(recovery_event) => {
|
||||
// The device is power cycled or switched off.
|
||||
if matches!(
|
||||
recovery_event,
|
||||
RecoveryEvent::Started | RecoveryEvent::ThresholdExceeded
|
||||
) {
|
||||
self.shared_mgm_set.lock().unwrap().valid = false;
|
||||
}
|
||||
mgm::Event::Recovery(recovery_event)
|
||||
}
|
||||
};
|
||||
self.send_event(event);
|
||||
}
|
||||
}
|
||||
|
||||
fn switch_off_faulty_device(&mut self) {
|
||||
// Do not restart an already pending Off transition, which would reset the transition
|
||||
// state machine before it can finish.
|
||||
if self.switch_and_mode_helper.target() != Some(DeviceMode::Off) {
|
||||
log::warn!("{}: commanding device off due to fault", self.id.str());
|
||||
self.start_transition(DeviceMode::Off, None);
|
||||
}
|
||||
}
|
||||
|
||||
/// Starts a power cycle if the health is [satrs::health::HealthState::NeedsRecovery]. The health is set
|
||||
/// either by the FDIR or by ground.
|
||||
fn check_needs_recovery(&mut self) {
|
||||
if self.switch_and_mode_helper.power_cycle_active()
|
||||
|| self.switch_and_mode_helper.target().is_some()
|
||||
|| !self.fdir.needs_recovery()
|
||||
{
|
||||
return;
|
||||
}
|
||||
if self.mode() == DeviceMode::Off {
|
||||
// Nothing to power cycle, the next switch-on is a fresh start anyway.
|
||||
log::info!("{}: device is off, no recovery required", self.id.str());
|
||||
self.fdir.recovery_done();
|
||||
return;
|
||||
}
|
||||
self.start_recovery(self.mode());
|
||||
}
|
||||
|
||||
fn start_recovery(&mut self, restore_mode: DeviceMode) {
|
||||
log::warn!("{}: starting power cycle recovery", self.id.str());
|
||||
self.shared_mgm_set.lock().unwrap().valid = false;
|
||||
self.switch_and_mode_helper
|
||||
.start_power_cycle(restore_mode, self.recovery_off_duration);
|
||||
self.send_event(mgm::Event::Recovery(RecoveryEvent::Started));
|
||||
}
|
||||
|
||||
fn handle_recovery_done(&mut self) {
|
||||
log::info!("{}: power cycle recovery done", self.id.str());
|
||||
// Faults registered while the device was switched off do not count anymore.
|
||||
self.spi_fault_counter.clear();
|
||||
self.fdir.recovery_done();
|
||||
self.send_event(mgm::Event::Recovery(RecoveryEvent::Done));
|
||||
}
|
||||
|
||||
/// A failed power cycle costs a recovery attempt like any other fault.
|
||||
fn handle_recovery_failure(&mut self, restore_mode: DeviceMode) {
|
||||
self.send_event(mgm::Event::Recovery(RecoveryEvent::Failed));
|
||||
match self.fdir.recovery_failed() {
|
||||
FaultResponse::Recover => {
|
||||
log::warn!("{}: power cycle recovery failed, retrying", self.id.str());
|
||||
self.start_recovery(restore_mode);
|
||||
}
|
||||
FaultResponse::SetFaulty => {
|
||||
log::error!(
|
||||
"{}: power cycle recovery failed too often, marking component faulty",
|
||||
self.id.str()
|
||||
);
|
||||
self.send_event(mgm::Event::Recovery(RecoveryEvent::ThresholdExceeded));
|
||||
self.switch_off_faulty_device();
|
||||
}
|
||||
// Ground changed the health during the recovery and is in charge now.
|
||||
FaultResponse::Ignored => (),
|
||||
}
|
||||
}
|
||||
|
||||
/// Mode commands from ground or the parent abort a running recovery.
|
||||
fn handle_mode_command(
|
||||
&mut self,
|
||||
target_mode: DeviceMode,
|
||||
tc_commander: Option<CcsdsPacketIdAndPsc>,
|
||||
) {
|
||||
if self.switch_and_mode_helper.power_cycle_active() {
|
||||
log::warn!(
|
||||
"{}: mode command aborts power cycle recovery",
|
||||
self.id.str()
|
||||
);
|
||||
// Otherwise, the recovery would restart right away.
|
||||
self.fdir.recovery_done();
|
||||
}
|
||||
self.fdir.handle_mode_command(&self.switch_and_mode_helper);
|
||||
self.start_transition(target_mode, tc_commander);
|
||||
}
|
||||
|
||||
@@ -662,6 +575,7 @@ mod tests {
|
||||
pcdu::{SwitchRequest, SwitchState, SwitchStateBinary},
|
||||
};
|
||||
|
||||
use crate::device_fdir::RECOVERY_THRESHOLD;
|
||||
use crate::eps::pcdu::{SharedSwitchSet, SwitchMap, SwitchSet};
|
||||
|
||||
use super::*;
|
||||
@@ -734,7 +648,7 @@ mod tests {
|
||||
health_table.clone(),
|
||||
event_tx,
|
||||
);
|
||||
handler.recovery_off_duration = Duration::ZERO;
|
||||
handler.fdir.recovery_off_duration = Duration::ZERO;
|
||||
Self {
|
||||
assembly_mode_request_tx,
|
||||
mode_report_rx,
|
||||
@@ -1132,7 +1046,7 @@ mod tests {
|
||||
mgm::Event::Recovery(RecoveryEvent::Done),
|
||||
]
|
||||
));
|
||||
assert_eq!(testbench.handler.spi_fault_counter.fault_count(), 0);
|
||||
assert_eq!(testbench.handler.fdir.fault_count(), 0);
|
||||
assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid);
|
||||
}
|
||||
|
||||
@@ -1343,7 +1257,7 @@ mod tests {
|
||||
testbench.handler.periodic_operation();
|
||||
testbench.set_switch_state(SwitchState::Off);
|
||||
// Keep the device off until the parent asked for its mode.
|
||||
testbench.handler.recovery_off_duration = Duration::from_secs(60);
|
||||
testbench.handler.fdir.recovery_off_duration = Duration::from_secs(60);
|
||||
testbench.handler.periodic_operation();
|
||||
assert_eq!(testbench.handler.mode(), DeviceMode::Off);
|
||||
testbench.mode_report_rx.try_iter().for_each(drop);
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
use std::collections::VecDeque;
|
||||
use std::time::Duration;
|
||||
|
||||
use satrs::fdir::{FaultCounterStd, FaultResponse, RecoveryEvent, RecoveryFdir};
|
||||
use satrs::health::{HealthState, HealthTableMapSync};
|
||||
use types::{ComponentId, DeviceMode};
|
||||
|
||||
use crate::device_mode::SwitchAndModeHelper;
|
||||
|
||||
// The component is marked faulty if it would be recovered more than RECOVERY_THRESHOLD times
|
||||
// before the counter is decremented again.
|
||||
pub const RECOVERY_THRESHOLD: u32 = 2;
|
||||
pub const RECOVERY_DECREMENT_AFTER: Duration = Duration::from_secs(60);
|
||||
/// Time the device stays unpowered during a power cycle, so it can fully discharge.
|
||||
pub const RECOVERY_OFF_DURATION: Duration = Duration::from_millis(500);
|
||||
|
||||
/// Generic FDIR events. The device handler maps them to its own event type.
|
||||
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
|
||||
pub enum FdirEvent {
|
||||
FaultThresholdExceeded,
|
||||
Recovery(RecoveryEvent),
|
||||
}
|
||||
|
||||
/// Fault counting and power cycle recovery for device handlers which own the power switch of
|
||||
/// their device.
|
||||
///
|
||||
/// The handler detects faults itself and reports them with [Self::register_fault]. This helper
|
||||
/// then decides whether the device is power cycled, or marked faulty and switched off, and drives
|
||||
/// the [SwitchAndModeHelper] of the handler accordingly. The handler retrieves the resulting
|
||||
/// events with [Self::next_event].
|
||||
pub struct DeviceFdir {
|
||||
name: &'static str,
|
||||
fault_counter: FaultCounterStd,
|
||||
recovery: RecoveryFdir<HealthTableMapSync>,
|
||||
pub recovery_off_duration: Duration,
|
||||
events: VecDeque<FdirEvent>,
|
||||
}
|
||||
|
||||
impl DeviceFdir {
|
||||
pub fn new(
|
||||
name: &'static str,
|
||||
component_id: ComponentId,
|
||||
health_table: HealthTableMapSync,
|
||||
fault_counter: FaultCounterStd,
|
||||
) -> Self {
|
||||
Self {
|
||||
name,
|
||||
fault_counter,
|
||||
recovery: RecoveryFdir::new(
|
||||
component_id.into(),
|
||||
health_table,
|
||||
RECOVERY_THRESHOLD,
|
||||
RECOVERY_DECREMENT_AFTER,
|
||||
),
|
||||
recovery_off_duration: RECOVERY_OFF_DURATION,
|
||||
events: VecDeque::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn fault_count(&self) -> u32 {
|
||||
self.fault_counter.fault_count()
|
||||
}
|
||||
|
||||
pub fn set_health(&mut self, health: HealthState) {
|
||||
self.recovery.set_health(health);
|
||||
}
|
||||
|
||||
pub fn next_event(&mut self) -> Option<FdirEvent> {
|
||||
self.events.pop_front()
|
||||
}
|
||||
|
||||
/// Should be called once per cycle, before the mode transition is handled. Starts a power
|
||||
/// cycle if the health was set to [HealthState::NeedsRecovery] by the FDIR or by ground.
|
||||
pub fn periodic_operation(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
|
||||
self.recovery.periodic_operation();
|
||||
self.check_needs_recovery(modes);
|
||||
}
|
||||
|
||||
pub fn register_success(&mut self) {
|
||||
self.fault_counter.try_decrement();
|
||||
}
|
||||
|
||||
/// If the fault threshold is exceeded, the device is power cycled. If it was power cycled
|
||||
/// too often, the component is marked faulty and commanded off instead.
|
||||
pub fn register_fault(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
|
||||
if !self.fault_counter.increment_and_check() {
|
||||
return;
|
||||
}
|
||||
match self.recovery.handle_fault() {
|
||||
FaultResponse::Ignored => {
|
||||
log::info!(
|
||||
"{}: fault threshold exceeded, but component is already faulty, \
|
||||
recovering or externally controlled",
|
||||
self.name
|
||||
);
|
||||
}
|
||||
FaultResponse::Recover => {
|
||||
log::warn!(
|
||||
"{}: fault threshold exceeded, power cycling device",
|
||||
self.name
|
||||
);
|
||||
self.events.push_back(FdirEvent::FaultThresholdExceeded);
|
||||
self.check_needs_recovery(modes);
|
||||
}
|
||||
FaultResponse::SetFaulty => {
|
||||
log::error!(
|
||||
"{}: fault threshold exceeded after too many recoveries, marking \
|
||||
component faulty",
|
||||
self.name
|
||||
);
|
||||
self.events.push_back(FdirEvent::FaultThresholdExceeded);
|
||||
self.events
|
||||
.push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded));
|
||||
self.switch_off_faulty_device(modes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mode commands from ground or the parent abort a running recovery. Must be called before
|
||||
/// the commanded transition is started.
|
||||
pub fn handle_mode_command(&mut self, modes: &SwitchAndModeHelper<DeviceMode>) {
|
||||
if modes.power_cycle_active() {
|
||||
log::warn!("{}: mode command aborts power cycle recovery", self.name);
|
||||
// Otherwise, the recovery would restart right away.
|
||||
self.recovery.recovery_done();
|
||||
}
|
||||
}
|
||||
|
||||
pub fn handle_power_cycle_done(&mut self) {
|
||||
log::info!("{}: power cycle recovery done", self.name);
|
||||
// Faults registered while the device was switched off do not count anymore.
|
||||
self.fault_counter.clear();
|
||||
self.recovery.recovery_done();
|
||||
self.events
|
||||
.push_back(FdirEvent::Recovery(RecoveryEvent::Done));
|
||||
}
|
||||
|
||||
/// A failed power cycle costs a recovery attempt like any other fault.
|
||||
pub fn handle_power_cycle_failed(
|
||||
&mut self,
|
||||
modes: &mut SwitchAndModeHelper<DeviceMode>,
|
||||
restore_mode: DeviceMode,
|
||||
) {
|
||||
self.events
|
||||
.push_back(FdirEvent::Recovery(RecoveryEvent::Failed));
|
||||
match self.recovery.recovery_failed() {
|
||||
FaultResponse::Recover => {
|
||||
log::warn!("{}: power cycle recovery failed, retrying", self.name);
|
||||
self.start_recovery(modes, restore_mode);
|
||||
}
|
||||
FaultResponse::SetFaulty => {
|
||||
log::error!(
|
||||
"{}: power cycle recovery failed too often, marking component faulty",
|
||||
self.name
|
||||
);
|
||||
self.events
|
||||
.push_back(FdirEvent::Recovery(RecoveryEvent::ThresholdExceeded));
|
||||
self.switch_off_faulty_device(modes);
|
||||
}
|
||||
// Ground changed the health during the recovery and is in charge now.
|
||||
FaultResponse::Ignored => (),
|
||||
}
|
||||
}
|
||||
|
||||
fn check_needs_recovery(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
|
||||
if modes.power_cycle_active() || modes.target().is_some() || !self.recovery.needs_recovery()
|
||||
{
|
||||
return;
|
||||
}
|
||||
if modes.mode() == DeviceMode::Off {
|
||||
// Nothing to power cycle, the next switch-on is a fresh start anyway.
|
||||
log::info!("{}: device is off, no recovery required", self.name);
|
||||
self.recovery.recovery_done();
|
||||
return;
|
||||
}
|
||||
let restore_mode = modes.mode();
|
||||
self.start_recovery(modes, restore_mode);
|
||||
}
|
||||
|
||||
fn start_recovery(
|
||||
&mut self,
|
||||
modes: &mut SwitchAndModeHelper<DeviceMode>,
|
||||
restore_mode: DeviceMode,
|
||||
) {
|
||||
log::warn!("{}: starting power cycle recovery", self.name);
|
||||
modes.start_power_cycle(restore_mode, self.recovery_off_duration);
|
||||
self.events
|
||||
.push_back(FdirEvent::Recovery(RecoveryEvent::Started));
|
||||
}
|
||||
|
||||
fn switch_off_faulty_device(&mut self, modes: &mut SwitchAndModeHelper<DeviceMode>) {
|
||||
// Do not restart an already pending Off transition, which would reset the transition
|
||||
// state machine before it can finish.
|
||||
if modes.target() != Some(DeviceMode::Off) {
|
||||
log::warn!("{}: commanding device off due to fault", self.name);
|
||||
modes.start_transition(DeviceMode::Off, None);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -48,6 +48,7 @@ use crate::{
|
||||
mod acs;
|
||||
mod ccsds;
|
||||
mod controller;
|
||||
mod device_fdir;
|
||||
mod device_mode;
|
||||
mod eps;
|
||||
mod event_manager;
|
||||
|
||||
Reference in New Issue
Block a user