feat: add FDIR fault counter and wire it into the MGM device handler
Add satrs::fdir::FaultCounter, an FSFW-style error threshold counter: counts faults, decrements over time when faults stop, and reports when a threshold is exceeded. Two variants for now, mirroring the hk.rs helper pattern: - FaultCounterStd, backed by std::time::Instant - FaultCounterEmbassy, backed by embassy_time::Instant (embassy-time feature), with an optional defmt::Format impl gated on the defmt feature Add satrs::health::HealthTableMapSync::default() for easy construction of a shared, global health table. Wire both into the example app's MGM device handler as the first real FDIR use case: - the minisim MGM model gains SpiFaultMode (None/AllZeros/AllOnes) and a SetSpiFault request, so a stuck SPI bus can be injected for testing, independent of switch state - MgmHandlerLis3Mdl::poll_sensor checks the SPI transfer result: a comm timeout or an all-1s stuck-bus reply (the same pattern the sim already uses for "device off") counts as a fault. Above threshold, the component is marked Faulty in a HealthTableMapSync shared from main.rs. This logic lives in the device handler, not the SPI comm layer, since deciding what a failed transfer means for FDIR is a handler concern. - an all-0s reply is deliberately not treated as a fault, since it collides with a legitimate zero-field reading Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BaKjBjnxaHJ6vzficJcjN4
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
877312d3c3
commit
70345e5e94
@@ -7,7 +7,7 @@ use nexosim::{
|
||||
use satrs_minisim::{
|
||||
acs::{
|
||||
lis3mdl::MgmLis3MdlReply, MgmReplyCommon, MgmReplyProvider, MgmSensorValuesMicroTesla,
|
||||
MgtDipole, MgtHkSet, MgtReply, MGT_GEN_MAGNETIC_FIELD,
|
||||
MgtDipole, MgtHkSet, MgtReply, SpiFaultMode, MGT_GEN_MAGNETIC_FIELD,
|
||||
},
|
||||
SimReply,
|
||||
};
|
||||
@@ -34,6 +34,7 @@ pub struct MagnetometerModel<ReplyProvider: MgmReplyProvider> {
|
||||
#[allow(dead_code)]
|
||||
pub periodicity: Duration,
|
||||
pub external_mag_field: Option<MgmSensorValuesMicroTesla>,
|
||||
pub spi_fault: SpiFaultMode,
|
||||
pub reply_sender: mpsc::Sender<SimReply>,
|
||||
pub phatom: std::marker::PhantomData<ReplyProvider>,
|
||||
}
|
||||
@@ -44,6 +45,7 @@ impl MagnetometerModel<MgmLis3MdlReply> {
|
||||
switch_state: SwitchStateBinary::Off,
|
||||
periodicity,
|
||||
external_mag_field: None,
|
||||
spi_fault: SpiFaultMode::None,
|
||||
reply_sender,
|
||||
phatom: std::marker::PhantomData,
|
||||
}
|
||||
@@ -55,12 +57,21 @@ impl<ReplyProvider: MgmReplyProvider> MagnetometerModel<ReplyProvider> {
|
||||
self.switch_state = switch_state;
|
||||
}
|
||||
|
||||
/// Force (or clear) a stuck-bus SPI fault, for FDIR testing purposes.
|
||||
pub async fn set_spi_fault(&mut self, fault_mode: SpiFaultMode) {
|
||||
self.spi_fault = fault_mode;
|
||||
}
|
||||
|
||||
pub async fn send_sensor_values(&mut self, _: (), scheduler: &mut Context<Self>) {
|
||||
self.reply_sender
|
||||
.send(ReplyProvider::create_mgm_reply(MgmReplyCommon {
|
||||
switch_state: self.switch_state,
|
||||
sensor_values: self.calculate_current_mgm_tuple(current_millis(scheduler.time())),
|
||||
}))
|
||||
.send(ReplyProvider::create_mgm_reply(
|
||||
MgmReplyCommon {
|
||||
switch_state: self.switch_state,
|
||||
sensor_values: self
|
||||
.calculate_current_mgm_tuple(current_millis(scheduler.time())),
|
||||
},
|
||||
self.spi_fault,
|
||||
))
|
||||
.expect("sending MGM sensor values failed");
|
||||
}
|
||||
|
||||
@@ -182,7 +193,7 @@ pub mod tests {
|
||||
use satrs_minisim::{
|
||||
acs::{
|
||||
lis3mdl::{self, MgmLis3MdlReply},
|
||||
MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest,
|
||||
MgmRequestLis3Mdl, MgtDipole, MgtHkSet, MgtReply, MgtRequest, SpiFaultMode,
|
||||
},
|
||||
SerializableSimMsgPayload, SimComponent, SimMessageProvider, SimRequest,
|
||||
};
|
||||
@@ -211,6 +222,37 @@ pub mod tests {
|
||||
assert_eq!(reply.common.sensor_values.z, 0.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mgm_spi_fault_injection_all_ones() {
|
||||
let mut sim_testbench = SimTestbench::new();
|
||||
switch_device_on(&mut sim_testbench, SwitchId::Mgm0);
|
||||
|
||||
let fault_request =
|
||||
SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::SetSpiFault(SpiFaultMode::AllOnes));
|
||||
sim_testbench
|
||||
.send_request(fault_request)
|
||||
.expect("sending MGM fault injection request failed");
|
||||
sim_testbench.handle_sim_requests_time_agnostic();
|
||||
sim_testbench.step().unwrap();
|
||||
|
||||
let data_request = SimRequest::new_with_epoch_time(MgmRequestLis3Mdl::RequestSensorData);
|
||||
sim_testbench
|
||||
.send_request(data_request)
|
||||
.expect("sending MGM request failed");
|
||||
sim_testbench.handle_sim_requests_time_agnostic();
|
||||
sim_testbench.step().unwrap();
|
||||
let sim_reply = sim_testbench
|
||||
.try_receive_next_reply()
|
||||
.expect("no MGM reply received");
|
||||
let reply = MgmLis3MdlReply::from_sim_message(&sim_reply)
|
||||
.expect("failed to deserialize MGM sensor values");
|
||||
// Even though the device is switched on, the injected fault forces a stuck-bus reply.
|
||||
assert_eq!(reply.common.switch_state, SwitchStateBinary::On);
|
||||
assert_eq!(reply.raw.x, -1);
|
||||
assert_eq!(reply.raw.y, -1);
|
||||
assert_eq!(reply.raw.z, -1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_basic_mgm_request_switched_on() {
|
||||
let mut sim_testbench = SimTestbench::new();
|
||||
|
||||
@@ -153,6 +153,17 @@ impl SimController {
|
||||
.process_event(MagnetometerModel::send_sensor_values, (), addr)
|
||||
.expect("event execution error for mgm");
|
||||
}
|
||||
MgmRequestLis3Mdl::SetSpiFault(fault_mode) => {
|
||||
let addr = match mgm_idx {
|
||||
0 => &self.addr_wrapper.mgm_0_addr,
|
||||
1 => &self.addr_wrapper.mgm_1_addr,
|
||||
|
||||
_ => panic!("invalid mgm index"),
|
||||
};
|
||||
self.simulation
|
||||
.process_event(MagnetometerModel::set_spi_fault, fault_mode, addr)
|
||||
.expect("event execution error for mgm");
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -202,12 +202,27 @@ pub mod acs {
|
||||
use super::*;
|
||||
|
||||
pub trait MgmReplyProvider: Send + 'static {
|
||||
fn create_mgm_reply(common: MgmReplyCommon) -> SimReply;
|
||||
fn create_mgm_reply(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> SimReply;
|
||||
}
|
||||
|
||||
/// Fault mode injected on the simulated SPI bus, independent of the switch state.
|
||||
///
|
||||
/// Models the classic symptom of a stuck SPI bus: an undriven MISO line commonly reads
|
||||
/// back as all-1s, a shorted/grounded one as all-0s.
|
||||
#[derive(Debug, Default, Copy, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum SpiFaultMode {
|
||||
#[default]
|
||||
None,
|
||||
AllZeros,
|
||||
AllOnes,
|
||||
}
|
||||
|
||||
#[derive(Debug, Copy, Clone, Serialize, Deserialize)]
|
||||
pub enum MgmRequestLis3Mdl {
|
||||
RequestSensorData,
|
||||
/// Force the raw register reply into a stuck-bus pattern, regardless of switch state.
|
||||
/// Used to test FDIR handling of SPI bus faults.
|
||||
SetSpiFault(SpiFaultMode),
|
||||
}
|
||||
|
||||
impl SerializableSimMsgPayload<SimRequest> for MgmRequestLis3Mdl {
|
||||
@@ -236,6 +251,7 @@ pub mod acs {
|
||||
z: 30.0,
|
||||
};
|
||||
pub const ALL_ONES_SENSOR_VAL: i16 = 0xffff_u16 as i16;
|
||||
pub const ALL_ZEROS_SENSOR_VAL: i16 = 0;
|
||||
|
||||
pub mod lis3mdl {
|
||||
use super::*;
|
||||
@@ -263,7 +279,30 @@ pub mod acs {
|
||||
}
|
||||
|
||||
impl MgmLis3MdlReply {
|
||||
pub fn new(common: MgmReplyCommon) -> Self {
|
||||
pub fn new(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> Self {
|
||||
match fault_mode {
|
||||
SpiFaultMode::AllZeros => {
|
||||
return Self {
|
||||
common,
|
||||
raw: MgmLis3RawValues {
|
||||
x: ALL_ZEROS_SENSOR_VAL,
|
||||
y: ALL_ZEROS_SENSOR_VAL,
|
||||
z: ALL_ZEROS_SENSOR_VAL,
|
||||
},
|
||||
};
|
||||
}
|
||||
SpiFaultMode::AllOnes => {
|
||||
return Self {
|
||||
common,
|
||||
raw: MgmLis3RawValues {
|
||||
x: ALL_ONES_SENSOR_VAL,
|
||||
y: ALL_ONES_SENSOR_VAL,
|
||||
z: ALL_ONES_SENSOR_VAL,
|
||||
},
|
||||
};
|
||||
}
|
||||
SpiFaultMode::None => (),
|
||||
}
|
||||
match common.switch_state {
|
||||
SwitchStateBinary::Off => Self {
|
||||
common,
|
||||
@@ -306,8 +345,8 @@ pub mod acs {
|
||||
}
|
||||
|
||||
impl MgmReplyProvider for MgmLis3MdlReply {
|
||||
fn create_mgm_reply(common: MgmReplyCommon) -> SimReply {
|
||||
SimReply::new(&Self::new(common))
|
||||
fn create_mgm_reply(common: MgmReplyCommon, fault_mode: SpiFaultMode) -> SimReply {
|
||||
SimReply::new(&Self::new(common, fault_mode))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
use satrs::fdir::FaultCounterStd;
|
||||
use satrs::health::{HealthState, HealthTableMapSync, HealthTableProvider};
|
||||
use satrs::spacepackets::CcsdsPacketIdAndPsc;
|
||||
use satrs_example::{HkHelperSingleSet, ModeHelper, TimestampHelper, TmtcQueues};
|
||||
use satrs_minisim::acs::MgmRequestLis3Mdl;
|
||||
@@ -26,6 +28,11 @@ pub const X_LOWBYTE_IDX: usize = 9;
|
||||
pub const Y_LOWBYTE_IDX: usize = 11;
|
||||
pub const Z_LOWBYTE_IDX: usize = 13;
|
||||
|
||||
// FDIR configuration for SPI communication faults (timeouts, stuck bus). Chosen so a handful of
|
||||
// transient errors are tolerated but a persistently faulty bus is caught quickly.
|
||||
pub const SPI_FAULT_THRESHOLD: u32 = 2;
|
||||
pub const SPI_FAULT_DECREMENT_AFTER: Duration = Duration::from_secs(30);
|
||||
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
pub enum MgmId {
|
||||
_0,
|
||||
@@ -63,10 +70,11 @@ pub struct SpiDummyInterface {
|
||||
}
|
||||
|
||||
impl SpiDummyInterface {
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) {
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool {
|
||||
rx[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.x.to_le_bytes());
|
||||
rx[Y_LOWBYTE_IDX..Y_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.y.to_be_bytes());
|
||||
rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2].copy_from_slice(&self.dummy_values.z.to_be_bytes());
|
||||
true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -77,11 +85,12 @@ pub struct TestSpiInterface {
|
||||
}
|
||||
|
||||
impl TestSpiInterface {
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) {
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool {
|
||||
rx[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.x.to_le_bytes());
|
||||
rx[Y_LOWBYTE_IDX..Y_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.y.to_le_bytes());
|
||||
rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2].copy_from_slice(&self.next_mgm_data.z.to_le_bytes());
|
||||
self.call_count += 1;
|
||||
true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -92,7 +101,9 @@ pub struct SpiSimInterface {
|
||||
|
||||
impl SpiSimInterface {
|
||||
// Right now, we only support requesting sensor data and not configuration of the sensor.
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) {
|
||||
// Returns whether a reply was received in time. This is a communication-layer concern only;
|
||||
// it is up to the caller to decide what a failed transfer means for FDIR purposes.
|
||||
fn transfer(&mut self, _tx: &[u8], rx: &mut [u8]) -> bool {
|
||||
let mgm_sensor_request = MgmRequestLis3Mdl::RequestSensorData;
|
||||
if let Err(e) = self
|
||||
.sim_request_tx
|
||||
@@ -110,9 +121,11 @@ impl SpiSimInterface {
|
||||
.copy_from_slice(&sim_reply_lis3.raw.y.to_le_bytes());
|
||||
rx[Z_LOWBYTE_IDX..Z_LOWBYTE_IDX + 2]
|
||||
.copy_from_slice(&sim_reply_lis3.raw.z.to_le_bytes());
|
||||
true
|
||||
}
|
||||
Err(e) => {
|
||||
log::warn!("MGM LIS3 SIM reply timeout: {e}");
|
||||
false
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -126,7 +139,8 @@ pub enum SpiCommunication {
|
||||
}
|
||||
|
||||
impl SpiCommunication {
|
||||
fn transfer(&mut self, tx: &[u8], rx: &mut [u8]) {
|
||||
/// Performs the transfer, returning whether it succeeded.
|
||||
fn transfer(&mut self, tx: &[u8], rx: &mut [u8]) -> bool {
|
||||
match self {
|
||||
SpiCommunication::Dummy(dummy) => dummy.transfer(tx, rx),
|
||||
SpiCommunication::Sim(sim_if) => sim_if.transfer(tx, rx),
|
||||
@@ -159,9 +173,12 @@ pub struct MgmHandlerLis3Mdl {
|
||||
hk_helper: HkHelperSingleSet,
|
||||
mode_helpers: ModeHelper<DeviceMode, TransitionState>,
|
||||
mode_leaf_helper: ModeLeafHelper,
|
||||
spi_fault_counter: FaultCounterStd,
|
||||
health_table: HealthTableMapSync,
|
||||
}
|
||||
|
||||
impl MgmHandlerLis3Mdl {
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn new(
|
||||
id: MgmId,
|
||||
tmtc_queues: TmtcQueues,
|
||||
@@ -170,6 +187,7 @@ impl MgmHandlerLis3Mdl {
|
||||
shared_mgm_set: Arc<Mutex<SensorData>>,
|
||||
mode_leaf_helper: ModeLeafHelper,
|
||||
mode_timeout: Duration,
|
||||
health_table: HealthTableMapSync,
|
||||
) -> Self {
|
||||
Self {
|
||||
id,
|
||||
@@ -182,6 +200,11 @@ impl MgmHandlerLis3Mdl {
|
||||
stamp_helper: TimestampHelper::default(),
|
||||
hk_helper: HkHelperSingleSet::new(false, Duration::from_millis(200)),
|
||||
mode_leaf_helper,
|
||||
spi_fault_counter: FaultCounterStd::new(
|
||||
SPI_FAULT_THRESHOLD,
|
||||
SPI_FAULT_DECREMENT_AFTER,
|
||||
),
|
||||
health_table,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -339,10 +362,14 @@ impl MgmHandlerLis3Mdl {
|
||||
pub fn poll_sensor(&mut self) {
|
||||
// Communicate with the device. This is actually how to read the data from the LIS3 device
|
||||
// SPI interface.
|
||||
self.spi_com.transfer(
|
||||
let transfer_ok = self.spi_com.transfer(
|
||||
&self.buffers.tx_buf[0..NR_OF_DATA_AND_CFG_REGISTERS + 1],
|
||||
&mut self.buffers.rx_buf[0..NR_OF_DATA_AND_CFG_REGISTERS + 1],
|
||||
);
|
||||
if !transfer_ok {
|
||||
self.register_spi_fault();
|
||||
return;
|
||||
}
|
||||
let x_raw = i16::from_le_bytes(
|
||||
self.buffers.rx_buf[X_LOWBYTE_IDX..X_LOWBYTE_IDX + 2]
|
||||
.try_into()
|
||||
@@ -358,6 +385,16 @@ impl MgmHandlerLis3Mdl {
|
||||
.try_into()
|
||||
.unwrap(),
|
||||
);
|
||||
// A stuck-high SPI bus (undriven MISO) reads back as all-1s on every register,
|
||||
// regardless of what was actually requested. This is the pattern this codebase already
|
||||
// uses for "no real device behind the bus" (see the switched-off MGM sim reply). An
|
||||
// all-0s reading is not used here, since it collides with a legitimate zero-field
|
||||
// reading and would cause false positives.
|
||||
if x_raw == -1 && y_raw == -1 && z_raw == -1 {
|
||||
self.register_spi_fault();
|
||||
return;
|
||||
}
|
||||
self.spi_fault_counter.try_decrement();
|
||||
// Simple scaling to retrieve the float value, assuming the best sensor resolution.
|
||||
let mut mgm_guard = self.shared_mgm_set.lock().unwrap();
|
||||
mgm_guard.x = x_raw as f32 * GAUSS_TO_MICROTESLA_FACTOR as f32 * FIELD_LSB_PER_GAUSS_4_SENS;
|
||||
@@ -367,6 +404,36 @@ impl MgmHandlerLis3Mdl {
|
||||
drop(mgm_guard);
|
||||
}
|
||||
|
||||
/// Registers one SPI communication fault (timeout or stuck-bus data) with the FDIR fault
|
||||
/// counter, invalidating the current sensor set. If the failure threshold is exceeded, the
|
||||
/// component is marked faulty in the global health table.
|
||||
fn register_spi_fault(&mut self) {
|
||||
log::warn!("{}: SPI communication fault", self.id.str());
|
||||
self.shared_mgm_set.lock().unwrap().valid = false;
|
||||
if !self.spi_fault_counter.increment_and_check() {
|
||||
return;
|
||||
}
|
||||
// Ground may have taken manual control, or already given up on this component.
|
||||
// Autonomous FDIR should not override that decision.
|
||||
let component_id = self.id.component_id().into();
|
||||
match self.health_table.health(component_id) {
|
||||
Some(HealthState::ExternalControl) | Some(HealthState::PermanentFaulty) => {
|
||||
log::info!(
|
||||
"{}: SPI fault threshold exceeded, but health is externally controlled, \
|
||||
not overriding",
|
||||
self.id.str()
|
||||
);
|
||||
}
|
||||
_ => {
|
||||
log::error!(
|
||||
"{}: SPI fault threshold exceeded, marking component faulty",
|
||||
self.id.str()
|
||||
);
|
||||
self.health_table.set_health(component_id, HealthState::Faulty);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn start_transition(&mut self, target_mode: DeviceMode, _forced: bool) {
|
||||
log::info!("{}: transitioning to mode {:?}", self.id.str(), target_mode);
|
||||
if target_mode == DeviceMode::Off {
|
||||
@@ -507,6 +574,7 @@ mod tests {
|
||||
pub tc_tx: mpsc::SyncSender<CcsdsTcPacketOwned>,
|
||||
pub tm_rx: mpsc::Receiver<CcsdsTmPacketOwned>,
|
||||
pub switch_rx: mpsc::Receiver<GenericMessage<SwitchRequest>>,
|
||||
pub health_table: HealthTableMapSync,
|
||||
pub handler: MgmHandlerLis3Mdl,
|
||||
}
|
||||
|
||||
@@ -526,6 +594,7 @@ mod tests {
|
||||
switch_map.insert(SwitchId::Mgm0, SwitchState::Off);
|
||||
let switch_map = SwitchSet::new(switch_map);
|
||||
let shared_switch_set = SharedSwitchSet::new(Mutex::new(switch_map));
|
||||
let health_table = HealthTableMapSync::default();
|
||||
let handler = MgmHandlerLis3Mdl::new(
|
||||
MgmId::_0,
|
||||
TmtcQueues { tc_rx, tm_tx },
|
||||
@@ -534,18 +603,37 @@ mod tests {
|
||||
shared_mgm_set,
|
||||
mode_leaf_helper,
|
||||
Duration::from_millis(100),
|
||||
health_table.clone(),
|
||||
);
|
||||
Self {
|
||||
assembly_mode_request_tx,
|
||||
mode_report_rx,
|
||||
shared_switch_set,
|
||||
switch_rx,
|
||||
health_table,
|
||||
handler,
|
||||
tm_rx,
|
||||
tc_tx,
|
||||
}
|
||||
}
|
||||
|
||||
/// Switches the MGM to `Normal` mode, completing the power-switch handshake.
|
||||
pub fn switch_to_normal(&mut self) {
|
||||
self.tc_tx
|
||||
.send(create_request_tc(
|
||||
MgmSelect::_0,
|
||||
mgm::request::Request::Mode(ModeRequest::SetMode(DeviceMode::Normal)),
|
||||
))
|
||||
.unwrap();
|
||||
self.handler.periodic_operation();
|
||||
self.shared_switch_set
|
||||
.lock()
|
||||
.unwrap()
|
||||
.set_switch_state(SwitchId::Mgm0, SwitchState::On);
|
||||
self.handler.periodic_operation();
|
||||
assert_eq!(self.handler.mode(), DeviceMode::Normal);
|
||||
}
|
||||
|
||||
pub fn test_spi_interface(&mut self) -> &mut TestSpiInterface {
|
||||
match &mut self.handler.spi_com {
|
||||
SpiCommunication::Dummy(_) | SpiCommunication::Sim(_) => {
|
||||
@@ -767,4 +855,84 @@ mod tests {
|
||||
|
||||
matches!(testbench.tm_rx.try_recv(), Err(TryRecvError::Empty));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_spi_fault_below_threshold_stays_healthy() {
|
||||
let mut testbench = MgmTestbench::new();
|
||||
testbench.switch_to_normal();
|
||||
testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues {
|
||||
x: -1,
|
||||
y: -1,
|
||||
z: -1,
|
||||
};
|
||||
// One stuck-bus reading should not be enough to trip SPI_FAULT_THRESHOLD.
|
||||
testbench.handler.periodic_operation();
|
||||
assert_eq!(
|
||||
testbench.health_table.health(ComponentId::AcsMgm0.into()),
|
||||
None,
|
||||
"component should not be marked faulty yet"
|
||||
);
|
||||
assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_spi_fault_above_threshold_marks_component_faulty() {
|
||||
let mut testbench = MgmTestbench::new();
|
||||
testbench.switch_to_normal();
|
||||
testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues {
|
||||
x: -1,
|
||||
y: -1,
|
||||
z: -1,
|
||||
};
|
||||
// SPI_FAULT_THRESHOLD is exceeded on the (threshold + 1)-th stuck-bus reading.
|
||||
for _ in 0..SPI_FAULT_THRESHOLD + 1 {
|
||||
testbench.handler.periodic_operation();
|
||||
}
|
||||
assert_eq!(
|
||||
testbench.health_table.health(ComponentId::AcsMgm0.into()),
|
||||
Some(HealthState::Faulty)
|
||||
);
|
||||
assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_spi_fault_does_not_override_external_control() {
|
||||
let mut testbench = MgmTestbench::new();
|
||||
testbench.switch_to_normal();
|
||||
testbench
|
||||
.health_table
|
||||
.set_health(ComponentId::AcsMgm0.into(), HealthState::ExternalControl);
|
||||
testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues {
|
||||
x: -1,
|
||||
y: -1,
|
||||
z: -1,
|
||||
};
|
||||
for _ in 0..SPI_FAULT_THRESHOLD + 1 {
|
||||
testbench.handler.periodic_operation();
|
||||
}
|
||||
// Ground took manual control; autonomous FDIR must not override that decision.
|
||||
assert_eq!(
|
||||
testbench.health_table.health(ComponentId::AcsMgm0.into()),
|
||||
Some(HealthState::ExternalControl)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recovering_from_spi_fault_clears_invalid_data_flag() {
|
||||
let mut testbench = MgmTestbench::new();
|
||||
testbench.switch_to_normal();
|
||||
testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues {
|
||||
x: -1,
|
||||
y: -1,
|
||||
z: -1,
|
||||
};
|
||||
testbench.handler.periodic_operation();
|
||||
assert!(!testbench.handler.shared_mgm_set.lock().unwrap().valid);
|
||||
|
||||
// Bus recovers before the threshold is exceeded.
|
||||
testbench.test_spi_interface().next_mgm_data = MgmLis3RawValues::default();
|
||||
testbench.handler.periodic_operation();
|
||||
assert_eq!(testbench.health_table.health(ComponentId::AcsMgm0.into()), None);
|
||||
assert!(testbench.handler.shared_mgm_set.lock().unwrap().valid);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -156,6 +156,9 @@ fn main() {
|
||||
let (switch_request_tx, switch_request_rx) = mpsc::sync_channel(20);
|
||||
let switch_helper = PowerSwitchHelper::new(switch_request_tx, shared_switch_set.clone());
|
||||
|
||||
// Global FDIR health table, shared by all software objects.
|
||||
let health_table = satrs::health::HealthTableMapSync::default();
|
||||
|
||||
let shared_mgm_0_set = Arc::default();
|
||||
let shared_mgm_1_set = Arc::default();
|
||||
let (mgm_0_spi_interface, mgm_1_spi_interface) =
|
||||
@@ -194,6 +197,7 @@ fn main() {
|
||||
report_tx: mgm_0_mode_report_tx,
|
||||
},
|
||||
Duration::from_millis(1000),
|
||||
health_table.clone(),
|
||||
);
|
||||
let mut mgm_1_handler = mgm::MgmHandlerLis3Mdl::new(
|
||||
mgm::MgmId::_1,
|
||||
@@ -209,6 +213,7 @@ fn main() {
|
||||
report_tx: mgm_1_mode_report_tx,
|
||||
},
|
||||
Duration::from_millis(1000),
|
||||
health_table.clone(),
|
||||
);
|
||||
let mut mgm_assembly = mgm_assembly::Assembly::new(
|
||||
mgm_assembly::ParentQueueHelper {
|
||||
|
||||
+1
-1
@@ -72,7 +72,7 @@ alloc = [
|
||||
]
|
||||
serde = ["dep:serde", "spacepackets/serde", "satrs-shared/serde"]
|
||||
crossbeam = ["crossbeam-channel"]
|
||||
defmt = ["dep:defmt", "spacepackets/defmt"]
|
||||
defmt = ["dep:defmt", "spacepackets/defmt", "embassy-time?/defmt"]
|
||||
embassy-time = ["dep:embassy-time"]
|
||||
test_util = []
|
||||
|
||||
|
||||
@@ -0,0 +1,292 @@
|
||||
//! # FDIR (Fault Detection, Isolation and Recovery) helpers
|
||||
//!
|
||||
//! A fault counter tracks a monotonic fault count, decrements it over time when faults stop
|
||||
//! occurring, and reports when a configured failure threshold has been exceeded. This is the
|
||||
//! typical building block used to turn a stream of transient error reports into a single
|
||||
//! "component is faulty" decision without reacting to the first isolated error.
|
||||
//!
|
||||
//! The design follows the FSFW `FaultCounter`:
|
||||
//! <https://egit.irs.uni-stuttgart.de/KSat/fsfw/src/branch/main/src/fsfw/fdir/FaultCounter.h>
|
||||
//!
|
||||
//! Pick a variant based on what clock is available:
|
||||
//!
|
||||
//! - [FaultCounterStd]: `std::time::Instant`, behind the `std` feature.
|
||||
#![cfg_attr(
|
||||
feature = "embassy-time",
|
||||
doc = "- [FaultCounterEmbassy]: `embassy_time::Instant`, behind the `embassy-time` feature."
|
||||
)]
|
||||
#![deny(missing_docs)]
|
||||
|
||||
/// Fault counter backed by [std::time::Instant].
|
||||
#[cfg(feature = "std")]
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct FaultCounterStd {
|
||||
fault_count: u32,
|
||||
failure_threshold: u32,
|
||||
decrement_after: core::time::Duration,
|
||||
last_decrement: Option<std::time::Instant>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl FaultCounterStd {
|
||||
/// Create a new [`FaultCounterStd`].
|
||||
///
|
||||
/// - `failure_threshold`: threshold above which [`Self::above_threshold`] returns `true` and
|
||||
/// resets the internal count.
|
||||
/// - `decrement_after`: minimum duration between automatic decrements performed by
|
||||
/// [`Self::try_decrement`].
|
||||
pub fn new(failure_threshold: u32, decrement_after: core::time::Duration) -> Self {
|
||||
Self {
|
||||
fault_count: 0,
|
||||
failure_threshold,
|
||||
decrement_after,
|
||||
last_decrement: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Current fault count.
|
||||
pub fn fault_count(&self) -> u32 {
|
||||
self.fault_count
|
||||
}
|
||||
|
||||
/// Increase the fault count by `1`.
|
||||
///
|
||||
/// If the counter was previously `0`, this starts a new decrement clock.
|
||||
pub fn increment(&mut self) {
|
||||
if self.fault_count == 0 {
|
||||
self.last_decrement = Some(std::time::Instant::now());
|
||||
}
|
||||
self.fault_count += 1;
|
||||
}
|
||||
|
||||
/// Increase the fault count by `n`.
|
||||
pub fn increment_n(&mut self, n: u32) {
|
||||
for _ in 0..n {
|
||||
self.increment();
|
||||
}
|
||||
}
|
||||
|
||||
fn has_decrement_timedout(&self) -> bool {
|
||||
match self.last_decrement {
|
||||
Some(last_decrement) => last_decrement.elapsed() >= self.decrement_after,
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Decrease the fault count by `1` if the decrement timeout elapsed.
|
||||
///
|
||||
/// Returns `true` if a decrement was performed, `false` otherwise. A decrement is only
|
||||
/// performed when the counter is non-zero and at least `decrement_after` has elapsed since
|
||||
/// the last decrement.
|
||||
pub fn try_decrement(&mut self) -> bool {
|
||||
if self.fault_count == 0 || !self.has_decrement_timedout() {
|
||||
return false;
|
||||
}
|
||||
self.last_decrement = Some(std::time::Instant::now());
|
||||
self.fault_count -= 1;
|
||||
true
|
||||
}
|
||||
|
||||
/// Check whether the counter exceeded the failure threshold.
|
||||
///
|
||||
/// Returns `true` when `fault_count > failure_threshold`. In that case, the counter is reset
|
||||
/// to `0`.
|
||||
pub fn above_threshold(&mut self) -> bool {
|
||||
if self.fault_count > self.failure_threshold {
|
||||
self.fault_count = 0;
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Convenience helper to increment once and immediately check the threshold.
|
||||
pub fn increment_and_check(&mut self) -> bool {
|
||||
self.increment();
|
||||
self.above_threshold()
|
||||
}
|
||||
|
||||
/// Clear the counter and decrement timing state.
|
||||
pub fn clear(&mut self) {
|
||||
self.fault_count = 0;
|
||||
self.last_decrement = None;
|
||||
}
|
||||
|
||||
/// Update the failure threshold used by [`Self::above_threshold`].
|
||||
pub fn set_failure_threshold(&mut self, threshold: u32) {
|
||||
self.failure_threshold = threshold;
|
||||
}
|
||||
|
||||
/// Update the minimum interval between automatic decrements.
|
||||
pub fn set_decrement_after(&mut self, duration: core::time::Duration) {
|
||||
self.decrement_after = duration;
|
||||
}
|
||||
}
|
||||
|
||||
/// Fault counter backed by [embassy_time::Instant].
|
||||
#[cfg(feature = "embassy-time")]
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
#[cfg_attr(feature = "defmt", derive(defmt::Format))]
|
||||
pub struct FaultCounterEmbassy {
|
||||
fault_count: u32,
|
||||
failure_threshold: u32,
|
||||
decrement_after: embassy_time::Duration,
|
||||
last_decrement: Option<embassy_time::Instant>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "embassy-time")]
|
||||
impl FaultCounterEmbassy {
|
||||
/// Create a new [`FaultCounterEmbassy`].
|
||||
///
|
||||
/// - `failure_threshold`: threshold above which [`Self::above_threshold`] returns `true` and
|
||||
/// resets the internal count.
|
||||
/// - `decrement_after`: minimum duration between automatic decrements performed by
|
||||
/// [`Self::try_decrement`].
|
||||
pub fn new(failure_threshold: u32, decrement_after: embassy_time::Duration) -> Self {
|
||||
Self {
|
||||
fault_count: 0,
|
||||
failure_threshold,
|
||||
decrement_after,
|
||||
last_decrement: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Current fault count.
|
||||
pub fn fault_count(&self) -> u32 {
|
||||
self.fault_count
|
||||
}
|
||||
|
||||
/// Increase the fault count by `1`.
|
||||
///
|
||||
/// If the counter was previously `0`, this starts a new decrement clock.
|
||||
pub fn increment(&mut self) {
|
||||
if self.fault_count == 0 {
|
||||
self.last_decrement = Some(embassy_time::Instant::now());
|
||||
}
|
||||
self.fault_count += 1;
|
||||
}
|
||||
|
||||
/// Increase the fault count by `n`.
|
||||
pub fn increment_n(&mut self, n: u32) {
|
||||
for _ in 0..n {
|
||||
self.increment();
|
||||
}
|
||||
}
|
||||
|
||||
fn has_decrement_timedout(&self) -> bool {
|
||||
match self.last_decrement {
|
||||
Some(last_decrement) => {
|
||||
embassy_time::Instant::now().duration_since(last_decrement) >= self.decrement_after
|
||||
}
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Decrease the fault count by `1` if the decrement timeout elapsed.
|
||||
///
|
||||
/// Returns `true` if a decrement was performed, `false` otherwise. A decrement is only
|
||||
/// performed when the counter is non-zero and at least `decrement_after` has elapsed since
|
||||
/// the last decrement.
|
||||
pub fn try_decrement(&mut self) -> bool {
|
||||
if self.fault_count == 0 || !self.has_decrement_timedout() {
|
||||
return false;
|
||||
}
|
||||
self.last_decrement = Some(embassy_time::Instant::now());
|
||||
self.fault_count -= 1;
|
||||
true
|
||||
}
|
||||
|
||||
/// Check whether the counter exceeded the failure threshold.
|
||||
///
|
||||
/// Returns `true` when `fault_count > failure_threshold`. In that case, the counter is reset
|
||||
/// to `0`.
|
||||
pub fn above_threshold(&mut self) -> bool {
|
||||
if self.fault_count > self.failure_threshold {
|
||||
self.fault_count = 0;
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Convenience helper to increment once and immediately check the threshold.
|
||||
pub fn increment_and_check(&mut self) -> bool {
|
||||
self.increment();
|
||||
self.above_threshold()
|
||||
}
|
||||
|
||||
/// Clear the counter and decrement timing state.
|
||||
pub fn clear(&mut self) {
|
||||
self.fault_count = 0;
|
||||
self.last_decrement = None;
|
||||
}
|
||||
|
||||
/// Update the failure threshold used by [`Self::above_threshold`].
|
||||
pub fn set_failure_threshold(&mut self, threshold: u32) {
|
||||
self.failure_threshold = threshold;
|
||||
}
|
||||
|
||||
/// Update the minimum interval between automatic decrements.
|
||||
pub fn set_decrement_after(&mut self, duration: embassy_time::Duration) {
|
||||
self.decrement_after = duration;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "std"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::thread;
|
||||
use std::time::Duration;
|
||||
|
||||
#[test]
|
||||
fn threshold_not_exceeded_below_limit() {
|
||||
let mut fc = FaultCounterStd::new(2, Duration::from_secs(60));
|
||||
assert!(!fc.increment_and_check());
|
||||
assert!(!fc.increment_and_check());
|
||||
assert_eq!(fc.fault_count(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn threshold_exceeded_resets_counter() {
|
||||
let mut fc = FaultCounterStd::new(2, Duration::from_secs(60));
|
||||
fc.increment_n(3);
|
||||
assert!(fc.above_threshold());
|
||||
assert_eq!(fc.fault_count(), 0);
|
||||
assert!(!fc.above_threshold());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decrement_only_after_timeout() {
|
||||
let mut fc = FaultCounterStd::new(5, Duration::from_millis(20));
|
||||
fc.increment();
|
||||
assert!(!fc.try_decrement());
|
||||
thread::sleep(Duration::from_millis(30));
|
||||
assert!(fc.try_decrement());
|
||||
assert_eq!(fc.fault_count(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decrement_noop_when_empty() {
|
||||
let mut fc = FaultCounterStd::new(5, Duration::from_millis(1));
|
||||
thread::sleep(Duration::from_millis(2));
|
||||
assert!(!fc.try_decrement());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn increment_after_empty_resets_decrement_timing() {
|
||||
let mut fc = FaultCounterStd::new(5, Duration::from_millis(20));
|
||||
fc.increment();
|
||||
thread::sleep(Duration::from_millis(30));
|
||||
assert!(fc.try_decrement());
|
||||
// Counter is 0 again, incrementing should require a fresh decrement_after wait.
|
||||
fc.increment();
|
||||
assert!(!fc.try_decrement());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clear_resets_state() {
|
||||
let mut fc = FaultCounterStd::new(1, Duration::from_secs(60));
|
||||
fc.increment_n(2);
|
||||
fc.clear();
|
||||
assert_eq!(fc.fault_count(), 0);
|
||||
assert!(!fc.above_threshold());
|
||||
}
|
||||
}
|
||||
@@ -27,6 +27,15 @@ impl HealthTableMapSync {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl Default for HealthTableMapSync {
|
||||
/// Creates an empty, shared health table. Absent entries are up to the consumer to
|
||||
/// interpret, for example as [HealthState::Healthy] by default.
|
||||
fn default() -> Self {
|
||||
Self::new(hashbrown::HashMap::new())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl HealthTableProvider for HealthTableMapSync {
|
||||
fn health(&self, id: ComponentId) -> Option<HealthState> {
|
||||
|
||||
@@ -15,6 +15,7 @@ pub mod ccsds;
|
||||
pub mod encoding;
|
||||
#[cfg(feature = "std")]
|
||||
pub mod executable;
|
||||
pub mod fdir;
|
||||
pub mod hal;
|
||||
pub mod health;
|
||||
/// Helpers to track when housekeeping sets need to be regenerated.
|
||||
|
||||
Reference in New Issue
Block a user