FDIR extensions and improvements for MGM device handler

This commit is contained in:
Robin Mueller
2026-09-23 17:26:28 +02:00
parent 1c6e777d24
commit ea31e89433
9 changed files with 1116 additions and 131 deletions
+3
View File
@@ -11,6 +11,9 @@ and this project adheres to [Semantic Versioning](http://semver.org/).
- Added `hk` module helpers to track whether a single HK set needs regeneration:
`SingleSetHkHelperStd` (`std`), `SingleSetHkHelperEmbassy` (new `embassy-time` feature),
and `SingleSetHkHelperCountdown`, generic over the existing `Countdown` trait.
- Added `fdir::RecoveryFdir` (`std`), which escalates component faults to a power cycle
recovery first and to a faulty component if it has to be recovered too often.
- Added `fdir::RecoveryEvent` for recovery related events.
# [v0.3.0-alpha.3] 2025-11-06
+201
View File
@@ -15,8 +15,45 @@
feature = "embassy-time",
doc = "- [FaultCounterEmbassy]: `embassy_time::Instant`, behind the `embassy-time` feature."
)]
//!
//! [RecoveryFdir] builds on top of that. It decides whether a component is power cycled or
//! marked faulty when one of its fault counters exceeds its threshold, and keeps the health
//! table up to date during the recovery. It follows the FSFW `DeviceHandlerFailureIsolation`.
#![deny(missing_docs)]
#[cfg(feature = "std")]
use crate::health::{HealthState, HealthTableProvider};
/// Events related to the recovery of a component. Components are expected to embed this into
/// their own event type.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
#[cfg_attr(feature = "defmt", derive(defmt::Format))]
pub enum RecoveryEvent {
/// The component health is [crate::health::HealthState::NeedsRecovery] and it is being power cycled.
Started,
/// The power cycle completed and the component is healthy again.
Done,
/// The power cycle failed, the component was marked faulty.
Failed,
/// The component was recovered too often, it was marked faulty.
ThresholdExceeded,
}
/// Outcome of [RecoveryFdir::handle_fault].
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
pub enum FaultResponse {
/// The component is already faulty, recovering or externally controlled, so nothing was
/// changed.
Ignored,
/// The health was set to [crate::health::HealthState::NeedsRecovery]. The component should
/// be power cycled.
Recover,
/// The component was recovered too often and its health was set to [crate::health::HealthState::Faulty].
/// The component should be switched off.
SetFaulty,
}
/// Fault counter backed by [std::time::Instant].
#[cfg(feature = "std")]
#[derive(Debug, Clone)]
@@ -230,6 +267,98 @@ impl FaultCounterEmbassy {
}
}
/// Escalates faults of a component to a power cycle recovery first, and to a faulty
/// component if it has to be recovered too often.
///
/// The component itself runs the power cycle while [Self::needs_recovery] returns `true` and
/// reports the outcome with [Self::recovery_done] or [Self::recovery_failed]. Setting
/// [HealthState::NeedsRecovery] from outside, for example by ground, triggers a recovery as well.
#[cfg(feature = "std")]
#[derive(Debug, Clone)]
pub struct RecoveryFdir<HealthTable: HealthTableProvider> {
id: crate::ComponentId,
health_table: HealthTable,
recovery_counter: FaultCounterStd,
}
#[cfg(feature = "std")]
impl<HealthTable: HealthTableProvider> RecoveryFdir<HealthTable> {
/// Create a new [RecoveryFdir] for component `id`.
///
/// The component is marked faulty when it would be recovered more than `recovery_threshold`
/// times, with the recovery count being decremented every `recovery_decrement_after`.
pub fn new(
id: crate::ComponentId,
health_table: HealthTable,
recovery_threshold: u32,
recovery_decrement_after: core::time::Duration,
) -> Self {
Self {
id,
health_table,
recovery_counter: FaultCounterStd::new(recovery_threshold, recovery_decrement_after),
}
}
/// Health of the component. Absent entries are returned as `None`.
pub fn health(&self) -> Option<HealthState> {
self.health_table.health(self.id)
}
/// Set the health of the component.
pub fn set_health(&mut self, health: HealthState) {
self.health_table.set_health(self.id, health);
}
/// Should be called periodically to decrement the recovery counter.
pub fn periodic_operation(&mut self) {
self.recovery_counter.try_decrement();
}
/// Should be called when a fault counter of the component exceeded its threshold.
pub fn handle_fault(&mut self) -> FaultResponse {
// Ground may have taken manual control, or already given up on this component.
// Autonomous FDIR should not override that decision. An already faulty or recovering
// component must not be escalated again. For example, this would allow a faulty component
// to become healthy again, because the recovery counter was reset when it became faulty.
if matches!(
self.health(),
Some(HealthState::ExternalControl)
| Some(HealthState::PermanentFaulty)
| Some(HealthState::Faulty)
| Some(HealthState::NeedsRecovery)
) {
return FaultResponse::Ignored;
}
if self.recovery_counter.increment_and_check() {
self.set_health(HealthState::Faulty);
return FaultResponse::SetFaulty;
}
self.set_health(HealthState::NeedsRecovery);
FaultResponse::Recover
}
/// The component should be power cycled.
pub fn needs_recovery(&self) -> bool {
self.health() == Some(HealthState::NeedsRecovery)
}
/// The power cycle completed, or was not required. Sets the health back to healthy.
pub fn recovery_done(&mut self) {
// The health might have changed during the recovery.
if self.needs_recovery() {
self.set_health(HealthState::Healthy);
}
}
/// The power cycle failed. Marks the component faulty.
pub fn recovery_failed(&mut self) {
if self.needs_recovery() {
self.set_health(HealthState::Faulty);
}
}
}
#[cfg(all(test, feature = "std"))]
mod tests {
use super::*;
@@ -281,6 +410,78 @@ mod tests {
assert!(!fc.try_decrement());
}
fn recovery_fdir() -> RecoveryFdir<crate::health::HealthTableMapSync> {
RecoveryFdir::new(
1,
crate::health::HealthTableMapSync::default(),
1,
Duration::from_secs(60),
)
}
#[test]
fn first_fault_triggers_recovery() {
let mut fdir = recovery_fdir();
assert_eq!(fdir.handle_fault(), FaultResponse::Recover);
assert!(fdir.needs_recovery());
fdir.recovery_done();
assert_eq!(fdir.health(), Some(HealthState::Healthy));
}
#[test]
fn repeated_recovery_sets_faulty() {
let mut fdir = recovery_fdir();
assert_eq!(fdir.handle_fault(), FaultResponse::Recover);
fdir.recovery_done();
assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty);
assert_eq!(fdir.health(), Some(HealthState::Faulty));
}
#[test]
fn faulty_component_stays_faulty() {
let mut fdir = recovery_fdir();
fdir.handle_fault();
fdir.recovery_done();
assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty);
// The recovery counter was reset, but this must not trigger a new recovery.
assert_eq!(fdir.handle_fault(), FaultResponse::Ignored);
assert_eq!(fdir.health(), Some(HealthState::Faulty));
}
#[test]
fn failed_recovery_sets_faulty() {
let mut fdir = recovery_fdir();
fdir.handle_fault();
fdir.recovery_failed();
assert_eq!(fdir.health(), Some(HealthState::Faulty));
}
#[test]
fn health_is_not_overridden() {
let mut fdir = recovery_fdir();
for health in [
HealthState::ExternalControl,
HealthState::PermanentFaulty,
HealthState::Faulty,
HealthState::NeedsRecovery,
] {
fdir.set_health(health);
assert_eq!(fdir.handle_fault(), FaultResponse::Ignored);
assert_eq!(fdir.health(), Some(health));
}
}
#[test]
fn health_changed_during_recovery_is_kept() {
let mut fdir = recovery_fdir();
fdir.handle_fault();
fdir.set_health(HealthState::ExternalControl);
fdir.recovery_done();
assert_eq!(fdir.health(), Some(HealthState::ExternalControl));
fdir.recovery_failed();
assert_eq!(fdir.health(), Some(HealthState::ExternalControl));
}
#[test]
fn clear_resets_state() {
let mut fc = FaultCounterStd::new(1, Duration::from_secs(60));