FDIR extensions and improvements for MGM device handler
This commit is contained in:
@@ -11,6 +11,9 @@ and this project adheres to [Semantic Versioning](http://semver.org/).
|
||||
- Added `hk` module helpers to track whether a single HK set needs regeneration:
|
||||
`SingleSetHkHelperStd` (`std`), `SingleSetHkHelperEmbassy` (new `embassy-time` feature),
|
||||
and `SingleSetHkHelperCountdown`, generic over the existing `Countdown` trait.
|
||||
- Added `fdir::RecoveryFdir` (`std`), which escalates component faults to a power cycle
|
||||
recovery first and to a faulty component if it has to be recovered too often.
|
||||
- Added `fdir::RecoveryEvent` for recovery related events.
|
||||
|
||||
# [v0.3.0-alpha.3] 2025-11-06
|
||||
|
||||
|
||||
@@ -15,8 +15,45 @@
|
||||
feature = "embassy-time",
|
||||
doc = "- [FaultCounterEmbassy]: `embassy_time::Instant`, behind the `embassy-time` feature."
|
||||
)]
|
||||
//!
|
||||
//! [RecoveryFdir] builds on top of that. It decides whether a component is power cycled or
|
||||
//! marked faulty when one of its fault counters exceeds its threshold, and keeps the health
|
||||
//! table up to date during the recovery. It follows the FSFW `DeviceHandlerFailureIsolation`.
|
||||
#![deny(missing_docs)]
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
use crate::health::{HealthState, HealthTableProvider};
|
||||
|
||||
/// Events related to the recovery of a component. Components are expected to embed this into
|
||||
/// their own event type.
|
||||
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
|
||||
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
|
||||
#[cfg_attr(feature = "defmt", derive(defmt::Format))]
|
||||
pub enum RecoveryEvent {
|
||||
/// The component health is [crate::health::HealthState::NeedsRecovery] and it is being power cycled.
|
||||
Started,
|
||||
/// The power cycle completed and the component is healthy again.
|
||||
Done,
|
||||
/// The power cycle failed, the component was marked faulty.
|
||||
Failed,
|
||||
/// The component was recovered too often, it was marked faulty.
|
||||
ThresholdExceeded,
|
||||
}
|
||||
|
||||
/// Outcome of [RecoveryFdir::handle_fault].
|
||||
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
|
||||
pub enum FaultResponse {
|
||||
/// The component is already faulty, recovering or externally controlled, so nothing was
|
||||
/// changed.
|
||||
Ignored,
|
||||
/// The health was set to [crate::health::HealthState::NeedsRecovery]. The component should
|
||||
/// be power cycled.
|
||||
Recover,
|
||||
/// The component was recovered too often and its health was set to [crate::health::HealthState::Faulty].
|
||||
/// The component should be switched off.
|
||||
SetFaulty,
|
||||
}
|
||||
|
||||
/// Fault counter backed by [std::time::Instant].
|
||||
#[cfg(feature = "std")]
|
||||
#[derive(Debug, Clone)]
|
||||
@@ -230,6 +267,98 @@ impl FaultCounterEmbassy {
|
||||
}
|
||||
}
|
||||
|
||||
/// Escalates faults of a component to a power cycle recovery first, and to a faulty
|
||||
/// component if it has to be recovered too often.
|
||||
///
|
||||
/// The component itself runs the power cycle while [Self::needs_recovery] returns `true` and
|
||||
/// reports the outcome with [Self::recovery_done] or [Self::recovery_failed]. Setting
|
||||
/// [HealthState::NeedsRecovery] from outside, for example by ground, triggers a recovery as well.
|
||||
#[cfg(feature = "std")]
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RecoveryFdir<HealthTable: HealthTableProvider> {
|
||||
id: crate::ComponentId,
|
||||
health_table: HealthTable,
|
||||
recovery_counter: FaultCounterStd,
|
||||
}
|
||||
|
||||
#[cfg(feature = "std")]
|
||||
impl<HealthTable: HealthTableProvider> RecoveryFdir<HealthTable> {
|
||||
/// Create a new [RecoveryFdir] for component `id`.
|
||||
///
|
||||
/// The component is marked faulty when it would be recovered more than `recovery_threshold`
|
||||
/// times, with the recovery count being decremented every `recovery_decrement_after`.
|
||||
pub fn new(
|
||||
id: crate::ComponentId,
|
||||
health_table: HealthTable,
|
||||
recovery_threshold: u32,
|
||||
recovery_decrement_after: core::time::Duration,
|
||||
) -> Self {
|
||||
Self {
|
||||
id,
|
||||
health_table,
|
||||
recovery_counter: FaultCounterStd::new(recovery_threshold, recovery_decrement_after),
|
||||
}
|
||||
}
|
||||
|
||||
/// Health of the component. Absent entries are returned as `None`.
|
||||
pub fn health(&self) -> Option<HealthState> {
|
||||
self.health_table.health(self.id)
|
||||
}
|
||||
|
||||
/// Set the health of the component.
|
||||
pub fn set_health(&mut self, health: HealthState) {
|
||||
self.health_table.set_health(self.id, health);
|
||||
}
|
||||
|
||||
/// Should be called periodically to decrement the recovery counter.
|
||||
pub fn periodic_operation(&mut self) {
|
||||
self.recovery_counter.try_decrement();
|
||||
}
|
||||
|
||||
/// Should be called when a fault counter of the component exceeded its threshold.
|
||||
pub fn handle_fault(&mut self) -> FaultResponse {
|
||||
// Ground may have taken manual control, or already given up on this component.
|
||||
// Autonomous FDIR should not override that decision. An already faulty or recovering
|
||||
// component must not be escalated again. For example, this would allow a faulty component
|
||||
// to become healthy again, because the recovery counter was reset when it became faulty.
|
||||
if matches!(
|
||||
self.health(),
|
||||
Some(HealthState::ExternalControl)
|
||||
| Some(HealthState::PermanentFaulty)
|
||||
| Some(HealthState::Faulty)
|
||||
| Some(HealthState::NeedsRecovery)
|
||||
) {
|
||||
return FaultResponse::Ignored;
|
||||
}
|
||||
if self.recovery_counter.increment_and_check() {
|
||||
self.set_health(HealthState::Faulty);
|
||||
return FaultResponse::SetFaulty;
|
||||
}
|
||||
self.set_health(HealthState::NeedsRecovery);
|
||||
FaultResponse::Recover
|
||||
}
|
||||
|
||||
/// The component should be power cycled.
|
||||
pub fn needs_recovery(&self) -> bool {
|
||||
self.health() == Some(HealthState::NeedsRecovery)
|
||||
}
|
||||
|
||||
/// The power cycle completed, or was not required. Sets the health back to healthy.
|
||||
pub fn recovery_done(&mut self) {
|
||||
// The health might have changed during the recovery.
|
||||
if self.needs_recovery() {
|
||||
self.set_health(HealthState::Healthy);
|
||||
}
|
||||
}
|
||||
|
||||
/// The power cycle failed. Marks the component faulty.
|
||||
pub fn recovery_failed(&mut self) {
|
||||
if self.needs_recovery() {
|
||||
self.set_health(HealthState::Faulty);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "std"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -281,6 +410,78 @@ mod tests {
|
||||
assert!(!fc.try_decrement());
|
||||
}
|
||||
|
||||
fn recovery_fdir() -> RecoveryFdir<crate::health::HealthTableMapSync> {
|
||||
RecoveryFdir::new(
|
||||
1,
|
||||
crate::health::HealthTableMapSync::default(),
|
||||
1,
|
||||
Duration::from_secs(60),
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn first_fault_triggers_recovery() {
|
||||
let mut fdir = recovery_fdir();
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::Recover);
|
||||
assert!(fdir.needs_recovery());
|
||||
fdir.recovery_done();
|
||||
assert_eq!(fdir.health(), Some(HealthState::Healthy));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_recovery_sets_faulty() {
|
||||
let mut fdir = recovery_fdir();
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::Recover);
|
||||
fdir.recovery_done();
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty);
|
||||
assert_eq!(fdir.health(), Some(HealthState::Faulty));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn faulty_component_stays_faulty() {
|
||||
let mut fdir = recovery_fdir();
|
||||
fdir.handle_fault();
|
||||
fdir.recovery_done();
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::SetFaulty);
|
||||
// The recovery counter was reset, but this must not trigger a new recovery.
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::Ignored);
|
||||
assert_eq!(fdir.health(), Some(HealthState::Faulty));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_recovery_sets_faulty() {
|
||||
let mut fdir = recovery_fdir();
|
||||
fdir.handle_fault();
|
||||
fdir.recovery_failed();
|
||||
assert_eq!(fdir.health(), Some(HealthState::Faulty));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn health_is_not_overridden() {
|
||||
let mut fdir = recovery_fdir();
|
||||
for health in [
|
||||
HealthState::ExternalControl,
|
||||
HealthState::PermanentFaulty,
|
||||
HealthState::Faulty,
|
||||
HealthState::NeedsRecovery,
|
||||
] {
|
||||
fdir.set_health(health);
|
||||
assert_eq!(fdir.handle_fault(), FaultResponse::Ignored);
|
||||
assert_eq!(fdir.health(), Some(health));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn health_changed_during_recovery_is_kept() {
|
||||
let mut fdir = recovery_fdir();
|
||||
fdir.handle_fault();
|
||||
fdir.set_health(HealthState::ExternalControl);
|
||||
fdir.recovery_done();
|
||||
assert_eq!(fdir.health(), Some(HealthState::ExternalControl));
|
||||
fdir.recovery_failed();
|
||||
assert_eq!(fdir.health(), Some(HealthState::ExternalControl));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clear_resets_state() {
|
||||
let mut fc = FaultCounterStd::new(1, Duration::from_secs(60));
|
||||
|
||||
Reference in New Issue
Block a user