Forge documentation
Library referenceRust

forge-health

ANVIL health profiles, lifecycle state machine, and monitoring for the Forge SDK

ANVIL health profiles, lifecycle state machine, and monitoring for the Forge SDK

Package contract

FieldValue
Languagerust
Source version0.2.0
Manifestforge-rs/crates/forge-health/Cargo.toml
Source files11
EvidenceSource reference; registry publication and runtime conformance are separate checks

Import boundary

use forge_health;

Use a source checkout or your verified private registry. Manifest coordinates identify the package; they do not establish that a public registry release exists.

Crate boundary

The following entries are taken from src/lib.rs. Feature conditions in the exact source still apply.

pub mod checkpoint;

pub mod degradation;

pub mod error;

pub mod events;

pub mod latency;

pub mod lifecycle;

pub mod monitoring;

pub mod profile;

pub mod reporting;

pub mod summary;

pub mod prelude;

pub use crate::checkpoint::{Checkpoint, CheckpointStore, InMemoryCheckpointStore};

pub use crate::degradation::{DegradationDetector, DegradationSignal, DegradationThresholds};

pub use crate::error::{ForgeHealthError, ForgeHealthResult};

pub use crate::events::LifecycleEvent;

pub use crate::latency::{LatencyStats, LatencyTracker};

pub use crate::lifecycle::{LifecycleManager, LifecycleState, LifecycleTransition};

pub use crate::monitoring::{HealthMonitor, HealthThresholds};

pub use crate::profile::HealthProfile;

pub use crate::reporting::{HealthReport, HealthStatus};

pub use crate::summary::{AgentHealthSnapshot, RuntimeHealthSummary};

Source reference

Download package reference JSON. Each original source file and generated declaration artifact has its own SHA-256 digest. Function bodies and constant values are omitted from downloads. These are source declaration inventories, not compiler-resolved rustdoc, TypeDoc, DocC, or Dokka output. Private modules can contain public declarations that are not reachable through the package boundary; consult the entry point before importing.

checkpoint.rs

Read declaration text · 8 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Checkpoint {
/// Unique identifier for the operation being checkpointed.

///

/// This should be stable across restarts so that a resumed agent can

/// find the checkpoint for its in-progress task.

pub operation_id: String,
/// The step or phase identifier within the operation.

///

/// Agent-defined; examples: "phase-2", "page-17", "batch-3-of-10".

pub step: String,
/// Arbitrary JSON state saved at this checkpoint.

///

/// The agent is responsible for interpreting this state on resume.

pub state: serde_json::Value,
/// Monotonically increasing sequence number for this operation.

///

/// Each successive checkpoint for the same `operation_id` should use

/// a higher sequence number.

pub sequence: u64,
/// Timestamp when this checkpoint was created.

pub created_at: Timestamp
}

pub fn new(operation_id: &str, step: &str, state: serde_json::Value) -> Self;

pub fn with_sequence(
        operation_id: &str,
        step: &str,
        state: serde_json::Value,
        sequence: u64,
    ) -> Self;

pub trait CheckpointStore {
    /// Saves a checkpoint, replacing any existing checkpoint for the same
    /// `operation_id`.
    ///
    /// # Arguments
    ///
    /// * `checkpoint` - The [`Checkpoint`] to persist.
    fn save(&mut self, checkpoint: &Checkpoint);

    /// Loads the most recent checkpoint for the given operation.
    ///
    /// # Arguments
    ///
    /// * `operation_id` - The operation identifier to look up.
    ///
    /// # Returns
    ///
    /// `Some(Checkpoint)` if a checkpoint exists, `None` otherwise.
    fn load(&self, operation_id: &str) -> Option<Checkpoint>;

    /// Deletes all checkpoints for the given operation.
    ///
    /// # Arguments
    ///
    /// * `operation_id` - The operation identifier to delete.
    ///
    /// # Returns
    ///
    /// `true` if a checkpoint was found and deleted, `false` otherwise.
    fn delete(&mut self, operation_id: &str) -> bool;

    /// Lists all operation IDs that have stored checkpoints.
    ///
    /// # Returns
    ///
    /// A vector of operation ID strings.
    fn list(&self) -> Vec<String>;
}

#[derive(Debug, Clone, Default)]
pub struct InMemoryCheckpointStore {

}

pub fn new() -> Self;

pub fn len(&self) -> usize;

pub fn is_empty(&self) -> bool;

degradation.rs

Read declaration text · 7 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct DegradationThresholds {
/// Maximum acceptable mean latency in microseconds.

/// Exceeding this triggers a `LatencySpike` signal.

pub latency_spike_threshold_us: u64,
/// Maximum acceptable P99 latency in microseconds.

/// Exceeding this triggers a `TailLatencyBlowup` signal.

pub p99_threshold_us: u64,
/// Minimum acceptable success rate as a fraction (0.0 to 1.0).

/// Dropping below this triggers a `SuccessRateDrop` signal.

pub min_success_rate: f64,
/// Maximum acceptable number of recent failures in the window.

/// Exceeding this triggers an `ErrorBurst` signal.

pub max_recent_failures: u64
}

#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub enum DegradationSignal {
    /// Mean latency exceeds the configured threshold.
    LatencySpike {
        /// Observed mean latency in microseconds.
        mean_latency_us: u64,
        /// Configured threshold in microseconds.
        threshold_us: u64,
    },

    /// P99 latency exceeds the configured threshold.
    TailLatencyBlowup {
        /// Observed P99 latency in microseconds.
        p99_latency_us: u64,
        /// Configured threshold in microseconds.
        threshold_us: u64,
    },

    /// Success rate has dropped below the minimum threshold.
    SuccessRateDrop {
        /// Observed success rate (0.0 to 1.0).
        success_rate: f64,
        /// Configured minimum success rate.
        min_threshold: f64,
    },

    /// Too many failures in the recent window.
    ErrorBurst {
        /// Number of failures in the current window.
        failure_count: u64,
        /// Configured maximum allowed failures.
        max_allowed: u64,
    },
}

pub fn description(&self) -> String;

#[derive(Debug, Clone)]
pub struct DegradationDetector {

}

pub fn new(thresholds: DegradationThresholds) -> Self;

pub fn thresholds(&self) -> &DegradationThresholds;

pub fn detect(&self, stats: &LatencyStats) -> Vec<DegradationSignal>;

error.rs

Read declaration text · 2 declaration entries

#[derive(Debug, Error)]
pub enum ForgeHealthError {
    /// An invalid lifecycle state transition was attempted.
    ///
    /// The ANVIL lifecycle state machine defines exactly which transitions are
    /// valid from each state. This error is returned when a transition is
    /// attempted that violates the state machine rules.
    ///
    /// See ANVIL Spec section 13.2 -- Lifecycle State Machine
    /// for the complete transition table.
    #[error("invalid lifecycle transition from {from} to {to}: {reason} (see ANVIL Spec section 5.1, Appendix B)")]
    InvalidTransition {
        /// The current lifecycle state.
        from: LifecycleState,
        /// The target state that was attempted.
        to: LifecycleState,
        /// A human-readable explanation of why this transition is invalid.
        reason: String,
    },

    /// A lifecycle operation was attempted on a state that does not support it.
    ///
    /// For example, attempting to resume an agent that is in the Terminated
    /// state, or performing any operation on a terminal state.
    #[error("invalid operation on lifecycle state {state}: {reason}")]
    InvalidState {
        /// The current lifecycle state where the operation was attempted.
        state: LifecycleState,
        /// A human-readable explanation of why this operation is invalid.
        reason: String,
    },

    /// A health monitoring operation failed.
    ///
    /// This covers failures in health check evaluation, threshold configuration,
    /// or profile update operations.
    #[error("health monitor error: {reason}")]
    MonitorError {
        /// A human-readable explanation of what went wrong.
        reason: String,
    },
}

pub type ForgeHealthResult<T> = Result<T, ForgeHealthError>;

events.rs

Read declaration text · 4 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LifecycleEvent {
/// The lifecycle transition that triggered this event.

pub transition: LifecycleTransition,
/// The agent's DID, if identity is bound.

///

/// This is `None` for agents running in legacy mode (without OAS identity).

/// When present, the DID is included in telemetry spans and audit trail

/// entries.

pub agent_did: Option<String>,
/// Optional metadata attached to this event.

///

/// Common uses:

/// - Error details when transitioning to the Error state

/// - Initialization parameters when transitioning to Initializing

/// - Shutdown reason when transitioning to Terminated

pub metadata: Option<serde_json::Value>
}

pub fn new(transition: LifecycleTransition) -> Self;

pub fn with_agent_did(transition: LifecycleTransition, agent_did: String) -> Self;

pub fn with_metadata(mut self, metadata: serde_json::Value) -> Self;

latency.rs

Read declaration text · 11 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LatencyStats {
/// Total number of operations in the window.

pub total_count: u64,
/// Number of successful operations in the window.

pub success_count: u64,
/// Number of failed operations in the window.

pub failure_count: u64,
/// Success rate as a fraction (0.0 to 1.0). Zero if no operations recorded.

pub success_rate: f64,
/// Mean latency in microseconds across all operations in the window.

/// Zero if no operations recorded.

pub mean_latency_us: f64,
/// P50 (median) latency in microseconds. Zero if no operations recorded.

pub p50_latency_us: u64,
/// P95 latency in microseconds. Zero if no operations recorded.

pub p95_latency_us: u64,
/// P99 latency in microseconds. Zero if no operations recorded.

pub p99_latency_us: u64,
/// Maximum latency observed in the window in microseconds.

pub max_latency_us: u64,
/// Minimum latency observed in the window in microseconds.

/// Zero if no operations recorded.

pub min_latency_us: u64
}

pub fn empty() -> Self;

#[derive(Debug, Clone)]
pub struct LatencyTracker {

}

pub fn new(window_size: usize) -> Self;

pub fn record_success(&mut self, latency_us: u64);

pub fn record_failure(&mut self, latency_us: u64);

pub fn stats(&self) -> LatencyStats;

pub fn len(&self) -> usize;

pub fn is_empty(&self) -> bool;

pub fn capacity(&self) -> usize;

pub fn clear(&mut self);

lib.rs

Read declaration text · 21 declaration entries

pub mod checkpoint;

pub mod degradation;

pub mod error;

pub mod events;

pub mod latency;

pub mod lifecycle;

pub mod monitoring;

pub mod profile;

pub mod reporting;

pub mod summary;

pub mod prelude;

pub use crate::checkpoint::{Checkpoint, CheckpointStore, InMemoryCheckpointStore};

pub use crate::degradation::{DegradationDetector, DegradationSignal, DegradationThresholds};

pub use crate::error::{ForgeHealthError, ForgeHealthResult};

pub use crate::events::LifecycleEvent;

pub use crate::latency::{LatencyStats, LatencyTracker};

pub use crate::lifecycle::{LifecycleManager, LifecycleState, LifecycleTransition};

pub use crate::monitoring::{HealthMonitor, HealthThresholds};

pub use crate::profile::HealthProfile;

pub use crate::reporting::{HealthReport, HealthStatus};

pub use crate::summary::{AgentHealthSnapshot, RuntimeHealthSummary};

lifecycle.rs

Read declaration text · 12 declaration entries

#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
pub enum LifecycleState {
    /// Agent is loading configuration, identity, and capabilities.
    ///
    /// This is the initial state for all agents. During initialization, the
    /// agent loads its OAS identity, Arsenal ACT, provider configuration, and
    /// tool definitions. Valid transitions are to Ready (on success), Error
    /// (on failure), or Terminated (abort).
    Initializing,

    /// Agent has completed initialization and is ready to accept work.
    ///
    /// The agent's identity, configuration, and tools have been loaded
    /// successfully. Valid transitions are to Running (start processing)
    /// or Terminated (shutdown before starting).
    Ready,

    /// Agent is actively processing requests.
    ///
    /// This is the primary operational state. The agent can process tool loops,
    /// generate text, and handle messages. Valid transitions are to Ready
    /// (return to idle), Paused (pause), Error (fatal failure), or Terminated
    /// (graceful shutdown).
    Running,

    /// Agent is temporarily paused and not processing requests.
    ///
    /// A paused agent retains its state and can resume. Valid transitions
    /// are to Running (resume) or Terminated (shutdown while paused).
    Paused,

    /// Agent encountered a fatal error.
    ///
    /// An agent in the Error state can attempt recovery by transitioning to
    /// Ready (after re-initialization), or give up by transitioning to
    /// Terminated.
    Error,

    /// Agent has completed shutdown. This is a **terminal state** with no
    /// outgoing transitions.
    ///
    /// Once an agent reaches Terminated, it cannot be reused. A new agent
    /// instance must be created.
    Terminated,
}

pub fn valid_transitions(&self) -> &'static [LifecycleState];

pub fn is_terminal(&self) -> bool;

pub fn is_operational(&self) -> bool;

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LifecycleTransition {
/// The state before the transition.

pub from: LifecycleState,
/// The state after the transition.

pub to: LifecycleState,
/// The UTC timestamp when the transition occurred.

pub timestamp: Timestamp
}

#[derive(Debug, Clone)]
pub struct LifecycleManager {

}

pub fn new() -> Self;

pub fn state(&self) -> LifecycleState;

pub fn transition(
        &mut self,
        target: LifecycleState,
    ) -> Result<LifecycleTransition, ForgeHealthError>;

pub fn can_transition_to(&self, target: LifecycleState) -> bool;

pub fn valid_transitions(&self) -> Vec<LifecycleState>;

pub fn history(&self) -> &[LifecycleTransition];

monitoring.rs

Read declaration text · 7 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct HealthThresholds {
/// Maximum acceptable error rate in errors per minute.

///

/// When the computed error rate exceeds this threshold, the agent's

/// health status becomes `Degraded` or `Critical`.

pub max_error_rate: f64,
/// Maximum acceptable CPU usage percentage (0.0 to 100.0).

///

/// When CPU usage exceeds this threshold, the agent's health status

/// becomes `Degraded`.

pub max_cpu_percent: f64,
/// Maximum acceptable memory usage in bytes.

///

/// When memory usage exceeds this threshold, the agent's health status

/// becomes `Critical`.

pub max_memory_bytes: u64,
/// Maximum acceptable inference latency in milliseconds.

///

/// This threshold is informational — it is reported in the health status

/// reasons but does not directly cause status changes since latency is

/// not tracked in the profile (it would require per-call timing).

pub max_inference_latency_ms: u64
}

#[derive(Debug, Clone)]
pub struct HealthMonitor {

}

pub fn new(thresholds: HealthThresholds) -> Self;

pub fn profile(&self) -> &HealthProfile;

pub fn profile_mut(&mut self) -> &mut HealthProfile;

pub fn thresholds(&self) -> &HealthThresholds;

pub fn check_health(&self) -> HealthStatus;

profile.rs

Read declaration text · 11 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct HealthProfile {
/// Total uptime of the agent in seconds.

///

/// This is updated externally by the monitoring system based on the

/// elapsed time since the agent entered the Active state.

pub uptime_seconds: u64,
/// Total number of errors recorded during the agent's lifetime.

///

/// This counter is monotonically increasing.

pub error_count: u64,
/// Total number of tool invocations executed by the agent.

///

/// Includes both successful and failed invocations.

pub tool_invocations: u64,
/// Total number of inference calls (LLM generation requests) made.

pub inference_calls: u64,
/// Total number of tokens consumed across all inference calls.

pub inference_tokens: u64,
/// Current CPU usage as a percentage (0.0 to 100.0).

///

/// This is a point-in-time measurement updated by `update_resources`.

pub cpu_usage_percent: f64,
/// Current memory usage in bytes.

///

/// This is a point-in-time measurement updated by `update_resources`.

pub memory_usage_bytes: u64,
/// Number of currently active (in-flight) tasks.

///

/// Incremented by `record_task_started`, decremented by `record_task_completed`.

/// This gauge can reach zero and will not go below zero (saturating subtraction).

pub active_tasks: u32,
/// Total number of tasks completed during the agent's lifetime.

///

/// This counter is monotonically increasing.

pub completed_tasks: u64,
/// Current error rate as a fraction (0.0 to 1.0).

///

/// Computed as `errors / (errors + successes)` across all generation calls.

/// Updated by `record_generation_success` and `record_generation_failure`.

pub error_rate: f64,
/// Average latency of completed operations in milliseconds.

///

/// Updated externally by the monitoring system.

pub avg_latency_ms: f64,
/// Tool invocation success rate as a fraction (0.0 to 1.0).

///

/// Computed as `successes / total` across all tool invocations.

pub tool_success_rate: f64,
/// Generation success rate as a fraction (0.0 to 1.0).

///

/// Computed as `successes / total` across all generation calls.

/// Updated by `record_generation_success` and `record_generation_failure`.

pub generation_success_rate: f64,
/// Timestamp of the last profile update.

pub last_updated: Timestamp
}

pub fn new() -> Self;

pub fn record_tool_invocation(&mut self);

pub fn record_inference(&mut self, tokens: u64);

pub fn record_error(&mut self);

pub fn update_resources(&mut self, cpu: f64, memory: u64);

pub fn update_uptime(&mut self, seconds: u64);

pub fn record_task_started(&mut self);

pub fn record_task_completed(&mut self);

pub fn record_generation_success(&mut self);

pub fn record_generation_failure(&mut self);

reporting.rs

Read declaration text · 7 declaration entries

#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(tag = "status", rename_all = "snake_case")]
pub enum HealthStatus {
    /// All metrics are within acceptable thresholds.
    Healthy,

    /// One or more metrics are approaching thresholds.
    ///
    /// The agent is functional but may need attention. Each reason describes
    /// a specific metric that is outside the normal range.
    Degraded {
        /// Human-readable descriptions of the metrics causing degradation.
        reasons: Vec<String>,
    },

    /// One or more metrics have exceeded critical thresholds.
    ///
    /// The agent may be unable to perform work reliably. Each reason describes
    /// a specific metric that has exceeded its critical threshold.
    Critical {
        /// Human-readable descriptions of the metrics causing critical status.
        reasons: Vec<String>,
    },
}

pub fn is_healthy(&self) -> bool;

pub fn is_critical(&self) -> bool;

pub fn reasons(&self) -> &[String];

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct HealthReport {
/// The assessed health status.

pub status: HealthStatus,
/// The current health profile metrics.

pub profile: HealthProfile,
/// The current lifecycle state of the agent.

pub lifecycle_state: LifecycleState,
/// Timestamp when this report was generated.

pub generated_at: Timestamp
}

pub fn new(
        status: HealthStatus,
        profile: HealthProfile,
        lifecycle_state: LifecycleState,
    ) -> Self;

pub fn derive_health_status(profile: &HealthProfile) -> HealthStatus;

summary.rs

Read declaration text · 8 declaration entries

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct AgentHealthSnapshot {
/// The agent's identifier (OAS DID or human-readable name).

pub agent_id: String,
/// The agent's current lifecycle state.

pub lifecycle_state: LifecycleState,
/// The agent's current health status.

pub health_status: HealthStatus,
/// Number of currently active tasks.

pub active_tasks: u32,
/// Current success rate (0.0 to 1.0) from the agent's latency tracker.

pub success_rate: f64
}

#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct RuntimeHealthSummary {
/// Total number of agents included in this summary.

pub total_agents: u32,
/// Number of agents with `HealthStatus::Healthy`.

pub healthy_count: u32,
/// Number of agents with `HealthStatus::Degraded`.

pub degraded_count: u32,
/// Number of agents with `HealthStatus::Critical`.

pub critical_count: u32,
/// Count of agents per lifecycle state.

pub lifecycle_counts: LifecycleStateCounts,
/// Total active tasks across all agents.

pub total_active_tasks: u64,
/// Mean success rate across all agents (0.0 to 1.0).

/// NaN-safe: returns 0.0 if no agents are included.

pub mean_success_rate: f64,
/// Agent IDs currently in degraded state.

pub degraded_agents: Vec<String>,
/// Agent IDs currently in critical state.

pub critical_agents: Vec<String>,
/// Timestamp when this summary was generated.

pub generated_at: Timestamp
}

#[derive(Debug, Clone, Default, Serialize, Deserialize)]
pub struct LifecycleStateCounts {
/// Agents in `Initializing` state.

pub initializing: u32,
/// Agents in `Ready` state.

pub ready: u32,
/// Agents in `Running` state.

pub running: u32,
/// Agents in `Paused` state.

pub paused: u32,
/// Agents in `Error` state.

pub error: u32,
/// Agents in `Terminated` state.

pub terminated: u32
}

pub fn empty() -> Self;

pub fn empty() -> Self;

pub fn from_snapshots(snapshots: &[AgentHealthSnapshot]) -> Self;

pub fn all_healthy(&self) -> bool;

pub fn has_critical(&self) -> bool;

Continue

On this page