e4b0d9b9f6
Lightweight: - GPU collection moves to a dedicated worker thread that owns the gfxinfo handle for the process lifetime. gfxinfo's active_gpu() runs a full NVML init/teardown (~20ms, blocking) and we were paying it on the async runtime for every collect — measured at ~80% of the agent's entire active CPU on a GPU machine. The handle holds Rc<Nvml> (not Send), so a thread + mpsc/oneshot channel pair confines it; a zero-total-VRAM reply is treated as a dead session (driver reload) and re-probed. - journalctl now runs via tokio::process instead of blocking one of the two runtime workers for the duration of the subprocess. - TtlCell (state.rs) replaces the four hand-rolled static TTL caches; a cached negative result now counts as fresh, so hosts with no matching temp sensor or GPU stop rescanning every request. Single lock+clone on the GPU cache hit path (was two). Correctness: - Process/child CPU times are now microseconds as documented; they were milliseconds, rendering 1000x too small next to (correct) thread times. - Non-Linux per-process CPU%% clamps AFTER dividing by core count; a 4-cores-busy process on an 8-core box reported 12.5% instead of 50%. - Journal timestamps are real RFC 3339 UTC plus an additive timestamp_us field (sorting is now numeric); the old strings were Debug-formatted SystemTime mangled by string replace. - Partition detection uses /sys/block on Linux: whole-disk filesystems on names like nvme0n1 or zram1 are no longer misclassified as partitions. One shared parent_disk_name() replaces two inline copies. - New sampled_at_ms on the metrics payload (additive) records when the snapshot was actually collected, so clients can compute exact rates across the agent's TTL cache. Security/robustness: - key.pem is created 0600 (was umask default 0644, world-readable) and pre-1.51 keys are tightened on startup. - Per-PID detail/journal caches now evict (60s max age, 64 entries max); they previously grew without bound under PID-walking clients. - The two per-PID ws handlers collapse into one generic helper. - /proc/<pid>/stat parsing unified in one comm-safe module. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
194 lines
6.2 KiB
Rust
194 lines
6.2 KiB
Rust
//! Shared agent state: sysinfo handles and hot JSON cache.
|
|
|
|
use std::collections::HashMap;
|
|
use std::sync::Arc;
|
|
use std::sync::atomic::{AtomicBool, AtomicUsize};
|
|
use std::time::{Duration, Instant};
|
|
use sysinfo::{Components, Disks, Networks, System};
|
|
use tokio::sync::Mutex;
|
|
|
|
pub type SharedSystem = Arc<Mutex<System>>;
|
|
pub type SharedComponents = Arc<Mutex<Components>>;
|
|
pub type SharedDisks = Arc<Mutex<Disks>>;
|
|
pub type SharedNetworks = Arc<Mutex<Networks>>;
|
|
|
|
#[cfg(target_os = "linux")]
|
|
#[derive(Default)]
|
|
pub struct ProcCpuTracker {
|
|
pub last_total: u64,
|
|
pub last_per_pid: HashMap<u32, u64>,
|
|
/// PID → process name cache. Mirrors the non-Linux `ProcessCache.names`.
|
|
/// On a Pi with ~150-300 mostly-stable processes this avoids re-allocating
|
|
/// the same `String`s on every processes poll (~once per 1.5s).
|
|
pub names: HashMap<u32, String>,
|
|
}
|
|
|
|
#[cfg(not(target_os = "linux"))]
|
|
pub struct ProcessCache {
|
|
pub names: HashMap<u32, String>,
|
|
pub reusable_vec: Vec<crate::types::ProcessInfo>,
|
|
}
|
|
|
|
#[cfg(not(target_os = "linux"))]
|
|
impl Default for ProcessCache {
|
|
fn default() -> Self {
|
|
Self {
|
|
names: HashMap::with_capacity(1000), // Pre-allocate for typical modern system process count
|
|
reusable_vec: Vec::with_capacity(1000),
|
|
}
|
|
}
|
|
}
|
|
|
|
#[derive(Clone)]
|
|
pub struct AppState {
|
|
pub sys: SharedSystem,
|
|
pub components: SharedComponents,
|
|
pub disks: SharedDisks,
|
|
pub networks: SharedNetworks,
|
|
pub hostname: String,
|
|
|
|
// For correct per-process CPU% using /proc deltas (Linux only path uses this tracker)
|
|
#[cfg(target_os = "linux")]
|
|
pub proc_cpu: Arc<Mutex<ProcCpuTracker>>,
|
|
|
|
// Process name caching and vector reuse for non-Linux to reduce allocations
|
|
#[cfg(not(target_os = "linux"))]
|
|
pub proc_cache: Arc<Mutex<ProcessCache>>,
|
|
|
|
// Connection tracking (to allow future idle sleeps if desired)
|
|
pub client_count: Arc<AtomicUsize>,
|
|
|
|
pub auth_token: Option<String>,
|
|
// GPU negative cache (probe once). gpu_checked=true after first attempt; gpu_present reflects result.
|
|
pub gpu_checked: Arc<AtomicBool>,
|
|
pub gpu_present: Arc<AtomicBool>,
|
|
|
|
// Lightweight on-demand caches (TTL based) to cap CPU under bursty polling.
|
|
pub cache_metrics: Arc<Mutex<CacheEntry<crate::types::Metrics>>>,
|
|
pub cache_disks: Arc<Mutex<CacheEntry<Vec<crate::types::DiskInfo>>>>,
|
|
pub cache_processes: Arc<Mutex<CacheEntry<crate::types::ProcessesPayload>>>,
|
|
|
|
// Process detail caches (per-PID)
|
|
pub cache_process_metrics:
|
|
Arc<Mutex<HashMap<u32, CacheEntry<crate::types::ProcessMetricsResponse>>>>,
|
|
pub cache_journal_entries: Arc<Mutex<HashMap<u32, CacheEntry<crate::types::JournalResponse>>>>,
|
|
}
|
|
|
|
/// TTL-gated value behind a std Mutex, for `static` caches on hot paths.
|
|
/// Replaces the hand-rolled TempCache/GpuCache/refresh-timestamp statics
|
|
/// that each reimplemented the same at/value pair.
|
|
pub struct TtlCell<T> {
|
|
inner: std::sync::Mutex<CacheEntry<T>>,
|
|
}
|
|
|
|
impl<T: Clone> Default for TtlCell<T> {
|
|
fn default() -> Self {
|
|
Self::new()
|
|
}
|
|
}
|
|
|
|
impl<T: Clone> TtlCell<T> {
|
|
pub const fn new() -> Self {
|
|
Self {
|
|
inner: std::sync::Mutex::new(CacheEntry::new()),
|
|
}
|
|
}
|
|
/// The stored value, only while fresh. Poisoned lock reads as a miss.
|
|
pub fn get_fresh(&self, ttl: Duration) -> Option<T> {
|
|
let g = self.inner.lock().ok()?;
|
|
if g.is_fresh(ttl) {
|
|
g.value.clone()
|
|
} else {
|
|
None
|
|
}
|
|
}
|
|
pub fn set(&self, v: T) {
|
|
if let Ok(mut g) = self.inner.lock() {
|
|
g.set(v);
|
|
}
|
|
}
|
|
/// True exactly once per TTL window: restamps and tells the caller to do
|
|
/// the refresh. Atomic check-and-stamp so concurrent callers don't both
|
|
/// refresh.
|
|
pub fn claim_stale(&self, ttl: Duration) -> bool {
|
|
let Ok(mut g) = self.inner.lock() else {
|
|
return false;
|
|
};
|
|
if g.at.is_none_or(|t| t.elapsed() >= ttl) {
|
|
g.at = Some(Instant::now());
|
|
true
|
|
} else {
|
|
false
|
|
}
|
|
}
|
|
}
|
|
|
|
#[derive(Clone, Debug)]
|
|
pub struct CacheEntry<T> {
|
|
pub at: Option<Instant>,
|
|
pub value: Option<T>,
|
|
}
|
|
|
|
impl<T> Default for CacheEntry<T> {
|
|
fn default() -> Self {
|
|
Self::new()
|
|
}
|
|
}
|
|
|
|
impl<T> CacheEntry<T> {
|
|
pub const fn new() -> Self {
|
|
Self {
|
|
at: None,
|
|
value: None,
|
|
}
|
|
}
|
|
pub fn is_fresh(&self, ttl: Duration) -> bool {
|
|
self.at.is_some_and(|t| t.elapsed() < ttl) && self.value.is_some()
|
|
}
|
|
pub fn set(&mut self, v: T) {
|
|
self.value = Some(v);
|
|
self.at = Some(Instant::now());
|
|
}
|
|
pub fn get(&self) -> Option<&T> {
|
|
self.value.as_ref()
|
|
}
|
|
}
|
|
|
|
impl Default for AppState {
|
|
fn default() -> Self {
|
|
Self::new()
|
|
}
|
|
}
|
|
|
|
impl AppState {
|
|
pub fn new() -> Self {
|
|
let sys = System::new();
|
|
let components = Components::new_with_refreshed_list();
|
|
let disks = Disks::new_with_refreshed_list();
|
|
let networks = Networks::new_with_refreshed_list();
|
|
|
|
Self {
|
|
sys: Arc::new(Mutex::new(sys)),
|
|
components: Arc::new(Mutex::new(components)),
|
|
disks: Arc::new(Mutex::new(disks)),
|
|
networks: Arc::new(Mutex::new(networks)),
|
|
hostname: System::host_name().unwrap_or_else(|| "unknown".into()),
|
|
#[cfg(target_os = "linux")]
|
|
proc_cpu: Arc::new(Mutex::new(ProcCpuTracker::default())),
|
|
#[cfg(not(target_os = "linux"))]
|
|
proc_cache: Arc::new(Mutex::new(ProcessCache::default())),
|
|
client_count: Arc::new(AtomicUsize::new(0)),
|
|
auth_token: std::env::var("SOCKTOP_TOKEN")
|
|
.ok()
|
|
.filter(|s| !s.is_empty()),
|
|
gpu_checked: Arc::new(AtomicBool::new(false)),
|
|
gpu_present: Arc::new(AtomicBool::new(false)),
|
|
cache_metrics: Arc::new(Mutex::new(CacheEntry::new())),
|
|
cache_disks: Arc::new(Mutex::new(CacheEntry::new())),
|
|
cache_processes: Arc::new(Mutex::new(CacheEntry::new())),
|
|
cache_process_metrics: Arc::new(Mutex::new(HashMap::new())),
|
|
cache_journal_entries: Arc::new(Mutex::new(HashMap::new())),
|
|
}
|
|
}
|
|
}
|