fix(agent): GPU worker thread, async journalctl, correctness + cache fixes
Lightweight: - GPU collection moves to a dedicated worker thread that owns the gfxinfo handle for the process lifetime. gfxinfo's active_gpu() runs a full NVML init/teardown (~20ms, blocking) and we were paying it on the async runtime for every collect — measured at ~80% of the agent's entire active CPU on a GPU machine. The handle holds Rc<Nvml> (not Send), so a thread + mpsc/oneshot channel pair confines it; a zero-total-VRAM reply is treated as a dead session (driver reload) and re-probed. - journalctl now runs via tokio::process instead of blocking one of the two runtime workers for the duration of the subprocess. - TtlCell (state.rs) replaces the four hand-rolled static TTL caches; a cached negative result now counts as fresh, so hosts with no matching temp sensor or GPU stop rescanning every request. Single lock+clone on the GPU cache hit path (was two). Correctness: - Process/child CPU times are now microseconds as documented; they were milliseconds, rendering 1000x too small next to (correct) thread times. - Non-Linux per-process CPU%% clamps AFTER dividing by core count; a 4-cores-busy process on an 8-core box reported 12.5% instead of 50%. - Journal timestamps are real RFC 3339 UTC plus an additive timestamp_us field (sorting is now numeric); the old strings were Debug-formatted SystemTime mangled by string replace. - Partition detection uses /sys/block on Linux: whole-disk filesystems on names like nvme0n1 or zram1 are no longer misclassified as partitions. One shared parent_disk_name() replaces two inline copies. - New sampled_at_ms on the metrics payload (additive) records when the snapshot was actually collected, so clients can compute exact rates across the agent's TTL cache. Security/robustness: - key.pem is created 0600 (was umask default 0644, world-readable) and pre-1.51 keys are tightened on startup. - Per-PID detail/journal caches now evict (60s max age, 64 entries max); they previously grew without bound under PID-walking clients. - The two per-PID ws handlers collapse into one generic helper. - /proc/<pid>/stat parsing unified in one comm-safe module. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
+72
-17
@@ -1,6 +1,4 @@
|
||||
// gpu.rs
|
||||
#[cfg(feature = "gpu")]
|
||||
use gfxinfo::active_gpu;
|
||||
|
||||
#[derive(Debug, Clone, serde::Serialize)]
|
||||
pub struct GpuMetrics {
|
||||
@@ -10,23 +8,80 @@ pub struct GpuMetrics {
|
||||
pub mem_total_bytes: u64,
|
||||
}
|
||||
|
||||
/// Collect metrics for the active GPU. `None` when there is no usable GPU.
|
||||
///
|
||||
/// Runs on a dedicated worker thread (see `worker`): gfxinfo's handle holds
|
||||
/// an `Rc<Nvml>` (not `Send`), and *creating* it runs a full NVML library
|
||||
/// init — ~20ms of blocking work that used to execute on the async runtime
|
||||
/// for every collection. The worker owns one handle for the process lifetime,
|
||||
/// so steady-state collection is just NVML queries. Measured on an RTX 5080
|
||||
/// box, re-initing per collect was ~80% of the agent's entire active CPU.
|
||||
#[cfg(feature = "gpu")]
|
||||
pub fn collect_all_gpus() -> Result<Vec<GpuMetrics>, Box<dyn std::error::Error>> {
|
||||
let gpu = active_gpu()?; // Use ? to unwrap Result
|
||||
let info = gpu.info();
|
||||
|
||||
let metrics = GpuMetrics {
|
||||
name: gpu.model().to_string(),
|
||||
utilization_gpu_pct: info.load_pct() as u32,
|
||||
mem_used_bytes: info.used_vram(),
|
||||
mem_total_bytes: info.total_vram(),
|
||||
};
|
||||
|
||||
Ok(vec![metrics])
|
||||
pub async fn collect_all_gpus() -> Option<Vec<GpuMetrics>> {
|
||||
worker::collect().await
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "gpu"))]
|
||||
pub fn collect_all_gpus() -> Result<Vec<GpuMetrics>, Box<dyn std::error::Error>> {
|
||||
// GPU support not available on this platform
|
||||
Ok(vec![])
|
||||
pub async fn collect_all_gpus() -> Option<Vec<GpuMetrics>> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(feature = "gpu")]
|
||||
mod worker {
|
||||
use super::GpuMetrics;
|
||||
use once_cell::sync::OnceCell;
|
||||
use std::sync::mpsc;
|
||||
|
||||
type Reply = tokio::sync::oneshot::Sender<Option<Vec<GpuMetrics>>>;
|
||||
static TX: OnceCell<mpsc::Sender<Reply>> = OnceCell::new();
|
||||
|
||||
pub async fn collect() -> Option<Vec<GpuMetrics>> {
|
||||
let tx = TX.get_or_init(spawn);
|
||||
let (reply_tx, reply_rx) = tokio::sync::oneshot::channel();
|
||||
tx.send(reply_tx).ok()?;
|
||||
reply_rx.await.ok().flatten()
|
||||
}
|
||||
|
||||
fn spawn() -> mpsc::Sender<Reply> {
|
||||
let (tx, rx) = mpsc::channel::<Reply>();
|
||||
std::thread::Builder::new()
|
||||
.name("socktop-gpu".into())
|
||||
.spawn(move || run(rx))
|
||||
.expect("spawn gpu worker thread");
|
||||
tx
|
||||
}
|
||||
|
||||
fn run(rx: mpsc::Receiver<Reply>) {
|
||||
let mut handle: Option<Box<dyn gfxinfo::Gpu>> = None;
|
||||
// Probing failed: remember and answer None without re-initing the GPU
|
||||
// stack per request. The agent's negative cache stops asking anyway.
|
||||
let mut probe_failed = false;
|
||||
while let Ok(reply) = rx.recv() {
|
||||
if handle.is_none() && !probe_failed {
|
||||
match gfxinfo::active_gpu() {
|
||||
Ok(g) => handle = Some(g),
|
||||
Err(_) => probe_failed = true,
|
||||
}
|
||||
}
|
||||
let out = handle.as_ref().map(|gpu| {
|
||||
let info = gpu.info();
|
||||
vec![GpuMetrics {
|
||||
name: gpu.model().to_string(),
|
||||
utilization_gpu_pct: info.load_pct().clamp(0, 100),
|
||||
mem_used_bytes: info.used_vram(),
|
||||
mem_total_bytes: info.total_vram(),
|
||||
}]
|
||||
});
|
||||
// A live GPU cannot report 0 total VRAM; gfxinfo returns zeros
|
||||
// when the underlying session died (e.g. driver reload). Drop the
|
||||
// handle so the next request re-probes.
|
||||
if let Some(v) = &out
|
||||
&& !v.is_empty()
|
||||
&& v.iter().all(|g| g.mem_total_bytes == 0)
|
||||
{
|
||||
handle = None;
|
||||
}
|
||||
let _ = reply.send(out.filter(|v| !v.is_empty()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user