From b42be6ea0d3fd9922cb9333a837da631baca284d Mon Sep 17 00:00:00 2001 From: Qijun Date: Mon, 5 Oct 2026 00:49:09 -0700 Subject: [PATCH] What a host uses beyond its workloads HostOverview gains other_cpu_cores and other_memory_bytes: the host's busy CPUs and used memory less what the running workloads of its latest report use, never below zero. A workload without a reading (its first minute) counts as none; one gone that the server still lists until it is archived is left out, since it keeps its last readings. CPU is null until the host has a rate. Memory is a lower bound: workloads' page cache is not in the host's used memory. The host preview and page show it after the largest CPU users; the skill tells an agent to point at the host when it is most of the use. --- README.md | 2 +- crates/core/src/view.rs | 9 ++++ crates/server/src/api/handlers.rs | 5 ++- crates/server/src/api/views.rs | 41 +++++++++++++++-- crates/server/src/api/views_tests.rs | 67 ++++++++++++++++++++++++++-- crates/view/src/ui/preview.rs | 14 ++++++ crates/view/src/ui/tests.rs | 6 +++ docs/api.md | 2 +- skill/SKILL.md | 2 +- 9 files changed, 136 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index a82b52d..d81098c 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,7 @@ Thresholds, severities, and when an incident opens or resolves are decided by sk ## A look -`skym-view` opens on the problems tab: every open problem, named by its application (or by its host, for the host's own). New problems come first; `≥` means the problem was already there when skym started watching, so it has lasted at least that long. `⇥` moves to the applications and the hosts. Applications are grouped by environment, host or tag (`g`), each group folded to those in trouble; `/` filters by words and by `host:`, `env:`, `tag:` or `!ok`. Every application shows the CPU and memory its services use; `s` sorts by either, and a host's page names its largest users of each. Hosts show how busy their CPUs are (and iowait) and their network traffic. +`skym-view` opens on the problems tab: every open problem, named by its application (or by its host, for the host's own). New problems come first; `≥` means the problem was already there when skym started watching, so it has lasted at least that long. `⇥` moves to the applications and the hosts. Applications are grouped by environment, host or tag (`g`), each group folded to those in trouble; `/` filters by words and by `host:`, `env:`, `tag:` or `!ok`. Every application shows the CPU and memory its services use; `s` sorts by either, and a host's page names its largest users of each. Hosts show how busy their CPUs are (and iowait), their network traffic, and what they use beyond their workloads. ```text skym · skym.example.com [Problems] Apps Hosts ✗ 1 ! 2 updated 3s ago diff --git a/crates/core/src/view.rs b/crates/core/src/view.rs index 3f3af30..8e74df1 100644 --- a/crates/core/src/view.rs +++ b/crates/core/src/view.rs @@ -204,6 +204,15 @@ pub struct HostOverview { pub net_rx_bytes_per_s: Option, #[serde(default)] pub net_tx_bytes_per_s: Option, + /// What the host uses beyond the workloads its latest report listed (other processes, + /// the kernel): busy CPUs and memory less what its running workloads report, never below + /// zero. A workload without a reading counts as none; CPU is `None` while the host has no + /// rate yet. Memory is a lower bound: page cache charged to workloads (all of it, for + /// systemd units) is not in the host's used memory. + #[serde(default)] + pub other_cpu_cores: Option, + #[serde(default)] + pub other_memory_bytes: Option, /// From the configuration, sorted. #[serde(default)] pub tags: Vec, diff --git a/crates/server/src/api/handlers.rs b/crates/server/src/api/handlers.rs index 97daa58..208d95f 100644 --- a/crates/server/src/api/handlers.rs +++ b/crates/server/src/api/handlers.rs @@ -78,14 +78,15 @@ pub async fn overview(State(s): State) -> ApiResult { let snap = Snapshot::read(&s).await?; let open = snap.views(&s, &snap.open); let apps = snap.apps(&s, &open); - Ok(Json(views::overview(&s.cfg, &snap.hosts, &apps, &open, snap.now))) + Ok(Json(views::overview(&s.cfg, &snap.hosts, &apps, &snap.workloads, &open, snap.now))) } pub async fn hosts(State(s): State) -> ApiResult { let snap = Snapshot::read(&s).await?; let open = snap.views(&s, &snap.open); let apps = snap.apps(&s, &open); - Ok(Json(HostList { hosts: views::hosts(&s.cfg, &snap.hosts, &apps, &open, snap.now) })) + let hosts = views::hosts(&s.cfg, &snap.hosts, &apps, &snap.workloads, &open, snap.now); + Ok(Json(HostList { hosts })) } pub async fn host(State(s): State, Path(host): Path) -> ApiResult { diff --git a/crates/server/src/api/views.rs b/crates/server/src/api/views.rs index ff6a6fa..49a6bf7 100644 --- a/crates/server/src/api/views.rs +++ b/crates/server/src/api/views.rs @@ -6,7 +6,7 @@ use crate::store::hosts::{HostRow, WorkloadRow}; use crate::store::incidents::LogEntry; use crate::store::probes::ProbeRow; use jiff::{SignedDuration, Timestamp}; -use skym_core::model::{Event, ExceptionGroup}; +use skym_core::model::{Event, ExceptionGroup, RunState}; use skym_core::rules::{IncidentCode, Severity}; use skym_core::subject::{AppKey, HostId, Subject, WorkloadKey, encode}; use skym_core::time::format_duration; @@ -86,6 +86,33 @@ pub fn summary(w: &WorkloadRow, incidents: &[IncidentView]) -> WorkloadSummary { workload_summary(w.key.clone(), w.facts.as_ref(), &w.state, incidents, links) } +/// The running workloads of these applications that the host's latest report listed: one +/// gone keeps its last state, and its last readings, until it is archived. +fn reported<'a>( + host: &HostRow, + apps: &[&'a AppSummary], + rows: &[WorkloadRow], +) -> Vec<&'a WorkloadSummary> { + let listed = + |w: &WorkloadSummary| rows.iter().any(|r| r.key == w.key && r.last_seen >= host.last_seen); + let workloads = apps.iter().flat_map(|a| &a.workloads); + workloads.filter(|w| w.run == RunState::Running && listed(w)).collect() +} + +/// The CPUs a host keeps busy beyond these workloads, never below zero (the readings are +/// moments apart). One without a reading (its first minute) counts as none. +fn other_cpu(row: &HostRow, workloads: &[&WorkloadSummary]) -> Option { + let busy = row.state.cpu_percent? / 100.0 * row.facts.as_ref()?.cpu_count as f32; + let used: f32 = workloads.iter().filter_map(|w| w.cpu_cores).sum(); + Some((busy - used).max(0.0)) +} + +/// The memory a host uses beyond these workloads, counted as for CPU. +fn other_memory(row: &HostRow, workloads: &[&WorkloadSummary]) -> u64 { + let used: u64 = workloads.iter().filter_map(|w| w.memory_used_bytes).sum(); + row.state.memory_used_bytes.saturating_sub(used) +} + /// Where to look next, so an agent never builds URLs itself. pub fn links(subject: &Subject) -> BTreeMap { let host_links = |h: &str| { @@ -180,6 +207,7 @@ pub fn host_overview( tags: &[String], row: Option<&HostRow>, apps: &[AppSummary], + workloads: &[WorkloadRow], open: &[IncidentView], now: Timestamp, ) -> HostOverview { @@ -230,6 +258,8 @@ pub fn host_overview( steal_percent: row.and_then(|r| r.state.steal_percent), net_rx_bytes_per_s: row.and_then(|r| r.state.net_rx_bytes_per_s), net_tx_bytes_per_s: row.and_then(|r| r.state.net_tx_bytes_per_s), + other_cpu_cores: row.and_then(|r| other_cpu(r, &reported(r, &own, workloads))), + other_memory_bytes: row.map(|r| other_memory(r, &reported(r, &own, workloads))), disks, apps: own.len() as u32, apps_in_trouble: own.iter().filter(|a| a.status != Status::Ok).count() as u32, @@ -251,13 +281,17 @@ pub fn hosts( cfg: &ServerConfig, rows: &[HostRow], apps: &[AppSummary], + workloads: &[WorkloadRow], open: &[IncidentView], now: Timestamp, ) -> Vec { let mut hosts: Vec = cfg .hosts .iter() - .map(|h| host_overview(&h.id, &h.tags, rows.iter().find(|r| r.id == h.id), apps, open, now)) + .map(|h| { + let row = rows.iter().find(|r| r.id == h.id); + host_overview(&h.id, &h.tags, row, apps, workloads, open, now) + }) .collect(); hosts.sort_by(|a, b| b.status.cmp(&a.status).then_with(|| a.id.cmp(&b.id))); hosts @@ -269,10 +303,11 @@ pub fn overview( cfg: &ServerConfig, rows: &[HostRow], apps: &[AppSummary], + workloads: &[WorkloadRow], open: &[IncidentView], now: Timestamp, ) -> Overview { - let hosts = hosts(cfg, rows, apps, open, now); + let hosts = hosts(cfg, rows, apps, workloads, open, now); let mut problems: Vec = open.iter().filter(|i| !i.muted).cloned().collect(); problems .sort_by(|a, b| b.severity.cmp(&a.severity).then_with(|| a.opened_at.cmp(&b.opened_at))); diff --git a/crates/server/src/api/views_tests.rs b/crates/server/src/api/views_tests.rs index 668ba50..ab40346 100644 --- a/crates/server/src/api/views_tests.rs +++ b/crates/server/src/api/views_tests.rs @@ -205,7 +205,7 @@ fn the_overview_lists_problems_flat_and_hosts_by_urgency() { summary("b/other", Status::Ok), summary("a/web", Status::Ok), ]; - let o = overview(&cfg, &rows, &apps, &open, at(5)); + let o = overview(&cfg, &rows, &apps, &[], &open, at(5)); let problems: Vec = o.problems.iter().map(|p| p.subject.to_string()).collect(); assert_eq!( problems, @@ -242,7 +242,7 @@ fn a_host_overview_shows_its_resources_and_which_disk_fills_up() { .iter() .map(|i| Views::new().of(i, at(5))) .collect(); - let h = host_overview(&"x".to_string(), &[], Some(&r), &[], &open, at(5)); + let h = host_overview(&"x".to_string(), &[], Some(&r), &[], &[], &open, at(5)); assert_eq!(h.load_1m, Some(r.state.load_1m)); assert_eq!(h.memory_used_bytes, Some(r.state.memory_used_bytes)); assert_eq!(h.memory_total_bytes, r.facts.as_ref().map(|f| f.memory_total_bytes)); @@ -294,11 +294,11 @@ fn a_host_overview_shows_its_resources_and_which_disk_fills_up() { let mut seen = r.clone(); seen.remote_addr = Some("203.0.113.7".into()); assert_eq!( - host_overview(&"x".to_string(), &[], Some(&seen), &[], &[], at(5)).ip.as_deref(), + host_overview(&"x".to_string(), &[], Some(&seen), &[], &[], &[], at(5)).ip.as_deref(), Some("203.0.113.7") ); let tags = ["cn".to_string(), "acme".to_string(), "cn".to_string()]; - let silent = host_overview(&"y".to_string(), &tags, None, &[], &[], at(5)); + let silent = host_overview(&"y".to_string(), &tags, None, &[], &[], &[], at(5)); assert_eq!((silent.status, silent.disks.len(), silent.load_1m), (Status::Unknown, 0, None)); assert_eq!((silent.cpu_percent, silent.net_rx_bytes_per_s), (None, None)); assert_eq!(silent.tags, ["acme", "cn"], "sorted, once each, even before it reports"); @@ -356,3 +356,62 @@ fn timeline_merges_newest_first_and_marks_truncation() { let cut = timeline(vec![change(3, Change::Severity), change(1, Change::Reopened)], vec![], 1); assert_eq!((cut.entries.len(), cut.truncated), (1, true)); } + +#[test] +fn a_host_shows_what_it_uses_beyond_its_workloads() { + use skym_core::model::RunState::{Exited, Running}; + let r = row("x", 0); // 23.5% of 8 CPUs busy, 8 GiB used, last seen at 0 + let (busy, used) = (0.235 * 8.0_f32, r.state.memory_used_bytes); + let workload = |key: &str, cores, memory, run| WorkloadSummary { + cpu_cores: cores, + memory_used_bytes: memory, + run, + ..skym_core::fixtures::workload(key) + }; + let app = |key: &str, workloads| AppSummary { workloads, ..skym_core::fixtures::app(key) }; + // What the latest report listed (seen at 0), and one gone since (last seen before). + let seen = |key: &str, minute| { + let Ok(Subject::Workload(key)) = format!("workload:{key}").parse() else { unreachable!() }; + let state = skym_core::fixtures::full_report().workloads[0].state.clone(); + WorkloadRow { key, facts: None, state, last_seen: at(minute) } + }; + let rows = [seen("x/a/s", 0), seen("x/a/t", 0), seen("x/gone/s", -60)]; + let other = |row: Option<&HostRow>, apps: &[AppSummary]| { + let h = host_overview(&"x".to_string(), &[], row, apps, &rows, &[], at(5)); + (h.other_cpu_cores, h.other_memory_bytes) + }; + let apps = [ + app( + "x/a", + vec![ + workload("x/a/s", Some(0.5), Some(1_000), Running), + workload("x/a/t", Some(2.0), Some(2_000), Exited), + ], + ), + app("x/gone", vec![workload("x/gone/s", Some(1.0), Some(5_000), Running)]), + app("y/b", vec![workload("y/b/s", Some(9.0), Some(u64::MAX), Running)]), + ]; + assert_eq!( + other(Some(&r), &apps), + (Some(busy - 0.5), Some(used - 1_000)), + "less its running workloads; stopped, gone and another host's left out" + ); + let unread = [app( + "x/a", + vec![ + workload("x/a/s", Some(0.5), Some(1_000), Running), + workload("x/a/t", None, None, Running), + ], + )]; + assert_eq!( + other(Some(&r), &unread), + (Some(busy - 0.5), Some(used - 1_000)), + "one without a reading counts as none" + ); + let mut no_rate = r.clone(); + no_rate.state.cpu_percent = None; + assert_eq!(other(Some(&no_rate), &apps).0, None, "the host's own first pass"); + let more = [app("x/a", vec![workload("x/a/s", Some(9.0), Some(u64::MAX), Running)])]; + assert_eq!(other(Some(&r), &more), (Some(0.0), Some(0)), "never below zero"); + assert_eq!(other(None, &apps), (None, None), "a host that never reported"); +} diff --git a/crates/view/src/ui/preview.rs b/crates/view/src/ui/preview.rs index e388db3..ab3096e 100644 --- a/crates/view/src/ui/preview.rs +++ b/crates/view/src/ui/preview.rs @@ -100,6 +100,19 @@ fn top( (!parts.is_empty()).then(|| Line::from(vec![label(name, theme), Span::raw(parts.join(" · "))])) } +/// `other 0.80 cores · 1.2G beyond the workloads`: what runs outside them, when known. +fn other(h: &HostOverview, theme: Theme) -> Option> { + let parts: Vec = + [h.other_cpu_cores.map(|c| format!("{} cores", cores(c))), h.other_memory_bytes.map(size)] + .into_iter() + .flatten() + .collect(); + (!parts.is_empty()).then(|| { + let text = format!("{} beyond the workloads", parts.join(" · ")); + Line::from(vec![label("other", theme), Span::raw(text)]) + }) +} + /// `tags acme · billing`, when it has any. fn tags(tags: &[String], theme: Theme) -> Option> { (!tags.is_empty()).then(|| Line::from(vec![label("tags", theme), Span::raw(tags.join(" · "))])) @@ -324,6 +337,7 @@ pub fn host_body(app: &App, h: &HostOverview, now: Timestamp, theme: Theme) -> V lines.extend(top("top mem", &on_host, memory, |b| size(b as u64), theme)); let cpu = |a: &AppSummary| apps::used_cpu(a).map(f64::from); lines.extend(top("top cpu", &on_host, cpu, |c| cores(c as f32), theme)); + lines.extend(other(h, theme)); lines } diff --git a/crates/view/src/ui/tests.rs b/crates/view/src/ui/tests.rs index c68944c..e01a12a 100644 --- a/crates/view/src/ui/tests.rs +++ b/crates/view/src/ui/tests.rs @@ -590,6 +590,12 @@ fn the_apps_and_hosts_tabs_preview_their_selection() { ); let top_cpu = line_of(&lines, "top cpu Shop 1.20"); assert!(top_cpu > line_of(&lines, "top mem"), "the first to go when there is no room"); + assert!(!lines.iter().any(|l| l.contains(" other ")), "not known yet: no line"); + let x = &mut app.overview.value.as_mut().unwrap().hosts[0]; + (x.other_cpu_cores, x.other_memory_bytes) = (Some(0.8), Some(1_200_000_000)); + let lines = render(&app, Theme { color: false }, 120, 40); + let other = line_of(&lines, " other 0.80 cores · 1.2G beyond the workloads"); + assert!(other > line_of(&lines, "top cpu"), "after the largest users"); let data = &lines[line_of(&lines, "/data ")..]; assert!(data.iter().any(|l| l.contains("filling up")), "the disk in full, flagged"); } diff --git a/docs/api.md b/docs/api.md index 1cb0851..68f8ef3 100644 --- a/docs/api.md +++ b/docs/api.md @@ -39,7 +39,7 @@ token_sha256 = "…" | `GET /api/overview` | Where is something wrong right now? | `problems`: every open, unmuted incident, worst and oldest first, each with its application (`app`, else it is its host's own); `hosts`: every host with its status, last report age, load, memory, disks and applications, most urgent first | | `GET /api/apps` | Which applications exist, where, and are they up? | Every application (compose project, or lone container or systemd unit) with its name, environment, note, status, URLs, open incidents, its `workloads` (image, run state, restart policy, ports, memory against its limit, CPUs kept busy as `cpu_cores`, restarts in the last hour), its latest `deploys`, `exceptions_1h` and `tags` (its host's and its own); most urgent first | | `GET /api/apps/{host}/{project}` | What is this application, and how is it? | The application and its workloads; lone containers and systemd units at `/api/apps/{host}/{project}/{service}` | -| `GET /api/hosts` | Which hosts are there, and how loaded? | The overview's `hosts`: also the system (`os`, `kernel`, `arch`, `cpu_count`, `boot_time`, `docker_version`, `agent_version`), the address it reports from (`ip`), its `tags`, how busy its CPUs were (`cpu_percent`, `iowait_percent`, `steal_percent`: shares of all CPUs since the previous report), its network traffic (`net_rx_bytes_per_s`, `net_tx_bytes_per_s`, physical interfaces only) and every disk's size, free space and inodes | +| `GET /api/hosts` | Which hosts are there, and how loaded? | The overview's `hosts`: also the system (`os`, `kernel`, `arch`, `cpu_count`, `boot_time`, `docker_version`, `agent_version`), the address it reports from (`ip`), its `tags`, how busy its CPUs were (`cpu_percent`, `iowait_percent`, `steal_percent`: shares of all CPUs since the previous report), its network traffic (`net_rx_bytes_per_s`, `net_tx_bytes_per_s`, physical interfaces only), what it uses beyond its workloads (`other_cpu_cores`, `other_memory_bytes`: busy CPUs and memory less what the running workloads of its latest report use, never below zero, a workload without a reading counting as none; the memory is a lower bound, as workloads' page cache is not in the host's used memory) and every disk's size, free space and inodes | | `GET /api/hosts/{host}` | What is going on with this host? | Facts, state, its applications, workloads with their status, open incidents | | `GET /api/hosts/{host}/workloads/{project}/{service}` | What is going on with this application? | Facts, state, recent exception groups, recent events | | `GET /api/timeline` | When did it start, and what else happened? | Incident changes and events, merged and sorted by time | diff --git a/skill/SKILL.md b/skill/SKILL.md index eddf377..4fada4d 100644 --- a/skill/SKILL.md +++ b/skill/SKILL.md @@ -15,7 +15,7 @@ curl -fsS -H "Authorization: Bearer $SKYM_TOKEN" "$SKYM_URL/api/overview" ## How to read it -1. Start at `/api/overview`. `problems` lists every open, unmuted incident, worst and oldest first; each names its application in `app` (none for a host's own: heartbeat, disks, logs). Name problems by their application, then the host. A workload's problem carries the workload (`workload`: image, ports, memory against its limit, restarts, a failing check's output), and each application its `workloads`, latest `deploys` and `exceptions_1h`: answer from these before following `links`. `hosts` (also `/api/hosts`) gives each machine's load, memory and disks, how busy its CPUs were since the last report (`cpu_percent`; high `iowait_percent` means waiting on disk, `steal_percent` on the hypervisor) and its network traffic. To say which application keeps a host busy, sum its `workloads[].cpu_cores` (CPUs kept busy, 1.5 = one and a half) from `/api/apps`. A URL of an application down (`ENDPOINT_DOWN`, `CERT_EXPIRING`) while its services are fine points at what lies between: a reverse proxy, DNS, a certificate. +1. Start at `/api/overview`. `problems` lists every open, unmuted incident, worst and oldest first; each names its application in `app` (none for a host's own: heartbeat, disks, logs). Name problems by their application, then the host. A workload's problem carries the workload (`workload`: image, ports, memory against its limit, restarts, a failing check's output), and each application its `workloads`, latest `deploys` and `exceptions_1h`: answer from these before following `links`. `hosts` (also `/api/hosts`) gives each machine's load, memory and disks, how busy its CPUs were since the last report (`cpu_percent`; high `iowait_percent` means waiting on disk, `steal_percent` on the hypervisor) and its network traffic. To say which application keeps a host busy, sum its `workloads[].cpu_cores` (CPUs kept busy, 1.5 = one and a half) from `/api/apps`. A host's `other_cpu_cores` and `other_memory_bytes` are what it uses outside its workloads (other processes, the kernel): when they are most of its use, say so and point at the host rather than at an application. A small `other_memory_bytes` does not prove nothing else uses memory: it is a lower bound. A URL of an application down (`ENDPOINT_DOWN`, `CERT_EXPIRING`) while its services are fine points at what lies between: a reverse proxy, DNS, a certificate. 2. Asked which applications there are, or about one by name: start at `/api/apps`. Each has a `name`, an `env` and a `note` saying what it is; quote the note. Point out applications without an `env` or `note`, so someone describes them. Asked about a customer or a group, select by `tags` (on applications and hosts); about one machine, by the host in the application's `key` (`/...`). `APP_MISSING` means a configured application has nothing running on its host. 3. `HEARTBEAT_LOST` comes first: nothing else about that host is current. When the details say "all hosts silent", suspect the server or its network, not every host. 4. Follow `links` to drill down (host → workload → timeline, exceptions, incidents); build a URL only from what `GET /api` lists.