diff --git a/README.MD b/README.MD index 84a26e0..848b296 100644 --- a/README.MD +++ b/README.MD @@ -11,8 +11,9 @@ SyMon is a self-hosted monitoring tool for Linux servers, home labs and Raspberr **Hosts** - CPU, overall and per core, load average, memory and swap - Disk space, disk IO and busy time, network traffic, TCP connections +- When each disk will be full, at the rate it grew over the last week - Pressure stall information and temperature sensors -- The top processes by CPU and memory, at any point in time +- The top processes by CPU and memory at any point in time, and the programs that used the most over any range - Whether chosen systemd services are running **Containers** @@ -23,18 +24,19 @@ SyMon is a self-hosted monitoring tool for Linux servers, home labs and Raspberr - Send any number from a script or cron job and get a chart for it **Alerts** -- Rules for CPU, memory, swap, disks, services, custom metrics, silent hosts and HTTP endpoints +- Rules for CPU, memory, swap, disks, disks filling up, services, custom metrics, silent hosts and HTTP endpoints - Warning and critical levels, shown on the dashboard and sent by email, Slack or PagerDuty **Dashboard** - Every host at a glance, and a page per host with charts from 15 minutes to 30 days, or any custom range -- Drag across a chart to zoom in, and switch any chart to a table +- Drag across a chart to zoom in, click a point to see the processes running then, and switch any chart to a table - Follows the system light or dark theme **Running it** - Add a host with one command, with a single-use token - Raw data kept for 7 days, 1 minute averages for 30 days and 1 hour averages for a year, all adjustable - Each host has its own key, and components can talk over TLS +- A Prometheus endpoint, for Grafana or a Prometheus you already run ## Screenshots @@ -96,7 +98,7 @@ Components talk over gRPC, so other tools can read from or push into them. See t The Client exposes a JSON API under `/api/v1`. Times are unix seconds. Errors return a JSON body `{"error": "..."}` with a 4xx or 5xx status. * `GET /api/v1/fleet` - * Every host with its status, latest usage, number of running containers and number of open alerts + * Every host with its status, latest usage, number of running containers, number of open alerts, and `diskFullDays`, the days until its first disk is full (null when none is filling up) * `GET /api/v1/hosts/{host}` * The host's latest snapshot as the agent sent it, with `up` and `lastSeen` * `GET /api/v1/hosts/{host}/series?metric=cpu&from=&to=` @@ -104,7 +106,13 @@ The Client exposes a JSON API under `/api/v1`. Times are unix seconds. Errors re * Metrics: `cpu`, `cpu_core`, `load1`, `load5`, `load15`, `memory`, `memory_used`, `swap`, `swap_used`, `psi_cpu`, `psi_memory`, `psi_memory_full`, `psi_io`, `psi_io_full`, `tcp_established`, `tcp_time_wait`, `tcp_close_wait`, `tcp_listen`, `tcp_total`, `disk_used`, `disk_inodes`, `disk_read`, `disk_write`, `disk_util`, `net_rx`, `net_tx`, `temperature`, `custom`, `container_cpu`, `container_memory`, `container_rx`, `container_tx`, `container_io_read`, `container_io_write` * `GET /api/v1/hosts/{host}/processes?at=` * Top processes by CPU and by memory at or before `at`, or the latest +* `GET /api/v1/hosts/{host}/process-usage?from=&to=` + * The programs that used the most CPU and memory over the range, 24 hours at most, with their average, peak and how often they were among the top processes. Processes with the same name are added up +* `GET /api/v1/hosts/{host}/disk-forecasts` + * Each disk's growth per day over the last week, and `daysToFull`, or null when the disk is not filling up * `GET /api/v1/hosts/{host}/custom-metrics` * Names of the host's custom metrics * `GET /api/v1/alerts?host=&open=1&from=&to=` * Alerts, newest first. `open=1` leaves out resolved ones + +The Client also serves `GET /metrics`, every host's latest values in the Prometheus text format. See [Prometheus and Grafana](docs/install.md#prometheus-and-grafana). diff --git a/client/internal/server/metrics.go b/client/internal/server/metrics.go new file mode 100644 index 0000000..3dc4ec4 --- /dev/null +++ b/client/internal/server/metrics.go @@ -0,0 +1,234 @@ +package server + +import ( + "encoding/json" + "fmt" + "io" + "net/http" + "strconv" + "strings" + + "github.com/dhamith93/SyMon/internal/api" + "github.com/dhamith93/SyMon/internal/logger" + "github.com/dhamith93/SyMon/internal/monitor" +) + +// getMetrics serves every host's latest values in the Prometheus text +// format, so Prometheus can scrape the dashboard +func (s *server) getMetrics(w http.ResponseWriter, r *http.Request) { + response, err := s.collector.Snapshots(r.Context(), &api.Void{}) + if err != nil { + writeGRPCError(w, "snapshots", err) + return + } + + metrics := newMetricsWriter() + for _, host := range response.Hosts { + metrics.gauge("symon_up", "1 when the host reported within the last minute.", boolValue(host.Up), "host", host.Host) + if host.LastSeen > 0 { + metrics.gauge("symon_last_seen_timestamp_seconds", "When the host was last heard from.", float64(host.LastSeen), "host", host.Host) + } + // a host that stopped reporting only has old values, which would + // look current to Prometheus + if !host.Up || host.SnapshotJson == "" { + continue + } + var data monitor.MonitorData + if err := json.Unmarshal([]byte(host.SnapshotJson), &data); err != nil { + logger.Log("error", "cannot read the snapshot of "+host.Host+": "+err.Error()) + continue + } + addHostMetrics(metrics, host.Host, &data) + } + + w.Header().Set("Content-Type", "text/plain; version=0.0.4; charset=utf-8") + if err := metrics.writeTo(w); err != nil { + logger.Log("error", "cannot write metrics: "+err.Error()) + } +} + +const mib = 1024 * 1024 + +func addHostMetrics(m *metricsWriter, host string, data *monitor.MonitorData) { + m.gauge("symon_uptime_seconds", "How long the host has been running.", data.System.UpTimeSeconds, "host", host) + // LoadAvg keeps its old name, it is the CPU usage + m.gauge("symon_cpu_usage_percent", "CPU usage.", float64(data.ProcUsage.LoadAvg), "host", host) + m.gauge("symon_load1", "Load average over 1 minute.", data.ProcUsage.Load1, "host", host) + m.gauge("symon_load5", "Load average over 5 minutes.", data.ProcUsage.Load5, "host", host) + m.gauge("symon_load15", "Load average over 15 minutes.", data.ProcUsage.Load15, "host", host) + + // the agent sends memory and swap in MiB + m.gauge("symon_memory_used_percent", "Memory in use.", data.Memory.PercentageUsed, "host", host) + m.gauge("symon_memory_total_bytes", "Total memory.", float64(data.Memory.Total)*mib, "host", host) + m.gauge("symon_memory_available_bytes", "Memory available to new programs.", float64(data.Memory.Available)*mib, "host", host) + m.gauge("symon_swap_used_percent", "Swap in use.", data.Swap.PercentageUsed, "host", host) + m.gauge("symon_swap_total_bytes", "Total swap.", float64(data.Swap.Total)*mib, "host", host) + + for _, disk := range data.Disk { + labels := []string{"host", host, "device", disk.FileSystem, "mount", disk.MountedOn} + if pct, ok := parsePercent(disk.Usage.Usage); ok { + m.gauge("symon_disk_used_percent", "Disk space in use, as df shows it.", pct, labels...) + } + m.gauge("symon_disk_size_bytes", "Disk size.", float64(disk.Usage.Size), labels...) + m.gauge("symon_disk_used_bytes", "Disk space in use.", float64(disk.Usage.Used), labels...) + if pct, ok := parsePercent(disk.Inodes.Usage); ok { + m.gauge("symon_disk_inodes_used_percent", "Inodes in use.", pct, labels...) + } + } + for _, diskIO := range data.DiskIO { + // named like the disks above + labels := []string{"host", host, "device", "/dev/" + diskIO.Device} + m.gauge("symon_disk_read_bytes_per_second", "Disk reads.", diskIO.ReadBytesPerSec, labels...) + m.gauge("symon_disk_write_bytes_per_second", "Disk writes.", diskIO.WriteBytesPerSec, labels...) + m.gauge("symon_disk_util_percent", "Time the disk was busy.", diskIO.UtilPercent, labels...) + } + for _, network := range data.Networks { + labels := []string{"host", host, "iface", network.Interface} + m.counter("symon_network_receive_bytes_total", "Bytes received since boot.", float64(network.Usage.RxBytes), labels...) + m.counter("symon_network_transmit_bytes_total", "Bytes sent since boot.", float64(network.Usage.TxBytes), labels...) + } + + if tcp := data.TCPStates; tcp != nil { + states := []struct { + name string + count int + }{ + {"established", tcp.Established}, {"syn_sent", tcp.SynSent}, {"syn_recv", tcp.SynRecv}, + {"fin_wait1", tcp.FinWait1}, {"fin_wait2", tcp.FinWait2}, {"time_wait", tcp.TimeWait}, + {"close", tcp.Close}, {"close_wait", tcp.CloseWait}, {"last_ack", tcp.LastAck}, + {"listen", tcp.Listen}, {"closing", tcp.Closing}, + } + for _, state := range states { + m.gauge("symon_tcp_connections", "TCP connections by state.", float64(state.count), "host", host, "state", state.name) + } + } + + if p := data.Pressure; p != nil { + resources := []struct { + name string + pressure monitor.ResourcePressure + }{{"cpu", p.CPU}, {"memory", p.Memory}, {"io", p.IO}} + for _, r := range resources { + m.gauge("symon_pressure_some_percent", "Share of the last 10 seconds some tasks waited on the resource.", r.pressure.Some.Avg10, "host", host, "resource", r.name) + if r.pressure.FullAvailable { + m.gauge("symon_pressure_full_percent", "Share of the last 10 seconds all tasks waited on the resource.", r.pressure.Full.Avg10, "host", host, "resource", r.name) + } + } + } + + for _, temp := range data.Temperatures { + sensor := temp.Name + if temp.Label != "" { + sensor += "/" + temp.Label + } + m.gauge("symon_temperature_celsius", "Sensor temperature.", temp.Celsius, "host", host, "sensor", sensor) + } + for _, service := range data.Services { + m.gauge("symon_service_up", "1 when the service is running.", boolValue(service.Running), "host", host, "service", service.Name) + } + + for _, container := range data.Containers { + name := container.Name + if name == "" { + name = container.ShortID + } + labels := []string{"host", host, "container", name, "project", container.ComposeProject} + m.gauge("symon_container_cpu_percent", "Container CPU usage, as a share of the host.", container.CPU.PercentOfHost, labels...) + m.gauge("symon_container_memory_bytes", "Container memory in use.", container.Memory.Used, labels...) + if rates := container.Rates; rates != nil { + // no traffic rates for containers on the host network + if rates.RxBytesPerSec != nil { + m.gauge("symon_container_receive_bytes_per_second", "Container network traffic received.", *rates.RxBytesPerSec, labels...) + } + if rates.TxBytesPerSec != nil { + m.gauge("symon_container_transmit_bytes_per_second", "Container network traffic sent.", *rates.TxBytesPerSec, labels...) + } + m.gauge("symon_container_read_bytes_per_second", "Container disk reads.", rates.ReadBytesPerSec, labels...) + m.gauge("symon_container_write_bytes_per_second", "Container disk writes.", rates.WriteBytesPerSec, labels...) + } + } +} + +// parsePercent turns "40%" into 40 +func parsePercent(value string) (float64, bool) { + pct, err := strconv.ParseFloat(strings.TrimSuffix(strings.TrimSpace(value), "%"), 64) + return pct, err == nil +} + +func boolValue(b bool) float64 { + if b { + return 1 + } + return 0 +} + +// metricsWriter builds the Prometheus text format. Each metric's HELP, +// TYPE and samples have to be written together, so samples are grouped by +// metric and written at the end. +type metricsWriter struct { + families []*metricFamily + byName map[string]*metricFamily +} + +type metricFamily struct { + name string + kind string + help string + samples []string + // labels already written, a repeated set would make the output invalid + seen map[string]bool +} + +func newMetricsWriter() *metricsWriter { + return &metricsWriter{byName: map[string]*metricFamily{}} +} + +func (w *metricsWriter) gauge(name string, help string, value float64, labels ...string) { + w.add(name, "gauge", help, value, labels) +} + +func (w *metricsWriter) counter(name string, help string, value float64, labels ...string) { + w.add(name, "counter", help, value, labels) +} + +var labelEscaper = strings.NewReplacer(`\`, `\\`, `"`, `\"`, "\n", `\n`) + +// add records one sample. labels are name and value pairs. +func (w *metricsWriter) add(name string, kind string, help string, value float64, labels []string) { + family, ok := w.byName[name] + if !ok { + family = &metricFamily{name: name, kind: kind, help: help, seen: map[string]bool{}} + w.byName[name] = family + w.families = append(w.families, family) + } + + var set strings.Builder + for i := 0; i+1 < len(labels); i += 2 { + if i > 0 { + set.WriteByte(',') + } + fmt.Fprintf(&set, `%s="%s"`, labels[i], labelEscaper.Replace(labels[i+1])) + } + if family.seen[set.String()] { + return + } + family.seen[set.String()] = true + + sample := name + if set.Len() > 0 { + sample += "{" + set.String() + "}" + } + family.samples = append(family.samples, sample+" "+strconv.FormatFloat(value, 'g', -1, 64)+"\n") +} + +func (w *metricsWriter) writeTo(out io.Writer) error { + var b strings.Builder + for _, family := range w.families { + fmt.Fprintf(&b, "# HELP %s %s\n# TYPE %s %s\n", family.name, family.help, family.name, family.kind) + for _, sample := range family.samples { + b.WriteString(sample) + } + } + _, err := io.WriteString(out, b.String()) + return err +} diff --git a/client/internal/server/metrics_test.go b/client/internal/server/metrics_test.go new file mode 100644 index 0000000..c953e76 --- /dev/null +++ b/client/internal/server/metrics_test.go @@ -0,0 +1,92 @@ +package server + +import ( + "context" + "encoding/json" + "net/http/httptest" + "strings" + "testing" + + "github.com/dhamith93/SyMon/internal/api" + "github.com/dhamith93/SyMon/internal/monitor" +) + +// Snapshots has web1 reporting, db1 gone quiet and new1 not sent anything yet +func (f *fakeCollector) Snapshots(ctx context.Context, in *api.Void) (*api.SnapshotList, error) { + snapshot, err := json.Marshal(monitor.MonitorData{ + System: monitor.System{UpTimeSeconds: 3600}, + ProcUsage: monitor.CPU{LoadAvg: 37, Load1: 0.5}, + Memory: monitor.Memory{PercentageUsed: 42.5, Total: 16000, Available: 9000}, + Disk: []monitor.Disk{ + {FileSystem: "/dev/sda1", MountedOn: "/", Usage: monitor.DiskUsage{Size: 1000, Used: 400, Usage: "40%"}, Inodes: monitor.InodeUsage{Usage: "10%"}}, + }, + Networks: []monitor.Network{{Interface: "eth0", Usage: monitor.NetworkUsage{RxBytes: 1000, TxBytes: 2000}}}, + Services: []monitor.Service{{Name: "nginx", Running: true}}, + Containers: []monitor.Container{ + {ShortID: "aaa111", Name: "web", ComposeProject: "shop", CPU: monitor.ContainerCPU{PercentOfHost: 12}, + Rates: &monitor.ContainerRates{RxBytesPerSec: floatPtr(2048), ReadBytesPerSec: 10}}, + }, + }) + if err != nil { + return nil, err + } + return &api.SnapshotList{Hosts: []*api.HostSnapshot{ + {Host: "web1", Up: true, LastSeen: 1700000000, Time: 1700000000, SnapshotJson: string(snapshot)}, + {Host: "db1", LastSeen: 1690000000, Time: 1690000000, SnapshotJson: string(snapshot)}, + {Host: "new1", Up: true, LastSeen: 1700000000}, + }}, nil +} + +func TestMetrics(t *testing.T) { + s, _ := newTestServer(t, nil) + rec := httptest.NewRecorder() + s.routes().ServeHTTP(rec, httptest.NewRequest("GET", "/metrics", nil)) + body := rec.Body.String() + + if rec.Code != 200 || rec.Header().Get("Content-Type") != "text/plain; version=0.0.4; charset=utf-8" { + t.Fatalf("unexpected response %d %q: %s", rec.Code, rec.Header().Get("Content-Type"), body) + } + for _, line := range []string{ + "# TYPE symon_up gauge\n" + `symon_up{host="web1"} 1` + "\n" + `symon_up{host="db1"} 0` + "\n" + `symon_up{host="new1"} 1`, + `symon_last_seen_timestamp_seconds{host="db1"} 1.69e+09`, + `symon_cpu_usage_percent{host="web1"} 37`, + `symon_memory_total_bytes{host="web1"} 1.6777216e+10`, + `symon_disk_used_percent{host="web1",device="/dev/sda1",mount="/"} 40`, + "# TYPE symon_network_receive_bytes_total counter\n" + `symon_network_receive_bytes_total{host="web1",iface="eth0"} 1000`, + `symon_service_up{host="web1",service="nginx"} 1`, + `symon_container_receive_bytes_per_second{host="web1",container="web",project="shop"} 2048`, + } { + if !strings.Contains(body, line+"\n") { + t.Errorf("expected %q in:\n%s", line, body) + } + } + // db1 stopped reporting, so its old values are left out + if strings.Contains(body, `{host="db1",`) || strings.Contains(body, `symon_cpu_usage_percent{host="db1"}`) { + t.Errorf("expected only up and last seen for db1:\n%s", body) + } + // web1 has no transmit rate, as if it were on the host network + if strings.Contains(body, "symon_container_transmit_bytes_per_second") { + t.Errorf("expected no transmit rate:\n%s", body) + } +} + +func TestMetricsWriter(t *testing.T) { + w := newMetricsWriter() + w.gauge("a", "First.", 1, "name", `quote " backslash \ newline`+"\n") + w.counter("b_total", "Second.", 2) + w.gauge("a", "First.", 3, "name", "other") + // a repeated label set is dropped, Prometheus rejects the whole scrape otherwise + w.gauge("a", "First.", 4, "name", "other") + + var out strings.Builder + if err := w.writeTo(&out); err != nil { + t.Fatal(err) + } + want := "# HELP a First.\n# TYPE a gauge\n" + + `a{name="quote \" backslash \\ newline\n"} 1` + "\n" + + `a{name="other"} 3` + "\n" + + "# HELP b_total Second.\n# TYPE b_total counter\nb_total 2\n" + if out.String() != want { + t.Errorf("got:\n%s\nwant:\n%s", out.String(), want) + } +} diff --git a/client/internal/server/server.go b/client/internal/server/server.go index ca0a9c9..184fe5d 100644 --- a/client/internal/server/server.go +++ b/client/internal/server/server.go @@ -67,11 +67,14 @@ func (s *server) routes() http.Handler { mux.HandleFunc("GET /api/v1/hosts/{host}", s.getHost) mux.HandleFunc("GET /api/v1/hosts/{host}/series", s.getSeries) mux.HandleFunc("GET /api/v1/hosts/{host}/processes", s.getProcesses) + mux.HandleFunc("GET /api/v1/hosts/{host}/process-usage", s.getProcessUsage) mux.HandleFunc("GET /api/v1/hosts/{host}/custom-metrics", s.getCustomMetrics) + mux.HandleFunc("GET /api/v1/hosts/{host}/disk-forecasts", s.getDiskForecasts) mux.HandleFunc("GET /api/v1/alerts", s.getAlerts) mux.HandleFunc("/api/", func(w http.ResponseWriter, r *http.Request) { writeError(w, http.StatusNotFound, "no such endpoint") }) + mux.HandleFunc("GET /metrics", s.getMetrics) mux.HandleFunc("GET /install.sh", s.getInstallScript) mux.HandleFunc("GET /downloads/{file}", s.getDownload) mux.Handle("/", s.app()) @@ -98,6 +101,8 @@ type hostSummary struct { ActiveAlerts int32 `json:"activeAlerts"` WorstSeverity int32 `json:"worstSeverity"` Containers int32 `json:"containers"` + // null when no disk is filling up + DiskFullDays *float64 `json:"diskFullDays"` } func (s *server) getFleet(w http.ResponseWriter, r *http.Request) { @@ -124,6 +129,7 @@ func (s *server) getFleet(w http.ResponseWriter, r *http.Request) { ActiveAlerts: h.ActiveAlerts, WorstSeverity: h.WorstSeverity, Containers: h.Containers, + DiskFullDays: h.DiskFullDays, }) } writeJSON(w, map[string]any{"hosts": hosts}) @@ -206,6 +212,47 @@ func (s *server) getProcesses(w http.ResponseWriter, r *http.Request) { }) } +type processUsage struct { + Name string `json:"name"` + CPUAvg float64 `json:"cpuAvg"` + CPUPeak float64 `json:"cpuPeak"` + MemAvg float64 `json:"memAvg"` + MemPeak float64 `json:"memPeak"` + SeenPct float64 `json:"seenPct"` +} + +// getProcessUsage returns the busiest programs over a range +func (s *server) getProcessUsage(w http.ResponseWriter, r *http.Request) { + host := r.PathValue("host") + query := r.URL.Query() + from, to, err := timeRange(query.Get("from"), query.Get("to")) + if err != nil { + writeError(w, http.StatusBadRequest, err.Error()) + return + } + response, err := s.collector.ProcessUsage(r.Context(), &api.ProcessUsageRequest{Host: host, From: from, To: to}) + if err != nil { + writeGRPCError(w, "process usage of "+host, err) + return + } + processes := make([]processUsage, 0, len(response.Processes)) + for _, p := range response.Processes { + processes = append(processes, processUsage{ + Name: p.Name, + CPUAvg: p.CpuAvg, + CPUPeak: p.CpuPeak, + MemAvg: p.MemAvg, + MemPeak: p.MemPeak, + SeenPct: p.SeenPct, + }) + } + writeJSON(w, map[string]any{ + "snapshots": response.Snapshots, + "firstTime": response.FirstTime, + "processes": processes, + }) +} + func (s *server) getCustomMetrics(w http.ResponseWriter, r *http.Request) { host := r.PathValue("host") names, err := s.collector.CustomMetricNames(r.Context(), &api.HostRequest{Host: host}) @@ -216,6 +263,37 @@ func (s *server) getCustomMetrics(w http.ResponseWriter, r *http.Request) { writeJSON(w, map[string]any{"names": nonNil(names.Names)}) } +type diskForecast struct { + Device string `json:"device"` + Mount string `json:"mount"` + UsedPct float64 `json:"usedPct"` + PctPerDay float64 `json:"pctPerDay"` + BytesPerDay float64 `json:"bytesPerDay"` + // null when the disk is not filling up + DaysToFull *float64 `json:"daysToFull"` +} + +func (s *server) getDiskForecasts(w http.ResponseWriter, r *http.Request) { + host := r.PathValue("host") + response, err := s.collector.DiskForecasts(r.Context(), &api.HostRequest{Host: host}) + if err != nil { + writeGRPCError(w, "disk forecasts of "+host, err) + return + } + disks := make([]diskForecast, 0, len(response.Disks)) + for _, d := range response.Disks { + disks = append(disks, diskForecast{ + Device: d.Device, + Mount: d.Mount, + UsedPct: d.UsedPct, + PctPerDay: d.PctPerDay, + BytesPerDay: d.BytesPerDay, + DaysToFull: d.DaysToFull, + }) + } + writeJSON(w, map[string]any{"disks": disks}) +} + type alert struct { ID int64 `json:"id"` Host string `json:"host"` diff --git a/client/internal/server/server_test.go b/client/internal/server/server_test.go index 7b2389d..1d03129 100644 --- a/client/internal/server/server_test.go +++ b/client/internal/server/server_test.go @@ -22,11 +22,15 @@ import ( // fakeCollector knows one host, web1, and fails for the host "broken" type fakeCollector struct { api.UnimplementedMonitorDataServiceServer - lastSeries *api.SeriesRequest + lastSeries *api.SeriesRequest + lastProcessUsage *api.ProcessUsageRequest } func (f *fakeCollector) Fleet(ctx context.Context, in *api.Void) (*api.FleetSummary, error) { - return &api.FleetSummary{Hosts: []*api.HostSummary{{Name: "web1", Up: true, CpuPct: 37, ActiveAlerts: 2}}}, nil + return &api.FleetSummary{Hosts: []*api.HostSummary{ + {Name: "web1", Up: true, CpuPct: 37, ActiveAlerts: 2, DiskFullDays: floatPtr(12.5)}, + {Name: "db1", Up: true}, + }}, nil } func (f *fakeCollector) Snapshot(ctx context.Context, in *api.HostRequest) (*api.HostSnapshot, error) { @@ -55,6 +59,24 @@ func (f *fakeCollector) CustomMetricNames(ctx context.Context, in *api.HostReque return &api.NameList{}, nil } +func (f *fakeCollector) DiskForecasts(ctx context.Context, in *api.HostRequest) (*api.DiskForecastList, error) { + return &api.DiskForecastList{Disks: []*api.DiskForecast{ + {Device: "/dev/sda1", Mount: "/", UsedPct: 40}, + {Device: "/dev/sdb1", Mount: "/data", UsedPct: 60, PctPerDay: 2, BytesPerDay: 2e7, DaysToFull: floatPtr(20)}, + }}, nil +} + +func (f *fakeCollector) ProcessUsage(ctx context.Context, in *api.ProcessUsageRequest) (*api.ProcessUsageList, error) { + f.lastProcessUsage = in + return &api.ProcessUsageList{Snapshots: 4, FirstTime: 1700000000, Processes: []*api.ProcessUsage{ + {Name: "php-fpm", CpuAvg: 17.5, CpuPeak: 40, MemAvg: 3.75, MemPeak: 10, SeenPct: 50}, + }}, nil +} + +func floatPtr(v float64) *float64 { + return &v +} + func (f *fakeCollector) Alerts(ctx context.Context, in *api.AlertsRequest) (*api.AlertList, error) { return &api.AlertList{Alerts: []*api.AlertRecord{{Id: 7, Host: "web1", Rule: "CPU", Severity: 2, StartedAt: 1700000000}}}, nil } @@ -106,7 +128,34 @@ func TestFleet(t *testing.T) { if err := json.Unmarshal([]byte(body), &out); err != nil { t.Fatal(err) } - if code != 200 || len(out.Hosts) != 1 || out.Hosts[0].Name != "web1" || out.Hosts[0].CPUPct != 37 || out.Hosts[0].ActiveAlerts != 2 { + if code != 200 || len(out.Hosts) != 2 || out.Hosts[0].Name != "web1" || out.Hosts[0].CPUPct != 37 || out.Hosts[0].ActiveAlerts != 2 { + t.Errorf("unexpected response %d: %s", code, body) + } + if !strings.Contains(body, `"diskFullDays":12.5`) || !strings.Contains(body, `"diskFullDays":null`) { + t.Errorf("expected a forecast for web1 and null for db1: %s", body) + } +} + +func TestProcessUsage(t *testing.T) { + s, fake := newTestServer(t, nil) + code, body, _ := get(t, s, "/api/v1/hosts/web1/process-usage?from=1700000000&to=1700003600") + want := `{"firstTime":1700000000,"processes":[` + + `{"name":"php-fpm","cpuAvg":17.5,"cpuPeak":40,"memAvg":3.75,"memPeak":10,"seenPct":50}],"snapshots":4}` + if code != 200 || strings.TrimSpace(body) != want { + t.Errorf("unexpected response %d: %s", code, body) + } + if fake.lastProcessUsage.Host != "web1" || fake.lastProcessUsage.From != 1700000000 || fake.lastProcessUsage.To != 1700003600 { + t.Errorf("unexpected request %+v", fake.lastProcessUsage) + } +} + +func TestDiskForecasts(t *testing.T) { + s, _ := newTestServer(t, nil) + code, body, _ := get(t, s, "/api/v1/hosts/web1/disk-forecasts") + want := `{"disks":[` + + `{"device":"/dev/sda1","mount":"/","usedPct":40,"pctPerDay":0,"bytesPerDay":0,"daysToFull":null},` + + `{"device":"/dev/sdb1","mount":"/data","usedPct":60,"pctPerDay":2,"bytesPerDay":20000000,"daysToFull":20}]}` + if code != 200 || strings.TrimSpace(body) != want { t.Errorf("unexpected response %d: %s", code, body) } } diff --git a/client/web/src/components/BusiestProcesses.svelte b/client/web/src/components/BusiestProcesses.svelte new file mode 100644 index 0000000..4f53c64 --- /dev/null +++ b/client/web/src/components/BusiestProcesses.svelte @@ -0,0 +1,150 @@ + + +
+
+

Busiest over this range

+
+ + +
+
+

+ {#if snapshots > 0} + {start > from ? 'The last 24 hours of this range, from' : 'From'} + {snapshots.toLocaleString()} snapshots since {formatDateTime(firstTime)}. Each keeps only the top 10 processes, so averages + are a lower bound. + {:else if !error && !loading} + No process lists recorded in this range. + {/if} +

+ + {#if error} +

{error}

+ {:else if rows.length > 0} +
+ + + + + + + + + + + {#each rows as process (process.name)} + + + + + + + {/each} + +
ProgramAveragePeakSeen
{process.name}{formatPercent(view === 'CPU' ? process.cpuAvg : process.memAvg, 1)}{formatPercent(view === 'CPU' ? process.cpuPeak : process.memPeak, 1)}{formatPercent(process.seenPct)}
+
+ {/if} +
+ + diff --git a/client/web/src/components/ChartCard.svelte b/client/web/src/components/ChartCard.svelte index f3f0928..f157862 100644 --- a/client/web/src/components/ChartCard.svelte +++ b/client/web/src/components/ChartCard.svelte @@ -19,13 +19,15 @@ note?: string; onzoom?: (from: number, to: number) => void; onpick?: (time: number) => void; + // a picked time to mark on the chart, 0 for none + marker?: number; // replaces the line chart, like the per core heatmap body?: Snippet; // rows for the table view when body replaces the chart table?: Snippet; } - let { title, unit, data, loading = false, error = '', yMax, syncKey, from, to, area = false, note, onzoom, onpick, body, table }: Props = $props(); + let { title, unit, data, loading = false, error = '', yMax, syncKey, from, to, area = false, note, onzoom, onpick, marker = 0, body, table }: Props = $props(); let showTable = $state(false); const colors = ['--series-1', '--series-2', '--series-3', '--series-4', '--series-5', '--series-6', '--series-7', '--series-8']; @@ -98,7 +100,7 @@ {:else if body} {@render body()} {:else if data} - + {/if} {#if data && data.hidden.length > 0} diff --git a/client/web/src/components/TimeChart.svelte b/client/web/src/components/TimeChart.svelte index f450c7e..4864b4c 100644 --- a/client/web/src/components/TimeChart.svelte +++ b/client/web/src/components/TimeChart.svelte @@ -21,9 +21,11 @@ height?: number; onzoom?: (from: number, to: number) => void; onpick?: (time: number) => void; + // a picked time to mark with a line, 0 for none + marker?: number; } - let { data, unit, yMax, syncKey, from, to, area = false, height = 180, onzoom, onpick }: Props = $props(); + let { data, unit, yMax, syncKey, from, to, area = false, height = 180, onzoom, onpick, marker = 0 }: Props = $props(); let wrapper: HTMLDivElement; // uPlot owns this element, Svelte owns the tooltip next to it @@ -94,6 +96,7 @@ })), ], hooks: { + draw: [drawMarker], setCursor: [(u) => updateTooltip(u, colors)], setSelect: [ (u) => { @@ -122,6 +125,23 @@ u.over.style.cursor = onpick ? 'crosshair' : 'default'; } + // a dashed line at the picked time, the moment the process table shows + function drawMarker(u: uPlot) { + if (!marker) return; + const x = Math.round(u.valToPos(marker, 'x', true)); + if (x < u.bbox.left || x > u.bbox.left + u.bbox.width) return; + const ctx = u.ctx; + ctx.save(); + ctx.strokeStyle = cssVar('--accent'); + ctx.lineWidth = uPlot.pxRatio; + ctx.setLineDash([4 * uPlot.pxRatio, 3 * uPlot.pxRatio]); + ctx.beginPath(); + ctx.moveTo(x, u.bbox.top); + ctx.lineTo(x, u.bbox.top + u.bbox.height); + ctx.stroke(); + ctx.restore(); + } + function updateTooltip(u: uPlot, colors: string[]) { const idx = u.cursor.idx; if (!pointerInside || idx == null || u.cursor.left == null || u.cursor.left < 0) { @@ -163,6 +183,11 @@ } }); + $effect(() => { + void marker; + plot?.redraw(false); + }); + onMount(() => { resizeObserver = new ResizeObserver(() => plot?.setSize({ width: wrapper.clientWidth, height })); resizeObserver.observe(wrapper); diff --git a/client/web/src/lib/api.ts b/client/web/src/lib/api.ts index 79dd1a9..912e4d0 100644 --- a/client/web/src/lib/api.ts +++ b/client/web/src/lib/api.ts @@ -18,8 +18,24 @@ export interface HostSummary { worstSeverity: number; // running containers at the latest snapshot containers: number; + // days until the first disk is full, null when none is filling up + diskFullDays: number | null; } +// A disk's growth over the last week and when it fills up at that rate +export interface DiskForecast { + device: string; + mount: string; + usedPct: number; + pctPerDay: number; + bytesPerDay: number; + // null when the disk is not filling up + daysToFull: number | null; +} + +// disks that fill up sooner than this many days are shown as warnings +export const diskFullSoonDays = 30; + export interface SeriesData { label: string; // [unix seconds, value] @@ -60,6 +76,17 @@ export interface Process { Threads: number; } +// One program's share of a time range, with its processes added up +export interface ProcessUsage { + name: string; + cpuAvg: number; + cpuPeak: number; + memAvg: number; + memPeak: number; + // share of snapshots the program was in the top lists + seenPct: number; +} + export interface Processes { CPU: Process[] | null; Memory: Process[] | null; @@ -174,7 +201,10 @@ export const api = { series: (name: string, metric: string, from: number, to: number, options: { label?: string; maxPoints?: number; max?: boolean } = {}) => get(`${host(name)}/series`, { metric, from, to, ...options }), processes: (name: string, at?: number) => get<{ time: number; processes: Processes }>(`${host(name)}/processes`, { at }), + processUsage: (name: string, from: number, to: number) => + get<{ snapshots: number; firstTime: number; processes: ProcessUsage[] }>(`${host(name)}/process-usage`, { from, to }), customMetrics: (name: string) => get<{ names: string[] }>(`${host(name)}/custom-metrics`), + diskForecasts: (name: string) => get<{ disks: DiskForecast[] }>(`${host(name)}/disk-forecasts`), alerts: (filter: { host?: string; open?: boolean; from?: number; to?: number } = {}) => get<{ alerts: AlertRecord[] }>('/api/v1/alerts', filter), }; diff --git a/client/web/src/lib/format.test.ts b/client/web/src/lib/format.test.ts index a63b02c..94debba 100644 --- a/client/web/src/lib/format.test.ts +++ b/client/web/src/lib/format.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest'; -import { formatAgo, formatBytes, formatDuration, formatNumber, formatPercent, formatRate, formatValue } from './format'; +import { formatAgo, formatBytes, formatDays, formatDuration, formatNumber, formatPercent, formatRate, formatValue } from './format'; describe('formatBytes', () => { it('uses binary units', () => { @@ -45,3 +45,20 @@ describe('formatDuration', () => { expect(formatAgo(1000, 1000 + 180)).toBe('3m ago'); }); }); + +describe('formatDays', () => { + it('rounds to a unit that fits', () => { + expect(formatDays(0)).toBe('under a day'); + expect(formatDays(0.4)).toBe('under a day'); + expect(formatDays(1.2)).toBe('about 1 day'); + expect(formatDays(9.4)).toBe('about 9 days'); + expect(formatDays(20.4)).toBe('about 20 days'); + expect(formatDays(35)).toBe('about 5 weeks'); + expect(formatDays(120)).toBe('about 4 months'); + }); + + it('handles bad input', () => { + expect(formatDays(-1)).toBe('–'); + expect(formatDays(NaN)).toBe('–'); + }); +}); diff --git a/client/web/src/lib/format.ts b/client/web/src/lib/format.ts index fae3d28..27b27ee 100644 --- a/client/web/src/lib/format.ts +++ b/client/web/src/lib/format.ts @@ -64,6 +64,17 @@ export function formatAgo(unixSeconds: number, now = Date.now() / 1000): string return `${formatDuration(diff)} ago`; } +// how long until something happens: "under a day", "about 9 days", +// "about 5 weeks", "about 4 months" +export function formatDays(days: number): string { + if (!Number.isFinite(days) || days < 0) return '–'; + if (days < 1) return 'under a day'; + if (days < 1.5) return 'about 1 day'; + if (days < 21) return `about ${Math.round(days)} days`; + if (days < 60) return `about ${Math.round(days / 7)} weeks`; + return `about ${Math.round(days / 30)} months`; +} + export function formatDateTime(unixSeconds: number): string { if (!unixSeconds) return '–'; return new Date(unixSeconds * 1000).toLocaleString(undefined, { diff --git a/client/web/src/pages/Alerts.svelte b/client/web/src/pages/Alerts.svelte index 3c4fadc..f57a988 100644 --- a/client/web/src/pages/Alerts.svelte +++ b/client/web/src/pages/Alerts.svelte @@ -1,7 +1,7 @@