Files
statuspage/example.config.yaml
aziis98 c988ddf22f ui: uptime chip, #0004 shadows, tick marks, raw metrics; server: config meta injection
- ui: add uptime% chip (green >=99, red below) after ssh; remove "up" word
- ui: all shadows use #0004 shadow-color; modal shadow-lg derives from it
- ui: status modal draws a datetime tick mark every 20 history entries from the newest
- server: drop max_points/downsampling/bucketing; metrics served raw, plots share a ~24h window
- docs: fix ps --no-headers in top example; document raw metrics API
- server: inject <meta name=title/description> into index.html from config title + per-group host patterns (shared with --check recap)
2026-08-02 02:06:27 +02:00

135 lines
5.7 KiB
YAML

# statuspage example configuration
#
# Everything in this file is commented out on purpose. Uncomment (and adjust)
# the bits you need. See the README for the full reference.
#
# Save a copy as config.local.yaml and start the server with:
#
# ./bin/statuspage -c config.local.yaml
#
# or point Docker at it (see docker-compose.yml).
# title: "My machines" # shown in the header and browser tab
# # default: "Status Page"
# interactive: true # allow on-demand probes via the refresh
# # button on each machine card
# # default: false
# shared_metric_window: false # align all metric plot x-axes to the same
# # window (the last ~24h) so shapes can be
# # compared directly
# # default: false
# ---------- global ping defaults ----------
# ping:
# interval: 10s # how often to probe default: 10s
# timeout: 2s # per-probe timeout default: 2s
# tcp_port: 22 # port to TCP-connect to default: 22
# # (only IPv4 hosts are probed)
# ---------- global ssh defaults ----------
# ssh:
# interval: 10m # how often to run the script default: 10m
# user: root # ssh user default: root
# port: 22 # ssh port default: 22
# key: | # PEM private key (the only supported auth)
# -----BEGIN OPENSSH PRIVATE KEY-----
# ...
# -----END OPENSSH PRIVATE KEY-----
# script: | # shell script to run over ssh
# # status chips (name:status:info)
# echo "hostname:on:$(hostname)"
# echo "kernel:on:$(uname -sr)"
# echo "vulkan:on:$(vulkaninfo --summary 2>/dev/null | sed -n '/deviceName/p' | head -1 | cut -d: -f2-)"
# echo "docker:$(systemctl is-active docker 2>/dev/null | sed 's/active/on/; s/inactive/off/; s/failed/down/'):docker daemon"
# echo "nfs:$(systemctl is-active nfs-server 2>/dev/null | sed 's/active/on/; s/inactive/off/; s/failed/down/'):nfs-server"
#
# # metrics (name:metric:value)
# echo "cpu_pct:metric:$(top -bn1 2>/dev/null | awk '/%Cpu/ {printf "%.1f", 100-$8}')"
# echo "load1:metric:$(cut -d' ' -f1 /proc/loadavg)"
# echo "ram_pct:metric:$(free -m 2>/dev/null | awk '/^Mem:/ {printf "%.1f", $3/$2*100}')"
# echo "ram_used:metric:$(free -m 2>/dev/null | awk '/^Mem:/ {print $3}')MB"
# echo "disk_root:metric:$(df -h / 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')"
# echo "boot_days:metric:$(awk '{printf "%.2f", $1/86400}' /proc/uptime)"
# echo "cpu_temp:metric:$(sensors 2>/dev/null | awk '/Package id 0:/ {print $4; exit}' | tr -d '+°C')"
# echo "gpu_temp:metric:$(nvidia-smi --query-gpu=temperature.gpu --format=csv,noheader 2>/dev/null | tr -d ' ')"
# echo "gpu_util:metric:$(nvidia-smi --query-gpu=utilization.gpu --format=csv,noheader 2>/dev/null | tr -d ' %')"
#
# # free-form section, shown raw in the machine modal
# echo "---"
# echo "uptime: $(uptime -p)"
# echo "load: $(cat /proc/loadavg)"
# echo "mem: $(free -h | awk '/^Mem:/ {print $3 "/" $2}')"
# echo "disk: $(df -h / | awk 'NR==2 {print $4 " free"}')"
# echo "top: $(ps --no-headers -eo %cpu,comm --sort=-%cpu | head -4 | tr '\n' ';')"
#
# Script format, one line per item:
# name:status:info -> a check chip (status: on / off / down / anything)
# name:metric:value -> a numeric metric, accumulated into the history db
# --- -> everything after the first such line is shown raw
#
# If key or script is empty, the ssh check is skipped for that machine.
# ---------- suggested y-axis bounds for metric plots ----------
# metrics:
# cpu_pct:
# min: 0 # soft bounds: the plot only extends the
# max: 100 # axis if the data goes outside
# ram_pct:
# min: 0
# max: 100
# disk_root:
# min: 0
# max: 100
# gpu_util:
# min: 0
# max: 100
# gpu_temp:
# min: 0
# max: 105
# boot_days:
# min: 0
# ---------- machines grouped by section ----------
# groups:
# - name: Aula 3
# machines:
# - host: a3-dott1.example.net
# - host: a3-dott2.example.net
# name: dottorandi 2 # optional display name (defaults to host)
# ping: # per-machine overrides, inherit the rest
# interval: 30s
# tcp_port: 22
# ssh:
# interval: 1m
# user: root
#
# # host patterns are expanded at load time. Inside a {...} group:
# # {1,2,3} enumeration -> 1, 2, 3
# # {2..5,8..10} inclusive numeric range -> 2, 3, 4, 5, 8, 9, 10
# # {a3,a4} labels (any non-range item is kept literally)
# # several groups multiply in cartesian fashion:
# # server-{a3,a4}-{1..2}.example.org
# # -> server-a3-1.example.org, server-a3-2.example.org,
# # server-a4-1.example.org, server-a4-2.example.org
# # leading zeros are preserved: {01..03} -> 01, 02, 03
# # a name: applies to every expanded machine.
# #
# # Bad patterns fail at startup: unmatched/nested braces, ranges with
# # non-integer endpoints or extra dots (e.g. {1..5..10}), and descending
# # ranges. A single pattern may expand to at most 10000 machines.
# - name: Server room
# machines:
# - host: server-{a3,a4}-{1..10}.example.org
# ssh:
# interval: 1m
# ---------- machines without a group (shown under "Machines") ----------
# machines:
# - host: router.example.net
# name: router
# ping:
# interval: 5s
# tcp_port: 80