feat: comprehensive observability stack
- Add restic_backups and mattermost scrape jobs to Prometheus config - Add UnPoller deployment+service+secret for UniFi metrics (unpoller.yaml) - Add unifi scrape job targeting unpoller in-cluster - Add 4 Grafana dashboards as NixOS-managed etc files on hosea: - system-health: CPU, memory, disk, network per host - backup-health: restic backup freshness and duration - kubernetes: pod counts, restart rates, node resources - network: DNS queries, DHCP leases, UniFi ports, ping latency - Add grafana-api-token agenix secret placeholder on hosea - Open firewall ports 9100/9427/9558 for prometheus exporters in baseline.nix
This commit is contained in:
@@ -114,6 +114,12 @@
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
networking.firewall.allowedTCPPorts = [
|
||||||
|
9100 # prometheus node exporter
|
||||||
|
9427 # prometheus ping exporter
|
||||||
|
9558 # prometheus systemd exporter
|
||||||
|
];
|
||||||
|
|
||||||
security.sudo.extraRules = [
|
security.sudo.extraRules = [
|
||||||
{
|
{
|
||||||
users = [ "greg" ];
|
users = [ "greg" ];
|
||||||
|
|||||||
@@ -148,6 +148,269 @@ in
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
# After first deploy: create a Grafana service account + API token for Klaatu
|
||||||
|
# via the Grafana UI, then encrypt it: agenix -e secrets/grafana-api-token.age
|
||||||
|
age.secrets.grafana-api-token = {
|
||||||
|
file = ../../../secrets/grafana-api-token.age;
|
||||||
|
owner = "grafana";
|
||||||
|
};
|
||||||
|
|
||||||
|
environment.etc = {
|
||||||
|
"grafana-dashboards/system-health.json".text = ''
|
||||||
|
{
|
||||||
|
"uid": "system-health",
|
||||||
|
"title": "System Health Overview",
|
||||||
|
"schemaVersion": 38,
|
||||||
|
"version": 1,
|
||||||
|
"refresh": "30s",
|
||||||
|
"time": {"from": "now-3h", "to": "now"},
|
||||||
|
"panels": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Hosts Up",
|
||||||
|
"gridPos": {"x": 0, "y": 0, "w": 6, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(up{job=\"node\"} == 1)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Hosts Down",
|
||||||
|
"gridPos": {"x": 6, "y": 0, "w": 6, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(up{job=\"node\"} == 0) or vector(0)", "refId": "A"}],
|
||||||
|
"fieldConfig": {
|
||||||
|
"defaults": {
|
||||||
|
"thresholds": {
|
||||||
|
"mode": "absolute",
|
||||||
|
"steps": [{"color": "green", "value": null}, {"color": "red", "value": 1}]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "CPU Usage %",
|
||||||
|
"gridPos": {"x": 0, "y": 4, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "100 - (avg by(instance) (rate(node_cpu_seconds_total{mode=\"idle\"}[5m])) * 100)", "legendFormat": "{{instance}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 4,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Memory Usage %",
|
||||||
|
"gridPos": {"x": 12, "y": 4, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "(1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)) * 100", "legendFormat": "{{instance}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 5,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Disk Usage % (/)",
|
||||||
|
"gridPos": {"x": 0, "y": 12, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "100 - ((node_filesystem_avail_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"} / node_filesystem_size_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"}) * 100)", "legendFormat": "{{instance}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 6,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Network RX bytes/s",
|
||||||
|
"gridPos": {"x": 0, "y": 20, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(node_network_receive_bytes_total{device!~\"lo|tailscale.*\"}[5m])", "legendFormat": "{{instance}} {{device}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 7,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Network TX bytes/s",
|
||||||
|
"gridPos": {"x": 12, "y": 20, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(node_network_transmit_bytes_total{device!~\"lo|tailscale.*\"}[5m])", "legendFormat": "{{instance}} {{device}}", "refId": "A"}]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
|
||||||
|
"grafana-dashboards/backup-health.json".text = ''
|
||||||
|
{
|
||||||
|
"uid": "backup-health",
|
||||||
|
"title": "Backup Health",
|
||||||
|
"schemaVersion": 38,
|
||||||
|
"version": 1,
|
||||||
|
"refresh": "30s",
|
||||||
|
"time": {"from": "now-3h", "to": "now"},
|
||||||
|
"panels": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Backup Jobs Reporting",
|
||||||
|
"gridPos": {"x": 0, "y": 0, "w": 8, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(restic_last_success_timestamp_seconds) or vector(0)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Stale Backups (>26h)",
|
||||||
|
"gridPos": {"x": 8, "y": 0, "w": 8, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(time() - restic_last_success_timestamp_seconds > 93600) or vector(0)", "refId": "A"}],
|
||||||
|
"fieldConfig": {
|
||||||
|
"defaults": {
|
||||||
|
"thresholds": {
|
||||||
|
"mode": "absolute",
|
||||||
|
"steps": [{"color": "green", "value": null}, {"color": "red", "value": 1}]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"type": "table",
|
||||||
|
"title": "Last Backup Times",
|
||||||
|
"gridPos": {"x": 0, "y": 4, "w": 24, "h": 10},
|
||||||
|
"options": {"instant": true},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "restic_last_success_timestamp_seconds", "instant": true, "refId": "A"}],
|
||||||
|
"fieldConfig": {"defaults": {"unit": "dateTimeAsLocal"}}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 4,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Backup Duration (seconds)",
|
||||||
|
"gridPos": {"x": 0, "y": 14, "w": 24, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "restic_last_run_duration_seconds", "refId": "A"}]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
|
||||||
|
"grafana-dashboards/kubernetes.json".text = ''
|
||||||
|
{
|
||||||
|
"uid": "kubernetes-overview",
|
||||||
|
"title": "Kubernetes Overview",
|
||||||
|
"schemaVersion": 38,
|
||||||
|
"version": 1,
|
||||||
|
"refresh": "30s",
|
||||||
|
"time": {"from": "now-3h", "to": "now"},
|
||||||
|
"panels": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Running Pods",
|
||||||
|
"gridPos": {"x": 0, "y": 0, "w": 6, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Running\"}) or vector(0)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Failed Pods",
|
||||||
|
"gridPos": {"x": 6, "y": 0, "w": 6, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Failed\"}) or vector(0)", "refId": "A"}],
|
||||||
|
"fieldConfig": {
|
||||||
|
"defaults": {
|
||||||
|
"thresholds": {
|
||||||
|
"mode": "absolute",
|
||||||
|
"steps": [{"color": "green", "value": null}, {"color": "red", "value": 1}]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "Pending Pods",
|
||||||
|
"gridPos": {"x": 12, "y": 0, "w": 6, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Pending\"}) or vector(0)", "refId": "A"}],
|
||||||
|
"fieldConfig": {
|
||||||
|
"defaults": {
|
||||||
|
"thresholds": {
|
||||||
|
"mode": "absolute",
|
||||||
|
"steps": [{"color": "green", "value": null}, {"color": "yellow", "value": 1}]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 4,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Pod Restart Rate",
|
||||||
|
"gridPos": {"x": 0, "y": 4, "w": 24, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "topk(10, rate(kube_pod_container_status_restarts_total[15m]) * 900)", "legendFormat": "{{namespace}}/{{pod}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 5,
|
||||||
|
"type": "table",
|
||||||
|
"title": "All Pods",
|
||||||
|
"gridPos": {"x": 0, "y": 12, "w": 24, "h": 10},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kube_pod_status_phase", "instant": true, "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 6,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Node CPU Usage",
|
||||||
|
"gridPos": {"x": 0, "y": 22, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "100 - (avg by(node) (rate(node_cpu_seconds_total{mode=\"idle\",job=\"node\"}[5m])) * 100)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 7,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Node Memory Usage %",
|
||||||
|
"gridPos": {"x": 12, "y": 22, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "(1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)) * 100", "refId": "A"}]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
|
||||||
|
"grafana-dashboards/network.json".text = ''
|
||||||
|
{
|
||||||
|
"uid": "network-overview",
|
||||||
|
"title": "Network & UniFi",
|
||||||
|
"schemaVersion": 38,
|
||||||
|
"version": 1,
|
||||||
|
"refresh": "30s",
|
||||||
|
"time": {"from": "now-3h", "to": "now"},
|
||||||
|
"panels": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "DNS Queries/s",
|
||||||
|
"gridPos": {"x": 0, "y": 0, "w": 8, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(dnsmasq_queries_total[5m]) or vector(0)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "DHCP Leases",
|
||||||
|
"gridPos": {"x": 8, "y": 0, "w": 8, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kea_dhcp4_addresses_assigned_total or vector(0)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"type": "stat",
|
||||||
|
"title": "UniFi Devices",
|
||||||
|
"gridPos": {"x": 16, "y": 0, "w": 8, "h": 4},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(unifipoller_device_uptime_seconds) or vector(0)", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 4,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "UniFi Port RX (bytes/s)",
|
||||||
|
"gridPos": {"x": 0, "y": 4, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_port_receive_bytes_total[5m])", "legendFormat": "{{port_id}} {{name}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 5,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "UniFi Port TX (bytes/s)",
|
||||||
|
"gridPos": {"x": 12, "y": 4, "w": 12, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_port_transmit_bytes_total[5m])", "legendFormat": "{{port_id}} {{name}}", "refId": "A"}]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 6,
|
||||||
|
"type": "timeseries",
|
||||||
|
"title": "Ping Latency (ms)",
|
||||||
|
"gridPos": {"x": 0, "y": 12, "w": 24, "h": 8},
|
||||||
|
"targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "probe_duration_seconds{job=\"ping\"} * 1000", "legendFormat": "{{instance}}", "refId": "A"}]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
users.users = {
|
users.users = {
|
||||||
greg.extraGroups = [ "vboxusers" ];
|
greg.extraGroups = [ "vboxusers" ];
|
||||||
|
|
||||||
|
|||||||
@@ -56,6 +56,22 @@ data:
|
|||||||
static_configs:
|
static_configs:
|
||||||
- targets:
|
- targets:
|
||||||
- "hosea.shire-zebra.ts.net:9108"
|
- "hosea.shire-zebra.ts.net:9108"
|
||||||
|
# Restic rest-server backup metrics
|
||||||
|
- job_name: restic_backups
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- "nas1.shire-zebra.ts.net:30248"
|
||||||
|
# Mattermost metrics
|
||||||
|
- job_name: mattermost
|
||||||
|
metrics_path: /metrics
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- "mattermost.thehellings.lan:8065"
|
||||||
|
# UniFi metrics via unpoller
|
||||||
|
- job_name: unifi
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- "unpoller.monitoring.svc.cluster.local:9130"
|
||||||
# This scrapes metrics from the Kubernetes kubelets
|
# This scrapes metrics from the Kubernetes kubelets
|
||||||
- job_name: kubelet
|
- job_name: kubelet
|
||||||
kubernetes_sd_configs:
|
kubernetes_sd_configs:
|
||||||
|
|||||||
@@ -9,3 +9,4 @@ resources:
|
|||||||
- deployment.yaml
|
- deployment.yaml
|
||||||
- service.yaml
|
- service.yaml
|
||||||
- ingress.yaml
|
- ingress.yaml
|
||||||
|
- unpoller.yaml
|
||||||
|
|||||||
@@ -0,0 +1,65 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: unpoller-credentials
|
||||||
|
namespace: monitoring
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
# Replace with actual UniFi controller password for the 'unpoller' read-only user
|
||||||
|
password: "CHANGEME"
|
||||||
|
---
|
||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: unpoller
|
||||||
|
namespace: monitoring
|
||||||
|
labels:
|
||||||
|
app: unpoller
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: unpoller
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app: unpoller
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: unpoller
|
||||||
|
image: ghcr.io/unpoller/unpoller:latest
|
||||||
|
ports:
|
||||||
|
- name: metrics
|
||||||
|
containerPort: 9130
|
||||||
|
env:
|
||||||
|
- name: UP_UNIFI_DEFAULT_URL
|
||||||
|
value: "https://10.42.1.1:8443"
|
||||||
|
- name: UP_UNIFI_DEFAULT_VERIFY_SSL
|
||||||
|
value: "false"
|
||||||
|
- name: UP_UNIFI_DEFAULT_USER
|
||||||
|
value: "unpoller"
|
||||||
|
- name: UP_UNIFI_DEFAULT_PASS
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: unpoller-credentials
|
||||||
|
key: password
|
||||||
|
- name: UP_PROMETHEUS_DISABLE
|
||||||
|
value: "false"
|
||||||
|
- name: UP_INFLUXDB_DISABLE
|
||||||
|
value: "true"
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Service
|
||||||
|
metadata:
|
||||||
|
name: unpoller
|
||||||
|
namespace: monitoring
|
||||||
|
labels:
|
||||||
|
app: unpoller
|
||||||
|
spec:
|
||||||
|
selector:
|
||||||
|
app: unpoller
|
||||||
|
ports:
|
||||||
|
- name: metrics
|
||||||
|
port: 9130
|
||||||
|
targetPort: 9130
|
||||||
@@ -135,4 +135,6 @@ in
|
|||||||
];
|
];
|
||||||
|
|
||||||
"compose/attic.env.age".publicKeys = everyone;
|
"compose/attic.env.age".publicKeys = everyone;
|
||||||
|
|
||||||
|
"grafana-api-token.age".publicKeys = everyone;
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user