diff --git a/hosts/unstable/hosea/default.nix b/hosts/unstable/hosea/default.nix index 624b7aa..ac81c3e 100644 --- a/hosts/unstable/hosea/default.nix +++ b/hosts/unstable/hosea/default.nix @@ -285,7 +285,7 @@ in "uid": "kubernetes-overview", "title": "Kubernetes Overview", "schemaVersion": 38, - "version": 1, + "version": 2, "refresh": "30s", "time": {"from": "now-3h", "to": "now"}, "panels": [ @@ -294,14 +294,14 @@ in "type": "stat", "title": "Running Pods", "gridPos": {"x": 0, "y": 0, "w": 6, "h": 4}, - "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Running\"}) or vector(0)", "refId": "A"}] + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Running\"} == 1) or vector(0)", "refId": "A"}] }, { "id": 2, "type": "stat", "title": "Failed Pods", "gridPos": {"x": 6, "y": 0, "w": 6, "h": 4}, - "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Failed\"}) or vector(0)", "refId": "A"}], + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Failed\"} == 1) or vector(0)", "refId": "A"}], "fieldConfig": { "defaults": { "thresholds": { @@ -316,7 +316,7 @@ in "type": "stat", "title": "Pending Pods", "gridPos": {"x": 12, "y": 0, "w": 6, "h": 4}, - "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Pending\"}) or vector(0)", "refId": "A"}], + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_pod_status_phase{phase=\"Pending\"} == 1) or vector(0)", "refId": "A"}], "fieldConfig": { "defaults": { "thresholds": { @@ -326,6 +326,21 @@ in } } }, + { + "id": 8, + "type": "stat", + "title": "Nodes Ready", + "gridPos": {"x": 18, "y": 0, "w": 6, "h": 4}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(kube_node_status_condition{condition=\"Ready\",status=\"true\"} == 1) or vector(0)", "refId": "A"}], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [{"color": "red", "value": null}, {"color": "green", "value": 1}] + } + } + } + }, { "id": 4, "type": "timeseries", @@ -333,25 +348,47 @@ in "gridPos": {"x": 0, "y": 4, "w": 24, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "topk(10, rate(kube_pod_container_status_restarts_total[15m]) * 900)", "legendFormat": "{{namespace}}/{{pod}}", "refId": "A"}] }, + { + "id": 9, + "type": "timeseries", + "title": "Container CPU Usage (cores)", + "gridPos": {"x": 0, "y": 12, "w": 12, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "topk(10, rate(container_cpu_usage_seconds_total{container!=\"\",container!=\"POD\"}[5m]))", "legendFormat": "{{namespace}}/{{pod}}/{{container}}", "refId": "A"}] + }, + { + "id": 10, + "type": "timeseries", + "title": "Container Memory Usage", + "gridPos": {"x": 12, "y": 12, "w": 12, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "topk(10, container_memory_working_set_bytes{container!=\"\",container!=\"POD\"})", "legendFormat": "{{namespace}}/{{pod}}/{{container}}", "refId": "A"}], + "fieldConfig": {"defaults": {"unit": "bytes"}} + }, { "id": 5, "type": "table", "title": "All Pods", - "gridPos": {"x": 0, "y": 12, "w": 24, "h": 10}, - "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kube_pod_status_phase", "instant": true, "refId": "A"}] + "gridPos": {"x": 0, "y": 20, "w": 24, "h": 10}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kube_pod_status_phase == 1", "instant": true, "refId": "A"}] + }, + { + "id": 11, + "type": "table", + "title": "Deployments", + "gridPos": {"x": 0, "y": 30, "w": 24, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kube_deployment_status_replicas_available", "instant": true, "refId": "A"}] }, { "id": 6, "type": "timeseries", "title": "Node CPU Usage", - "gridPos": {"x": 0, "y": 22, "w": 12, "h": 8}, + "gridPos": {"x": 0, "y": 38, "w": 12, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "100 - (avg by(node) (rate(node_cpu_seconds_total{mode=\"idle\",job=\"node\"}[5m])) * 100)", "refId": "A"}] }, { "id": 7, "type": "timeseries", "title": "Node Memory Usage %", - "gridPos": {"x": 12, "y": 22, "w": 12, "h": 8}, + "gridPos": {"x": 12, "y": 38, "w": 12, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "(1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)) * 100", "refId": "A"}] } ] @@ -363,7 +400,7 @@ in "uid": "network-overview", "title": "Network & UniFi", "schemaVersion": 38, - "version": 1, + "version": 2, "refresh": "30s", "time": {"from": "now-3h", "to": "now"}, "panels": [ @@ -371,43 +408,82 @@ in "id": 1, "type": "stat", "title": "DNS Queries/s", - "gridPos": {"x": 0, "y": 0, "w": 8, "h": 4}, + "gridPos": {"x": 0, "y": 0, "w": 6, "h": 4}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(dnsmasq_queries_total[5m]) or vector(0)", "refId": "A"}] }, - { - "id": 2, - "type": "stat", - "title": "DHCP Leases", - "gridPos": {"x": 8, "y": 0, "w": 8, "h": 4}, - "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "kea_dhcp4_addresses_assigned_total or vector(0)", "refId": "A"}] - }, { "id": 3, "type": "stat", "title": "UniFi Devices", - "gridPos": {"x": 16, "y": 0, "w": 8, "h": 4}, + "gridPos": {"x": 6, "y": 0, "w": 6, "h": 4}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(unifipoller_device_uptime_seconds) or vector(0)", "refId": "A"}] }, + { + "id": 7, + "type": "stat", + "title": "WiFi Clients", + "gridPos": {"x": 12, "y": 0, "w": 6, "h": 4}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(unifipoller_client_wifi_tx_rate_bps) or vector(0)", "refId": "A"}] + }, + { + "id": 8, + "type": "stat", + "title": "Wired Clients", + "gridPos": {"x": 18, "y": 0, "w": 6, "h": 4}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "count(unifipoller_client_wired_tx_rate_bps) or vector(0)", "refId": "A"}] + }, + { + "id": 9, + "type": "timeseries", + "title": "WAN RX (bytes/s)", + "gridPos": {"x": 0, "y": 4, "w": 12, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_device_wan_receive_bytes_total[5m])", "legendFormat": "{{name}}", "refId": "A"}], + "fieldConfig": {"defaults": {"unit": "Bps"}} + }, + { + "id": 10, + "type": "timeseries", + "title": "WAN TX (bytes/s)", + "gridPos": {"x": 12, "y": 4, "w": 12, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_device_wan_transmit_bytes_total[5m])", "legendFormat": "{{name}}", "refId": "A"}], + "fieldConfig": {"defaults": {"unit": "Bps"}} + }, { "id": 4, "type": "timeseries", "title": "UniFi Port RX (bytes/s)", - "gridPos": {"x": 0, "y": 4, "w": 12, "h": 8}, + "gridPos": {"x": 0, "y": 12, "w": 12, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_port_receive_bytes_total[5m])", "legendFormat": "{{port_id}} {{name}}", "refId": "A"}] }, { "id": 5, "type": "timeseries", "title": "UniFi Port TX (bytes/s)", - "gridPos": {"x": 12, "y": 4, "w": 12, "h": 8}, + "gridPos": {"x": 12, "y": 12, "w": 12, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "rate(unifipoller_port_transmit_bytes_total[5m])", "legendFormat": "{{port_id}} {{name}}", "refId": "A"}] }, + { + "id": 11, + "type": "timeseries", + "title": "Top Client Throughput (bytes/s)", + "gridPos": {"x": 0, "y": 20, "w": 24, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "topk(10, rate(unifipoller_client_receive_bytes_total[5m]) + rate(unifipoller_client_transmit_bytes_total[5m]))", "legendFormat": "{{name}} {{ip}}", "refId": "A"}], + "fieldConfig": {"defaults": {"unit": "Bps"}} + }, { "id": 6, "type": "timeseries", "title": "Ping Latency (ms)", - "gridPos": {"x": 0, "y": 12, "w": 24, "h": 8}, + "gridPos": {"x": 0, "y": 28, "w": 12, "h": 8}, "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "probe_duration_seconds{job=\"ping\"} * 1000", "legendFormat": "{{instance}}", "refId": "A"}] + }, + { + "id": 12, + "type": "table", + "title": "Device Status", + "gridPos": {"x": 12, "y": 28, "w": 12, "h": 8}, + "targets": [{"datasource": {"type": "prometheus", "uid": "prometheus"}, "expr": "unifipoller_device_uptime_seconds", "instant": true, "refId": "A"}], + "fieldConfig": {"defaults": {"unit": "s"}} } ] } diff --git a/manifests/monitoring/config.yaml b/manifests/monitoring/config.yaml index 0e20d18..8cfd7d1 100644 --- a/manifests/monitoring/config.yaml +++ b/manifests/monitoring/config.yaml @@ -75,10 +75,14 @@ data: static_configs: - targets: - "hosea.shire-zebra.ts.net:9108" - # Restic rest-server backup metrics + # Restic rest-server backup metrics (HTTPS with basic auth) - job_name: restic_backups scheme: https tls_config: + insecure_skip_verify: true + basic_auth: + username_file: /etc/prometheus/secrets/restic/username + password_file: /etc/prometheus/secrets/restic/password static_configs: - targets: - "nas1.shire-zebra.ts.net:30248" @@ -93,6 +97,11 @@ data: static_configs: - targets: - "unpoller.monitoring.svc.cluster.local:9130" + # kube-state-metrics + - job_name: kube_state_metrics + static_configs: + - targets: + - "kube-state-metrics.monitoring.svc.cluster.local:8080" # This scrapes metrics from the Kubernetes kubelets - job_name: kubelet kubernetes_sd_configs: @@ -111,8 +120,6 @@ data: ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt metrics_path: /metrics/cadvisor bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - #authorization: - # credentials_file: /var/run/secrets/kubernetes.io/serviceaccount/token # This scrapes Service endpoints in Kubernetes - job_name: 'k8sservices' kubernetes_sd_configs: diff --git a/manifests/monitoring/deployment.yaml b/manifests/monitoring/deployment.yaml index eaf4b4f..46dd010 100644 --- a/manifests/monitoring/deployment.yaml +++ b/manifests/monitoring/deployment.yaml @@ -41,6 +41,9 @@ spec: subPath: alerts.yml - name: prometheus-storage-volume mountPath: /prometheus + - name: restic-credentials + mountPath: /etc/prometheus/secrets/restic + readOnly: true restartPolicy: Always volumes: - name: prometheus-config-values @@ -54,3 +57,6 @@ spec: - name: prometheus-storage-volume persistentVolumeClaim: claimName: prometheus-data + - name: restic-credentials + secret: + secretName: restic-credentials diff --git a/manifests/monitoring/kube-state-metrics.yaml b/manifests/monitoring/kube-state-metrics.yaml new file mode 100644 index 0000000..a6fa7b1 --- /dev/null +++ b/manifests/monitoring/kube-state-metrics.yaml @@ -0,0 +1,102 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: kube-state-metrics + namespace: monitoring + labels: + app.kubernetes.io/name: kube-state-metrics + app.kubernetes.io/version: "2.15.0" +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: kube-state-metrics + labels: + app.kubernetes.io/name: kube-state-metrics + app.kubernetes.io/version: "2.15.0" +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: kube-state-metrics +subjects: + - kind: ServiceAccount + name: kube-state-metrics + namespace: monitoring +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: kube-state-metrics + namespace: monitoring + labels: + app.kubernetes.io/name: kube-state-metrics + app.kubernetes.io/version: "2.15.0" +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: kube-state-metrics + template: + metadata: + labels: + app.kubernetes.io/name: kube-state-metrics + app.kubernetes.io/version: "2.15.0" + spec: + serviceAccountName: kube-state-metrics + securityContext: + runAsNonRoot: true + runAsUser: 65534 + fsGroup: 65534 + containers: + - name: kube-state-metrics + image: registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.15.0 + ports: + - name: metrics + containerPort: 8080 + - name: telemetry + containerPort: 8081 + securityContext: + readOnlyRootFilesystem: true + allowPrivilegeEscalation: false + capabilities: + drop: + - ALL + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 100m + memory: 128Mi + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + timeoutSeconds: 5 + readinessProbe: + httpGet: + path: / + port: 8081 + initialDelaySeconds: 5 + timeoutSeconds: 5 +--- +apiVersion: v1 +kind: Service +metadata: + name: kube-state-metrics + namespace: monitoring + labels: + app.kubernetes.io/name: kube-state-metrics + app.kubernetes.io/version: "2.15.0" +spec: + selector: + app.kubernetes.io/name: kube-state-metrics + ports: + - name: metrics + port: 8080 + targetPort: metrics + - name: telemetry + port: 8081 + targetPort: telemetry + clusterIP: None diff --git a/manifests/monitoring/kustomization.yaml b/manifests/monitoring/kustomization.yaml index ebf4798..a781015 100644 --- a/manifests/monitoring/kustomization.yaml +++ b/manifests/monitoring/kustomization.yaml @@ -10,3 +10,5 @@ resources: - service.yaml - ingress.yaml - unpoller.yaml + - kube-state-metrics.yaml + - restic-secret.yaml diff --git a/manifests/monitoring/restic-secret.yaml b/manifests/monitoring/restic-secret.yaml new file mode 100644 index 0000000..83018e3 --- /dev/null +++ b/manifests/monitoring/restic-secret.yaml @@ -0,0 +1,22 @@ +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: + name: restic-credentials + namespace: monitoring +spec: + refreshInterval: 1h + secretStoreRef: + name: bitwarden-login + kind: ClusterSecretStore + target: + name: restic-credentials + creationPolicy: Owner + data: + - secretKey: username + remoteRef: + key: cb04cb1b-1037-40d6-a187-b375015e7ef6 + property: username + - secretKey: password + remoteRef: + key: cb04cb1b-1037-40d6-a187-b375015e7ef6 + property: password