# cluster-watchdog.yaml
# ----------------------------------------------------------------------------
# Tells a human when the cluster is quietly broken: a node NotReady past a grace
# period, or a Longhorn volume faulted / degraded past a grace period. Written
# after rpi02 sat dead for four weeks (2026-08-14 -> 2026-09-11) with every
# volume degraded and nobody told -- see ../cluster-watchdog.md.
#
# Every 5 minutes a Job runs watchdog.py (mounted from the ConfigMap that
# kustomization.yaml generates from the file next to this one). It keeps its
# dedup state in the ConfigMap `cluster-watchdog-state` in this namespace and
# sends through Azure Communication Services (email and/or SMS) using the
# optional `acs-notify` Secret. Without the Secret it logs what it would send.
#
# Apply:  kubectl apply -k docs/k3s/cluster-watchdog/
---
apiVersion: v1
kind: Namespace
metadata:
  name: cluster-watchdog
---
apiVersion: v1
kind: ServiceAccount
metadata:
  name: cluster-watchdog
  namespace: cluster-watchdog
automountServiceAccountToken: true
---
# Cluster-wide READ ONLY: nodes and Longhorn volumes. Nothing here can mutate.
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
  name: cluster-watchdog
rules:
  - apiGroups: [""]
    resources: ["nodes"]
    verbs: ["get", "list"]
  - apiGroups: ["longhorn.io"]
    resources: ["volumes"]
    verbs: ["get", "list"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
  name: cluster-watchdog
roleRef:
  apiGroup: rbac.authorization.k8s.io
  kind: ClusterRole
  name: cluster-watchdog
subjects:
  - kind: ServiceAccount
    name: cluster-watchdog
    namespace: cluster-watchdog
---
# The ONLY write grant: the dedup-state ConfigMap in its own namespace. `create`
# cannot be name-scoped, so it is a separate rule; get/update are pinned to the
# one ConfigMap by resourceNames.
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
  name: cluster-watchdog-state
  namespace: cluster-watchdog
rules:
  - apiGroups: [""]
    resources: ["configmaps"]
    verbs: ["get", "update"]
    resourceNames: ["cluster-watchdog-state"]
  - apiGroups: [""]
    resources: ["configmaps"]
    verbs: ["create"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding
metadata:
  name: cluster-watchdog-state
  namespace: cluster-watchdog
roleRef:
  apiGroup: rbac.authorization.k8s.io
  kind: Role
  name: cluster-watchdog-state
subjects:
  - kind: ServiceAccount
    name: cluster-watchdog
    namespace: cluster-watchdog
---
apiVersion: batch/v1
kind: CronJob
metadata:
  name: cluster-watchdog
  namespace: cluster-watchdog
  labels:
    app: cluster-watchdog
spec:
  schedule: "*/5 * * * *"
  timeZone: "Europe/Copenhagen"
  concurrencyPolicy: Forbid
  startingDeadlineSeconds: 120
  successfulJobsHistoryLimit: 3
  failedJobsHistoryLimit: 3
  jobTemplate:
    spec:
      # One retry covers a transient API hiccup; the next scheduled run is 5 min away anyway.
      backoffLimit: 1
      activeDeadlineSeconds: 180
      template:
        metadata:
          labels:
            app: cluster-watchdog
        spec:
          serviceAccountName: cluster-watchdog
          restartPolicy: Never
          securityContext:
            runAsNonRoot: true
            runAsUser: 65534
            runAsGroup: 65534
            seccompProfile:
              type: RuntimeDefault
          containers:
            - name: watchdog
              # Pinned by digest for a reproducible on-call tool (repo convention; see
              # matter_server_maintenance.yaml). python:3.12-slim as of 2026-09-14.
              image: python:3.12-slim@sha256:78387bc3881b8273120a12ebe6c1ab22b018ccc2c9adf565ae1ac9b536e184ea
              command: ["python3", "/app/watchdog.py"]
              env:
                - name: STATE_NAMESPACE
                  valueFrom:
                    fieldRef:
                      fieldPath: metadata.namespace
                - name: STATE_CONFIGMAP
                  value: cluster-watchdog-state
                - name: NODE_GRACE_MINUTES
                  value: "10"
                - name: LONGHORN_GRACE_MINUTES
                  value: "60"
                - name: REMIND_HOURS
                  value: "24"
                # Delivery channels. Every key is optional: with none set the job
                # runs in log-only mode; email needs the first four, SMS the
                # endpoint + key + the two SMS keys. See ../cluster-watchdog.md.
                - name: ACS_ENDPOINT
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_ENDPOINT, optional: true }
                - name: ACS_ACCESS_KEY
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_ACCESS_KEY, optional: true }
                - name: ACS_EMAIL_FROM
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_EMAIL_FROM, optional: true }
                - name: ACS_EMAIL_TO
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_EMAIL_TO, optional: true }
                - name: ACS_SMS_FROM
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_SMS_FROM, optional: true }
                - name: ACS_SMS_TO
                  valueFrom:
                    secretKeyRef: { name: acs-notify, key: ACS_SMS_TO, optional: true }
              securityContext:
                allowPrivilegeEscalation: false
                readOnlyRootFilesystem: true
                capabilities:
                  drop: ["ALL"]
              resources:
                requests:
                  cpu: 10m
                  memory: 32Mi
                limits:
                  cpu: 200m
                  memory: 128Mi
              volumeMounts:
                - name: app
                  mountPath: /app
                  readOnly: true
          volumes:
            - name: app
              configMap:
                name: cluster-watchdog-script
