Compare commits

..

1 Commits

Author SHA1 Message Date
dependabot[bot] d0f8cdfb32 build(deps): bump jellyfin/jellyfin
Bumps jellyfin/jellyfin from 2026070606 to 2026072708.

---
updated-dependencies:
- dependency-name: jellyfin/jellyfin
  dependency-version: '2026072708'
  dependency-type: direct:production
  update-type: version-update:semver-major
...

Signed-off-by: dependabot[bot] <support@github.com>
2026-07-28 13:50:01 +00:00
12 changed files with 276 additions and 598 deletions
+47 -35
View File
@@ -1,53 +1,43 @@
// Grafana Alloy config for the docker host (replaces Promtail, EOL since
// 2026-03-02). Tails Docker container logs via docker_sd discovery, attaches
// container_name/compose_service/stream labels, and pushes them to the
// in-cluster Loki at loki-internal.lan (LAN-only Traefik Ingress on port 80).
// River config produced by `alloy convert -f promtail` from an equivalent
// Promtail YAML (so component names and argument shapes are guaranteed
// correct for Alloy v1.18).
// Grafana Alloy config for the docker host
// Tails Docker container logs via the docker_sd discovery +
// file streaming, applies container_name/image labels, and pushes them to
// the in-cluster Loki at loki-internal.lan (LAN-only Traefik Ingress).
discovery.docker "docker" {
host = "unix:///var/run/docker.sock"
refresh_interval = "5s"
logging {
level = "info"
format = "logfmt"
}
loki.process "docker" {
forward_to = [loki.write.default.receiver]
// Parse docker json-file log lines
// ({"log":"...","stream":"stdout","time":"..."}).
stage.json {
expressions = {
log = "log",
stream = "stream",
}
}
stage.labels {
values = {
stream = null,
}
}
// Discover running Docker containers via the host socket.
discovery.docker "containers" {
host = "unix:///var/run/docker.sock"
}
discovery.relabel "docker" {
targets = []
// Relabel rules: strip the leading slash from __meta_docker_container_name and
// expose useful labels (container name, compose service, image).
discovery.relabel "containers" {
targets = discovery.docker.containers.targets
// Strip leading / from container name.
rule {
source_labels = ["__meta_docker_container_name"]
regex = "/(.*)"
target_label = "container_name"
replacement = "$1"
}
// Compose service name (if label is set).
rule {
source_labels = ["__meta_docker_container_label_com_docker_compose_service"]
target_label = "compose_service"
}
// Log stream (stdout/stderr).
rule {
source_labels = ["__meta_docker_container_log_stream"]
target_label = "stream"
}
// Mark the source so dashboards/alerts can distinguish the docker host.
rule {
target_label = "source"
@@ -55,17 +45,39 @@ discovery.relabel "docker" {
}
}
loki.source.docker "docker" {
host = "unix:///var/run/docker.sock"
targets = discovery.docker.docker.targets
forward_to = [loki.process.docker.receiver]
relabel_rules = discovery.relabel.docker.rules
refresh_interval = "5s"
// Tail the container log files based on the docker_sd discovery. Each target
// exposes a `__path__`idders pointing at /var/lib/docker/containers/<id>/<id>-json.log
// (the default docker json-file driver path; we mount that host path read-only).
local.file "container_logs" {
targets = discovery.relabel.containers.output
tail_from_end = true
}
// Parse docker json-file logs ({"log":"...","stream":"stdout","time":"..."}).
loki.process "containers" {
forward_to = [loki.write.default.receiver]
stage.json {
expressions = {
log = "log",
stream = "stream",
time = "time",
}
}
stage.labels {
values = {
stream = "stream",
}
}
stage.timestamp {
source = "time"
format = "RFC3339Nano"
}
}
// Push to Loki in-cluster via the LAN-only Traefik Ingress (port 80).
loki.write "default" {
endpoint {
url = "http://loki-internal.lan/loki/api/v1/push"
}
external_labels = {}
}
@@ -62,7 +62,8 @@ services:
command:
- "run"
- "/etc/alloy/config.river"
- "--server.http.listen-addr=0.0.0.0:12345"
- "--server.http.listen.address=0.0.0.0"
- "--server.http.listen.port=12345"
volumes:
- type: bind
source: /root/homeprod/docker/infrastructure/observability/alloy.river
@@ -1,6 +1,6 @@
services:
jellyfin:
image: jellyfin/jellyfin:2026070606
image: jellyfin/jellyfin:2026072708
container_name: jellyfin
networks:
- default
-214
View File
@@ -1,214 +0,0 @@
# Talos control-plane node for the Raspberry Pi 4 — joins the r740 "kube" cluster
# as a third etcd member to restore quorum (2-of-3 majority). Unlike the p330
# failover node, this node is tainted "quorum" so no user workloads are ever
# scheduled on it; only essential DaemonSets (Cilium, etc.) that tolerate the
# taint land here for cluster networking.
#
# Secret handling: the cluster machine secrets are provided via
# var.machine_secrets_file (a local, gitignored JSON file in the provider's
# machine_secrets format). They are consumed by EPHEMERAL resources and
# WRITE-ONLY attributes so they never land in Terraform state. See
# variables.tf and scripts/extract-talos-secrets.sh for how to produce the
# file from the live r740 node.
terraform {
required_providers {
talos = {
source = "siderolabs/talos"
version = "0.11.0"
}
null = {
source = "hashicorp/null"
version = "3.2.3"
}
}
}
locals {
# Load the machine secrets from the gitignored JSON file. This local is only
# ever referenced by ephemeral resources / write-only attributes, so the
# values are never persisted to state.
machine_secrets = jsondecode(file(var.machine_secrets_file))
# Network config: static if node_subnet is provided, otherwise Talos DHCPs.
# The rpi4 uses DHCP (node_subnet = null), so only nameservers are patched in.
static_network = var.node_subnet == null ? {} : {
interfaces = [{
interface = var.network_interface
addresses = [var.node_subnet]
routes = var.node_gateway == null ? [] : [{ gateway = var.node_gateway }]
}]
}
network_patch = {
nameservers = var.nameservers
}
network_patch_merged = merge(local.network_patch, local.static_network)
machine_patch = {
install = {
image = var.installer_image
disk = var.install_disk
}
network = local.network_patch_merged
# NOTE: no Longhorn iSCSI/ext4 kernel modules here. This is a quorum-only
# node: the quorum taint keeps user workloads (and Longhorn replicas) off
# it, so the storage stack is not needed. Essential DaemonSets such as
# Cilium still run here for cluster networking and tolerate the taint.
sysctls = {
"fs.inotify.max_user_instances" = "1024"
"fs.inotify.max_user_watches" = "1048576"
}
kubelet = {
# Register the node already tainted so the scheduler never admits user
# workloads even before the null_resource below runs. NoSchedule is
# sufficient: essential DaemonSets (Cilium, etc.) tolerate it, but no
# user pods are admitted.
extraArgs = {
"register-with-taints" = "${var.quorum_taint_key}=${var.quorum_taint_value}:${var.quorum_taint_effect}"
}
}
}
}
# --- Ephemeral resources: secrets never stored in state ---------------------
#
# talos_machine_configuration generates the control-plane join config from the
# provided machine_secrets. The output (machine_configuration) is an ephemeral
# value — it can only flow into write-only attributes or provisioners, never
# into a persisted resource attribute.
ephemeral "talos_machine_configuration" "rpi4" {
cluster_name = var.cluster_name
machine_type = "controlplane"
cluster_endpoint = var.cluster_endpoint
machine_secrets = local.machine_secrets
config_patches = [
yamlencode({
machine = local.machine_patch
}),
# Pin the Kubernetes node name via a HostnameConfig document (Talos v1.13+).
# The old machine.network.hostname field conflicts with the default
# HostnameConfig document ("static hostname is already set"), so we use the
# document-based config with auto: off + an explicit hostname instead.
yamlencode({
apiVersion = "v1alpha1"
kind = "HostnameConfig"
hostname = var.rpi4_node_name
auto = "off"
})
]
}
# talos_client_configuration generates a Talos client config (talosconfig) from
# the machine_secrets, scoped to the rpi4 node. Also ephemeral — used only to
# drive the write-only client_configuration_wo on the apply resource.
ephemeral "talos_client_configuration" "rpi4" {
cluster_name = var.cluster_name
machine_secrets = local.machine_secrets
nodes = [var.rpi4_host]
}
# --- Apply the config to the node (write-only attrs → no secrets in state) --
#
# machine_configuration_input_wo and client_configuration_wo are write-only:
# Terraform uses them during apply but does NOT persist them to state. Only a
# hash of the machine config (machine_configuration_hash) is stored, for drift
# detection. Because the config patch contains a `machine.install` block, when
# Talos receives this config on a node booted from the SD card (maintenance)
# image it installs itself to install.disk and reboots into the installed
# system. As a controlplane node it then joins the existing etcd cluster as a
# new member and runs the control-plane components. With r740 + p330 + rpi4 the
# etcd cluster reaches 3 members → 2-of-3 quorum.
resource "talos_machine_configuration_apply" "rpi4" {
node = var.rpi4_host
client_configuration_wo = ephemeral.talos_client_configuration.rpi4.client_configuration
machine_configuration_input_wo = ephemeral.talos_machine_configuration.rpi4.machine_configuration
}
# --- Write the rendered config to disk for manual use ----------------------
#
# local_file.content cannot accept an ephemeral value (it would persist to
# state), so we use a null_resource local-exec provisioner instead —
# provisioners do not persist their arguments to state. This writes rpi4.yaml
# so the config can also be applied manually with
# `talosctl apply-config --nodes <rpi4_host> --file rpi4.yaml` if needed.
resource "null_resource" "rpi4_machine_config_file" {
triggers = {
# Re-run only when the (non-secret) inputs that shape the config change.
node = var.rpi4_node_name
install_disk = var.install_disk
installer_image = var.installer_image
taint = "${var.quorum_taint_key}=${var.quorum_taint_value}:${var.quorum_taint_effect}"
}
provisioner "local-exec" {
command = <<-EOT
set -euo pipefail
cat > "${path.module}/rpi4.yaml" <<'YAMLEOF'
${ephemeral.talos_machine_configuration.rpi4.machine_configuration}
YAMLEOF
echo "Wrote ${path.module}/rpi4.yaml"
EOT
}
depends_on = [talos_machine_configuration_apply.rpi4]
}
# --- Wait for the node, then label + taint ---------------------------------
#
# Wait for the node to register with Kubernetes (kubelet creates the Node
# object after Talos installs and reboots), then label it and (re)apply the
# quorum taint. This is idempotent: kubectl exits 0 if the label/taint already
# exists. The taint is also set via kubelet `register-with-taints`, so this
# null_resource is a safety net for manual edits / drift. The kubeconfig path
# is only used inside the provisioner (not persisted to state).
resource "null_resource" "rpi4_node_label_and_taint" {
triggers = {
node = var.rpi4_node_name
key = var.quorum_taint_key
value = var.quorum_taint_value
effect = var.quorum_taint_effect
kubeconfig = var.kubeconfig_path
}
provisioner "local-exec" {
# Wait for the node to show up, then label + taint. The wait loop is bounded
# by kubectl --timeout; tune it via TF_LOG / re-run if the node is slow to
# join (a controlplane node must first complete the etcd join handshake).
command = <<-EOT
set -euo pipefail
KUBECONFIG="${var.kubeconfig_path}"
export KUBECONFIG
NODE="${var.rpi4_node_name}"
echo "Waiting for node $NODE to be registered (kubelet creates the Node object once Talos has installed, rebooted and joined etcd)..."
# kubectl wait --for=condition=Ready fails instantly with NotFound if the
# node object doesn't exist yet, so poll for existence first.
# /bin/sh (dash) has no $SECONDS, so count iterations with a bounded loop.
tries=240 # 240 * 5s = 20 minutes max
until kubectl get node "$NODE" >/dev/null 2>&1; do
tries=$((tries - 1))
if [ "$tries" -le 0 ]; then
echo "Timed out waiting for node $NODE to register." >&2
exit 1
fi
sleep 5
done
echo "Node $NODE registered. Waiting for it to become Ready..."
# Now wait for Ready (a controlplane node needs etcd joined + apiserver up).
kubectl wait --for=condition=Ready "node/$NODE" --timeout=20m || \
kubectl wait --for=jsonpath='{.status.conditions[?(@.reason=="KubeletReady")].status}'=True "node/$NODE" --timeout=20m
# Quorum marker + taint (applied to the controlplane node).
kubectl label --overwrite node "$NODE" homeprod.io/quorum=true
# Apply the taint idempotently (kubectl taint --overwrite is a no-op if it exists).
kubectl taint --overwrite node "$NODE" \
"${var.quorum_taint_key}=${var.quorum_taint_value}:${var.quorum_taint_effect}"
echo "Node $NODE ready, labeled and tainted for quorum-only scheduling."
EOT
}
depends_on = [talos_machine_configuration_apply.rpi4]
}
-146
View File
@@ -1,146 +0,0 @@
# Variables for the Raspberry Pi 4 Talos control-plane node that joins the r740
# "kube" cluster as a third etcd member to restore quorum (2-of-3 majority).
#
# Secret handling: the cluster machine secrets (cluster id/secret, etcd/k8s
# certs, bootstrap token) are NOT read from terraform state (the r740 state is
# stale) and are NOT generated here (that would create a new, incompatible
# cluster). Instead they are provided via `machine_secrets_file` — a local,
# gitignored JSON file in the Talos provider's machine_secrets format. The
# file is produced once from the live r740 node (see
# scripts/extract-talos-secrets.sh) and stored in a real secret manager
# (Bitwarden); you paste it back to disk when running this module. Ephemeral
# resources + write-only attributes ensure the secrets never land in Terraform
# state.
variable "rpi4_host" {
description = "Reachable IP/hostname of the rpi4 Talos node (for Talos API access). With DHCP this is the leased IP (e.g. 10.1.2.135)."
type = string
}
variable "rpi4_node_name" {
description = "Kubernetes/Talos node name for the rpi4 (e.g. rpi4). Pinned via machine.network.hostname so the node registers with this name regardless of DHCP."
type = string
default = "rpi4"
}
# --- Cluster identity (no terraform_remote_state — state is stale) ----------
variable "cluster_name" {
description = "Name of the existing Talos cluster the rpi4 joins. Must match the cluster the r740 bootstrapped (kube-r740)."
type = string
default = "kube-r740"
}
variable "cluster_endpoint" {
description = "Endpoint (host:port) of the Talos/Kubernetes API on the cluster. Must match the r740 bootstrap endpoint."
type = string
default = "https://kube-r740.lan:6443"
}
# --- Secrets (provided manually, never in state) ---------------------------
variable "machine_secrets_file" {
description = <<EOT
Path to a local, gitignored JSON file containing the cluster machine secrets in
the Talos provider's machine_secrets format (cluster.id, cluster.secret, certs,
secrets.bootstrap_token, secrets.secretbox_encryption_secret, trustdinfo.token).
Generate it once from the live r740 node with
scripts/extract-talos-secrets.sh, store the contents in Bitwarden, and paste it
back to this file when running this module. The file MUST be gitignored — it
contains the cluster root of trust.
EOT
type = string
default = "secrets.json"
}
# --- Install / network -----------------------------------------------------
variable "installer_image" {
description = <<EOT
Talos installer image to use on the rpi4 (bare metal, ARM64).
Must be an ARM64 Image Factory build (schematic generated at
https://factory.talos.dev) for the Raspberry Pi 4 platform. Unlike the x86
control-plane nodes, this quorum node does NOT need the iSCSI/Longhorn
extensions because no user workloads or Longhorn replicas are scheduled on it
(the quorum taint keeps it empty); only essential DaemonSets (Cilium, etc.)
land here.
EOT
type = string
default = "factory.talos.dev/installer/ee21ef4a5ef808a9b7484cc0dda0f25075021691c8c09a276591eedb638ea1f9:v1.13.6"
}
variable "install_disk" {
description = "Block device path to install Talos on. For the rpi4 booting from the SD card this is /dev/mmcblk0."
type = string
default = "/dev/mmcblk0"
}
variable "node_subnet" {
description = <<EOT
Static IPv4 address in CIDR notation for the rpi4 node (e.g. 10.1.2.135/24).
Set to null (default) to use DHCP. The rpi4 uses DHCP, so a static address is
not required; the node registers with Kubernetes under rpi4_node_name regardless
of the leased IP.
EOT
type = string
default = null
}
variable "node_gateway" {
description = "IPv4 gateway for the rpi4 node. Ignored when node_subnet is null (DHCP)."
type = string
default = null
}
variable "network_interface" {
description = <<EOT
Primary network interface name on the rpi4. The built-in Ethernet port is eth0.
EOT
type = string
default = "eth0"
}
variable "nameservers" {
description = "DNS nameservers configured on the node (must work independently of kube)."
type = list(string)
default = ["10.1.2.148", "1.1.1.1"]
}
# --- Quorum taint ----------------------------------------------------------
variable "quorum_taint_key" {
description = "Taint key applied to the node to reserve it as a quorum-only member (no user workloads)."
type = string
default = "dedicated"
}
variable "quorum_taint_value" {
description = "Taint value applied to the node."
type = string
default = "quorum"
}
variable "quorum_taint_effect" {
description = "Taint effect applied to the node. NoSchedule is sufficient: essential DaemonSets (Cilium, etc.) tolerate it for networking, but no user workloads are admitted."
type = string
default = "NoSchedule"
validation {
condition = contains(["NoSchedule", "PreferNoSchedule", "NoExecute"], var.quorum_taint_effect)
error_message = "quorum_taint_effect must be NoSchedule, PreferNoSchedule or NoExecute."
}
}
# --- Kubeconfig for the label/taint null_resource --------------------------
variable "kubeconfig_path" {
description = <<EOT
Path to a kubeconfig for the cluster, used by the null_resource that waits for
the node and applies the quorum label/taint. This is NOT stored in state — it is
only referenced inside a local-exec provisioner. Point it at the r740 kube
module's kubeconfig (../../r740/kube/kubeconfig) or any valid kubeconfig for the
cluster.
EOT
type = string
default = "../../r740/kube/kubeconfig"
}
@@ -0,0 +1,27 @@
apiVersion: v1
kind: Secret
metadata:
name: alertmanager-config
namespace: observability
labels:
app.kubernetes.io/name: alertmanager
app.kubernetes.io/component: notifiers
type: Opaque
stringData:
alertmanager.yaml: |
global:
resolve_timeout: 5m
route:
receiver: default
group_by: ["alertname", "severity", "instance"]
group_wait: 30s
group_interval: 5m
repeat_interval: 4h
receivers:
- name: default
webhook_configs:
# n8n workflow webhook
- url: "http://n8n.lan/webhook/observability-alert"
send_resolved: true
@@ -12,57 +12,82 @@ data:
// Grafana Alloy — log collection only (metrics cluster+host collection is
// handled by vmagent/node-exporter/cAdvisor elsewhere).
// Forwards pod logs to the in-cluster Loki single-binary.
// River config produced by `alloy convert -f promtail` from an equivalent
// Promtail YAML (so component names and argument shapes are guaranteed
// correct for Alloy v1.18).
// -----------------------------------------------------------------------------
discovery.kubernetes "kubernetes_pods" {
role = "pod"
logging {
level = "info"
format = "logfmt"
}
selectors {
role = "pod"
field = "spec.nodeName=" + coalesce(sys.env("HOSTNAME"), constants.hostname)
// Tail every pod's log files written under /var/log/pods/* on the host.
local.file_match "pods" {
path_targets = [{
__path__ = "/var/log/pods/*/*/*.log",
}]
}
// Discover Kubernetes metadata for the tailed files (pod_name, namespace,
// container_name, uid).
discovery.kubernetes "pods" {
role = "pod"
}
// Stream-stage relabel: read the filename to extract namespace/pod/container
// metadata in line with Loki/Promtail's expectations.
lrelabel "pods" {
targets = local.file_match.pods.targets
rule {
source_labels = ["__path__"]
regex = "/var/log/pods/(?P<namespace>[^/]+)/(?P<pod>[^/]+)/(?P<container>[^/]+)/(?P<uid>[^/]+)/.*"
target_label = "namespace"
replacement = "$namespace"
}
rule {
source_labels = ["__path__"]
regex = "/var/log/pods/([^/]+)/([^/]+)/([^/]+)/[^/]+/.*"
target_label = "pod"
replacement = "$pod"
}
rule {
source_labels = ["__path__"]
regex = "/var/log/pods/([^/]+)/([^/]+)/([^/]+)/[^/]+/.*"
target_label = "container"
replacement = "$container"
}
}
loki.process "kubernetes_pods" {
// Tail the actual log files.
local.file "pods_logs" {
targets = lrelabel.pods.output
read_from = "beginning"
tail_from_end = false
}
// Parse CRI-style lines ({"log":"...","stream":"stdout","time":"..."}).
loki.process "pods" {
forward_to = [loki.write.default.receiver]
// CRI-style log lines on the host: {"log":"...","stream":"stdout","time":"..."}
stage.cri { }
}
discovery.relabel "kubernetes_pods" {
targets = discovery.kubernetes.kubernetes_pods.targets
rule {
source_labels = ["__meta_kubernetes_namespace"]
target_label = "namespace"
stage.json {
expressions = {
log = "log",
stream = "stream",
time = "time",
}
}
rule {
source_labels = ["__meta_kubernetes_pod_name"]
target_label = "pod"
stage.labels {
values = {
stream = "stream",
}
}
rule {
source_labels = ["__meta_kubernetes_pod_container_name"]
target_label = "container"
stage.timestamp {
source = "time"
format = "RFC3339Nano"
}
}
loki.source.file "kubernetes_pods" {
targets = discovery.relabel.kubernetes_pods.output
forward_to = [loki.process.kubernetes_pods.receiver]
file_match {
enabled = true
}
legacy_positions_file = "/tmp/positions.yaml"
}
// Push to Loki in-cluster (no auth on .lan).
loki.write "default" {
endpoint {
url = "http://loki.observability.svc.cluster.local:3100/loki/api/v1/push"
}
external_labels = {}
}
@@ -9,7 +9,9 @@ resources:
- vmstack-release.yaml
- loki-release.yaml
- alloy-release.yaml
- vmalert-rules.yaml
- alloy-config.yaml
- alertmanager-config.yaml
- vm-internal-ingress.yaml
- loki-internal-ingress.yaml
secretGenerator:
@@ -1,29 +1,10 @@
# Loki Helm values
# Uses Monolithic deployment mode with filesystem storage on a Longhorn PVC.
# Loki single-binary Helm values
# SOPS-encrypted before commit (per .sops.yaml rules).
# Uses the new chart-v18 "Monolithic" deployment mode (SingleBinary mode is
# deprecated in Loki 4 / chart >=18.x). Bundles storage on a Longhorn PVC —
# no minio, no gateway. Resource-light; tuned for a 2-node home cluster.
loki:
deploymentMode: Monolithic
# Filesystem storage — chunks + rules on the PVC.
storage:
type: filesystem
filesystem:
chunks_directory: /var/loki/chunks
rules_directory: /var/loki/rules
# Required schema config. tsdb store + filesystem object store, schema v13.
# See https://grafana.com/docs/loki/latest/operations/storage/schema/.
schemaConfig:
configs:
- from: "2024-04-01"
store: tsdb
object_store: filesystem
schema: v13
index:
prefix: index_
period: 24h
# For monolithic mode, point tsdb_shipper at the local store (no index gateway).
storage_config:
tsdb_shipper:
active_index_directory: /var/loki/index
cache_location: /var/loki/index-cache
monolithic:
persistence:
storageClassName: longhorn
@@ -40,9 +21,10 @@ loki:
# 7d log retention (user can tune later).
limitsConfig:
retention_period: 7d
# Disable distributed components + gateway + minio (monolithic only).
# Disable subcharts we don't need in monolithic mode.
test:
enabled: false
# Single binary only — disable distributed components.
backend:
replicas: 0
read:
@@ -56,27 +38,27 @@ gateway:
minio:
enabled: false
sops:
lastmodified: "2026-08-01T20:19:42Z"
mac: ENC[AES256_GCM,data:KYZQGIimeeOlt2lltCrKttLXzk2b7DOT8KNOhpeVqF/VNNcVxP7Srz6nCR5J7dI6wFSePspof5gNJEZlzP3EGJFQaGBf34hOGhHBUtlcscX8vj2JYmmJ85B+pylHjQNPHaXqLP4tAwIFfhcJi368K+GEAu8tztkxJKk7dtOOyrg=,iv:V24R3eSzW0edwPnPSSwEQqqSfiNrK6+ZyQ368xpJKhg=,tag:LmaMhhGwRIIG+SxTUJf5gQ==,type:str]
lastmodified: "2026-07-28T13:46:53Z"
mac: ENC[AES256_GCM,data:bk1sxO2qIY/FT3+9dTCfpxKBbb9NlyEfTFuUihURUA/AmMJHV4XOx4EaBK6YFbSOVCLny5/aDbUDfF5DMudZTvKMNyU5sWq4X3+6BBZaL3vHSw2+UTA8pdpkNYpLl4uDq5aDFvAN2v/f/XwOpMIEte1L+yi7Rl96hq3Is9TYxjg=,iv:0Ayu7eiEVRAml+SycHWgmmP+rf4fZxhXAALor6d01rI=,tag:+aVf2mehHqak3K5xOHSJCA==,type:str]
pgp:
- created_at: "2026-08-01T20:19:42Z"
- created_at: "2026-07-28T13:46:53Z"
enc: |-
-----BEGIN PGP MESSAGE-----
hQIMA7uy4qQr71wiAQ/+PDeQtUHMysWOkTjqU8vNAl1IQzDrwHskeBmLXLI99kdc
hU0afTjzUbomFdMRGvf0QZrcFbnD5OHskw5KRleDbYxOp4aXjzX0RsqSe9elTFAj
ml4XG0UrEV4Ygi4mTNOGzmINaIjobcn0tQd6nZFPUUA//udBcHsczs9x9Rrf+YTv
XrUdNAhveRlowvbvfNrDIh1OsOgyWyHJEJsKiLbt2b7e47PKhhRQzET+bNwhXzTb
GMY/w3rVYPddqx1wi5HPtYlhL3emYlmMYLLi1ITr1fKvAsgGKSJpaGS6yu+YkyiB
AJHou+2aCUy1TVk8U/Ne9TSrhsV1AJSIbtr8+iyKAtSsSuodam0Pyzn59uqL4nNd
1S4Nkvm1UDYv97hmSQdAzTGs+Aql6HyB/Wyi+rkMZQ0fl7+hdX5/eEDYZzijD0Nu
Mqk1iFbTn/rXir71VRWNttycIM1PfGAOUmLoyOR+TrL5Ecy0Sx9yx2DS9DecnqtS
39YI6UNqiEYdvW8Y+XELSFI/B5cjw7C1A5S9tecUxoOIBGZZSeRQa2WDSWb1dYky
Rzvncoeemgjkp8CHQhSqrylB+I4s0Y1HMUWR/q0wxew9juNEIP8ZUkfKcueuCYlm
5ui/RaeYatMFo8hg+DrxEtQIkbBZ/ILss/QzyHZZG3RMq0i9s/PIn5yipVhOgorS
XAFEeOrVaKpN06LfNP7/HTI+D/5HJL++Twuyz9fwC2Eb5saIIB6gpywiICH8HcD5
MRmmhl18x2YU+JdmkAcWNfH5+JHVnx8o+D/y9JqeCf6sle/9LyToazJv58Tc
=2u7h
hQIMA7uy4qQr71wiAQ//cYdDJn/PZWg3nx7eNeHjByT8kpYiTgi28Hbx0MmaR5Dd
koqKVn47tSO8UkoIYkJ2uoImBJMmIqSLg5XjIhleUmSrZnrCLm0zhdgljMnkphME
N9NYHc2gd65inztRDoZi12zxeDdPLcSo1tKPBvIQi33tWODbT6E8TJ4YGKBEv1+X
NggSqgADxmYevrVYrzJtHCf7o369CUjHUedSkGPDEk7ZNKimb2kEdi4qKV+QFiX7
y4iyqgeEepTjmloXSF6diC1F4cak96MqsNeIpiJLU5L1vWGGvb5J9znijpgqL6eX
aompNyrpUrwrF2dbsfb32o/AW4FjdJsC+7RfLPWs9qBneoUtU7LPisZpN+xQ9kgt
cx0CVY966y/0WTPm3coIhzO90fzL1kAn7+Bl5W5WhMBFRGDq7K0oqYcgdQmcr45o
yegnYxV4uO6Vgf9NV9zEqCvq+UqRZUwJhQSdmaPkvWQDhx91y7lca0WP90xwYMhN
jUvhHVcpZ7b++sbxNnvbhmnBBkAlwesQHVc85WSvok4kb3KpM1wju/j0KmOP2/lb
YazrzeXu47diSttopRU+JphtUQyuOdTqQ+k5x/1gt7qd8R4IbfKAFMKWPC/KLBnH
d4UIwa+ln/lWhhuR+/PxaPZrXGlm5SidM3r2SlH+1iGJfRzq1USaeEn9Vfp6OVvS
XgFZngqHN4Wv9zdT71EUGeixMz142maTlcPBM1OnHgiIkf3gLoInizU1VrZz/f8s
OOB4rHLwFjqj//Lze896q0kaWtnvZOuKdvTRsE7ep8VqAG2pVKwzcpINqnyTZss=
=ww9K
-----END PGP MESSAGE-----
fp: DC6910268E657FF70BA7EC289974494E76938DDC
encrypted_regex: ^(password|value|ssh-key|api-key|user|username|privateKey|clientSecret|clientId|apiKey|extraArgs.*|.*Secret.*|extraEnvVars|.*SECRET.*|.*secret.*|key|.*Password|.*\.ya?ml)$
@@ -0,0 +1,37 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: vmalert-rules
namespace: observability
labels:
app.kubernetes.io/name: vmalert
app.kubernetes.io/component: alerting-rules
data:
node.rules: |
groups:
- name: node
rules:
- alert: HighNodeCPU
expr: 100 - (avg by(instance)(rate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 90
for: 5m
labels:
severity: warning
annotations:
summary: "High CPU on {{ $labels.instance }}"
description: "CPU usage above 90% for 5m on {{ $labels.instance }}."
- alert: HighNodeRAM
expr: 1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) > 0.9
for: 10m
labels:
severity: warning
annotations:
summary: "RAM >90% on {{ $labels.instance }}"
description: "Available RAM below 10% for 10m on {{ $labels.instance }}."
- alert: LowDiskSpace
expr: 1 - (node_filesystem_avail_bytes / node_filesystem_size_bytes{fstype!~"tmpfs|overlay"}) > 0.85
for: 10m
labels:
severity: warning
annotations:
summary: "Disk >85% full on {{ $labels.instance }} {{ $labels.mountpoint }}"
description: "Filesystem {{ $labels.mountpoint }} on {{ $labels.instance }} is more than 85% full for 10m."
@@ -12,7 +12,7 @@ spec:
name: vm
namespace: observability
chart: victoria-metrics-k8s-stack
version: "0.87.0"
version: "0.43.2"
interval: 1m
valuesFrom:
- kind: Secret
@@ -1,12 +1,14 @@
# victoria-metrics-k8s-stack Helm values
# The chart installs the VictoriaMetrics operator + CRDs and creates VMSingle,
# VMAgent, VMAlert, VMAlertmanager CRs.
# victoria-metrics-k8s-stack Helm values
# SOPS-encrypted before commit (per .sops.yaml rules).
# All values for this chart live here; vmagent, grafana, vmalert, alertmanager,
# kube-state-metrics and node-exporter are sub-charts toggled on/off below.
# ---------------------------------------------------------------------------
# VictoriaMetrics single-binary (the metrics database)
# ---------------------------------------------------------------------------
vmsingle:
enabled: true
spec:
# Keep 7d of metrics; user can tune later.
retentionPeriod: 7d
storage:
storageClassName: longhorn
@@ -22,30 +24,26 @@ vmsingle:
limits:
cpu: "1"
memory: 1Gi
# ---------------------------------------------------------------------------
# vmagent — scrapes node-exporter, kube-state-metrics, kubelet, etc.
# The docker host runs its own vmagent that remote_writes through
# vm-internal.lan (the Traefik Ingress → this vmsingle).
# ---------------------------------------------------------------------------
# vmagent scrapes the cluster (node-exporter, kube-state-metrics, cAdvisor).
# That same vmagent is reused for the inner cluster; the docker host runs its
# own vmagent that remote_writes through vm-internal.lan.
vmagent:
enabled: true
spec:
selectAllByDefault: true
scrapeInterval: 20s
tolerations:
- key: ENC[AES256_GCM,data:SyutB0y9QtqQ,iv:EKRXNGtUovQYsESHFQTGygP09O2UWtLuBEQQcYZm3Oc=,tag:zh7CYOZleSgbbQhYZvcT+w==,type:str]
value: ENC[AES256_GCM,data:So+8xmHYSxs=,iv:7sfUBAaMZd7imjVx14lusVn58cJQ4HpTkaPy5Eal2dY=,tag:hRBjT6wuEfG3Z+ZiPJhwQQ==,type:str]
operator: Equal
effect: NoSchedule
- key: ENC[AES256_GCM,data:lj9OurynU8pRt4/MhgyevjmV0b7Zq+WG/9+zRRvPeA+RYZsV+A==,iv:LjuhErDT0bDlLcpo/AlNykz8Xu89B+nx1ja+HkPNXwo=,tag:5xM6XHrCdEejE7ituQ8W4Q==,type:str]
operator: Exists
effect: NoSchedule
# Tolerate the failover taint so the scrape DaemonSet covers all nodes.
tolerations:
- key: ENC[AES256_GCM,data:pERZJq2cU/V7,iv:eMeR91aKxAJmKpKLEL6pA9rcIHOrd+Mg+RmA3gt1xGw=,tag:l2igrTTHLclHPEO3bCq/dg==,type:str]
value: ENC[AES256_GCM,data:m0pJQf4RLt4=,iv:Au3k95QkD2wv3OP/yDyNQKDN5ZrDIuNAwhXaMCEEn+g=,tag:mEcqTG4e0aYwoHr4TtmgTg==,type:str]
operator: Equal
effect: NoSchedule
- key: ENC[AES256_GCM,data:lDW9tKXBee7p5HjH/FWmzpwtnrazCuohAQm47B5htn++6QUYIA==,iv:yJJdF/XaFXghwBOwC2lByMRM+M5MNP51ho3qTfAKqq0=,tag:WFNE1PMRFss+M/I6j/8Nbg==,type:str]
operator: Exists
effect: NoSchedule
# ---------------------------------------------------------------------------
# kube-state-metrics + node-exporter (metrics sources)
# ---------------------------------------------------------------------------
kube-state-metrics:
kubeStateMetrics:
enabled: true
prometheus-node-exporter:
nodeExporter:
enabled: true
# ---------------------------------------------------------------------------
# Grafana (UI) — served behind ingress at grafana.lan.
@@ -60,20 +58,15 @@ grafana:
size: 5Gi
adminUser: admin
# SOPS encrypts this when the file is processed.
adminPassword: ENC[AES256_GCM,data:vpPvJA/PktMnsg/IovYfsD4lDBndUw==,iv:b4eLMEcUMJj4BwyeYw1kw2sZuieyr2xL9et+R7ss0BU=,tag:u0RcTp6mkQNAGkblG4/xww==,type:str]
# Provision a Loki datasource alongside the chart's default VictoriaMetrics
adminPassword: ENC[AES256_GCM,data:VXtozCJ6XZ7JqWuIrDTmZkfWDeglVXl9JZ9T0sY2OUDY9MM=,iv:PO4leV1zpTFZTB2/hWipitGec4zrHfW/gOkJBrGk24k=,tag:9GinyiqoA4qrbyeKJ33Iuw==,type:str]
# Provision both the chart's default VictoriaMetrics datasource AND a Loki
# datasource so metric/log correlation works in one UI.
datasources:
datasources.yaml:
apiVersion: ENC[AES256_GCM,data:Uw==,iv:KYTMBV53yMbojMaqNinYpm/Uj5ylV86jcdJTdxuy+G0=,tag:IbIpiSBQFw7YSTQg0WRGqg==,type:int]
datasources:
- name: ENC[AES256_GCM,data:BPfR8Q==,iv:MkMhH3IJfBZzbIcqQw7y6bfJB/4Ew4feZNugjovk5O0=,tag:jaa43tzIXkE8kWZp2R/kcg==,type:str]
type: ENC[AES256_GCM,data:WWncww==,iv:hPqJE0KCidSBnf11wxFH+Je+YKztWmeToBqi39Qf4ZQ=,tag:c9enJ2VgQqDJVbasvgG+Dw==,type:str]
url: ENC[AES256_GCM,data:YkS66tpWXpW62Dtful0KNCwoZ3rG3nQtDBUsNQUwdQ+eqtG38JTrAudQqnsRD51q,iv:Mu3Sc6T2xypdot9dA1DHc8nD7UI8372nqlveLITR82E=,tag:1VmmIg+xR9qjWcvPKVYsJQ==,type:str]
access: ENC[AES256_GCM,data:af3GpjE=,iv:MAYKfklF+wCC/XN+4snebjPAnW4KpYzaMMGYds7rG18=,tag:X+oFiDAiCb1f7cfREqFT7A==,type:str]
isDefault: ENC[AES256_GCM,data:Zgu4Hn8=,iv:nnVB/in75nPAQos0Dfl6abBrHwT8tCt5X7lcCYBZbLc=,tag:u1UvkSj6SjcMm06NzKMGDA==,type:bool]
jsonData:
maxLines: ENC[AES256_GCM,data:88JJqA==,iv:lKsZa024I6223UEu0SHuZ3N9LZrzw0WBTfC+L2/3ljI=,tag:YvqieXK36xMmC/sSJ11NIg==,type:int]
additionalDataSources:
- name: Loki
type: loki
url: http://loki.observability.svc.cluster.local:3100
access: proxy
isDefault: false
ingress:
enabled: true
ingressClassName: traefik
@@ -85,17 +78,19 @@ grafana:
pathType: Prefix
tls: []
# ---------------------------------------------------------------------------
# vmalert — evaluates VMRule CRs against VictoriaMetrics, forwards firing
# alerts to Alertmanager. selectAllByDefault picks up all VMRules in the
# namespace (including our vmalert-rules.yaml VMRule CR).
# vmalert — evaluates the node.rules ConfigMap (see vmalert-rules.yaml) against
# VictoriaMetrics, forwards firing alerts to Alertmanager below.
# ---------------------------------------------------------------------------
vmalert:
enabled: true
spec:
selectAllByDefault: true
evaluationInterval: 20s
notifiers:
- url: http://vm-victoria-metrics-k8s-stack-alertmanager.observability.svc.cluster.local:9093
rule:
configMap: vmalert-rules
configMapKeyRef:
key: ENC[AES256_GCM,data:7obbOvuMe8Cpdg==,iv:oIzOCzFFA0cptWQMPiIOi5b76dR3zp96Y8JrdKOkE8A=,tag:ArteBgOhdz/7N4wNpWE83g==,type:str]
name: vmalert-rules
notifier:
alertmanagerUrl: http://vm-victoria-metrics-k8s-stack-alertmanager.observability.svc.cluster.local:9093
resources:
requests:
cpu: 50m
@@ -104,95 +99,52 @@ vmalert:
cpu: 200m
memory: 128Mi
# ---------------------------------------------------------------------------
# Alertmanager — 1 replica; inline config with a single n8n webhook receiver.
# n8n runs on the docker host and fans out to email/Telegram/whatever.
# Alertmanager — 1 replica; reads its config from the alertmanager-config
# Secret (see alertmanager-config.yaml), which has a single n8n webhook receiver.
# ---------------------------------------------------------------------------
alertmanager:
enabled: true
spec:
replicaCount: 1
port: "9093"
selectAllByDefault: true
storage:
volumeClaimTemplate:
spec:
storageClassName: longhorn
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
resources:
requests:
cpu: 50m
memory: 64Mi
limits:
cpu: 200m
memory: 128Mi
config:
route:
receiver: n8n-webhook
group_wait: 30s
group_interval: 5m
repeat_interval: 4h
receivers:
- name: n8n-webhook
webhook_configs:
- url: http://n8n.lan/webhook/observability-alert
send_resolved: true
extraRules:
node-alerts:
groups:
- name: node
rules:
- alert: HighNodeCPU
expr: 100 - (avg by (instance) (rate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 80
for: 10m
labels:
severity: warning
annotations:
summary: High CPU on {{ $labels.instance }}
description: CPU usage above 80% for 10 minutes.
- alert: HighNodeRAM
expr: 100 - ((node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100) > 85
for: 10m
labels:
severity: warning
annotations:
summary: High RAM on {{ $labels.instance }}
description: RAM usage above 85% for 10 minutes.
- alert: LowDiskSpace
expr: |
100 - ((node_filesystem_avail_bytes{mountpoint!~"/run.*|/var/lib/docker.*"} /
node_filesystem_size_bytes{mountpoint!~"/run.*|/var/lib/docker.*"}) * 100) > 85
for: 10m
labels:
severity: warning
annotations:
summary: Low disk space on {{ $labels.instance }} {{ $labels.mountpoint }}
description: Disk usage above 85% for 10 minutes.
replicaCount: 1
storage:
storageClassName: longhorn
accessModes:
- ReadWriteOnce
size: 2Gi
configFromSecretRef: ENC[AES256_GCM,data:C6NDZCkX76QLaZLw/vY3B01+AA==,iv:RSpe3Zv1mPfo0nj73tWrf5IBd+Sfqo1NUrERvmVPBgs=,tag:4ZGHA2Kyj9zpGn6eylNbXw==,type:str]
configKey: alertmanager.yaml
resources:
requests:
cpu: 50m
memory: 64Mi
limits:
cpu: 200m
memory: 128Mi
# ---------------------------------------------------------------------------
# Storage class default
# ---------------------------------------------------------------------------
defaultStorageClass: longhorn
sops:
lastmodified: "2026-07-28T16:39:09Z"
mac: ENC[AES256_GCM,data:Ymxmmkui0Nuv35Yq6v4BXM4xGhvjAufexW7fIm9fUIdTTXZ3e10PitNCZXKiaDwwCJfdEkNZri7jPAYzhTzKMIdcTKnOMImsMjdiJZwRwwbKcYVay0A6KSLJ8eFjDRRYhlCXbsAIRlQHpFh4UkssKdu40mIiudU0qIVm04ppbno=,iv:R8kpITPYSbQ1yZw5IK0fArssxUtIjNmvzknOrLJQcuk=,tag:/CRcr1VZAIKIPbcag05uTw==,type:str]
lastmodified: "2026-07-28T13:46:53Z"
mac: ENC[AES256_GCM,data:ZyCYO9b3OMJo8Br9tzbK+bN3rZvlkk/scOEUSHwkAABeQxL7JDbcpjHL8WauSY57wLnl3sK9JQnUHS+/ghOiZu/QDuodbhpoRrE8jpdjo1+qZaqJBqjbG1NIYU/aLQnHr4bnIzh8oX0PYnsu2uSqkg1c0c1KCJmrvHtXlFwtWOs=,iv:zh8yVzExwT29JF+4jLcZgmuSPM6R/d/2IWTcYfK3Qn4=,tag:LYTYY6rJVEKJRsOp6G/T0g==,type:str]
pgp:
- created_at: "2026-07-28T16:39:08Z"
- created_at: "2026-07-28T13:46:53Z"
enc: |-
-----BEGIN PGP MESSAGE-----
hQIMA7uy4qQr71wiARAAmOxyx5eH5Jykp/GDL79REgf61iGujtPiUVy8CT3o7O5k
tTGma6rRXK5wZV/iGN3wOQehF19Oy2iYwMfxilRwpB69RFwyKPKdHDDWMWNXv5Th
wJnNjVjWZ0xiRautWPn/TjDOJZ6j6p7lQ3IcQ3vzuap1RVRIXv081DxJGR9JbeYl
khmrl1+zecwxS87JK+fLko9NIBb5JgFEtEf1RvLyFak8nZu5cNCu+PCykALvUaO7
V5NAaF1aV9cdz8G9TZWmHZsIlDDE9/oro2aLR7cr8uxnKQhWerttqvyDYNwYzNKW
qgsUplTxTmyuc9DdnLDfJ55Ncuv9qS5/4/p5GpsHrteD6oiPNjXaxR8cBVFLvGwp
kQb/VZgOGAnkYlMCUG8APiwRhbpAlyy51hKBqVh+a9iRlzZEWHKlxcvIrcFAQaHt
sDXyg36F0GCoHiMAqHKG0kRcbTV7Sq0JZVoLU/qxvaDfqvyCZO7X/U/YsSWNbWL0
emyBfeSvBX9h62NJx2aw8xKyiL+aeVfzSKAaXHMd/r4gKxBw/kVHAljKwRzbL85j
UHeLUvqVkXs1mtIqGgUnpCcAVuV83omgcOwF44FitG/f9GCx+UWTvScwaw5ywuG3
J/4fOaA3K0Z4e/mOejARd4jLSPktSRpA01k9kjYY8UBn7TzQ0LrMLj5dxh8BpWrS
XgECZapi+q6IjIses7+JqkWiP25hcglhqK3LAiUapA0c5UmAa5Vf7jfnkFm8eUdZ
bA7ZnhX8nxK9n7JdSxrZhrbPWMO8hF/XKi7KTZMvgXFlb4/qP96y7oeqzKYcnvE=
=pmSq
hQIMA7uy4qQr71wiAQ/+KP62YdvAUJjxvFoWH1skMELYx9Lj/5TRXnrHOtqQrpXC
UzKNgS3XFmgeUcpA3wXvhTEPhICtdC9KDvxkF0jPTATqM2WFPcF7V2vjhTjuAMtU
/g4RR44j6U2WRZJFjs0tEpVGVrw+HXiaACdgqy1P6OmnmgvTSBvZvj3AsBpHOHeM
teaE11couXeYnrZ9NJeCvhbcvJfYAkdpqyCLJ9UWEro3QFm4z6tvs6D4rq+nAhBb
MntlvArijXunkwbHLZVPh8SyfxgGooy9PBSEX+ymyuxPYmjTlarYgO7oBQJUWjRk
BqkKns9KKNPyaUP/37tXJA8Gvgog/Z9JKDT28lq5noe6CAeHbBbUUo4hZZA14zcD
/Lmkc03KdOKf/eLnEFGtoWjQJ38FD5+uAh5Zkixm1WW0wV/PdqevJrG8mRjn8FuP
6GU+nWo+77BxDyKvA1O13iOtTnGZtg/Z+lXVUbpcM8F7QCGrabhLiWVe51iWjYfU
/Ub7HZHRDXmiz7Q6Kem5Wleoy3ME/lxFblx9v0LbO3Dc0lj/85EVNm9os8tt5PD0
cvGE/cGsZfUfgwbjEVpPnRN0SzlQKfEr9KVQP4DGNAqDOzEl2AD4Dg3l29oAgSda
ZRXdYbEN/ZIAGUg1wNVfZJf2KoJA+wm8c4uk7KAAI0+bIeaongovU4vnylVCHIbS
XgGDxEIDZGOueHms+xEMIW6QmdazFdkaLumY0XrEInHMwqhKM0NOyZL2zd/liZtE
wb8xm/V+eOWhjnE2RynNBCwMI6VeVYUPkin2MKQcq+nr7SVpWQsm2+Jbj5YtWCk=
=yKiR
-----END PGP MESSAGE-----
fp: DC6910268E657FF70BA7EC289974494E76938DDC
encrypted_regex: ^(password|value|ssh-key|api-key|user|username|privateKey|clientSecret|clientId|apiKey|extraArgs.*|.*Secret.*|extraEnvVars|.*SECRET.*|.*secret.*|key|.*Password|.*\.ya?ml)$