diff --git a/.gitignore b/.gitignore index f69041e..eac96f9 100644 --- a/.gitignore +++ b/.gitignore @@ -9,7 +9,7 @@ secrets/ .vault_pass .vault_password ansible/inventory/hosts.yml -ansible/group_vars/**/vault.yml +ansible/inventory/group_vars/**/vault.yml wireguard/wg0.conf wireguard/peers/ .aws/ diff --git a/README.md b/README.md index 9007efa..867b99d 100644 --- a/README.md +++ b/README.md @@ -14,8 +14,10 @@ single Hetzner CPX42 in Nuremberg (nbg1). Built with Mimir, Loki, Tempo, and Gra ```bash # 1. Provision infrastructure cd tofu -cp terraform.tfvars.example terraform.tfvars # fill in your tokens -tofu init +cp backend.hcl.example backend.hcl # set your Cloudflare Account ID +cp terraform.tfvars.example terraform.tfvars # set Hetzner tokens + SSH key +set -a && source .env && set +a # load R2 credentials into env +tofu init -backend-config=backend.hcl tofu apply # 2. Configure server diff --git a/ansible/ansible.cfg b/ansible/ansible.cfg index e1c3e4b..76904c5 100644 --- a/ansible/ansible.cfg +++ b/ansible/ansible.cfg @@ -3,11 +3,12 @@ inventory = inventory/hosts.yml roles_path = roles host_key_checking = False retry_files_enabled = False -stdout_callback = yaml +stdout_callback = default +result_format = yaml forks = 5 interpreter_python = /usr/bin/python3 -vault_password_file = .vault_pass +vault_password_file = $HOME/.config/watchtower-observe/vault_pass [ssh_connection] pipelining = True -ssh_args = -o ControlMaster=auto -o ControlPersist=60s +ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o IdentityFile=~/.ssh/watchtower-observe -o IdentitiesOnly=yes diff --git a/ansible/group_vars/all/main.yml b/ansible/inventory/group_vars/all/main.yml similarity index 94% rename from ansible/group_vars/all/main.yml rename to ansible/inventory/group_vars/all/main.yml index e3b2a94..3d5754a 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/inventory/group_vars/all/main.yml @@ -48,8 +48,8 @@ observability_volume_id: "" # When observability_volume_id is set, use the stable Hetzner by-id path. # When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only). luks_device_source: >- - {{ (observability_volume_id | length > 0) - | ternary('/dev/disk/by-id/scsi-0HC_Volume_' + observability_volume_id, '/dev/sdb') }} + {{ ((observability_volume_id | string) | length > 0) + | ternary('/dev/disk/by-id/scsi-0HC_Volume_' + (observability_volume_id | string), '/dev/sdb') }} luks_mapper_name: observability_data # Container images (pin in production) diff --git a/ansible/group_vars/all/vault.yml.example b/ansible/inventory/group_vars/all/vault.yml.example similarity index 100% rename from ansible/group_vars/all/vault.yml.example rename to ansible/inventory/group_vars/all/vault.yml.example diff --git a/ansible/inventory/hosts.yml.example b/ansible/inventory/hosts.yml.example index 73599a2..058744b 100644 --- a/ansible/inventory/hosts.yml.example +++ b/ansible/inventory/hosts.yml.example @@ -4,6 +4,6 @@ all: ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4 ansible_user: root ansible_python_interpreter: /usr/bin/python3 - # Hetzner Volume ID for the observability data disk. + # Hetzner Volume ID for the observability data disk (quote to keep as string). # Populate from: tofu output -json | jq -r '.ansible_inventory.value' - observability_volume_id: REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID + observability_volume_id: "REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID" diff --git a/ansible/roles/base/tasks/main.yml b/ansible/roles/base/tasks/main.yml index a38f00a..be4bbe9 100644 --- a/ansible/roles/base/tasks/main.yml +++ b/ansible/roles/base/tasks/main.yml @@ -9,7 +9,6 @@ ansible.builtin.dnf: name: - podman - - podman-compose - wireguard-tools - cryptsetup - nftables @@ -27,6 +26,12 @@ community.general.timezone: name: "{{ timezone }}" +- name: Ensure journald drop-in directory exists + ansible.builtin.file: + path: /etc/systemd/journald.conf.d + state: directory + mode: "0755" + - name: Configure persistent journald with size cap ansible.builtin.copy: dest: /etc/systemd/journald.conf.d/persistent.conf diff --git a/ansible/roles/luks/tasks/main.yml b/ansible/roles/luks/tasks/main.yml index 16d4cfe..10bfe85 100644 --- a/ansible/roles/luks/tasks/main.yml +++ b/ansible/roles/luks/tasks/main.yml @@ -22,7 +22,7 @@ to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server and udev has settled (try: udevadm settle --timeout=30). when: - - observability_volume_id | length > 0 + - (observability_volume_id | string) | length > 0 - not luks_device_stat.stat.exists - name: Create loop-backed LUKS image (if no real device) diff --git a/ansible/roles/observability/tasks/main.yml b/ansible/roles/observability/tasks/main.yml index dc6a3b9..d442bf2 100644 --- a/ansible/roles/observability/tasks/main.yml +++ b/ansible/roles/observability/tasks/main.yml @@ -57,6 +57,13 @@ mode: "0644" notify: Restart mimir +- name: Render Mimir runtime overrides + ansible.builtin.template: + src: mimir-runtime.yaml.j2 + dest: "{{ observability_config_dir }}/mimir/runtime.yaml" + mode: "0644" + notify: Restart mimir + - name: Render Loki config ansible.builtin.template: src: loki.yaml.j2 @@ -75,7 +82,7 @@ ansible.builtin.template: src: grafana.ini.j2 dest: "{{ observability_config_dir }}/grafana/grafana.ini" - mode: "0640" + mode: "0644" notify: Restart grafana - name: Render Grafana datasource provisioning @@ -163,6 +170,25 @@ - name: Force handlers to flush before health checks ansible.builtin.meta: flush_handlers +# Handlers only restart services when the quadlet templates change. On +# subsequent runs (or partial first runs) the units exist but are stopped, so +# explicitly ensure each backend is started before the health checks below. +# Quadlet-generated units cannot be `enabled` (they're auto-wired by their +# WantedBy= line), so we only set state=started — systemd treats this as a +# no-op when the unit is already active. +- name: Ensure observability services are started + ansible.builtin.systemd: + name: "{{ item }}" + state: started + daemon_reload: true + loop: + - wg-easy.service + - mimir.service + - loki.service + - tempo.service + - grafana.service + - caddy.service + - name: Deploy disk-guard script ansible.builtin.template: src: disk-guard.sh.j2 diff --git a/ansible/roles/observability/templates/mimir-runtime.yaml.j2 b/ansible/roles/observability/templates/mimir-runtime.yaml.j2 new file mode 100644 index 0000000..a25839e --- /dev/null +++ b/ansible/roles/observability/templates/mimir-runtime.yaml.j2 @@ -0,0 +1,4 @@ +# Mimir runtime overrides (hot-reloaded). Empty by default; add per-tenant +# overrides here without restarting Mimir. +# Reference: https://grafana.com/docs/mimir/latest/configure/about-runtime-configuration/ +overrides: {} diff --git a/ansible/roles/observability/templates/mimir.container.j2 b/ansible/roles/observability/templates/mimir.container.j2 index b7d67b0..2834388 100644 --- a/ansible/roles/observability/templates/mimir.container.j2 +++ b/ansible/roles/observability/templates/mimir.container.j2 @@ -9,6 +9,7 @@ ContainerName=mimir Network=observability.network Volume=mimir-data.volume:/var/lib/mimir:Z Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z +Volume={{ observability_config_dir }}/mimir/runtime.yaml:/etc/mimir/runtime.yaml:ro,Z EnvironmentFile={{ observability_secrets_dir }}/s3.env # -config.expand-env enables ${AWS_*} substitution from EnvironmentFile. diff --git a/ansible/roles/observability/templates/mimir.yaml.j2 b/ansible/roles/observability/templates/mimir.yaml.j2 index e89212e..19db5ae 100644 --- a/ansible/roles/observability/templates/mimir.yaml.j2 +++ b/ansible/roles/observability/templates/mimir.yaml.j2 @@ -5,6 +5,12 @@ target: all,alertmanager,overrides-exporter multitenancy_enabled: true +# The activity tracker writes its log to ./metrics-activity.log by default. +# Inside the container the working directory is "/" which is not writable for +# the non-root user, so point it at the data volume. +activity_tracker: + filepath: /var/lib/mimir/metrics-activity.log + server: http_listen_port: 8080 grpc_listen_port: 9095 diff --git a/ansible/roles/wireguard/tasks/main.yml b/ansible/roles/wireguard/tasks/main.yml index 142db5f..9e105c6 100644 --- a/ansible/roles/wireguard/tasks/main.yml +++ b/ansible/roles/wireguard/tasks/main.yml @@ -12,6 +12,23 @@ owner: root group: root +# wg-easy runs `wg-quick up wg0` inside its container, which calls +# iptables-legacy to set up the NAT POSTROUTING rule. AlmaLinux 10 doesn't +# auto-load the legacy iptables kernel modules, so the call fails with +# "can't initialize iptables table 'nat': Table does not exist" and wg0 is +# torn down. Load the modules now and persist them across reboots. +- name: Load iptables kernel modules required by wg-easy + community.general.modprobe: + name: "{{ item }}" + state: present + persistent: present + loop: + - ip_tables + - iptable_filter + - iptable_nat + - nf_nat + - nf_conntrack + - name: Enable IPv4 forwarding ansible.posix.sysctl: name: net.ipv4.ip_forward diff --git a/build/caddy/Dockerfile b/build/caddy/Dockerfile index b2eb2d3..89b922a 100644 --- a/build/caddy/Dockerfile +++ b/build/caddy/Dockerfile @@ -7,10 +7,10 @@ # Rebuild whenever the ARG versions below change; Ansible will detect the # Dockerfile checksum change and re-run `podman build`. -FROM golang:1.23-bookworm AS builder +FROM golang:1.26-trixie AS builder ARG XCADDY_VERSION=v0.3.5 -ARG CADDY_VERSION=v2.9.1 +ARG CADDY_VERSION=v2.11.2 ARG CORAZA_CADDY_VERSION=v2.5.0 ARG CORAZA_CRS_VERSION=v4.7.0 @@ -21,7 +21,7 @@ RUN xcaddy build "${CADDY_VERSION}" \ --with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}" # ── Runtime image ────────────────────────────────────────────────────────────── -FROM debian:bookworm-slim +FROM debian:trixie-slim RUN apt-get update \ && apt-get install -y --no-install-recommends ca-certificates \ diff --git a/tofu/backend.hcl.example b/tofu/backend.hcl.example new file mode 100644 index 0000000..f5f5af6 --- /dev/null +++ b/tofu/backend.hcl.example @@ -0,0 +1,10 @@ +# Copy to backend.hcl (gitignored) and set your Cloudflare Account ID. +# Pass at init time: tofu init -backend-config=backend.hcl +# +# Credentials are read from env vars — set before running tofu: +# export AWS_ACCESS_KEY_ID= +# export AWS_SECRET_ACCESS_KEY= + +endpoints = { + s3 = "https://.r2.cloudflarestorage.com" +} diff --git a/tofu/buckets.tf b/tofu/buckets.tf index 1d0cdc7..9d692fa 100644 --- a/tofu/buckets.tf +++ b/tofu/buckets.tf @@ -1,17 +1,8 @@ locals { buckets = { - mimir = { - name = "${var.bucket_prefix}-mimir" - retention_days = 365 - } - loki = { - name = "${var.bucket_prefix}-loki" - retention_days = 90 - } - tempo = { - name = "${var.bucket_prefix}-tempo" - retention_days = 90 - } + mimir = { name = "${var.bucket_prefix}-mimir" } + loki = { name = "${var.bucket_prefix}-loki" } + tempo = { name = "${var.bucket_prefix}-tempo" } } } @@ -22,40 +13,11 @@ resource "aws_s3_bucket" "telemetry" { bucket = each.value.name # Hetzner does not yet support all S3 ACL operations, so keep this minimal. + # Note: Hetzner Object Storage does not support the S3 Lifecycle Configuration + # API (PutBucketLifecycleConfiguration / GetBucketLifecycleConfiguration). + # Retention is enforced at the application layer: Mimir, Loki, and Tempo each + # have native retention settings configured via their respective config files. lifecycle { prevent_destroy = true } } - -resource "aws_s3_bucket_versioning" "telemetry" { - for_each = local.buckets - provider = aws.hetzner - - bucket = aws_s3_bucket.telemetry[each.key].id - versioning_configuration { - status = "Suspended" - } -} - -# Lifecycle rule: hard-delete safety net behind application retention. -resource "aws_s3_bucket_lifecycle_configuration" "telemetry" { - for_each = local.buckets - provider = aws.hetzner - - bucket = aws_s3_bucket.telemetry[each.key].id - - rule { - id = "hard-delete-after-${each.value.retention_days}d" - status = "Enabled" - - filter {} - - expiration { - days = each.value.retention_days - } - - abort_incomplete_multipart_upload { - days_after_initiation = 7 - } - } -} diff --git a/tofu/server.tf b/tofu/server.tf index 1f17794..fa4b234 100644 --- a/tofu/server.tf +++ b/tofu/server.tf @@ -31,17 +31,17 @@ resource "hcloud_firewall" "watchtower" { } resource "hcloud_primary_ip" "ipv4" { - name = "${var.server_name}-ipv4" - type = "ipv4" - assignee_type = "server" - auto_delete = false + name = "${var.server_name}-ipv4" + type = "ipv4" + location = var.location + auto_delete = false } resource "hcloud_primary_ip" "ipv6" { - name = "${var.server_name}-ipv6" - type = "ipv6" - assignee_type = "server" - auto_delete = false + name = "${var.server_name}-ipv6" + type = "ipv6" + location = var.location + auto_delete = false } resource "hcloud_volume" "observability" {