working deployment

This commit is contained in:
Jason Ross
2026-05-10 22:05:12 -05:00
parent 924203c8d8
commit 1b3cb87ceb
17 changed files with 103 additions and 69 deletions
+1 -1
View File
@@ -9,7 +9,7 @@ secrets/
.vault_pass .vault_pass
.vault_password .vault_password
ansible/inventory/hosts.yml ansible/inventory/hosts.yml
ansible/group_vars/**/vault.yml ansible/inventory/group_vars/**/vault.yml
wireguard/wg0.conf wireguard/wg0.conf
wireguard/peers/ wireguard/peers/
.aws/ .aws/
+4 -2
View File
@@ -14,8 +14,10 @@ single Hetzner CPX42 in Nuremberg (nbg1). Built with Mimir, Loki, Tempo, and Gra
```bash ```bash
# 1. Provision infrastructure # 1. Provision infrastructure
cd tofu cd tofu
cp terraform.tfvars.example terraform.tfvars # fill in your tokens cp backend.hcl.example backend.hcl # set your Cloudflare Account ID
tofu init cp terraform.tfvars.example terraform.tfvars # set Hetzner tokens + SSH key
set -a && source .env && set +a # load R2 credentials into env
tofu init -backend-config=backend.hcl
tofu apply tofu apply
# 2. Configure server # 2. Configure server
+4 -3
View File
@@ -3,11 +3,12 @@ inventory = inventory/hosts.yml
roles_path = roles roles_path = roles
host_key_checking = False host_key_checking = False
retry_files_enabled = False retry_files_enabled = False
stdout_callback = yaml stdout_callback = default
result_format = yaml
forks = 5 forks = 5
interpreter_python = /usr/bin/python3 interpreter_python = /usr/bin/python3
vault_password_file = .vault_pass vault_password_file = $HOME/.config/watchtower-observe/vault_pass
[ssh_connection] [ssh_connection]
pipelining = True pipelining = True
ssh_args = -o ControlMaster=auto -o ControlPersist=60s ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o IdentityFile=~/.ssh/watchtower-observe -o IdentitiesOnly=yes
@@ -48,8 +48,8 @@ observability_volume_id: ""
# When observability_volume_id is set, use the stable Hetzner by-id path. # When observability_volume_id is set, use the stable Hetzner by-id path.
# When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only). # When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only).
luks_device_source: >- luks_device_source: >-
{{ (observability_volume_id | length > 0) {{ ((observability_volume_id | string) | length > 0)
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + observability_volume_id, '/dev/sdb') }} | ternary('/dev/disk/by-id/scsi-0HC_Volume_' + (observability_volume_id | string), '/dev/sdb') }}
luks_mapper_name: observability_data luks_mapper_name: observability_data
# Container images (pin in production) # Container images (pin in production)
+2 -2
View File
@@ -4,6 +4,6 @@ all:
ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4 ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4
ansible_user: root ansible_user: root
ansible_python_interpreter: /usr/bin/python3 ansible_python_interpreter: /usr/bin/python3
# Hetzner Volume ID for the observability data disk. # Hetzner Volume ID for the observability data disk (quote to keep as string).
# Populate from: tofu output -json | jq -r '.ansible_inventory.value' # Populate from: tofu output -json | jq -r '.ansible_inventory.value'
observability_volume_id: REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID observability_volume_id: "REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID"
+6 -1
View File
@@ -9,7 +9,6 @@
ansible.builtin.dnf: ansible.builtin.dnf:
name: name:
- podman - podman
- podman-compose
- wireguard-tools - wireguard-tools
- cryptsetup - cryptsetup
- nftables - nftables
@@ -27,6 +26,12 @@
community.general.timezone: community.general.timezone:
name: "{{ timezone }}" name: "{{ timezone }}"
- name: Ensure journald drop-in directory exists
ansible.builtin.file:
path: /etc/systemd/journald.conf.d
state: directory
mode: "0755"
- name: Configure persistent journald with size cap - name: Configure persistent journald with size cap
ansible.builtin.copy: ansible.builtin.copy:
dest: /etc/systemd/journald.conf.d/persistent.conf dest: /etc/systemd/journald.conf.d/persistent.conf
+1 -1
View File
@@ -22,7 +22,7 @@
to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server
and udev has settled (try: udevadm settle --timeout=30). and udev has settled (try: udevadm settle --timeout=30).
when: when:
- observability_volume_id | length > 0 - (observability_volume_id | string) | length > 0
- not luks_device_stat.stat.exists - not luks_device_stat.stat.exists
- name: Create loop-backed LUKS image (if no real device) - name: Create loop-backed LUKS image (if no real device)
+27 -1
View File
@@ -57,6 +57,13 @@
mode: "0644" mode: "0644"
notify: Restart mimir notify: Restart mimir
- name: Render Mimir runtime overrides
ansible.builtin.template:
src: mimir-runtime.yaml.j2
dest: "{{ observability_config_dir }}/mimir/runtime.yaml"
mode: "0644"
notify: Restart mimir
- name: Render Loki config - name: Render Loki config
ansible.builtin.template: ansible.builtin.template:
src: loki.yaml.j2 src: loki.yaml.j2
@@ -75,7 +82,7 @@
ansible.builtin.template: ansible.builtin.template:
src: grafana.ini.j2 src: grafana.ini.j2
dest: "{{ observability_config_dir }}/grafana/grafana.ini" dest: "{{ observability_config_dir }}/grafana/grafana.ini"
mode: "0640" mode: "0644"
notify: Restart grafana notify: Restart grafana
- name: Render Grafana datasource provisioning - name: Render Grafana datasource provisioning
@@ -163,6 +170,25 @@
- name: Force handlers to flush before health checks - name: Force handlers to flush before health checks
ansible.builtin.meta: flush_handlers ansible.builtin.meta: flush_handlers
# Handlers only restart services when the quadlet templates change. On
# subsequent runs (or partial first runs) the units exist but are stopped, so
# explicitly ensure each backend is started before the health checks below.
# Quadlet-generated units cannot be `enabled` (they're auto-wired by their
# WantedBy= line), so we only set state=started — systemd treats this as a
# no-op when the unit is already active.
- name: Ensure observability services are started
ansible.builtin.systemd:
name: "{{ item }}"
state: started
daemon_reload: true
loop:
- wg-easy.service
- mimir.service
- loki.service
- tempo.service
- grafana.service
- caddy.service
- name: Deploy disk-guard script - name: Deploy disk-guard script
ansible.builtin.template: ansible.builtin.template:
src: disk-guard.sh.j2 src: disk-guard.sh.j2
@@ -0,0 +1,4 @@
# Mimir runtime overrides (hot-reloaded). Empty by default; add per-tenant
# overrides here without restarting Mimir.
# Reference: https://grafana.com/docs/mimir/latest/configure/about-runtime-configuration/
overrides: {}
@@ -9,6 +9,7 @@ ContainerName=mimir
Network=observability.network Network=observability.network
Volume=mimir-data.volume:/var/lib/mimir:Z Volume=mimir-data.volume:/var/lib/mimir:Z
Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z
Volume={{ observability_config_dir }}/mimir/runtime.yaml:/etc/mimir/runtime.yaml:ro,Z
EnvironmentFile={{ observability_secrets_dir }}/s3.env EnvironmentFile={{ observability_secrets_dir }}/s3.env
# -config.expand-env enables ${AWS_*} substitution from EnvironmentFile. # -config.expand-env enables ${AWS_*} substitution from EnvironmentFile.
@@ -5,6 +5,12 @@ target: all,alertmanager,overrides-exporter
multitenancy_enabled: true multitenancy_enabled: true
# The activity tracker writes its log to ./metrics-activity.log by default.
# Inside the container the working directory is "/" which is not writable for
# the non-root user, so point it at the data volume.
activity_tracker:
filepath: /var/lib/mimir/metrics-activity.log
server: server:
http_listen_port: 8080 http_listen_port: 8080
grpc_listen_port: 9095 grpc_listen_port: 9095
+17
View File
@@ -12,6 +12,23 @@
owner: root owner: root
group: root group: root
# wg-easy runs `wg-quick up wg0` inside its container, which calls
# iptables-legacy to set up the NAT POSTROUTING rule. AlmaLinux 10 doesn't
# auto-load the legacy iptables kernel modules, so the call fails with
# "can't initialize iptables table 'nat': Table does not exist" and wg0 is
# torn down. Load the modules now and persist them across reboots.
- name: Load iptables kernel modules required by wg-easy
community.general.modprobe:
name: "{{ item }}"
state: present
persistent: present
loop:
- ip_tables
- iptable_filter
- iptable_nat
- nf_nat
- nf_conntrack
- name: Enable IPv4 forwarding - name: Enable IPv4 forwarding
ansible.posix.sysctl: ansible.posix.sysctl:
name: net.ipv4.ip_forward name: net.ipv4.ip_forward
+3 -3
View File
@@ -7,10 +7,10 @@
# Rebuild whenever the ARG versions below change; Ansible will detect the # Rebuild whenever the ARG versions below change; Ansible will detect the
# Dockerfile checksum change and re-run `podman build`. # Dockerfile checksum change and re-run `podman build`.
FROM golang:1.23-bookworm AS builder FROM golang:1.26-trixie AS builder
ARG XCADDY_VERSION=v0.3.5 ARG XCADDY_VERSION=v0.3.5
ARG CADDY_VERSION=v2.9.1 ARG CADDY_VERSION=v2.11.2
ARG CORAZA_CADDY_VERSION=v2.5.0 ARG CORAZA_CADDY_VERSION=v2.5.0
ARG CORAZA_CRS_VERSION=v4.7.0 ARG CORAZA_CRS_VERSION=v4.7.0
@@ -21,7 +21,7 @@ RUN xcaddy build "${CADDY_VERSION}" \
--with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}" --with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}"
# ── Runtime image ────────────────────────────────────────────────────────────── # ── Runtime image ──────────────────────────────────────────────────────────────
FROM debian:bookworm-slim FROM debian:trixie-slim
RUN apt-get update \ RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates \ && apt-get install -y --no-install-recommends ca-certificates \
+10
View File
@@ -0,0 +1,10 @@
# Copy to backend.hcl (gitignored) and set your Cloudflare Account ID.
# Pass at init time: tofu init -backend-config=backend.hcl
#
# Credentials are read from env vars — set before running tofu:
# export AWS_ACCESS_KEY_ID=<r2-access-key-id>
# export AWS_SECRET_ACCESS_KEY=<r2-secret-access-key>
endpoints = {
s3 = "https://<ACCOUNT_ID>.r2.cloudflarestorage.com"
}
+7 -45
View File
@@ -1,17 +1,8 @@
locals { locals {
buckets = { buckets = {
mimir = { mimir = { name = "${var.bucket_prefix}-mimir" }
name = "${var.bucket_prefix}-mimir" loki = { name = "${var.bucket_prefix}-loki" }
retention_days = 365 tempo = { name = "${var.bucket_prefix}-tempo" }
}
loki = {
name = "${var.bucket_prefix}-loki"
retention_days = 90
}
tempo = {
name = "${var.bucket_prefix}-tempo"
retention_days = 90
}
} }
} }
@@ -22,40 +13,11 @@ resource "aws_s3_bucket" "telemetry" {
bucket = each.value.name bucket = each.value.name
# Hetzner does not yet support all S3 ACL operations, so keep this minimal. # Hetzner does not yet support all S3 ACL operations, so keep this minimal.
# Note: Hetzner Object Storage does not support the S3 Lifecycle Configuration
# API (PutBucketLifecycleConfiguration / GetBucketLifecycleConfiguration).
# Retention is enforced at the application layer: Mimir, Loki, and Tempo each
# have native retention settings configured via their respective config files.
lifecycle { lifecycle {
prevent_destroy = true prevent_destroy = true
} }
} }
resource "aws_s3_bucket_versioning" "telemetry" {
for_each = local.buckets
provider = aws.hetzner
bucket = aws_s3_bucket.telemetry[each.key].id
versioning_configuration {
status = "Suspended"
}
}
# Lifecycle rule: hard-delete safety net behind application retention.
resource "aws_s3_bucket_lifecycle_configuration" "telemetry" {
for_each = local.buckets
provider = aws.hetzner
bucket = aws_s3_bucket.telemetry[each.key].id
rule {
id = "hard-delete-after-${each.value.retention_days}d"
status = "Enabled"
filter {}
expiration {
days = each.value.retention_days
}
abort_incomplete_multipart_upload {
days_after_initiation = 7
}
}
}
+8 -8
View File
@@ -31,17 +31,17 @@ resource "hcloud_firewall" "watchtower" {
} }
resource "hcloud_primary_ip" "ipv4" { resource "hcloud_primary_ip" "ipv4" {
name = "${var.server_name}-ipv4" name = "${var.server_name}-ipv4"
type = "ipv4" type = "ipv4"
assignee_type = "server" location = var.location
auto_delete = false auto_delete = false
} }
resource "hcloud_primary_ip" "ipv6" { resource "hcloud_primary_ip" "ipv6" {
name = "${var.server_name}-ipv6" name = "${var.server_name}-ipv6"
type = "ipv6" type = "ipv6"
assignee_type = "server" location = var.location
auto_delete = false auto_delete = false
} }
resource "hcloud_volume" "observability" { resource "hcloud_volume" "observability" {