Feat add scratch disk #1

Merged
JMR-dev merged 2 commits from feat-add-scratch-disk into main 2026-05-11 03:06:38 +00:00
40 changed files with 780 additions and 112 deletions
+1 -1
View File
@@ -9,7 +9,7 @@ secrets/
.vault_pass .vault_pass
.vault_password .vault_password
ansible/inventory/hosts.yml ansible/inventory/hosts.yml
ansible/group_vars/**/vault.yml ansible/inventory/group_vars/**/vault.yml
wireguard/wg0.conf wireguard/wg0.conf
wireguard/peers/ wireguard/peers/
.aws/ .aws/
+4 -2
View File
@@ -14,8 +14,10 @@ single Hetzner CPX42 in Nuremberg (nbg1). Built with Mimir, Loki, Tempo, and Gra
```bash ```bash
# 1. Provision infrastructure # 1. Provision infrastructure
cd tofu cd tofu
cp terraform.tfvars.example terraform.tfvars # fill in your tokens cp backend.hcl.example backend.hcl # set your Cloudflare Account ID
tofu init cp terraform.tfvars.example terraform.tfvars # set Hetzner tokens + SSH key
set -a && source .env && set +a # load R2 credentials into env
tofu init -backend-config=backend.hcl
tofu apply tofu apply
# 2. Configure server # 2. Configure server
+4 -3
View File
@@ -3,11 +3,12 @@ inventory = inventory/hosts.yml
roles_path = roles roles_path = roles
host_key_checking = False host_key_checking = False
retry_files_enabled = False retry_files_enabled = False
stdout_callback = yaml stdout_callback = default
result_format = yaml
forks = 5 forks = 5
interpreter_python = /usr/bin/python3 interpreter_python = /usr/bin/python3
vault_password_file = .vault_pass vault_password_file = $HOME/.config/watchtower-observe/vault_pass
[ssh_connection] [ssh_connection]
pipelining = True pipelining = True
ssh_args = -o ControlMaster=auto -o ControlPersist=60s ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o IdentityFile=~/.ssh/watchtower-observe -o IdentitiesOnly=yes
@@ -40,7 +40,16 @@ admin_allow_ipv6:
observability_data_root: /var/lib/observability observability_data_root: /var/lib/observability
observability_secrets_dir: /etc/observability/secrets observability_secrets_dir: /etc/observability/secrets
observability_config_dir: /etc/observability/config observability_config_dir: /etc/observability/config
luks_device_source: /dev/sdb # observability_volume_id is injected per-host from the Ansible inventory.
# Run `tofu output -json | jq -r '.ansible_inventory.value'` to regenerate hosts.yml.
# Leave empty in dev/staging — the LUKS role will fall back to an 80 GB loop file.
observability_volume_id: ""
# When observability_volume_id is set, use the stable Hetzner by-id path.
# When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only).
luks_device_source: >-
{{ ((observability_volume_id | string) | length > 0)
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + (observability_volume_id | string), '/dev/sdb') }}
luks_mapper_name: observability_data luks_mapper_name: observability_data
# Container images (pin in production) # Container images (pin in production)
@@ -48,11 +57,21 @@ image_mimir: "docker.io/grafana/mimir:2.14.3"
image_loki: "docker.io/grafana/loki:3.3.2" image_loki: "docker.io/grafana/loki:3.3.2"
image_tempo: "docker.io/grafana/tempo:2.7.1" image_tempo: "docker.io/grafana/tempo:2.7.1"
image_grafana: "docker.io/grafana/grafana:11.4.0" image_grafana: "docker.io/grafana/grafana:11.4.0"
image_caddy: "localhost/watchtower-caddy:latest"
# wg-easy — WireGuard peer management web UI.
# image built from ghcr.io; pin the tag in production.
image_wg_easy: "ghcr.io/wg-easy/wg-easy:14"
wg_easy_web_port: 51821 # TCP — web UI, wg0 only (covered by iifname wg0 accept in nftables)
wg_easy_wg_port: "{{ wireguard_listen_port }}"
# Indirection for vault secret
wg_easy_password_hash: "{{ vault_wg_easy_password_hash }}"
# Retention (per signal) # Retention (per signal)
mimir_retention_days: 90 mimir_retention_days: 90
loki_retention_days: 30 loki_retention_days: 30
tempo_retention_days: 14 tempo_retention_days: 30
# Tenants — override in production # Tenants — override in production
tenants: tenants:
@@ -12,3 +12,8 @@ vault_wireguard_peers:
- name: admin-laptop - name: admin-laptop
public_key: REPLACE_ME public_key: REPLACE_ME
allowed_ips: 10.8.0.10/32 allowed_ips: 10.8.0.10/32
# bcrypt hash of the wg-easy web-UI password.
# Generate with: python3 -c "import bcrypt; print(bcrypt.hashpw(b'YOURPASSWORD', bcrypt.gensalt(12)).decode())"
# or: htpasswd -bnBC 12 '' YOURPASSWORD | tr -d ':\n'
vault_wg_easy_password_hash: "REPLACE_ME_BCRYPT_HASH"
+3
View File
@@ -4,3 +4,6 @@ all:
ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4 ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4
ansible_user: root ansible_user: root
ansible_python_interpreter: /usr/bin/python3 ansible_python_interpreter: /usr/bin/python3
# Hetzner Volume ID for the observability data disk (quote to keep as string).
# Populate from: tofu output -json | jq -r '.ansible_inventory.value'
observability_volume_id: "REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID"
+6 -1
View File
@@ -9,7 +9,6 @@
ansible.builtin.dnf: ansible.builtin.dnf:
name: name:
- podman - podman
- podman-compose
- wireguard-tools - wireguard-tools
- cryptsetup - cryptsetup
- nftables - nftables
@@ -27,6 +26,12 @@
community.general.timezone: community.general.timezone:
name: "{{ timezone }}" name: "{{ timezone }}"
- name: Ensure journald drop-in directory exists
ansible.builtin.file:
path: /etc/systemd/journald.conf.d
state: directory
mode: "0755"
- name: Configure persistent journald with size cap - name: Configure persistent journald with size cap
ansible.builtin.copy: ansible.builtin.copy:
dest: /etc/systemd/journald.conf.d/persistent.conf dest: /etc/systemd/journald.conf.d/persistent.conf
+6
View File
@@ -10,3 +10,9 @@ fail2ban_sender: fail2ban@watchtower
fail2ban_recidive_bantime: 1w fail2ban_recidive_bantime: 1w
fail2ban_recidive_findtime: 1d fail2ban_recidive_findtime: 1d
fail2ban_recidive_maxretry: 5 fail2ban_recidive_maxretry: 5
# Coraza WAF jail — higher maxretry than sshd because CRS in DetectionOnly
# mode fires on benign requests too; tune these down when SecRuleEngine=On.
fail2ban_coraza_bantime: 1h
fail2ban_coraza_findtime: 10m
fail2ban_coraza_maxretry: 15
@@ -0,0 +1,34 @@
# /etc/fail2ban/action.d/nftables-forward-allports.conf
#
# Bans IPs in the nftables FORWARD hook instead of INPUT.
#
# Why FORWARD instead of INPUT:
# Rootful Podman publishes ports via nftables DNAT rules in PREROUTING.
# After DNAT, the routing decision sends the packet through FORWARD (not
# INPUT), so INPUT-based bans never see Podman-destined traffic.
# FORWARD priority -1 evaluates before the watchtower FORWARD chain (0),
# so banned WireGuard peer IPs (10.8.0.x) are dropped before the
# `iifname wg0 accept` rule in the watchtower table.
#
# Each jail using this action gets its own per-jail chain and set so that
# multiple jails can safely coexist without shared-chain lifecycle conflicts.
[Definition]
actionstart = nft add table inet f2b-table
nft add chain inet f2b-table f2b-<name>-fwd { type filter hook forward priority -1 \; }
nft add set inet f2b-table f2b-<name>-set { type ipv4_addr \; flags timeout \; }
nft add rule inet f2b-table f2b-<name>-fwd ip saddr @f2b-<name>-set drop
actionstop = nft flush chain inet f2b-table f2b-<name>-fwd
nft delete chain inet f2b-table f2b-<name>-fwd
nft delete set inet f2b-table f2b-<name>-set
actioncheck = nft list chain inet f2b-table f2b-<name>-fwd
actionban = nft add element inet f2b-table f2b-<name>-set { <ip> timeout <bantime>s }
actionunban = nft delete element inet f2b-table f2b-<name>-set { <ip> }
[Init]
name = default
@@ -0,0 +1,30 @@
# /etc/fail2ban/filter.d/coraza-waf.conf
#
# Matches the Section A transaction header in a Coraza serial audit log.
#
# Coraza serial audit log structure (SecAuditLogType Serial):
# --<boundary>--A-- ← boundary marker (ignored)
# [DD/Mon/YYYY:HH:MM:SS[.us] ±ZZZZ] id ip port dst_ip dst_port ← matched
# --<boundary>--B-- ← request headers (ignored)
# ...
# --<boundary>--H-- ← rule messages (ignored)
# --<boundary>--Z-- ← end marker (ignored)
#
# With SecAuditEngine RelevantOnly, ONLY transactions that matched a WAF rule
# are written to the log, so every match of the Section A header == a WAF hit.
#
# The timestamp format is [DD/Mon/YYYY:HH:MM:SS ±HHMM] (Apache Combined style)
# with an optional sub-second fraction. fail2ban extracts the date via
# datepattern; the failregex captures <HOST> (client IP) from the same line.
[INCLUDES]
before = common.conf
[Definition]
failregex = ^\[[\d]{2}/\w+/[\d]{4}:[\d]{2}:[\d]{2}:[\d]{2}(?:\.[\d]+)? [+-][\d]{4}\] \S+ <HOST> [\d]+ \S+ [\d]+\s*$
ignoreregex =
# Timestamp is at the start of the line, wrapped in square brackets.
datepattern = {^LN-BEG}\[%%d/%%b/%%Y:%%H:%%M:%%S %%z\]
@@ -0,0 +1,17 @@
module watchtower-fail2ban 1.0;
# Allow fail2ban to read Coraza WAF audit logs written by the containerised
# Caddy process. The audit log directory is bind-mounted with Podman's shared
# `:z` relabel option, which labels files as container_file_t:s0 (no private
# MCS category). fail2ban runs with s0 clearance, so the MCS check passes;
# only the type allow rule is missing from the base policy.
require {
type fail2ban_t;
type container_file_t;
class file { getattr open read };
class dir { getattr open read search };
}
allow fail2ban_t container_file_t:dir { getattr open read search };
allow fail2ban_t container_file_t:file { getattr open read };
+40 -1
View File
@@ -4,11 +4,12 @@
name: epel-release name: epel-release
state: present state: present
- name: Install fail2ban - name: Install fail2ban and checkpolicy
ansible.builtin.dnf: ansible.builtin.dnf:
name: name:
- fail2ban - fail2ban
- fail2ban-firewalld # provides the firewallcmd actions; harmless even if firewalld is masked - fail2ban-firewalld # provides the firewallcmd actions; harmless even if firewalld is masked
- checkpolicy # provides checkmodule + semodule_package for custom SELinux modules
state: present state: present
update_cache: true update_cache: true
@@ -21,6 +22,44 @@
state: absent state: absent
notify: Restart fail2ban notify: Restart fail2ban
# Deploy filter and action files BEFORE the jail is enabled so fail2ban can
# find them on its first start.
- name: Deploy fail2ban filter files
ansible.builtin.copy:
src: filter.d/
dest: /etc/fail2ban/filter.d/
mode: "0644"
notify: Restart fail2ban
- name: Deploy fail2ban action files
ansible.builtin.copy:
src: action.d/
dest: /etc/fail2ban/action.d/
mode: "0644"
notify: Restart fail2ban
# Install a custom SELinux policy module that allows fail2ban_t to read files
# labelled container_file_t:s0. The Coraza audit log directory is bind-mounted
# into Caddy with ':z' (shared label), which sets container_file_t:s0 — no
# private MCS categories — so fail2ban's s0 clearance can pass the MCS check,
# but the type allow rule is still required.
- name: Copy watchtower-fail2ban SELinux policy source
ansible.builtin.copy:
src: selinux/watchtower-fail2ban.te
dest: /tmp/watchtower-fail2ban.te
mode: "0644"
register: selinux_policy_src
- name: Compile and install watchtower-fail2ban SELinux policy module
ansible.builtin.shell: |
set -e
cd /tmp
checkmodule -M -m -o watchtower-fail2ban.mod watchtower-fail2ban.te
semodule_package -o watchtower-fail2ban.pp -m watchtower-fail2ban.mod
semodule -i watchtower-fail2ban.pp
when: selinux_policy_src.changed
notify: Restart fail2ban
- name: Render jail.local - name: Render jail.local
ansible.builtin.template: ansible.builtin.template:
src: jail.local.j2 src: jail.local.j2
@@ -41,3 +41,27 @@ banaction = nftables-allports
bantime = {{ fail2ban_recidive_bantime }} bantime = {{ fail2ban_recidive_bantime }}
findtime = {{ fail2ban_recidive_findtime }} findtime = {{ fail2ban_recidive_findtime }}
maxretry = {{ fail2ban_recidive_maxretry }} maxretry = {{ fail2ban_recidive_maxretry }}
# Coraza WAF audit log — bans WireGuard peer IPs (10.8.0.x) that trigger
# WAF rules through Caddy. Uses nftables-forward-allports because Podman
# DNATs published ports, so traffic traverses FORWARD (not INPUT) and
# INPUT-based ban actions never intercept it.
#
# ignoreip intentionally excludes wireguard_subnet_v4 so misbehaving VPN
# clients can be banned. The server's own address (127.0.0.1/::1) is still
# whitelisted to prevent self-banning.
#
# Tune fail2ban_coraza_maxretry down (and SecRuleEngine to "On") once you
# have reviewed the Coraza audit log and confirmed false-positive rate.
[coraza-waf]
enabled = true
port = 0:65535
protocol = tcp
logpath = {{ observability_data_root }}/caddy/logs/coraza-audit-grafana.log
backend = auto
filter = coraza-waf
banaction = nftables-forward-allports
ignoreip = 127.0.0.1/8 ::1
maxretry = {{ fail2ban_coraza_maxretry }}
findtime = {{ fail2ban_coraza_findtime }}
bantime = {{ fail2ban_coraza_bantime }}
@@ -55,6 +55,8 @@ table inet watchtower {
# Anything arriving on WireGuard is treated as trusted (only authenticated # Anything arriving on WireGuard is treated as trusted (only authenticated
# peers can reach this interface). Service ports are bound to 10.8.0.1. # peers can reach this interface). Service ports are bound to 10.8.0.1.
# This also implicitly allows the wg-easy web UI on TCP 51821, which
# is bound to 10.8.0.1 and therefore only reachable from VPN peers.
iifname "{{ wireguard_interface }}" accept iifname "{{ wireguard_interface }}" accept
# Podman bridge networks: allow the kernel to deliver packets the # Podman bridge networks: allow the kernel to deliver packets the
+10
View File
@@ -15,6 +15,16 @@
ansible.builtin.set_fact: ansible.builtin.set_fact:
luks_use_loop_file: "{{ not luks_device_stat.stat.exists }}" luks_use_loop_file: "{{ not luks_device_stat.stat.exists }}"
- name: Fail if Hetzner Volume expected but device not found
ansible.builtin.fail:
msg: >
Device {{ luks_device_source }} was not found, but observability_volume_id is set
to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server
and udev has settled (try: udevadm settle --timeout=30).
when:
- (observability_volume_id | string) | length > 0
- not luks_device_stat.stat.exists
- name: Create loop-backed LUKS image (if no real device) - name: Create loop-backed LUKS image (if no real device)
when: luks_use_loop_file when: luks_use_loop_file
block: block:
@@ -0,0 +1,8 @@
[Unit]
Description=Observability disk-guard: flush services when data volume is near-full
After=network-online.target wg-easy.service loki.service mimir.service tempo.service
Wants=network-online.target
[Service]
Type=oneshot
ExecStart=/usr/local/bin/disk-guard.sh
@@ -0,0 +1,10 @@
[Unit]
Description=Run observability disk-guard every 5 minutes
[Timer]
OnBootSec=5min
OnUnitActiveSec=5min
Persistent=true
[Install]
WantedBy=timers.target
@@ -27,13 +27,33 @@
state: restarted state: restarted
daemon_reload: true daemon_reload: true
- name: Restart caddy
ansible.builtin.systemd:
name: caddy.service
state: restarted
daemon_reload: true
- name: Restart wg-easy
ansible.builtin.systemd:
name: wg-easy.service
state: restarted
daemon_reload: true
- name: Restart all services - name: Restart all services
ansible.builtin.systemd: ansible.builtin.systemd:
name: "{{ item }}" name: "{{ item }}"
state: restarted state: restarted
daemon_reload: true daemon_reload: true
loop: loop:
- wg-easy.service
- caddy.service
- mimir.service - mimir.service
- loki.service - loki.service
- tempo.service - tempo.service
- grafana.service - grafana.service
- name: Restart disk-guard timer
ansible.builtin.systemd:
name: disk-guard.timer
state: restarted
daemon_reload: true
+118 -13
View File
@@ -16,8 +16,10 @@
- { path: "{{ observability_config_dir }}/grafana/provisioning", mode: "0755" } - { path: "{{ observability_config_dir }}/grafana/provisioning", mode: "0755" }
- { path: "{{ observability_config_dir }}/grafana/provisioning/datasources", mode: "0755" } - { path: "{{ observability_config_dir }}/grafana/provisioning/datasources", mode: "0755" }
- { path: "{{ observability_config_dir }}/grafana/provisioning/dashboards", mode: "0755" } - { path: "{{ observability_config_dir }}/grafana/provisioning/dashboards", mode: "0755" }
- { path: "{{ observability_config_dir }}/caddy", mode: "0755" }
- { path: /var/log/observability, mode: "0750" } - { path: /var/log/observability, mode: "0750" }
- { path: /etc/containers/systemd, mode: "0755" } - { path: /etc/containers/systemd, mode: "0755" }
- { path: /etc/observability/caddy-build, mode: "0755" }
# Container processes run as non-root UIDs. Bind-mounted data dirs must be # Container processes run as non-root UIDs. Bind-mounted data dirs must be
# owned by those UIDs or the services cannot create their WAL/index/DB files. # owned by those UIDs or the services cannot create their WAL/index/DB files.
@@ -29,10 +31,12 @@
owner: "{{ item.uid }}" owner: "{{ item.uid }}"
group: "{{ item.gid }}" group: "{{ item.gid }}"
loop: loop:
- { path: "{{ observability_data_root }}/mimir", uid: 10001, gid: 10001 } - { path: "{{ observability_data_root }}/mimir", uid: 10001, gid: 10001 }
- { path: "{{ observability_data_root }}/loki", uid: 10001, gid: 10001 } - { path: "{{ observability_data_root }}/loki", uid: 10001, gid: 10001 }
- { path: "{{ observability_data_root }}/tempo", uid: 10001, gid: 10001 } - { path: "{{ observability_data_root }}/tempo", uid: 10001, gid: 10001 }
- { path: "{{ observability_data_root }}/grafana", uid: 472, gid: 472 } - { path: "{{ observability_data_root }}/grafana", uid: 472, gid: 472 }
- { path: "{{ observability_data_root }}/caddy", uid: 1000, gid: 1000 }
- { path: "{{ observability_data_root }}/caddy/logs", uid: 1000, gid: 1000 }
- name: Write S3 credentials env file - name: Write S3 credentials env file
ansible.builtin.template: ansible.builtin.template:
@@ -53,6 +57,13 @@
mode: "0644" mode: "0644"
notify: Restart mimir notify: Restart mimir
- name: Render Mimir runtime overrides
ansible.builtin.template:
src: mimir-runtime.yaml.j2
dest: "{{ observability_config_dir }}/mimir/runtime.yaml"
mode: "0644"
notify: Restart mimir
- name: Render Loki config - name: Render Loki config
ansible.builtin.template: ansible.builtin.template:
src: loki.yaml.j2 src: loki.yaml.j2
@@ -71,7 +82,7 @@
ansible.builtin.template: ansible.builtin.template:
src: grafana.ini.j2 src: grafana.ini.j2
dest: "{{ observability_config_dir }}/grafana/grafana.ini" dest: "{{ observability_config_dir }}/grafana/grafana.ini"
mode: "0640" mode: "0644"
notify: Restart grafana notify: Restart grafana
- name: Render Grafana datasource provisioning - name: Render Grafana datasource provisioning
@@ -81,6 +92,22 @@
mode: "0644" mode: "0644"
notify: Restart grafana notify: Restart grafana
- name: Render Caddyfile
ansible.builtin.template:
src: Caddyfile.j2
dest: "{{ observability_config_dir }}/caddy/Caddyfile"
mode: "0644"
notify: Restart caddy
- name: Render wg-easy env file
ansible.builtin.template:
src: wg-easy.env.j2
dest: "{{ observability_secrets_dir }}/wg-easy.env"
mode: "0600"
owner: root
group: root
notify: Restart wg-easy
- name: Apply SELinux context to config tree - name: Apply SELinux context to config tree
ansible.builtin.command: restorecon -R /etc/observability ansible.builtin.command: restorecon -R /etc/observability
changed_when: false changed_when: false
@@ -102,8 +129,28 @@
- loki-data - loki-data
- tempo-data - tempo-data
- grafana-data - grafana-data
- caddy-data
notify: Reload systemd notify: Reload systemd
# Copy the Caddy build context to the server so podman build runs locally.
- name: Copy Caddy + Coraza Dockerfile to server
ansible.builtin.copy:
src: "{{ role_path }}/../../../build/caddy/Dockerfile"
dest: /etc/observability/caddy-build/Dockerfile
mode: "0644"
register: caddy_dockerfile_result
- name: Check if Caddy + Coraza image already exists
ansible.builtin.command: podman image inspect localhost/watchtower-caddy:latest
register: caddy_image_inspect
changed_when: false
failed_when: false
- name: Build Caddy + Coraza image
ansible.builtin.command: podman build -t localhost/watchtower-caddy:latest /etc/observability/caddy-build/
when: caddy_image_inspect.rc != 0 or caddy_dockerfile_result.changed
notify: Restart caddy
- name: Deploy container quadlets - name: Deploy container quadlets
ansible.builtin.template: ansible.builtin.template:
src: "{{ item }}.container.j2" src: "{{ item }}.container.j2"
@@ -114,6 +161,8 @@
- loki - loki
- tempo - tempo
- grafana - grafana
- wg-easy
- caddy
notify: notify:
- Reload systemd - Reload systemd
- Restart all services - Restart all services
@@ -121,6 +170,67 @@
- name: Force handlers to flush before health checks - name: Force handlers to flush before health checks
ansible.builtin.meta: flush_handlers ansible.builtin.meta: flush_handlers
# Handlers only restart services when the quadlet templates change. On
# subsequent runs (or partial first runs) the units exist but are stopped, so
# explicitly ensure each backend is started before the health checks below.
# Quadlet-generated units cannot be `enabled` (they're auto-wired by their
# WantedBy= line), so we only set state=started — systemd treats this as a
# no-op when the unit is already active.
- name: Ensure observability services are started
ansible.builtin.systemd:
name: "{{ item }}"
state: started
daemon_reload: true
loop:
- wg-easy.service
- mimir.service
- loki.service
- tempo.service
- grafana.service
- caddy.service
- name: Deploy disk-guard script
ansible.builtin.template:
src: disk-guard.sh.j2
dest: /usr/local/bin/disk-guard.sh
mode: "0750"
owner: root
group: root
- name: Deploy disk-guard systemd units
ansible.builtin.copy:
src: "{{ item }}"
dest: "/etc/systemd/system/{{ item }}"
mode: "0644"
owner: root
group: root
loop:
- disk-guard.service
- disk-guard.timer
notify:
- Reload systemd
- Restart disk-guard timer
- name: Enable and start disk-guard timer
ansible.builtin.systemd:
name: disk-guard.timer
enabled: true
state: started
daemon_reload: true
# Health checks run on the managed host (via SSH) so they work regardless of
# whether the Ansible controller is connected to the WireGuard VPN. The server
# can reach its own WireGuard IP (10.8.0.1) via the local wg0 interface.
- name: Wait for Caddy to be ready
ansible.builtin.uri:
url: "http://{{ wireguard_server_ip }}/api/health"
status_code: 200
register: caddy_ready
retries: 30
delay: 5
until: caddy_ready.status == 200
delegate_to: "{{ inventory_hostname }}"
- name: Wait for Mimir /ready - name: Wait for Mimir /ready
ansible.builtin.uri: ansible.builtin.uri:
url: "http://{{ wireguard_server_ip }}:8080/ready" url: "http://{{ wireguard_server_ip }}:8080/ready"
@@ -129,6 +239,7 @@
retries: 30 retries: 30
delay: 5 delay: 5
until: mimir_ready.status == 200 until: mimir_ready.status == 200
delegate_to: "{{ inventory_hostname }}"
- name: Wait for Loki /ready - name: Wait for Loki /ready
ansible.builtin.uri: ansible.builtin.uri:
@@ -138,6 +249,7 @@
retries: 30 retries: 30
delay: 5 delay: 5
until: loki_ready.status == 200 until: loki_ready.status == 200
delegate_to: "{{ inventory_hostname }}"
- name: Wait for Tempo /ready - name: Wait for Tempo /ready
ansible.builtin.uri: ansible.builtin.uri:
@@ -147,12 +259,5 @@
retries: 30 retries: 30
delay: 5 delay: 5
until: tempo_ready.status == 200 until: tempo_ready.status == 200
delegate_to: "{{ inventory_hostname }}"
- name: Wait for Grafana /api/health
ansible.builtin.uri:
url: "http://{{ wireguard_server_ip }}:3000/api/health"
status_code: 200
register: grafana_ready
retries: 30
delay: 5
until: grafana_ready.status == 200
@@ -0,0 +1,77 @@
{
# Disable the built-in ACME client; TLS is terminated inside the VPN.
auto_https off
# coraza_waf must be ordered before the reverse_proxy handler so WAF
# inspection runs before the request is forwarded upstream.
order coraza_waf first
# Send Caddy's own structured log to stdout; Podman + journald capture it.
log {
output stdout
format json
}
# Accept cleartext HTTP/2 (h2c) from OTLP gRPC clients on port 4317.
# The transport block in the :4317 site only controls Caddy → Tempo;
# this servers block makes Caddy accept h2c from the client side too.
servers :4317 {
protocols h1 h2c
}
}
# ── Grafana UI — browser-facing ───────────────────────────────────────────────
# Full OWASP CRS in detection-only mode. Switch SecRuleEngine to "On" after
# reviewing Coraza audit logs and tuning false-positives for your workload.
:80 {
coraza_waf {
load_owasp_crs
directives `
Include @coraza.conf-recommended
Include @crs-setup.conf.example
Include @owasp_crs/*.conf
SecRuleEngine DetectionOnly
SecRequestBodyAccess On
SecResponseBodyAccess Off
SecAuditEngine RelevantOnly
SecAuditLog /var/log/caddy/coraza-audit-grafana.log
SecAuditLogType Serial
`
}
reverse_proxy grafana:3000
}
# ── Loki push + query API ─────────────────────────────────────────────────────
# Machine-to-machine over WireGuard; no CRS (structured/binary payloads).
:3100 {
reverse_proxy loki:3100
}
# ── Mimir push + query API ────────────────────────────────────────────────────
:8080 {
reverse_proxy mimir:8080
}
# ── Tempo query API ───────────────────────────────────────────────────────────
:3200 {
reverse_proxy tempo:3200
}
# ── Tempo OTLP HTTP ingest ────────────────────────────────────────────────────
:4318 {
reverse_proxy tempo:4318
}
# ── Tempo OTLP gRPC ingest ────────────────────────────────────────────────────
# Tempo's gRPC receiver speaks h2c (plain HTTP/2 without TLS upgrade).
# The transport block disables TLS and forces HTTP/2 on the upstream connection.
:4317 {
reverse_proxy tempo:4317 {
transport http {
versions h2c
}
}
}
@@ -0,0 +1,7 @@
[Volume]
# Caddy uses this volume for its internal state (OCSP staples, Coraza audit
# logs, etc.). The host path sits on the 500 GB observability data disk.
VolumeName=caddy-data
Device={{ observability_data_root }}/caddy
Type=bind
Options=bind
@@ -0,0 +1,56 @@
[Unit]
Description=Caddy reverse proxy with Coraza WAF
# Caddy publishes ports on the WireGuard IP; wg0 must be up before the
# PublishPort binds succeed. wg-easy.service owns the wg0 interface.
Wants=network-online.target observability-network.service wg-easy.service
After=network-online.target observability-network.service wg-easy.service
[Container]
Image={{ image_caddy }}
ContainerName=caddy
# Same network as all backends; Caddy reaches them by container name.
Network=observability.network
Volume=caddy-data.volume:/var/lib/caddy:Z
Volume={{ observability_config_dir }}/caddy/Caddyfile:/etc/caddy/Caddyfile:ro,Z
# Separate bind mount with shared (':z') label so fail2ban on the host can
# read the Coraza audit log. Private (':Z') would assign a container-private
# MCS category that fail2ban cannot access even as root.
Volume={{ observability_data_root }}/caddy/logs:/var/log/caddy:z
# Expose each service port on the WireGuard IP only. Caddy listens on
# 0.0.0.0 inside the container; Podman's PublishPort restricts host exposure.
PublishPort={{ wireguard_server_ip }}:80:80
PublishPort={{ wireguard_server_ip }}:3100:3100
PublishPort={{ wireguard_server_ip }}:8080:8080
PublishPort={{ wireguard_server_ip }}:3200:3200
PublishPort={{ wireguard_server_ip }}:4317:4317
PublishPort={{ wireguard_server_ip }}:4318:4318
# Port 80 requires NET_BIND_SERVICE when running as non-root.
AddCapability=NET_BIND_SERVICE
User=1000
Group=1000
PodmanArgs=--memory=512m --memory-swap=512m
[Service]
Restart=on-failure
RestartSec=10
TimeoutStartSec=120
# Wait for wg0 to have its address before Caddy tries to bind PublishPorts on
# the WireGuard IP. Times out after 2 minutes and lets systemd mark the unit
# failed so the operator sees a clear error rather than a silent bind error.
ExecStartPre=/bin/bash -c \
'for i in $(seq 60); do \
ip -4 addr show {{ wireguard_interface }} 2>/dev/null \
| grep -q "{{ wireguard_server_ip }}" && exit 0; \
sleep 2; \
done; \
echo "Timed out waiting for {{ wireguard_interface }}"; exit 1'
[Install]
WantedBy=multi-user.target default.target
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
# Managed by Ansible — see roles/observability/templates/disk-guard.sh.j2
#
# Monitors disk usage on the observability data volume and calls each service's
# ingester-flush HTTP API when the disk is near capacity, accelerating the
# drain of in-memory chunks to Hetzner Object Storage.
#
# NOTE: /flush and /ingester/flush are admin-only endpoints registered outside
# the standard auth middleware in Loki, Mimir, and Tempo. They flush all tenants
# and do not require X-Scope-OrgID even when multitenancy is enabled.
#
# Thresholds:
# >= 80 % WARNING — flush all ingesters
# >= 90 % ERR — flush + journal critical (operator action required)
set -euo pipefail
DATA_ROOT="{{ observability_data_root }}"
MAPPER="{{ luks_mapper_name }}"
BASE_URL="http://{{ wireguard_server_ip }}"
# Verify the LUKS volume is actually mounted; if not, we'd be checking the
# wrong (root) filesystem and triggering spurious flushes.
if ! findmnt --source "/dev/mapper/${MAPPER}" --target "${DATA_ROOT}" > /dev/null 2>&1; then
systemd-cat -t disk-guard -p err \
printf 'LUKS volume /dev/mapper/%s is not mounted at %s — skipping disk check' \
"${MAPPER}" "${DATA_ROOT}"
exit 1
fi
USAGE=$(df --output=pcent "${DATA_ROOT}" | tail -1 | tr -d ' %')
if (( USAGE < 80 )); then
exit 0
fi
LEVEL="WARNING"
(( USAGE >= 90 )) && LEVEL="ERR"
systemd-cat -t disk-guard -p "${LEVEL,,}" \
printf '%s: observability data volume at %d%% — flushing ingesters to object storage' \
"${LEVEL}" "${USAGE}"
flush() {
local svc=$1 url=$2
local http_code
http_code=$(curl -s -o /dev/null -w "%{http_code}" -X POST "${url}" --max-time 10 || true)
if [[ "${http_code}" == "204" || "${http_code}" == "200" ]]; then
systemd-cat -t disk-guard -p info printf 'Flushed %s ingester (HTTP %s)' "${svc}" "${http_code}"
else
systemd-cat -t disk-guard -p warning printf 'Flush %s returned HTTP %s (check service logs)' "${svc}" "${http_code}"
fi
}
flush loki "${BASE_URL}:3100/flush"
flush mimir "${BASE_URL}:8080/ingester/flush"
flush tempo "${BASE_URL}:3200/flush"
if (( USAGE >= 90 )); then
systemd-cat -t disk-guard -p err \
printf 'ERR: observability volume at %d%% — manual intervention required (compact, scale, or extend volume)' \
"${USAGE}"
fi
@@ -1,8 +1,7 @@
[Unit] [Unit]
Description=Grafana Description=Grafana
Wants=network-online.target observability-network.service mimir.service loki.service tempo.service Wants=network-online.target observability-network.service mimir.service loki.service tempo.service
After=network-online.target observability-network.service mimir.service loki.service tempo.service wg-quick@{{ wireguard_interface }}.service After=network-online.target observability-network.service mimir.service loki.service tempo.service
Requires=wg-quick@{{ wireguard_interface }}.service
[Container] [Container]
Image={{ image_grafana }} Image={{ image_grafana }}
@@ -18,8 +17,6 @@ Environment=GF_PATHS_PLUGINS=/var/lib/grafana/plugins
Environment=GF_PATHS_PROVISIONING=/etc/grafana/provisioning Environment=GF_PATHS_PROVISIONING=/etc/grafana/provisioning
Environment=GF_SECURITY_ADMIN_PASSWORD={{ grafana_admin_password }} Environment=GF_SECURITY_ADMIN_PASSWORD={{ grafana_admin_password }}
PublishPort={{ wireguard_server_ip }}:3000:3000
User=472 User=472
Group=472 Group=472
@@ -11,7 +11,7 @@ protocol = http
http_addr = 0.0.0.0 http_addr = 0.0.0.0
http_port = 3000 http_port = 3000
domain = {{ wireguard_server_ip }} domain = {{ wireguard_server_ip }}
root_url = http://{{ wireguard_server_ip }}:3000/ root_url = http://{{ wireguard_server_ip }}/
enforce_domain = false enforce_domain = false
[security] [security]
@@ -1,8 +1,7 @@
[Unit] [Unit]
Description=Grafana Loki Description=Grafana Loki
Wants=network-online.target observability-network.service Wants=network-online.target observability-network.service
After=network-online.target observability-network.service wg-quick@{{ wireguard_interface }}.service After=network-online.target observability-network.service
Requires=wg-quick@{{ wireguard_interface }}.service
[Container] [Container]
Image={{ image_loki }} Image={{ image_loki }}
@@ -12,8 +11,6 @@ Volume=loki-data.volume:/var/lib/loki:Z
Volume={{ observability_config_dir }}/loki/loki.yaml:/etc/loki/loki.yaml:ro,Z Volume={{ observability_config_dir }}/loki/loki.yaml:/etc/loki/loki.yaml:ro,Z
EnvironmentFile={{ observability_secrets_dir }}/s3.env EnvironmentFile={{ observability_secrets_dir }}/s3.env
PublishPort={{ wireguard_server_ip }}:3100:3100
Exec=-config.file=/etc/loki/loki.yaml -config.expand-env=true Exec=-config.file=/etc/loki/loki.yaml -config.expand-env=true
User=10001 User=10001
@@ -0,0 +1,4 @@
# Mimir runtime overrides (hot-reloaded). Empty by default; add per-tenant
# overrides here without restarting Mimir.
# Reference: https://grafana.com/docs/mimir/latest/configure/about-runtime-configuration/
overrides: {}
@@ -1,8 +1,7 @@
[Unit] [Unit]
Description=Grafana Mimir Description=Grafana Mimir
Wants=network-online.target observability-network.service Wants=network-online.target observability-network.service
After=network-online.target observability-network.service wg-quick@{{ wireguard_interface }}.service After=network-online.target observability-network.service
Requires=wg-quick@{{ wireguard_interface }}.service
[Container] [Container]
Image={{ image_mimir }} Image={{ image_mimir }}
@@ -10,12 +9,9 @@ ContainerName=mimir
Network=observability.network Network=observability.network
Volume=mimir-data.volume:/var/lib/mimir:Z Volume=mimir-data.volume:/var/lib/mimir:Z
Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z
Volume={{ observability_config_dir }}/mimir/runtime.yaml:/etc/mimir/runtime.yaml:ro,Z
EnvironmentFile={{ observability_secrets_dir }}/s3.env EnvironmentFile={{ observability_secrets_dir }}/s3.env
# Bind only to WireGuard interface
PublishPort={{ wireguard_server_ip }}:8080:8080
PublishPort={{ wireguard_server_ip }}:9095:9095
# -config.expand-env enables ${AWS_*} substitution from EnvironmentFile. # -config.expand-env enables ${AWS_*} substitution from EnvironmentFile.
Exec=-config.file=/etc/mimir/mimir.yaml -config.expand-env=true Exec=-config.file=/etc/mimir/mimir.yaml -config.expand-env=true
@@ -5,6 +5,12 @@ target: all,alertmanager,overrides-exporter
multitenancy_enabled: true multitenancy_enabled: true
# The activity tracker writes its log to ./metrics-activity.log by default.
# Inside the container the working directory is "/" which is not writable for
# the non-root user, so point it at the data volume.
activity_tracker:
filepath: /var/lib/mimir/metrics-activity.log
server: server:
http_listen_port: 8080 http_listen_port: 8080
grpc_listen_port: 9095 grpc_listen_port: 9095
@@ -1,8 +1,7 @@
[Unit] [Unit]
Description=Grafana Tempo Description=Grafana Tempo
Wants=network-online.target observability-network.service Wants=network-online.target observability-network.service
After=network-online.target observability-network.service wg-quick@{{ wireguard_interface }}.service After=network-online.target observability-network.service
Requires=wg-quick@{{ wireguard_interface }}.service
[Container] [Container]
Image={{ image_tempo }} Image={{ image_tempo }}
@@ -12,10 +11,6 @@ Volume=tempo-data.volume:/var/lib/tempo:Z
Volume={{ observability_config_dir }}/tempo/tempo.yaml:/etc/tempo/tempo.yaml:ro,Z Volume={{ observability_config_dir }}/tempo/tempo.yaml:/etc/tempo/tempo.yaml:ro,Z
EnvironmentFile={{ observability_secrets_dir }}/s3.env EnvironmentFile={{ observability_secrets_dir }}/s3.env
PublishPort={{ wireguard_server_ip }}:3200:3200
PublishPort={{ wireguard_server_ip }}:4317:4317
PublishPort={{ wireguard_server_ip }}:4318:4318
Exec=-config.file=/etc/tempo/tempo.yaml -config.expand-env=true Exec=-config.file=/etc/tempo/tempo.yaml -config.expand-env=true
User=10001 User=10001
@@ -0,0 +1,37 @@
[Unit]
Description=wg-easy WireGuard peer management
# wg-easy manages the host wg0 interface directly; it must start after the
# network is up but has no dependency on any observability container.
Wants=network-online.target
After=network-online.target
[Container]
Image={{ image_wg_easy }}
ContainerName=wg-easy
# Host network required so wg-easy can create/configure the wg0 interface on
# the real host namespace. Container network isolation is intentionally bypassed.
Network=host
# Capabilities for WireGuard interface management
AddCapability=NET_ADMIN
AddCapability=NET_RAW
AddCapability=SYS_MODULE
# Disable SELinux labelling — the container needs to write /etc/wireguard on
# the host and bind /dev/net/tun; the default container policy would deny this.
SecurityLabelDisable=true
# /etc/wireguard is bind-mounted WITHOUT :Z so the host directory keeps its
# original SELinux context and `wg` tooling outside the container still works.
Volume=/etc/wireguard:/etc/wireguard
EnvironmentFile={{ observability_secrets_dir }}/wg-easy.env
[Service]
Restart=on-failure
RestartSec=10
TimeoutStartSec=120
[Install]
WantedBy=multi-user.target default.target
@@ -0,0 +1,25 @@
# wg-easy environment — rendered from Ansible vault; do not commit plaintext.
# This file is chmod 0600 and lives under observability_secrets_dir.
# Public hostname (or IP) that WireGuard peers use to reach this server.
# This is written into generated peer configs, not the bind address.
WG_HOST={{ wireguard_public_ip | default(ansible_default_ipv4.address) }}
# WireGuard listen port (must match the nftables allow rule)
WG_PORT={{ wg_easy_wg_port }}
# Web-UI port
PORT={{ wg_easy_web_port }}
# Bind the web UI only to the WireGuard interface IP so it is never reachable
# from the public internet (nftables is defense-in-depth on top of this).
HOST={{ wireguard_server_ip }}
# Default WireGuard subnet assigned to peers (x = incremented per peer)
WG_DEFAULT_ADDRESS=10.8.0.x
# bcrypt hash of the web-UI password (from vault_wg_easy_password_hash)
PASSWORD_HASH={{ wg_easy_password_hash }}
# DNS pushed to peers
WG_DEFAULT_DNS=1.1.1.1,1.0.0.1
+4 -3
View File
@@ -1,5 +1,6 @@
--- ---
# wg-quick is no longer used; wg-easy manages the WireGuard interface.
# Handler kept as a no-op stub so any lingering notify references don't fail.
- name: Restart wireguard - name: Restart wireguard
ansible.builtin.systemd: ansible.builtin.debug:
name: "wg-quick@{{ wireguard_interface }}" msg: "wg-quick is disabled; wg-easy manages {{ wireguard_interface }}"
state: restarted
+34 -11
View File
@@ -1,4 +1,9 @@
--- ---
- name: Install wireguard-tools
ansible.builtin.package:
name: wireguard-tools
state: present
- name: Ensure /etc/wireguard exists - name: Ensure /etc/wireguard exists
ansible.builtin.file: ansible.builtin.file:
path: /etc/wireguard path: /etc/wireguard
@@ -7,14 +12,22 @@
owner: root owner: root
group: root group: root
- name: Render wg0.conf # wg-easy runs `wg-quick up wg0` inside its container, which calls
ansible.builtin.template: # iptables-legacy to set up the NAT POSTROUTING rule. AlmaLinux 10 doesn't
src: wg0.conf.j2 # auto-load the legacy iptables kernel modules, so the call fails with
dest: "/etc/wireguard/{{ wireguard_interface }}.conf" # "can't initialize iptables table 'nat': Table does not exist" and wg0 is
mode: "0600" # torn down. Load the modules now and persist them across reboots.
owner: root - name: Load iptables kernel modules required by wg-easy
group: root community.general.modprobe:
notify: Restart wireguard name: "{{ item }}"
state: present
persistent: present
loop:
- ip_tables
- iptable_filter
- iptable_nat
- nf_nat
- nf_conntrack
- name: Enable IPv4 forwarding - name: Enable IPv4 forwarding
ansible.posix.sysctl: ansible.posix.sysctl:
@@ -24,8 +37,18 @@
state: present state: present
reload: true reload: true
- name: Enable wg-quick service - name: Enable src_valid_mark (required for WireGuard routing)
ansible.posix.sysctl:
name: net.ipv4.conf.all.src_valid_mark
value: "1"
sysctl_set: true
state: present
reload: true
# wg-easy takes over wg0 management; ensure wg-quick is not competing.
- name: Disable and stop wg-quick
ansible.builtin.systemd: ansible.builtin.systemd:
name: "wg-quick@{{ wireguard_interface }}" name: "wg-quick@{{ wireguard_interface }}"
enabled: true enabled: false
state: started state: stopped
failed_when: false
+38
View File
@@ -0,0 +1,38 @@
# syntax=docker/dockerfile:1
#
# Multi-stage build: xcaddy compiles Caddy with the Coraza WAF plugin and the
# OWASP Core Rule Set embedded at build time, then the final image is a minimal
# debian:bookworm-slim layer that ships only the compiled binary.
#
# Rebuild whenever the ARG versions below change; Ansible will detect the
# Dockerfile checksum change and re-run `podman build`.
FROM golang:1.26-trixie AS builder
ARG XCADDY_VERSION=v0.3.5
ARG CADDY_VERSION=v2.11.2
ARG CORAZA_CADDY_VERSION=v2.5.0
ARG CORAZA_CRS_VERSION=v4.7.0
RUN go install "github.com/caddyserver/xcaddy/cmd/xcaddy@${XCADDY_VERSION}"
RUN xcaddy build "${CADDY_VERSION}" \
--with "github.com/corazawaf/coraza-caddy/v2@${CORAZA_CADDY_VERSION}" \
--with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}"
# ── Runtime image ──────────────────────────────────────────────────────────────
FROM debian:trixie-slim
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates \
&& rm -rf /var/lib/apt/lists/*
COPY --from=builder /go/caddy /usr/bin/caddy
RUN groupadd --system --gid 1000 caddy \
&& useradd --system --uid 1000 --gid caddy --no-create-home caddy
EXPOSE 80 3100 3200 4317 4318 8080
ENTRYPOINT ["/usr/bin/caddy"]
CMD ["run", "--config", "/etc/caddy/Caddyfile", "--adapter", "caddyfile"]
+10
View File
@@ -0,0 +1,10 @@
# Copy to backend.hcl (gitignored) and set your Cloudflare Account ID.
# Pass at init time: tofu init -backend-config=backend.hcl
#
# Credentials are read from env vars — set before running tofu:
# export AWS_ACCESS_KEY_ID=<r2-access-key-id>
# export AWS_SECRET_ACCESS_KEY=<r2-secret-access-key>
endpoints = {
s3 = "https://<ACCOUNT_ID>.r2.cloudflarestorage.com"
}
+7 -45
View File
@@ -1,17 +1,8 @@
locals { locals {
buckets = { buckets = {
mimir = { mimir = { name = "${var.bucket_prefix}-mimir" }
name = "${var.bucket_prefix}-mimir" loki = { name = "${var.bucket_prefix}-loki" }
retention_days = 365 tempo = { name = "${var.bucket_prefix}-tempo" }
}
loki = {
name = "${var.bucket_prefix}-loki"
retention_days = 90
}
tempo = {
name = "${var.bucket_prefix}-tempo"
retention_days = 30
}
} }
} }
@@ -22,40 +13,11 @@ resource "aws_s3_bucket" "telemetry" {
bucket = each.value.name bucket = each.value.name
# Hetzner does not yet support all S3 ACL operations, so keep this minimal. # Hetzner does not yet support all S3 ACL operations, so keep this minimal.
# Note: Hetzner Object Storage does not support the S3 Lifecycle Configuration
# API (PutBucketLifecycleConfiguration / GetBucketLifecycleConfiguration).
# Retention is enforced at the application layer: Mimir, Loki, and Tempo each
# have native retention settings configured via their respective config files.
lifecycle { lifecycle {
prevent_destroy = true prevent_destroy = true
} }
} }
resource "aws_s3_bucket_versioning" "telemetry" {
for_each = local.buckets
provider = aws.hetzner
bucket = aws_s3_bucket.telemetry[each.key].id
versioning_configuration {
status = "Suspended"
}
}
# Lifecycle rule: hard-delete safety net behind application retention.
resource "aws_s3_bucket_lifecycle_configuration" "telemetry" {
for_each = local.buckets
provider = aws.hetzner
bucket = aws_s3_bucket.telemetry[each.key].id
rule {
id = "hard-delete-after-${each.value.retention_days}d"
status = "Enabled"
filter {}
expiration {
days = each.value.retention_days
}
abort_incomplete_multipart_upload {
days_after_initiation = 7
}
}
}
+7 -1
View File
@@ -12,6 +12,11 @@ output "server_id" {
value = hcloud_server.watchtower.id value = hcloud_server.watchtower.id
} }
output "observability_volume_id" {
description = "Hetzner Volume ID for the observability data disk. Used by Ansible to construct the stable by-id device path (scsi-0HC_Volume_<id>)."
value = hcloud_volume.observability.id
}
output "buckets" { output "buckets" {
description = "Telemetry bucket names." description = "Telemetry bucket names."
value = { for k, v in aws_s3_bucket.telemetry : k => v.bucket } value = { for k, v in aws_s3_bucket.telemetry : k => v.bucket }
@@ -23,7 +28,7 @@ output "s3_endpoint" {
} }
output "ansible_inventory" { output "ansible_inventory" {
description = "Drop-in inventory snippet for Ansible." description = "Drop-in inventory snippet for Ansible. Pipe to hosts.yml: tofu output -json | jq -r '.ansible_inventory.value'"
value = yamlencode({ value = yamlencode({
all = { all = {
hosts = { hosts = {
@@ -31,6 +36,7 @@ output "ansible_inventory" {
ansible_host = hcloud_primary_ip.ipv4.ip_address ansible_host = hcloud_primary_ip.ipv4.ip_address
ansible_user = "root" ansible_user = "root"
ansible_python_interpreter = "/usr/bin/python3" ansible_python_interpreter = "/usr/bin/python3"
observability_volume_id = hcloud_volume.observability.id
} }
} }
} }
+30 -8
View File
@@ -31,17 +31,39 @@ resource "hcloud_firewall" "watchtower" {
} }
resource "hcloud_primary_ip" "ipv4" { resource "hcloud_primary_ip" "ipv4" {
name = "${var.server_name}-ipv4" name = "${var.server_name}-ipv4"
type = "ipv4" type = "ipv4"
assignee_type = "server" location = var.location
auto_delete = false auto_delete = false
} }
resource "hcloud_primary_ip" "ipv6" { resource "hcloud_primary_ip" "ipv6" {
name = "${var.server_name}-ipv6" name = "${var.server_name}-ipv6"
type = "ipv6" type = "ipv6"
assignee_type = "server" location = var.location
auto_delete = false auto_delete = false
}
resource "hcloud_volume" "observability" {
name = "${var.server_name}-observability"
size = var.observability_volume_size_gb
location = var.location
labels = {
role = "observability-data"
managed_by = "opentofu"
environment = "prod"
}
lifecycle {
prevent_destroy = true
}
}
resource "hcloud_volume_attachment" "observability" {
volume_id = hcloud_volume.observability.id
server_id = hcloud_server.watchtower.id
automount = false
} }
resource "hcloud_server" "watchtower" { resource "hcloud_server" "watchtower" {
+6
View File
@@ -57,6 +57,12 @@ variable "admin_allow_ipv6" {
default = ["::/0"] default = ["::/0"]
} }
variable "observability_volume_size_gb" {
description = "Size in GB of the Hetzner Volume used for all observability service data (Loki WAL, Tempo WAL, Mimir TSDB, Grafana DB)."
type = number
default = 500
}
variable "bucket_prefix" { variable "bucket_prefix" {
description = "Prefix for the three telemetry buckets." description = "Prefix for the three telemetry buckets."
type = string type = string