working deployment
This commit is contained in:
+1
-1
@@ -9,7 +9,7 @@ secrets/
|
||||
.vault_pass
|
||||
.vault_password
|
||||
ansible/inventory/hosts.yml
|
||||
ansible/group_vars/**/vault.yml
|
||||
ansible/inventory/group_vars/**/vault.yml
|
||||
wireguard/wg0.conf
|
||||
wireguard/peers/
|
||||
.aws/
|
||||
|
||||
@@ -14,8 +14,10 @@ single Hetzner CPX42 in Nuremberg (nbg1). Built with Mimir, Loki, Tempo, and Gra
|
||||
```bash
|
||||
# 1. Provision infrastructure
|
||||
cd tofu
|
||||
cp terraform.tfvars.example terraform.tfvars # fill in your tokens
|
||||
tofu init
|
||||
cp backend.hcl.example backend.hcl # set your Cloudflare Account ID
|
||||
cp terraform.tfvars.example terraform.tfvars # set Hetzner tokens + SSH key
|
||||
set -a && source .env && set +a # load R2 credentials into env
|
||||
tofu init -backend-config=backend.hcl
|
||||
tofu apply
|
||||
|
||||
# 2. Configure server
|
||||
|
||||
+4
-3
@@ -3,11 +3,12 @@ inventory = inventory/hosts.yml
|
||||
roles_path = roles
|
||||
host_key_checking = False
|
||||
retry_files_enabled = False
|
||||
stdout_callback = yaml
|
||||
stdout_callback = default
|
||||
result_format = yaml
|
||||
forks = 5
|
||||
interpreter_python = /usr/bin/python3
|
||||
vault_password_file = .vault_pass
|
||||
vault_password_file = $HOME/.config/watchtower-observe/vault_pass
|
||||
|
||||
[ssh_connection]
|
||||
pipelining = True
|
||||
ssh_args = -o ControlMaster=auto -o ControlPersist=60s
|
||||
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o IdentityFile=~/.ssh/watchtower-observe -o IdentitiesOnly=yes
|
||||
|
||||
@@ -48,8 +48,8 @@ observability_volume_id: ""
|
||||
# When observability_volume_id is set, use the stable Hetzner by-id path.
|
||||
# When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only).
|
||||
luks_device_source: >-
|
||||
{{ (observability_volume_id | length > 0)
|
||||
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + observability_volume_id, '/dev/sdb') }}
|
||||
{{ ((observability_volume_id | string) | length > 0)
|
||||
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + (observability_volume_id | string), '/dev/sdb') }}
|
||||
luks_mapper_name: observability_data
|
||||
|
||||
# Container images (pin in production)
|
||||
@@ -4,6 +4,6 @@ all:
|
||||
ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4
|
||||
ansible_user: root
|
||||
ansible_python_interpreter: /usr/bin/python3
|
||||
# Hetzner Volume ID for the observability data disk.
|
||||
# Hetzner Volume ID for the observability data disk (quote to keep as string).
|
||||
# Populate from: tofu output -json | jq -r '.ansible_inventory.value'
|
||||
observability_volume_id: REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID
|
||||
observability_volume_id: "REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID"
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
ansible.builtin.dnf:
|
||||
name:
|
||||
- podman
|
||||
- podman-compose
|
||||
- wireguard-tools
|
||||
- cryptsetup
|
||||
- nftables
|
||||
@@ -27,6 +26,12 @@
|
||||
community.general.timezone:
|
||||
name: "{{ timezone }}"
|
||||
|
||||
- name: Ensure journald drop-in directory exists
|
||||
ansible.builtin.file:
|
||||
path: /etc/systemd/journald.conf.d
|
||||
state: directory
|
||||
mode: "0755"
|
||||
|
||||
- name: Configure persistent journald with size cap
|
||||
ansible.builtin.copy:
|
||||
dest: /etc/systemd/journald.conf.d/persistent.conf
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server
|
||||
and udev has settled (try: udevadm settle --timeout=30).
|
||||
when:
|
||||
- observability_volume_id | length > 0
|
||||
- (observability_volume_id | string) | length > 0
|
||||
- not luks_device_stat.stat.exists
|
||||
|
||||
- name: Create loop-backed LUKS image (if no real device)
|
||||
|
||||
@@ -57,6 +57,13 @@
|
||||
mode: "0644"
|
||||
notify: Restart mimir
|
||||
|
||||
- name: Render Mimir runtime overrides
|
||||
ansible.builtin.template:
|
||||
src: mimir-runtime.yaml.j2
|
||||
dest: "{{ observability_config_dir }}/mimir/runtime.yaml"
|
||||
mode: "0644"
|
||||
notify: Restart mimir
|
||||
|
||||
- name: Render Loki config
|
||||
ansible.builtin.template:
|
||||
src: loki.yaml.j2
|
||||
@@ -75,7 +82,7 @@
|
||||
ansible.builtin.template:
|
||||
src: grafana.ini.j2
|
||||
dest: "{{ observability_config_dir }}/grafana/grafana.ini"
|
||||
mode: "0640"
|
||||
mode: "0644"
|
||||
notify: Restart grafana
|
||||
|
||||
- name: Render Grafana datasource provisioning
|
||||
@@ -163,6 +170,25 @@
|
||||
- name: Force handlers to flush before health checks
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
# Handlers only restart services when the quadlet templates change. On
|
||||
# subsequent runs (or partial first runs) the units exist but are stopped, so
|
||||
# explicitly ensure each backend is started before the health checks below.
|
||||
# Quadlet-generated units cannot be `enabled` (they're auto-wired by their
|
||||
# WantedBy= line), so we only set state=started — systemd treats this as a
|
||||
# no-op when the unit is already active.
|
||||
- name: Ensure observability services are started
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ item }}"
|
||||
state: started
|
||||
daemon_reload: true
|
||||
loop:
|
||||
- wg-easy.service
|
||||
- mimir.service
|
||||
- loki.service
|
||||
- tempo.service
|
||||
- grafana.service
|
||||
- caddy.service
|
||||
|
||||
- name: Deploy disk-guard script
|
||||
ansible.builtin.template:
|
||||
src: disk-guard.sh.j2
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
# Mimir runtime overrides (hot-reloaded). Empty by default; add per-tenant
|
||||
# overrides here without restarting Mimir.
|
||||
# Reference: https://grafana.com/docs/mimir/latest/configure/about-runtime-configuration/
|
||||
overrides: {}
|
||||
@@ -9,6 +9,7 @@ ContainerName=mimir
|
||||
Network=observability.network
|
||||
Volume=mimir-data.volume:/var/lib/mimir:Z
|
||||
Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z
|
||||
Volume={{ observability_config_dir }}/mimir/runtime.yaml:/etc/mimir/runtime.yaml:ro,Z
|
||||
EnvironmentFile={{ observability_secrets_dir }}/s3.env
|
||||
|
||||
# -config.expand-env enables ${AWS_*} substitution from EnvironmentFile.
|
||||
|
||||
@@ -5,6 +5,12 @@ target: all,alertmanager,overrides-exporter
|
||||
|
||||
multitenancy_enabled: true
|
||||
|
||||
# The activity tracker writes its log to ./metrics-activity.log by default.
|
||||
# Inside the container the working directory is "/" which is not writable for
|
||||
# the non-root user, so point it at the data volume.
|
||||
activity_tracker:
|
||||
filepath: /var/lib/mimir/metrics-activity.log
|
||||
|
||||
server:
|
||||
http_listen_port: 8080
|
||||
grpc_listen_port: 9095
|
||||
|
||||
@@ -12,6 +12,23 @@
|
||||
owner: root
|
||||
group: root
|
||||
|
||||
# wg-easy runs `wg-quick up wg0` inside its container, which calls
|
||||
# iptables-legacy to set up the NAT POSTROUTING rule. AlmaLinux 10 doesn't
|
||||
# auto-load the legacy iptables kernel modules, so the call fails with
|
||||
# "can't initialize iptables table 'nat': Table does not exist" and wg0 is
|
||||
# torn down. Load the modules now and persist them across reboots.
|
||||
- name: Load iptables kernel modules required by wg-easy
|
||||
community.general.modprobe:
|
||||
name: "{{ item }}"
|
||||
state: present
|
||||
persistent: present
|
||||
loop:
|
||||
- ip_tables
|
||||
- iptable_filter
|
||||
- iptable_nat
|
||||
- nf_nat
|
||||
- nf_conntrack
|
||||
|
||||
- name: Enable IPv4 forwarding
|
||||
ansible.posix.sysctl:
|
||||
name: net.ipv4.ip_forward
|
||||
|
||||
@@ -7,10 +7,10 @@
|
||||
# Rebuild whenever the ARG versions below change; Ansible will detect the
|
||||
# Dockerfile checksum change and re-run `podman build`.
|
||||
|
||||
FROM golang:1.23-bookworm AS builder
|
||||
FROM golang:1.26-trixie AS builder
|
||||
|
||||
ARG XCADDY_VERSION=v0.3.5
|
||||
ARG CADDY_VERSION=v2.9.1
|
||||
ARG CADDY_VERSION=v2.11.2
|
||||
ARG CORAZA_CADDY_VERSION=v2.5.0
|
||||
ARG CORAZA_CRS_VERSION=v4.7.0
|
||||
|
||||
@@ -21,7 +21,7 @@ RUN xcaddy build "${CADDY_VERSION}" \
|
||||
--with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}"
|
||||
|
||||
# ── Runtime image ──────────────────────────────────────────────────────────────
|
||||
FROM debian:bookworm-slim
|
||||
FROM debian:trixie-slim
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates \
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Copy to backend.hcl (gitignored) and set your Cloudflare Account ID.
|
||||
# Pass at init time: tofu init -backend-config=backend.hcl
|
||||
#
|
||||
# Credentials are read from env vars — set before running tofu:
|
||||
# export AWS_ACCESS_KEY_ID=<r2-access-key-id>
|
||||
# export AWS_SECRET_ACCESS_KEY=<r2-secret-access-key>
|
||||
|
||||
endpoints = {
|
||||
s3 = "https://<ACCOUNT_ID>.r2.cloudflarestorage.com"
|
||||
}
|
||||
+7
-45
@@ -1,17 +1,8 @@
|
||||
locals {
|
||||
buckets = {
|
||||
mimir = {
|
||||
name = "${var.bucket_prefix}-mimir"
|
||||
retention_days = 365
|
||||
}
|
||||
loki = {
|
||||
name = "${var.bucket_prefix}-loki"
|
||||
retention_days = 90
|
||||
}
|
||||
tempo = {
|
||||
name = "${var.bucket_prefix}-tempo"
|
||||
retention_days = 90
|
||||
}
|
||||
mimir = { name = "${var.bucket_prefix}-mimir" }
|
||||
loki = { name = "${var.bucket_prefix}-loki" }
|
||||
tempo = { name = "${var.bucket_prefix}-tempo" }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,40 +13,11 @@ resource "aws_s3_bucket" "telemetry" {
|
||||
bucket = each.value.name
|
||||
|
||||
# Hetzner does not yet support all S3 ACL operations, so keep this minimal.
|
||||
# Note: Hetzner Object Storage does not support the S3 Lifecycle Configuration
|
||||
# API (PutBucketLifecycleConfiguration / GetBucketLifecycleConfiguration).
|
||||
# Retention is enforced at the application layer: Mimir, Loki, and Tempo each
|
||||
# have native retention settings configured via their respective config files.
|
||||
lifecycle {
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
resource "aws_s3_bucket_versioning" "telemetry" {
|
||||
for_each = local.buckets
|
||||
provider = aws.hetzner
|
||||
|
||||
bucket = aws_s3_bucket.telemetry[each.key].id
|
||||
versioning_configuration {
|
||||
status = "Suspended"
|
||||
}
|
||||
}
|
||||
|
||||
# Lifecycle rule: hard-delete safety net behind application retention.
|
||||
resource "aws_s3_bucket_lifecycle_configuration" "telemetry" {
|
||||
for_each = local.buckets
|
||||
provider = aws.hetzner
|
||||
|
||||
bucket = aws_s3_bucket.telemetry[each.key].id
|
||||
|
||||
rule {
|
||||
id = "hard-delete-after-${each.value.retention_days}d"
|
||||
status = "Enabled"
|
||||
|
||||
filter {}
|
||||
|
||||
expiration {
|
||||
days = each.value.retention_days
|
||||
}
|
||||
|
||||
abort_incomplete_multipart_upload {
|
||||
days_after_initiation = 7
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+8
-8
@@ -31,17 +31,17 @@ resource "hcloud_firewall" "watchtower" {
|
||||
}
|
||||
|
||||
resource "hcloud_primary_ip" "ipv4" {
|
||||
name = "${var.server_name}-ipv4"
|
||||
type = "ipv4"
|
||||
assignee_type = "server"
|
||||
auto_delete = false
|
||||
name = "${var.server_name}-ipv4"
|
||||
type = "ipv4"
|
||||
location = var.location
|
||||
auto_delete = false
|
||||
}
|
||||
|
||||
resource "hcloud_primary_ip" "ipv6" {
|
||||
name = "${var.server_name}-ipv6"
|
||||
type = "ipv6"
|
||||
assignee_type = "server"
|
||||
auto_delete = false
|
||||
name = "${var.server_name}-ipv6"
|
||||
type = "ipv6"
|
||||
location = var.location
|
||||
auto_delete = false
|
||||
}
|
||||
|
||||
resource "hcloud_volume" "observability" {
|
||||
|
||||
Reference in New Issue
Block a user