working deployment
This commit is contained in:
+1
-1
@@ -9,7 +9,7 @@ secrets/
|
|||||||
.vault_pass
|
.vault_pass
|
||||||
.vault_password
|
.vault_password
|
||||||
ansible/inventory/hosts.yml
|
ansible/inventory/hosts.yml
|
||||||
ansible/group_vars/**/vault.yml
|
ansible/inventory/group_vars/**/vault.yml
|
||||||
wireguard/wg0.conf
|
wireguard/wg0.conf
|
||||||
wireguard/peers/
|
wireguard/peers/
|
||||||
.aws/
|
.aws/
|
||||||
|
|||||||
@@ -14,8 +14,10 @@ single Hetzner CPX42 in Nuremberg (nbg1). Built with Mimir, Loki, Tempo, and Gra
|
|||||||
```bash
|
```bash
|
||||||
# 1. Provision infrastructure
|
# 1. Provision infrastructure
|
||||||
cd tofu
|
cd tofu
|
||||||
cp terraform.tfvars.example terraform.tfvars # fill in your tokens
|
cp backend.hcl.example backend.hcl # set your Cloudflare Account ID
|
||||||
tofu init
|
cp terraform.tfvars.example terraform.tfvars # set Hetzner tokens + SSH key
|
||||||
|
set -a && source .env && set +a # load R2 credentials into env
|
||||||
|
tofu init -backend-config=backend.hcl
|
||||||
tofu apply
|
tofu apply
|
||||||
|
|
||||||
# 2. Configure server
|
# 2. Configure server
|
||||||
|
|||||||
+4
-3
@@ -3,11 +3,12 @@ inventory = inventory/hosts.yml
|
|||||||
roles_path = roles
|
roles_path = roles
|
||||||
host_key_checking = False
|
host_key_checking = False
|
||||||
retry_files_enabled = False
|
retry_files_enabled = False
|
||||||
stdout_callback = yaml
|
stdout_callback = default
|
||||||
|
result_format = yaml
|
||||||
forks = 5
|
forks = 5
|
||||||
interpreter_python = /usr/bin/python3
|
interpreter_python = /usr/bin/python3
|
||||||
vault_password_file = .vault_pass
|
vault_password_file = $HOME/.config/watchtower-observe/vault_pass
|
||||||
|
|
||||||
[ssh_connection]
|
[ssh_connection]
|
||||||
pipelining = True
|
pipelining = True
|
||||||
ssh_args = -o ControlMaster=auto -o ControlPersist=60s
|
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o IdentityFile=~/.ssh/watchtower-observe -o IdentitiesOnly=yes
|
||||||
|
|||||||
@@ -48,8 +48,8 @@ observability_volume_id: ""
|
|||||||
# When observability_volume_id is set, use the stable Hetzner by-id path.
|
# When observability_volume_id is set, use the stable Hetzner by-id path.
|
||||||
# When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only).
|
# When empty, the LUKS role falls back to a loop-backed image on the root disk (dev/staging only).
|
||||||
luks_device_source: >-
|
luks_device_source: >-
|
||||||
{{ (observability_volume_id | length > 0)
|
{{ ((observability_volume_id | string) | length > 0)
|
||||||
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + observability_volume_id, '/dev/sdb') }}
|
| ternary('/dev/disk/by-id/scsi-0HC_Volume_' + (observability_volume_id | string), '/dev/sdb') }}
|
||||||
luks_mapper_name: observability_data
|
luks_mapper_name: observability_data
|
||||||
|
|
||||||
# Container images (pin in production)
|
# Container images (pin in production)
|
||||||
@@ -4,6 +4,6 @@ all:
|
|||||||
ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4
|
ansible_host: REPLACE_WITH_TOFU_OUTPUT_IPV4
|
||||||
ansible_user: root
|
ansible_user: root
|
||||||
ansible_python_interpreter: /usr/bin/python3
|
ansible_python_interpreter: /usr/bin/python3
|
||||||
# Hetzner Volume ID for the observability data disk.
|
# Hetzner Volume ID for the observability data disk (quote to keep as string).
|
||||||
# Populate from: tofu output -json | jq -r '.ansible_inventory.value'
|
# Populate from: tofu output -json | jq -r '.ansible_inventory.value'
|
||||||
observability_volume_id: REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID
|
observability_volume_id: "REPLACE_WITH_TOFU_OUTPUT_OBSERVABILITY_VOLUME_ID"
|
||||||
|
|||||||
@@ -9,7 +9,6 @@
|
|||||||
ansible.builtin.dnf:
|
ansible.builtin.dnf:
|
||||||
name:
|
name:
|
||||||
- podman
|
- podman
|
||||||
- podman-compose
|
|
||||||
- wireguard-tools
|
- wireguard-tools
|
||||||
- cryptsetup
|
- cryptsetup
|
||||||
- nftables
|
- nftables
|
||||||
@@ -27,6 +26,12 @@
|
|||||||
community.general.timezone:
|
community.general.timezone:
|
||||||
name: "{{ timezone }}"
|
name: "{{ timezone }}"
|
||||||
|
|
||||||
|
- name: Ensure journald drop-in directory exists
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/systemd/journald.conf.d
|
||||||
|
state: directory
|
||||||
|
mode: "0755"
|
||||||
|
|
||||||
- name: Configure persistent journald with size cap
|
- name: Configure persistent journald with size cap
|
||||||
ansible.builtin.copy:
|
ansible.builtin.copy:
|
||||||
dest: /etc/systemd/journald.conf.d/persistent.conf
|
dest: /etc/systemd/journald.conf.d/persistent.conf
|
||||||
|
|||||||
@@ -22,7 +22,7 @@
|
|||||||
to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server
|
to '{{ observability_volume_id }}'. Ensure the Hetzner Volume is attached to the server
|
||||||
and udev has settled (try: udevadm settle --timeout=30).
|
and udev has settled (try: udevadm settle --timeout=30).
|
||||||
when:
|
when:
|
||||||
- observability_volume_id | length > 0
|
- (observability_volume_id | string) | length > 0
|
||||||
- not luks_device_stat.stat.exists
|
- not luks_device_stat.stat.exists
|
||||||
|
|
||||||
- name: Create loop-backed LUKS image (if no real device)
|
- name: Create loop-backed LUKS image (if no real device)
|
||||||
|
|||||||
@@ -57,6 +57,13 @@
|
|||||||
mode: "0644"
|
mode: "0644"
|
||||||
notify: Restart mimir
|
notify: Restart mimir
|
||||||
|
|
||||||
|
- name: Render Mimir runtime overrides
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: mimir-runtime.yaml.j2
|
||||||
|
dest: "{{ observability_config_dir }}/mimir/runtime.yaml"
|
||||||
|
mode: "0644"
|
||||||
|
notify: Restart mimir
|
||||||
|
|
||||||
- name: Render Loki config
|
- name: Render Loki config
|
||||||
ansible.builtin.template:
|
ansible.builtin.template:
|
||||||
src: loki.yaml.j2
|
src: loki.yaml.j2
|
||||||
@@ -75,7 +82,7 @@
|
|||||||
ansible.builtin.template:
|
ansible.builtin.template:
|
||||||
src: grafana.ini.j2
|
src: grafana.ini.j2
|
||||||
dest: "{{ observability_config_dir }}/grafana/grafana.ini"
|
dest: "{{ observability_config_dir }}/grafana/grafana.ini"
|
||||||
mode: "0640"
|
mode: "0644"
|
||||||
notify: Restart grafana
|
notify: Restart grafana
|
||||||
|
|
||||||
- name: Render Grafana datasource provisioning
|
- name: Render Grafana datasource provisioning
|
||||||
@@ -163,6 +170,25 @@
|
|||||||
- name: Force handlers to flush before health checks
|
- name: Force handlers to flush before health checks
|
||||||
ansible.builtin.meta: flush_handlers
|
ansible.builtin.meta: flush_handlers
|
||||||
|
|
||||||
|
# Handlers only restart services when the quadlet templates change. On
|
||||||
|
# subsequent runs (or partial first runs) the units exist but are stopped, so
|
||||||
|
# explicitly ensure each backend is started before the health checks below.
|
||||||
|
# Quadlet-generated units cannot be `enabled` (they're auto-wired by their
|
||||||
|
# WantedBy= line), so we only set state=started — systemd treats this as a
|
||||||
|
# no-op when the unit is already active.
|
||||||
|
- name: Ensure observability services are started
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: started
|
||||||
|
daemon_reload: true
|
||||||
|
loop:
|
||||||
|
- wg-easy.service
|
||||||
|
- mimir.service
|
||||||
|
- loki.service
|
||||||
|
- tempo.service
|
||||||
|
- grafana.service
|
||||||
|
- caddy.service
|
||||||
|
|
||||||
- name: Deploy disk-guard script
|
- name: Deploy disk-guard script
|
||||||
ansible.builtin.template:
|
ansible.builtin.template:
|
||||||
src: disk-guard.sh.j2
|
src: disk-guard.sh.j2
|
||||||
|
|||||||
@@ -0,0 +1,4 @@
|
|||||||
|
# Mimir runtime overrides (hot-reloaded). Empty by default; add per-tenant
|
||||||
|
# overrides here without restarting Mimir.
|
||||||
|
# Reference: https://grafana.com/docs/mimir/latest/configure/about-runtime-configuration/
|
||||||
|
overrides: {}
|
||||||
@@ -9,6 +9,7 @@ ContainerName=mimir
|
|||||||
Network=observability.network
|
Network=observability.network
|
||||||
Volume=mimir-data.volume:/var/lib/mimir:Z
|
Volume=mimir-data.volume:/var/lib/mimir:Z
|
||||||
Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z
|
Volume={{ observability_config_dir }}/mimir/mimir.yaml:/etc/mimir/mimir.yaml:ro,Z
|
||||||
|
Volume={{ observability_config_dir }}/mimir/runtime.yaml:/etc/mimir/runtime.yaml:ro,Z
|
||||||
EnvironmentFile={{ observability_secrets_dir }}/s3.env
|
EnvironmentFile={{ observability_secrets_dir }}/s3.env
|
||||||
|
|
||||||
# -config.expand-env enables ${AWS_*} substitution from EnvironmentFile.
|
# -config.expand-env enables ${AWS_*} substitution from EnvironmentFile.
|
||||||
|
|||||||
@@ -5,6 +5,12 @@ target: all,alertmanager,overrides-exporter
|
|||||||
|
|
||||||
multitenancy_enabled: true
|
multitenancy_enabled: true
|
||||||
|
|
||||||
|
# The activity tracker writes its log to ./metrics-activity.log by default.
|
||||||
|
# Inside the container the working directory is "/" which is not writable for
|
||||||
|
# the non-root user, so point it at the data volume.
|
||||||
|
activity_tracker:
|
||||||
|
filepath: /var/lib/mimir/metrics-activity.log
|
||||||
|
|
||||||
server:
|
server:
|
||||||
http_listen_port: 8080
|
http_listen_port: 8080
|
||||||
grpc_listen_port: 9095
|
grpc_listen_port: 9095
|
||||||
|
|||||||
@@ -12,6 +12,23 @@
|
|||||||
owner: root
|
owner: root
|
||||||
group: root
|
group: root
|
||||||
|
|
||||||
|
# wg-easy runs `wg-quick up wg0` inside its container, which calls
|
||||||
|
# iptables-legacy to set up the NAT POSTROUTING rule. AlmaLinux 10 doesn't
|
||||||
|
# auto-load the legacy iptables kernel modules, so the call fails with
|
||||||
|
# "can't initialize iptables table 'nat': Table does not exist" and wg0 is
|
||||||
|
# torn down. Load the modules now and persist them across reboots.
|
||||||
|
- name: Load iptables kernel modules required by wg-easy
|
||||||
|
community.general.modprobe:
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: present
|
||||||
|
persistent: present
|
||||||
|
loop:
|
||||||
|
- ip_tables
|
||||||
|
- iptable_filter
|
||||||
|
- iptable_nat
|
||||||
|
- nf_nat
|
||||||
|
- nf_conntrack
|
||||||
|
|
||||||
- name: Enable IPv4 forwarding
|
- name: Enable IPv4 forwarding
|
||||||
ansible.posix.sysctl:
|
ansible.posix.sysctl:
|
||||||
name: net.ipv4.ip_forward
|
name: net.ipv4.ip_forward
|
||||||
|
|||||||
@@ -7,10 +7,10 @@
|
|||||||
# Rebuild whenever the ARG versions below change; Ansible will detect the
|
# Rebuild whenever the ARG versions below change; Ansible will detect the
|
||||||
# Dockerfile checksum change and re-run `podman build`.
|
# Dockerfile checksum change and re-run `podman build`.
|
||||||
|
|
||||||
FROM golang:1.23-bookworm AS builder
|
FROM golang:1.26-trixie AS builder
|
||||||
|
|
||||||
ARG XCADDY_VERSION=v0.3.5
|
ARG XCADDY_VERSION=v0.3.5
|
||||||
ARG CADDY_VERSION=v2.9.1
|
ARG CADDY_VERSION=v2.11.2
|
||||||
ARG CORAZA_CADDY_VERSION=v2.5.0
|
ARG CORAZA_CADDY_VERSION=v2.5.0
|
||||||
ARG CORAZA_CRS_VERSION=v4.7.0
|
ARG CORAZA_CRS_VERSION=v4.7.0
|
||||||
|
|
||||||
@@ -21,7 +21,7 @@ RUN xcaddy build "${CADDY_VERSION}" \
|
|||||||
--with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}"
|
--with "github.com/corazawaf/coraza-coreruleset@${CORAZA_CRS_VERSION}"
|
||||||
|
|
||||||
# ── Runtime image ──────────────────────────────────────────────────────────────
|
# ── Runtime image ──────────────────────────────────────────────────────────────
|
||||||
FROM debian:bookworm-slim
|
FROM debian:trixie-slim
|
||||||
|
|
||||||
RUN apt-get update \
|
RUN apt-get update \
|
||||||
&& apt-get install -y --no-install-recommends ca-certificates \
|
&& apt-get install -y --no-install-recommends ca-certificates \
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
# Copy to backend.hcl (gitignored) and set your Cloudflare Account ID.
|
||||||
|
# Pass at init time: tofu init -backend-config=backend.hcl
|
||||||
|
#
|
||||||
|
# Credentials are read from env vars — set before running tofu:
|
||||||
|
# export AWS_ACCESS_KEY_ID=<r2-access-key-id>
|
||||||
|
# export AWS_SECRET_ACCESS_KEY=<r2-secret-access-key>
|
||||||
|
|
||||||
|
endpoints = {
|
||||||
|
s3 = "https://<ACCOUNT_ID>.r2.cloudflarestorage.com"
|
||||||
|
}
|
||||||
+7
-45
@@ -1,17 +1,8 @@
|
|||||||
locals {
|
locals {
|
||||||
buckets = {
|
buckets = {
|
||||||
mimir = {
|
mimir = { name = "${var.bucket_prefix}-mimir" }
|
||||||
name = "${var.bucket_prefix}-mimir"
|
loki = { name = "${var.bucket_prefix}-loki" }
|
||||||
retention_days = 365
|
tempo = { name = "${var.bucket_prefix}-tempo" }
|
||||||
}
|
|
||||||
loki = {
|
|
||||||
name = "${var.bucket_prefix}-loki"
|
|
||||||
retention_days = 90
|
|
||||||
}
|
|
||||||
tempo = {
|
|
||||||
name = "${var.bucket_prefix}-tempo"
|
|
||||||
retention_days = 90
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -22,40 +13,11 @@ resource "aws_s3_bucket" "telemetry" {
|
|||||||
bucket = each.value.name
|
bucket = each.value.name
|
||||||
|
|
||||||
# Hetzner does not yet support all S3 ACL operations, so keep this minimal.
|
# Hetzner does not yet support all S3 ACL operations, so keep this minimal.
|
||||||
|
# Note: Hetzner Object Storage does not support the S3 Lifecycle Configuration
|
||||||
|
# API (PutBucketLifecycleConfiguration / GetBucketLifecycleConfiguration).
|
||||||
|
# Retention is enforced at the application layer: Mimir, Loki, and Tempo each
|
||||||
|
# have native retention settings configured via their respective config files.
|
||||||
lifecycle {
|
lifecycle {
|
||||||
prevent_destroy = true
|
prevent_destroy = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
resource "aws_s3_bucket_versioning" "telemetry" {
|
|
||||||
for_each = local.buckets
|
|
||||||
provider = aws.hetzner
|
|
||||||
|
|
||||||
bucket = aws_s3_bucket.telemetry[each.key].id
|
|
||||||
versioning_configuration {
|
|
||||||
status = "Suspended"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
# Lifecycle rule: hard-delete safety net behind application retention.
|
|
||||||
resource "aws_s3_bucket_lifecycle_configuration" "telemetry" {
|
|
||||||
for_each = local.buckets
|
|
||||||
provider = aws.hetzner
|
|
||||||
|
|
||||||
bucket = aws_s3_bucket.telemetry[each.key].id
|
|
||||||
|
|
||||||
rule {
|
|
||||||
id = "hard-delete-after-${each.value.retention_days}d"
|
|
||||||
status = "Enabled"
|
|
||||||
|
|
||||||
filter {}
|
|
||||||
|
|
||||||
expiration {
|
|
||||||
days = each.value.retention_days
|
|
||||||
}
|
|
||||||
|
|
||||||
abort_incomplete_multipart_upload {
|
|
||||||
days_after_initiation = 7
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+8
-8
@@ -31,17 +31,17 @@ resource "hcloud_firewall" "watchtower" {
|
|||||||
}
|
}
|
||||||
|
|
||||||
resource "hcloud_primary_ip" "ipv4" {
|
resource "hcloud_primary_ip" "ipv4" {
|
||||||
name = "${var.server_name}-ipv4"
|
name = "${var.server_name}-ipv4"
|
||||||
type = "ipv4"
|
type = "ipv4"
|
||||||
assignee_type = "server"
|
location = var.location
|
||||||
auto_delete = false
|
auto_delete = false
|
||||||
}
|
}
|
||||||
|
|
||||||
resource "hcloud_primary_ip" "ipv6" {
|
resource "hcloud_primary_ip" "ipv6" {
|
||||||
name = "${var.server_name}-ipv6"
|
name = "${var.server_name}-ipv6"
|
||||||
type = "ipv6"
|
type = "ipv6"
|
||||||
assignee_type = "server"
|
location = var.location
|
||||||
auto_delete = false
|
auto_delete = false
|
||||||
}
|
}
|
||||||
|
|
||||||
resource "hcloud_volume" "observability" {
|
resource "hcloud_volume" "observability" {
|
||||||
|
|||||||
Reference in New Issue
Block a user