infra, ansible, and required container changes

This commit is contained in:
Jason Ross
2026-04-29 16:39:00 -05:00
parent 5219e3131d
commit 45f2a9a073
35 changed files with 1379 additions and 12 deletions
+277
View File
@@ -0,0 +1,277 @@
name: deploy
# ----------------------------------------------------------------------------
# Pipeline shape:
#
# build-app ─┐
# build-caddy├──► gate ──► push-app ─┐
# infra ─┘ push-caddy ─┴──► deploy (Ansible)
#
# - build-app / build-caddy: build container images, save as tar artifacts.
# - infra: tofu fmt/validate/plan (PR) or apply (push/dispatch); emits
# instance_ip as a job output.
# - gate: fan-in checkpoint, fails fast if any upstream failed.
# - push-app / push-caddy: load tar artifact, tag, and push to GHCR.
# - deploy: run Ansible against the Vultr host, pulling images from GHCR.
# ----------------------------------------------------------------------------
on:
workflow_dispatch:
inputs:
action:
description: "Pipeline action"
required: true
default: deploy
type: choice
options: [plan, deploy, destroy]
push:
branches: [main]
pull_request:
permissions:
contents: read
packages: write
pull-requests: write
concurrency:
group: deploy-${{ github.ref }}
cancel-in-progress: false
env:
REGISTRY: ghcr.io
IMAGE_APP: ${{ github.repository }}/app
IMAGE_CADDY: ${{ github.repository }}/caddy
TAG: ${{ github.sha }}
jobs:
# ---------------------------------------------------------------------------
# Parallel build / plan stage
# ---------------------------------------------------------------------------
build-app:
name: Build app image
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Build image with Podman
run: |
podman build \
-t "app:${TAG}" \
-f Containerfile \
.
- name: Save image as OCI tar
run: podman save -o /tmp/app.tar "app:${TAG}"
- name: Upload artifact
uses: actions/upload-artifact@v4
with:
name: image-app
path: /tmp/app.tar
retention-days: 1
if-no-files-found: error
build-caddy:
name: Build caddy image
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Build image with Podman
run: |
podman build \
-t "caddy:${TAG}" \
-f caddy/Containerfile \
caddy
- name: Save image as OCI tar
run: podman save -o /tmp/caddy.tar "caddy:${TAG}"
- name: Upload artifact
uses: actions/upload-artifact@v4
with:
name: image-caddy
path: /tmp/caddy.tar
retention-days: 1
if-no-files-found: error
infra:
name: OpenTofu (Vultr / R2)
runs-on: ubuntu-latest
defaults:
run:
working-directory: infra
outputs:
instance_ip: ${{ steps.outputs.outputs.instance_ip }}
env:
TF_VAR_vultr_api_key: ${{ secrets.VULTR_API_KEY }}
AWS_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
AWS_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
R2_BUCKET: ${{ secrets.R2_BUCKET }}
R2_ACCOUNT_ID: ${{ secrets.R2_ACCOUNT_ID }}
steps:
- uses: actions/checkout@v4
- uses: opentofu/setup-opentofu@v1
with:
tofu_version: "1.8.5"
- name: tofu fmt
run: tofu fmt -check -recursive
- name: tofu init (R2 backend)
run: |
tofu init \
-backend-config="bucket=${R2_BUCKET}" \
-backend-config="endpoints={ s3 = \"https://${R2_ACCOUNT_ID}.r2.cloudflarestorage.com\" }"
- name: tofu validate
run: tofu validate
- name: tofu plan
if: github.event_name == 'pull_request' || inputs.action == 'plan'
run: tofu plan -input=false -no-color
- name: tofu apply
if: >-
github.event_name == 'push' ||
(github.event_name == 'workflow_dispatch' && inputs.action == 'deploy')
run: tofu apply -input=false -auto-approve
- name: tofu destroy
if: github.event_name == 'workflow_dispatch' && inputs.action == 'destroy'
run: tofu destroy -input=false -auto-approve
- name: Export instance IP
id: outputs
if: >-
github.event_name == 'push' ||
(github.event_name == 'workflow_dispatch' && inputs.action == 'deploy')
run: echo "instance_ip=$(tofu output -raw main_ip)" >> "$GITHUB_OUTPUT"
# ---------------------------------------------------------------------------
# Quality gate (fan-in)
# ---------------------------------------------------------------------------
gate:
name: Quality gate
needs: [build-app, build-caddy, infra]
runs-on: ubuntu-latest
steps:
- run: echo "build-app, build-caddy, and infra all succeeded."
# ---------------------------------------------------------------------------
# Push to GHCR (parallel, post-gate)
# ---------------------------------------------------------------------------
push-app:
name: Push app → GHCR
needs: gate
if: >-
github.event_name == 'push' ||
(github.event_name == 'workflow_dispatch' && inputs.action == 'deploy')
runs-on: ubuntu-latest
steps:
- uses: actions/download-artifact@v4
with:
name: image-app
path: /tmp
- name: Login to GHCR
run: echo "${{ secrets.GITHUB_TOKEN }}" | podman login "${REGISTRY}" -u "${{ github.actor }}" --password-stdin
- name: Tag and push
run: |
podman load -i /tmp/app.tar
podman tag "app:${TAG}" "${REGISTRY}/${IMAGE_APP}:${TAG}"
podman tag "app:${TAG}" "${REGISTRY}/${IMAGE_APP}:latest"
podman push "${REGISTRY}/${IMAGE_APP}:${TAG}"
podman push "${REGISTRY}/${IMAGE_APP}:latest"
push-caddy:
name: Push caddy → GHCR
needs: gate
if: >-
github.event_name == 'push' ||
(github.event_name == 'workflow_dispatch' && inputs.action == 'deploy')
runs-on: ubuntu-latest
steps:
- uses: actions/download-artifact@v4
with:
name: image-caddy
path: /tmp
- name: Login to GHCR
run: echo "${{ secrets.GITHUB_TOKEN }}" | podman login "${REGISTRY}" -u "${{ github.actor }}" --password-stdin
- name: Tag and push
run: |
podman load -i /tmp/caddy.tar
podman tag "caddy:${TAG}" "${REGISTRY}/${IMAGE_CADDY}:${TAG}"
podman tag "caddy:${TAG}" "${REGISTRY}/${IMAGE_CADDY}:latest"
podman push "${REGISTRY}/${IMAGE_CADDY}:${TAG}"
podman push "${REGISTRY}/${IMAGE_CADDY}:latest"
# ---------------------------------------------------------------------------
# Deploy with Ansible
# ---------------------------------------------------------------------------
deploy:
name: Ansible deploy
needs: [push-app, push-caddy, infra]
if: >-
github.event_name == 'push' ||
(github.event_name == 'workflow_dispatch' && inputs.action == 'deploy')
runs-on: ubuntu-latest
defaults:
run:
working-directory: ansible
env:
ANSIBLE_HOST_KEY_CHECKING: "False"
ANSIBLE_FORCE_COLOR: "1"
steps:
- uses: actions/checkout@v4
- name: Set up Python + Ansible
uses: actions/setup-python@v5
with:
python-version: "3.12"
- name: Install Ansible + collections
run: |
python -m pip install --upgrade pip
pip install ansible
ansible-galaxy collection install ansible.posix containers.podman
- name: Configure SSH
env:
SSH_PRIVATE_KEY: ${{ secrets.SSH_PRIVATE_KEY }}
INSTANCE_IP: ${{ needs.infra.outputs.instance_ip }}
run: |
mkdir -p ~/.ssh
printf '%s\n' "$SSH_PRIVATE_KEY" > ~/.ssh/id_ed25519
chmod 600 ~/.ssh/id_ed25519
ssh-keyscan -H "$INSTANCE_IP" >> ~/.ssh/known_hosts 2>/dev/null
- name: Render inventory
env:
INSTANCE_IP: ${{ needs.infra.outputs.instance_ip }}
run: |
cat > inventory.yml <<EOF
all:
children:
blog:
hosts:
dev-blog-prod:
ansible_host: ${INSTANCE_IP}
ansible_user: root
ansible_python_interpreter: /usr/bin/python3
EOF
- name: Run playbook
env:
GHCR_PAT: ${{ secrets.GHCR_PULL_TOKEN }}
run: |
ansible-playbook site.yml \
-e "ghcr_owner=${{ github.repository_owner }}" \
-e "ghcr_repo=${{ github.event.repository.name }}" \
-e "ghcr_user=${{ github.actor }}" \
-e "image_tag=${TAG}" \
-e "ghcr_pat=${GHCR_PAT}"
+90
View File
@@ -0,0 +1,90 @@
# Ansible — dev_blog production host
Provisions the AlmaLinux 10 instance created by `infra/` (Vultr) so it can
run the Podman-managed `dev-blog-app` + `dev-blog-caddy` stack with:
- **Automatic security updates** via `dnf-automatic` (`upgrade_type=security`,
`apply_updates=yes`, `reboot=never`) plus a separate `auto-reboot.timer`
that fires daily at **02:00 America/Chicago** and reboots only when
`dnf needs-restarting -r` says one is pending.
- **nftables** as the host firewall (allow `22/80/443/tcp` + `443/udp`,
rate-limited new SSH, `policy drop` everywhere else; firewalld masked).
- **fail2ban** with the `nftables-multiport` banaction, watching the SSH
journal and the Caddy / Coraza logs.
- **Podman + runtime deps** (`crun`, `netavark`, `aardvark-dns`,
`slirp4netns`, `fuse-overlayfs`, `passt`, `container-selinux`, `skopeo`,
`buildah`) and the Quadlet units from `../quadlet/` (with the `Image=`
lines rewritten to point at GHCR).
- **GHCR auth** via the `gh` CLI: a PAT (`ghcr_pat`) is used to authenticate
both `gh` and `podman` against `ghcr.io`, and the app/caddy images are
pulled before the Quadlet units start.
## Coraza → fail2ban → nftables wiring
```
[ Caddy + Coraza container ]
│ writes to /var/log/caddy (bind-mounted from host)
▼
/var/log/caddy/access.log (Caddy JSON access log – status=403 lines)
/var/log/caddy/coraza-audit.log (Coraza serial audit log – any --A-- entry)
│ tailed by
▼
[ fail2ban jail: caddy-coraza ]
│ banaction = nftables-multiport
▼
[ nftables table f2b-table / set addr-set-caddy-coraza ] → packets dropped
```
Three repo files were updated to enable that wiring:
- `quadlet/dev-blog-caddy.container` — adds `Volume=/var/log/caddy:/var/log/caddy:Z`.
- `caddy/Caddyfile` — sends access logs to `/var/log/caddy/access.log` (JSON).
- `caddy/coraza.conf` — enables `SecAuditLog` (serial) at
`/var/log/caddy/coraza-audit.log`, only on relevant 4xx/5xx.
The fail2ban filter (`roles/fail2ban/files/caddy-coraza.filter`) matches
either `"remote_ip":"…","status":403` lines from the access log or
`--xxxx-A--` headers in the audit log. Either pattern produces a hit on
`<HOST>`, so once a client crosses `maxretry` within `findtime`, fail2ban
drops it into the `f2b-table` nft set for `bantime`.
## Usage
```sh
cd ansible
cp inventory.yml.example inventory.yml # set ansible_host = your Vultr IP
# Quick connectivity / fact gathering:
ansible -m ping blog
# Full provision:
ansible-playbook site.yml
# Just one role:
ansible-playbook site.yml --tags fail2ban # (add `tags:` to roles if needed)
```
## Tunables (group_vars/all.yml)
| Var | Default | Notes |
| ------------------------------ | -------------- | ------------------------------- |
| `ssh_port` | `22` | |
| `http_port` / `https_port` | `80` / `443` | |
| `caddy_log_dir` | `/var/log/caddy` | host path bind-mounted into Caddy |
| `dnf_automatic_apply_updates` | `true` | |
| `dnf_automatic_upgrade_type` | `security` | `default` for full updates |
| `f2b_findtime` / `f2b_bantime` / `f2b_maxretry` | `10m / 1h / 5` | Caddy/Coraza jail thresholds |
| `ghcr_owner` / `ghcr_repo` | `OWNER / dev_blog` | GHCR namespace |
| `image_tag` | `latest` | Tag to pull (CI sets to commit SHA) |
## Notes
- Don't enable both `firewalld` and the `nftables` service — this playbook
masks `firewalld`. Podman's `netavark` writes to its own nft tables and
doesn't conflict.
- The `synchronize` task uses `ansible.posix.synchronize`. Install
`ansible-galaxy collection install ansible.posix` once on the controller.
- After first run, verify with:
- `systemctl status dnf-automatic.timer nftables fail2ban`
- `nft list ruleset`
- `fail2ban-client status caddy-coraza`
+11
View File
@@ -0,0 +1,11 @@
[defaults]
inventory = inventory.yml
roles_path = roles
host_key_checking = True
retry_files_enabled = False
stdout_callback = yaml
forks = 5
interpreter_python = auto_silent
[ssh_connection]
pipelining = True
+40
View File
@@ -0,0 +1,40 @@
---
# ----------------------------------------------------------------------------
# Shared variables for the dev_blog production host.
# ----------------------------------------------------------------------------
# SSH access
ssh_port: 22
# Public web ports terminated by Caddy.
http_port: 80
https_port: 443
# Where Caddy + Coraza write logs on the host. This directory is bind-mounted
# into the Caddy container, and is the file path that fail2ban tails.
caddy_log_dir: /var/log/caddy
# Automated updates: only security errata are applied automatically.
dnf_automatic_apply_updates: true
dnf_automatic_upgrade_type: security
# fail2ban thresholds for the Caddy/Coraza jail.
f2b_findtime: 10m
f2b_bantime: 1h
f2b_maxretry: 5
# Quadlet units shipped from the repo.
quadlet_src_dir: "{{ playbook_dir }}/../quadlet"
quadlet_dest_dir: /etc/containers/systemd
# ----------------------------------------------------------------------------
# Container images on GHCR. Set `image_tag`, `ghcr_owner`, `ghcr_repo` from
# CI extra-vars (`-e image_tag=<sha> -e ghcr_owner=...`). Defaults below let
# the playbook syntax-check without those being provided.
# ----------------------------------------------------------------------------
ghcr_owner: "OWNER"
ghcr_repo: "dev_blog"
image_tag: "latest"
image_app: "ghcr.io/{{ ghcr_owner }}/{{ ghcr_repo }}/app:{{ image_tag }}"
image_caddy: "ghcr.io/{{ ghcr_owner }}/{{ ghcr_repo }}/caddy:{{ image_tag }}"
+9
View File
@@ -0,0 +1,9 @@
all:
children:
blog:
hosts:
dev-blog-prod:
# Public IP of the Vultr instance (set me).
ansible_host: 203.0.113.10
ansible_user: root
ansible_python_interpreter: /usr/bin/python3
@@ -0,0 +1,10 @@
[Unit]
Description=Reboot host if pending updates require it
Documentation=man:dnf-needs-restarting(1)
ConditionPathExists=!/run/nologin
[Service]
Type=oneshot
# `dnf needs-restarting -r` exits 1 when a reboot is required, 0 otherwise.
# Only reboot in that case.
ExecStart=/bin/sh -c '/usr/bin/dnf -q needs-restarting -r; if [ $? -eq 1 ]; then /usr/bin/logger -t auto-reboot "Rebooting: pending updates require it"; /usr/bin/systemctl reboot; fi'
@@ -0,0 +1,12 @@
[Unit]
Description=Daily 02:00 America/Chicago reboot check (if updates require it)
[Timer]
# 02:00 Central time, DST-aware.
OnCalendar=*-*-* 02:00:00 America/Chicago
Persistent=true
RandomizedDelaySec=5m
Unit=auto-reboot.service
[Install]
WantedBy=timers.target
+9
View File
@@ -0,0 +1,9 @@
---
- name: Restart dnf-automatic timer
ansible.builtin.systemd:
name: dnf-automatic.timer
state: restarted
- name: Reload systemd for auto-reboot
ansible.builtin.systemd:
daemon_reload: true
+73
View File
@@ -0,0 +1,73 @@
---
- name: Install EPEL release
ansible.builtin.dnf:
name: epel-release
state: present
- name: Install base packages
ansible.builtin.dnf:
name:
- dnf-automatic
- chrony
- curl
- tar
- vim-enhanced
- policycoreutils-python-utils
- logrotate
- rsync
state: present
- name: Ensure chrony is running
ansible.builtin.systemd:
name: chronyd
enabled: true
state: started
- name: Configure dnf-automatic for security updates
ansible.builtin.copy:
dest: /etc/dnf/automatic.conf
owner: root
group: root
mode: "0644"
content: |
# Managed by Ansible (roles/common).
[commands]
upgrade_type = {{ dnf_automatic_upgrade_type }}
random_sleep = 360
network_online_timeout = 60
download_updates = yes
apply_updates = {{ 'yes' if dnf_automatic_apply_updates else 'no' }}
reboot = never
reboot_command = "shutdown -r now"
[emitters]
emit_via = stdio
[base]
debuglevel = 1
notify: Restart dnf-automatic timer
- name: Enable dnf-automatic timer
ansible.builtin.systemd:
name: dnf-automatic.timer
enabled: true
state: started
- name: Install auto-reboot service + timer (02:00 America/Chicago)
ansible.builtin.copy:
src: "{{ item }}"
dest: "/etc/systemd/system/{{ item }}"
owner: root
group: root
mode: "0644"
loop:
- auto-reboot.service
- auto-reboot.timer
notify: Reload systemd for auto-reboot
- name: Enable auto-reboot timer
ansible.builtin.systemd:
name: auto-reboot.timer
enabled: true
state: started
daemon_reload: true
@@ -0,0 +1,19 @@
---
- name: Reload systemd and restart dev_blog
ansible.builtin.systemd:
daemon_reload: true
listen: Reload systemd and restart dev_blog
- name: Restart Caddy service
ansible.builtin.systemd:
name: dev-blog-caddy.service
state: restarted
listen: Reload systemd and restart dev_blog
failed_when: false
- name: Restart app service
ansible.builtin.systemd:
name: dev-blog-app.service
state: restarted
listen: Reload systemd and restart dev_blog
failed_when: false
@@ -0,0 +1,63 @@
---
# ----------------------------------------------------------------------------
# Container runtime + Podman Quadlet deployment for the dev_blog stack.
# ----------------------------------------------------------------------------
- name: Install Podman + container runtime dependencies
ansible.builtin.dnf:
name:
- podman
- containers-common
- crun
- netavark
- aardvark-dns
- slirp4netns
- fuse-overlayfs
- passt
- skopeo
- buildah
- container-selinux
state: present
- name: Ensure /etc/containers/systemd exists for Quadlet units
ansible.builtin.file:
path: "{{ quadlet_dest_dir }}"
state: directory
owner: root
group: root
mode: "0755"
- name: Sync static Quadlet units (network + volumes)
ansible.posix.synchronize:
src: "{{ quadlet_src_dir }}/"
dest: "{{ quadlet_dest_dir }}/"
delete: false
rsync_opts:
- "--exclude=*.container"
- "--chmod=F0644,D0755"
- "--chown=root:root"
notify: Reload systemd and restart dev_blog
- name: Render app Quadlet (Image= → GHCR ref)
ansible.builtin.copy:
dest: "{{ quadlet_dest_dir }}/dev-blog-app.container"
owner: root
group: root
mode: "0644"
content: "{{ lookup('file', quadlet_src_dir ~ '/dev-blog-app.container') | regex_replace('(?m)^Image=.*$', 'Image=' ~ image_app) }}"
notify: Reload systemd and restart dev_blog
- name: Render Caddy Quadlet (Image= → GHCR ref)
ansible.builtin.copy:
dest: "{{ quadlet_dest_dir }}/dev-blog-caddy.container"
owner: root
group: root
mode: "0644"
content: "{{ lookup('file', quadlet_src_dir ~ '/dev-blog-caddy.container') | regex_replace('(?m)^Image=.*$', 'Image=' ~ image_caddy) }}"
notify: Reload systemd and restart dev_blog
- name: Ensure podman.socket is available (for healthchecks / API)
ansible.builtin.systemd:
name: podman.socket
enabled: true
state: started
@@ -0,0 +1,26 @@
# Managed by Ansible (roles/fail2ban).
#
# Matches two distinct sources written by the Caddy + Coraza container:
#
# 1. Caddy JSON access log entries with HTTP 403 (Coraza denies with 403):
# {"level":"info","ts":...,"logger":"http.log.access",
# "request":{"remote_ip":"203.0.113.42",...},"status":403, ...}
#
# 2. Coraza serial audit-log "A" section data line (the line that follows
# the `--xxxx-A--` boundary):
# [29/Apr/2026:16:00:00 +0000] unique-id 203.0.113.42 54321 10.0.0.1 80
#
# Either pattern produces a hit on <HOST>.
[INCLUDES]
before = common.conf
[Definition]
failregex = ^.*"remote_ip":"<HOST>"[^\n]*"status":403[^\n]*$
^\[[^\]]+\] \S+ <HOST> \d+ \S+ \d+\s*$
ignoreregex =
datepattern = ^\[%%d/%%b/%%Y:%%H:%%M:%%S %%z\]
{^LN-BEG}
+5
View File
@@ -0,0 +1,5 @@
---
- name: Restart fail2ban
ansible.builtin.systemd:
name: fail2ban
state: restarted
+52
View File
@@ -0,0 +1,52 @@
---
- name: Install fail2ban (from EPEL)
ansible.builtin.dnf:
name: fail2ban
state: present
- name: Ensure Caddy log directory exists (host-side, bind-mounted into container)
ansible.builtin.file:
path: "{{ caddy_log_dir }}"
state: directory
owner: root
group: root
mode: "0755"
setype: container_file_t
- name: Ensure log files exist so fail2ban can start tailing them
ansible.builtin.file:
path: "{{ caddy_log_dir }}/{{ item }}"
state: touch
owner: root
group: root
mode: "0644"
setype: container_file_t
modification_time: preserve
access_time: preserve
loop:
- access.log
- coraza-audit.log
- name: Deploy Caddy/Coraza fail2ban filter
ansible.builtin.copy:
src: caddy-coraza.filter
dest: /etc/fail2ban/filter.d/caddy-coraza.conf
owner: root
group: root
mode: "0644"
notify: Restart fail2ban
- name: Deploy fail2ban jail.local
ansible.builtin.template:
src: jail.local.j2
dest: /etc/fail2ban/jail.local
owner: root
group: root
mode: "0644"
notify: Restart fail2ban
- name: Enable fail2ban
ansible.builtin.systemd:
name: fail2ban
enabled: true
state: started
@@ -0,0 +1,37 @@
# Managed by Ansible (roles/fail2ban).
[DEFAULT]
banaction = nftables-multiport
banaction_allports = nftables-allports
backend = auto
findtime = {{ f2b_findtime }}
bantime = {{ f2b_bantime }}
maxretry = {{ f2b_maxretry }}
# Add private ranges and your own admin IPs here.
ignoreip = 127.0.0.1/8 ::1
# ----------------------------------------------------------------------------
# SSH brute-force protection.
# ----------------------------------------------------------------------------
[sshd]
enabled = true
port = {{ ssh_port }}
# ----------------------------------------------------------------------------
# Caddy + Coraza WAF.
#
# Two log sources are watched:
# - access.log : every Caddy response. We ban on repeated 4xx / 403.
# - coraza-audit.log : Coraza serial audit log; any entry here means a rule
# fired in blocking mode, so a single hit is enough on top of the access
# log threshold.
# ----------------------------------------------------------------------------
[caddy-coraza]
enabled = true
port = http,https
filter = caddy-coraza
logpath = {{ caddy_log_dir }}/access.log
{{ caddy_log_dir }}/coraza-audit.log
maxretry = {{ f2b_maxretry }}
findtime = {{ f2b_findtime }}
bantime = {{ f2b_bantime }}
+5
View File
@@ -0,0 +1,5 @@
---
- name: Reload nftables
ansible.builtin.systemd:
name: nftables
state: reloaded
+29
View File
@@ -0,0 +1,29 @@
---
- name: Install nftables
ansible.builtin.dnf:
name: nftables
state: present
- name: Disable firewalld (we manage nftables directly)
ansible.builtin.systemd:
name: firewalld
enabled: false
state: stopped
masked: true
failed_when: false
- name: Deploy base nftables ruleset
ansible.builtin.template:
src: nftables.conf.j2
dest: /etc/sysconfig/nftables.conf
owner: root
group: root
mode: "0644"
validate: "/usr/sbin/nft -c -f %s"
notify: Reload nftables
- name: Enable nftables
ansible.builtin.systemd:
name: nftables
enabled: true
state: started
@@ -0,0 +1,54 @@
#!/usr/sbin/nft -f
# Managed by Ansible (roles/nftables).
#
# Base ruleset for the dev_blog host. fail2ban dynamically adds a separate
# table (`f2b-table`) with sets of banned addresses; we don't need to manage
# those here.
flush ruleset
table inet filter {
set blackhole_v4 {
type ipv4_addr
flags interval
}
set blackhole_v6 {
type ipv6_addr
flags interval
}
chain input {
type filter hook input priority filter; policy drop;
# Loopback + established traffic.
iif "lo" accept
ct state established,related accept
ct state invalid drop
# Manual blackhole sets (for ops use).
ip saddr @blackhole_v4 drop
ip6 saddr @blackhole_v6 drop
# ICMP / ICMPv6 (rate-limited).
ip protocol icmp limit rate 5/second accept
ip6 nexthdr icmpv6 accept
# Public services.
tcp dport { {{ ssh_port }} } ct state new limit rate 10/minute accept
tcp dport { {{ http_port }}, {{ https_port }} } accept
udp dport { {{ https_port }} } accept # HTTP/3 (QUIC)
# Log a small sample of everything else, then drop (policy drop).
limit rate 5/minute log prefix "nft_drop: "
}
chain forward {
type filter hook forward priority filter; policy drop;
# Container traffic is handled by podman/netavark in its own tables.
ct state established,related accept
}
chain output {
type filter hook output priority filter; policy accept;
}
}
+19
View File
@@ -0,0 +1,19 @@
---
- name: Restart dev_blog services
ansible.builtin.systemd:
daemon_reload: true
listen: Restart dev_blog services
- name: Restart Caddy
ansible.builtin.systemd:
name: dev-blog-caddy.service
state: restarted
listen: Restart dev_blog services
failed_when: false
- name: Restart app
ansible.builtin.systemd:
name: dev-blog-app.service
state: restarted
listen: Restart dev_blog services
failed_when: false
+78
View File
@@ -0,0 +1,78 @@
---
# ----------------------------------------------------------------------------
# GHCR registry login + image pull.
#
# Required extra-vars (typically set by CI):
# ghcr_pat - GitHub PAT with `read:packages` scope
# ghcr_user - GitHub username that owns the PAT
# image_tag - container tag to pull (e.g. the commit SHA)
# ----------------------------------------------------------------------------
- name: Assert GHCR variables are set
ansible.builtin.assert:
that:
- ghcr_pat | default('') | length > 0
- ghcr_user | default('') | length > 0
- image_tag | default('') | length > 0
fail_msg: "ghcr_pat, ghcr_user, and image_tag must be provided as extra-vars."
- name: Add GitHub CLI dnf repo
ansible.builtin.yum_repository:
name: gh-cli
description: GitHub CLI
baseurl: https://cli.github.com/packages/rpm
gpgcheck: true
gpgkey: https://cli.github.com/packages/rpm/gh-cli.repo.gpg.key
enabled: true
- name: Install gh CLI
ansible.builtin.dnf:
name: gh
state: present
- name: Authenticate gh with the provided PAT
ansible.builtin.shell:
cmd: |
set -euo pipefail
printf '%s' "$GHCR_PAT" | gh auth login --hostname github.com --with-token
gh auth status --hostname github.com
environment:
GHCR_PAT: "{{ ghcr_pat }}"
register: gh_login
changed_when: false
no_log: true
- name: Ensure podman auth dir exists
ansible.builtin.file:
path: /root/.config/containers
state: directory
owner: root
group: root
mode: "0700"
- name: Log podman in to GHCR (uses the same PAT)
ansible.builtin.shell:
cmd: |
set -euo pipefail
printf '%s' "$GHCR_PAT" | podman login ghcr.io \
--username "{{ ghcr_user }}" \
--password-stdin \
--authfile /root/.config/containers/auth.json
environment:
GHCR_PAT: "{{ ghcr_pat }}"
changed_when: false
no_log: true
- name: Pull the application image from GHCR
containers.podman.podman_image:
name: "{{ image_app }}"
auth_file: /root/.config/containers/auth.json
force: true
notify: Restart dev_blog services
- name: Pull the Caddy image from GHCR
containers.podman.podman_image:
name: "{{ image_caddy }}"
auth_file: /root/.config/containers/auth.json
force: true
notify: Restart dev_blog services
+20
View File
@@ -0,0 +1,20 @@
---
- name: Configure dev_blog production host (AlmaLinux 10)
hosts: blog
become: true
gather_facts: true
pre_tasks:
- name: Assert AlmaLinux 10
ansible.builtin.assert:
that:
- ansible_facts['distribution'] == 'AlmaLinux'
- ansible_facts['distribution_major_version'] == '10'
fail_msg: "This playbook targets AlmaLinux 10, found {{ ansible_facts['distribution'] }} {{ ansible_facts['distribution_version'] }}"
roles:
- role: common
- role: nftables
- role: fail2ban
- role: registry
- role: container_host
+10 -4
View File
@@ -47,9 +47,11 @@
} }
# ------------------------------------------------------------------ # ------------------------------------------------------------------
# Reverse-proxy to the Astro container on the internal network # Reverse-proxy to the Astro container. Caddy is on the host network
# namespace, so we connect over loopback to the port the app container
# publishes on 127.0.0.1 / [::1]:4321.
# ------------------------------------------------------------------ # ------------------------------------------------------------------
reverse_proxy app:4321 { reverse_proxy 127.0.0.1:4321 {
header_up X-Real-IP {remote_host} header_up X-Real-IP {remote_host}
header_up X-Forwarded-Proto {scheme} header_up X-Forwarded-Proto {scheme}
} }
@@ -64,7 +66,11 @@
} }
log { log {
output stdout output file /var/log/caddy/access.log {
format console roll_size 10MiB
roll_keep 5
roll_keep_for 168h
}
format json
} }
} }
+10
View File
@@ -21,3 +21,13 @@ SecRule REQUEST_URI "@beginsWith /_astro/" \
"id:1000,phase:1,pass,nolog,ctl:ruleEngine=Off" "id:1000,phase:1,pass,nolog,ctl:ruleEngine=Off"
SecRule REQUEST_URI "@beginsWith /fonts/" \ SecRule REQUEST_URI "@beginsWith /fonts/" \
"id:1001,phase:1,pass,nolog,ctl:ruleEngine=Off" "id:1001,phase:1,pass,nolog,ctl:ruleEngine=Off"
# ---------------------------------------------------------------------------
# Audit log — every blocked request gets an entry in serial format that
# fail2ban tails on the host (mounted at /var/log/caddy/coraza-audit.log).
# ---------------------------------------------------------------------------
SecAuditEngine RelevantOnly
SecAuditLogRelevantStatus "^(?:5|4(?!04))"
SecAuditLogParts ABIJDEFHZ
SecAuditLogType Serial
SecAuditLog /var/log/caddy/coraza-audit.log
+140
View File
@@ -0,0 +1,140 @@
# Caddy on host networking + loopback-published app
This doc captures *why* the production deployment puts Caddy in the host
network namespace and reaches the app over loopback, and *exactly what*
that requires in the Quadlet units and the Caddyfile.
> Note: this file lives under `docs/` at the repo root. It is **not**
> part of the Astro site (`src/pages/` + the `blog` content collection
> are what get built and served). Do not move it under `src/`.
## Why
With Podman's default port-publish path (`PublishPort=80:80` on a
user-defined bridge), inbound packets are NAT'd by netavark before they
hit the container. Caddy then sees the bridge gateway as the source
address instead of the real client. That breaks two things we care about:
- **fail2ban / nftables bans** — the `nftables-multiport` action would
populate the f2b set with the gateway's address, banning nothing
useful.
- **Coraza's audit log** — every "blocked" entry would record the same
internal address, so manual triage is impossible.
Switching Caddy to the host network namespace removes that NAT hop:
Caddy binds 80/443 directly on the host's interfaces and sees the real
client IP for both v4 and v6.
## What changed
### 1. `quadlet/dev-blog-caddy.container`
Caddy now lives in the host network namespace.
```diff
-Network=dev-blog.network
-NetworkAlias=caddy
-
-PublishPort=80:80
-PublishPort=443:443
-PublishPort=443:443/udp
-PublishPort=[::]:80:80
-PublishPort=[::]:443:443
-PublishPort=[::]:443:443/udp
+Network=host
```
`Network=host` is mutually exclusive with both `Network=dev-blog.network`
and any `PublishPort=` directives — Caddy binds 80/443 directly on the
host's interfaces, so the user-defined network and port-forwarding are
gone for this container. `CAP_NET_BIND_SERVICE` is still kept because
the container's non-root user still has to bind privileged ports.
### 2. `quadlet/dev-blog-app.container`
App stays on the user-defined network *and* publishes on host loopback
only.
```diff
-# Not exposed publicly – Caddy reverse-proxies in over the internal network.
-# PublishPort=4321:4321
+# Caddy now runs on the host network namespace, so it cannot reach the app
+# via podman DNS (`app:4321`). Publish the app port on the host's *loopback*
+# only -- it is reachable to Caddy on the host but not from outside.
+PublishPort=127.0.0.1:4321:4321
+PublishPort=[::1]:4321:4321
```
The app keeps `Network=dev-blog.network` (so it could still talk to
future sidecars on that bridge), but now also surfaces on
`127.0.0.1:4321` and `[::1]:4321`. nftables already accepts
`iif "lo" accept`, so loopback traffic is unfiltered.
### 3. `caddy/Caddyfile`
Reverse-proxy target updated.
```diff
-reverse_proxy app:4321 {
+reverse_proxy 127.0.0.1:4321 {
header_up X-Real-IP {remote_host}
header_up X-Forwarded-Proto {scheme}
}
```
`{remote_host}` is now the **real** client IP (v4 or v6) because Caddy
sees the connection unmodified, instead of the bridge gateway. That
value flows into the JSON access log (`"remote_ip"`) and Coraza's serial
audit log, which is exactly what the fail2ban filter keys on — so the
`nftables-multiport` action populates the right addresses in
`f2b-table` and bans actually work.
## Topology, before vs. after
**Before**
```
Internet ─┐
(host nft) ─► podman PublishPort NAT ─► dev-blog (bridge, dual-stack)
├─ caddy (real IP lost)
└─ app (Caddy → app:4321 via podman DNS)
```
**After**
```
Internet ─┐
(host nft) ─► caddy in host netns (real client v4/v6 preserved)
│ reverse_proxy
▼
127.0.0.1:4321 / [::1]:4321 (loopback publish)
│
dev-blog (bridge, dual-stack)
└─ app
```
## Things that did *not* need to change
- `dev-blog.network` — still a dual-stack bridge; the app still uses it
(and any future sidecar can join it).
- `nftables.conf.j2` — already accepted `tcp dport { 22, 80, 443 }` and
`udp dport { 443 }` in the `inet filter` table, plus
`iif "lo" accept`, so both Caddy's public listeners and the loopback
hop to the app are covered.
- fail2ban filter regexes, Coraza config, Ansible roles, Quadlet
`[Install]` lines.
## Operational heads-up
- `compose.yaml` (local-dev convenience) was deliberately left alone —
it still wires Caddy through the bridge with `PublishPort`, which is
fine for local iteration where preserving real client IPs doesn't
matter.
- If you ever add another container that Caddy needs to reach, follow
the same pattern: have *that* container publish to `127.0.0.1:<port>`
(and `[::1]:<port>` for v6) and reference it from the Caddyfile via
loopback.
- Because the app's loopback publish is bound to `127.0.0.1` / `[::1]`
only, it is **not** reachable from any other host on the network even
if nftables were ever flushed — the kernel itself drops non-loopback
traffic destined to those addresses.
+14
View File
@@ -0,0 +1,14 @@
# Local state & secrets
*.tfstate
*.tfstate.*
*.tfstate.backup
.terraform/
.terraform.lock.hcl.bak
crash.log
crash.*.log
# User-specific configs (may contain secrets)
backend.hcl
*.auto.tfvars
terraform.tfvars
!terraform.tfvars.example
+71
View File
@@ -0,0 +1,71 @@
# Infra (OpenTofu → Vultr)
Provisions a single Vultr instance for the dev_blog:
| Setting | Value |
| -------------- | ---------------------------------- |
| Region | `sea` (Seattle) |
| Plan | `vc2-1c-2gb` |
| OS | AlmaLinux 10 (looked up via `data "vultr_os"`) |
| Backups | Automated, daily |
| IPv6 | Enabled |
State is stored in **Cloudflare R2** via OpenTofu's S3-compatible backend.
## One-time setup
### 1. R2 bucket
In the Cloudflare dashboard:
1. Create an R2 bucket, e.g. `dev-blog-tfstate`.
2. Create an R2 API token (Account → R2 → Manage API tokens) with **Object Read & Write** scoped to that bucket. Note the access key ID + secret.
3. Note your Cloudflare **Account ID** (R2 endpoint host).
### 2. GitHub repository secrets
Add these in *Settings → Secrets and variables → Actions*:
| Secret | Value |
| ----------------------- | ------------------------------------ |
| `VULTR_API_KEY` | Vultr API key |
| `R2_ACCESS_KEY_ID` | R2 token access key ID |
| `R2_SECRET_ACCESS_KEY` | R2 token secret access key |
| `R2_ACCOUNT_ID` | Cloudflare account ID |
| `R2_BUCKET` | `dev-blog-tfstate` |
## Running locally
```sh
cd infra
cp backend.hcl.example backend.hcl # fill in bucket + endpoint
cp terraform.tfvars.example terraform.tfvars
export AWS_ACCESS_KEY_ID=<r2-key-id>
export AWS_SECRET_ACCESS_KEY=<r2-secret>
export TF_VAR_vultr_api_key=<vultr-key>
tofu init -backend-config=backend.hcl
tofu plan
tofu apply
```
## CI/CD
The workflow [`.github/workflows/infra.yml`](../.github/workflows/infra.yml) runs:
- **`pull_request`** touching `infra/**` → `tofu plan` (read-only).
- **`workflow_dispatch`** → choose `plan`, `apply`, or `destroy`.
Backend init uses `-backend-config` flags so the bucket and R2 endpoint are
injected from secrets at runtime — no account-specific values are committed.
## Notes
- The Vultr provider's `vultr_instance` resource enables daily backups via
`backups = "enabled"` and a `backups_schedule { type = "daily" }` block.
- AlmaLinux 10 is resolved by name through `data "vultr_os"` so we don't have
to hard-code an OS ID that may change. Adjust `os_name_filter` in
`variables.tf` if Vultr renames it.
- The R2 backend uses `region = "auto"` and skips AWS-specific validations,
which is the standard configuration for R2 as an OpenTofu/Terraform S3 backend.
+16
View File
@@ -0,0 +1,16 @@
# Example backend configuration for Cloudflare R2.
# Copy to `backend.hcl` (gitignored) and fill in your values, then run:
# tofu init -backend-config=backend.hcl
#
# In CI, these are passed as -backend-config=... flags from secrets instead.
bucket = "dev-blog-tfstate"
endpoints = {
s3 = "https://<ACCOUNT_ID>.r2.cloudflarestorage.com"
}
# AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY env vars supply credentials.
# Alternatively, uncomment:
# access_key = "<R2_ACCESS_KEY_ID>"
# secret_key = "<R2_SECRET_ACCESS_KEY>"
+51
View File
@@ -0,0 +1,51 @@
data "vultr_os" "alma" {
filter {
name = "name"
values = [var.os_name_filter]
}
}
resource "vultr_instance" "blog" {
region = var.region
plan = var.plan
os_id = data.vultr_os.alma.id
hostname = var.hostname
label = var.hostname
tags = var.tags
ssh_key_ids = var.ssh_key_ids
backups = "enabled"
backups_schedule {
type = "daily"
hour = var.backup_hour_utc
}
enable_ipv6 = true
ddos_protection = false
activation_email = false
}
# ---------------------------------------------------------------------------
# Static (Reserved) IPs
#
# Reserved IPs survive instance replacement, so DNS records stay valid even
# if `vultr_instance.blog` is destroyed and recreated.
#
# - v4 reservation is a single /32, so `subnet` is the address itself.
# - v6 reservation is a /64; `subnet` is the network prefix and the instance
# takes an address inside it (exposed as `vultr_instance.blog.v6_main_ip`).
# ---------------------------------------------------------------------------
resource "vultr_reserved_ip" "v4" {
region = var.region
ip_type = "v4"
label = "${var.hostname}-v4"
instance_id = vultr_instance.blog.id
}
resource "vultr_reserved_ip" "v6" {
region = var.region
ip_type = "v6"
label = "${var.hostname}-v6"
instance_id = vultr_instance.blog.id
}
+27
View File
@@ -0,0 +1,27 @@
output "instance_id" {
value = vultr_instance.blog.id
}
# Static IPv4 (Vultr Reserved IP /32).
output "main_ip" {
value = vultr_reserved_ip.v4.subnet
}
# Static IPv6 prefix (Vultr Reserved IP /64). Use a host address inside this
# subnet for AAAA records (the instance's primary v6 is `ipv6_address`).
output "ipv6_subnet" {
value = "${vultr_reserved_ip.v6.subnet}/${vultr_reserved_ip.v6.subnet_size}"
}
output "ipv6_address" {
value = vultr_instance.blog.v6_main_ip
}
output "default_password" {
value = vultr_instance.blog.default_password
sensitive = true
}
output "os" {
value = data.vultr_os.alma.name
}
+6
View File
@@ -0,0 +1,6 @@
vultr_api_key = "REPLACE_ME"
region = "sea"
plan = "vc2-1c-2gb"
os_name_filter = "AlmaLinux 10"
hostname = "dev-blog"
ssh_key_ids = []
+47
View File
@@ -0,0 +1,47 @@
variable "vultr_api_key" {
description = "Vultr API key. Provide via TF_VAR_vultr_api_key env var."
type = string
sensitive = true
}
variable "region" {
description = "Vultr region code."
type = string
default = "sea" # Seattle, WA
}
variable "plan" {
description = "Vultr instance plan."
type = string
default = "vc2-1c-2gb"
}
variable "os_name_filter" {
description = "Substring to match an OS name in the Vultr OS catalog."
type = string
default = "AlmaLinux 10"
}
variable "hostname" {
description = "Hostname / label for the instance."
type = string
default = "dev-blog"
}
variable "ssh_key_ids" {
description = "List of pre-existing Vultr SSH key IDs to inject."
type = list(string)
default = []
}
variable "backup_hour_utc" {
description = "Hour of day (UTC, 0-23) for the daily automated backup."
type = number
default = 8
}
variable "tags" {
description = "Tags to apply to the instance."
type = list(string)
default = ["dev_blog", "managed-by=opentofu"]
}
+32
View File
@@ -0,0 +1,32 @@
terraform {
required_version = ">= 1.8.0"
required_providers {
vultr = {
source = "vultr/vultr"
version = "~> 2.21"
}
}
# Cloudflare R2 is S3-compatible, so we use the s3 backend with a custom
# endpoint. Backend values that depend on secrets/account-specific data are
# supplied at init time via `-backend-config=backend.hcl` (see README).
backend "s3" {
key = "dev_blog/terraform.tfstate"
region = "auto"
# R2 quirks: skip AWS-specific validations and use path-style URLs.
skip_credentials_validation = true
skip_metadata_api_check = true
skip_region_validation = true
skip_requesting_account_id = true
skip_s3_checksum = true
use_path_style = true
}
}
provider "vultr" {
api_key = var.vultr_api_key
rate_limit = 700
retry_limit = 3
}
+5 -2
View File
@@ -23,8 +23,11 @@ ReadOnly=true
DropCapability=ALL DropCapability=ALL
Tmpfs=/tmp:rw,size=64m,mode=1777 Tmpfs=/tmp:rw,size=64m,mode=1777
# Not exposed publicly – Caddy reverse-proxies in over the internal network. # Caddy now runs on the host network namespace, so it cannot reach the app
# PublishPort=4321:4321 # via podman DNS (`app:4321`). Publish the app port on the host's *loopback*
# only -- it is reachable to Caddy on the host but not from outside.
PublishPort=127.0.0.1:4321:4321
PublishPort=[::1]:4321:4321
[Service] [Service]
Restart=on-failure Restart=on-failure
+9 -6
View File
@@ -9,12 +9,11 @@ ContainerName=dev-blog-caddy
# Build with: podman build -t localhost/dev-blog-caddy:latest ./caddy # Build with: podman build -t localhost/dev-blog-caddy:latest ./caddy
Image=localhost/dev-blog-caddy:latest Image=localhost/dev-blog-caddy:latest
Network=dev-blog.network # Caddy uses the *host* network namespace so it sees real client IP addresses
NetworkAlias=caddy # (both v4 and v6). Required for fail2ban / Coraza to act on actual sources
# instead of the bridge gateway. With Network=host, PublishPort= is a no-op
PublishPort=80:80 # and is intentionally omitted -- Caddy binds 80/443 directly on the host.
PublishPort=443:443 Network=host
PublishPort=443:443/udp
# --- Configuration --- # --- Configuration ---
Environment=SITE_ADDRESS=https://example.com Environment=SITE_ADDRESS=https://example.com
@@ -33,6 +32,10 @@ Secret=gcp-dns-sa,type=mount,target=gcp-dns.json,mode=0400
Volume=caddy-data.volume:/data Volume=caddy-data.volume:/data
Volume=caddy-config.volume:/config Volume=caddy-config.volume:/config
# Bind-mount the host log directory so fail2ban can tail Caddy access logs
# and Coraza audit logs. The :Z suffix asks Podman to relabel for SELinux.
Volume=/var/log/caddy:/var/log/caddy:Z
# Hardening # Hardening
NoNewPrivileges=true NoNewPrivileges=true
DropCapability=ALL DropCapability=ALL
+3
View File
@@ -5,6 +5,9 @@ Description=Internal network for the dev-blog stack
NetworkName=dev-blog NetworkName=dev-blog
Driver=bridge Driver=bridge
DisableDNS=false DisableDNS=false
IPv6=true
Subnet=10.89.0.0/24
Subnet=fd00:dead:beef::/64
[Install] [Install]
WantedBy=multi-user.target default.target WantedBy=multi-user.target default.target