diff --git a/.gitignore b/.gitignore index 6879282..57b7194 100644 --- a/.gitignore +++ b/.gitignore @@ -25,6 +25,10 @@ ansible/.facts_cache/ *.swp *~ +# Local notes — untracked scratch files (review punch lists, TODO scratchpads) +*.local.md +REVIEW_DEFERRED.md + # OS .DS_Store Thumbs.db diff --git a/ansible/group_vars/all.yml b/ansible/group_vars/all.yml index 04f3b22..0f93f32 100644 --- a/ansible/group_vars/all.yml +++ b/ansible/group_vars/all.yml @@ -22,7 +22,9 @@ epel_packages: - msmtp-sendmail - wireguard-tools - fail2ban - - fail2ban-firewalld # for the nftables backend bits + # Deliberately NOT fail2ban-firewalld: it ships /etc/fail2ban/jail.d/00- + # firewalld.conf which overrides banaction to firewallcmd-rich-rules. + # We use nftables-multiport directly (set in jail.local). # Google Cloud Secret Manager — where Ansible pulls secrets from. The # controller (operator laptop or Cloud Build runner) must be authed to GCP @@ -30,3 +32,9 @@ epel_packages: # stacks live in different projects. gcp_project: REPLACE-with-gcp-project-id secret_prefix: vaultwarden- + +# Extra CIDRs allowed to reach SSH at the host nftables layer, on top of the +# WG subnet. Pass via --extra-vars from the controller (e.g. Cloud Build sets +# its outbound IP here for the duration of the deploy, matching the Pulumi- +# managed Cloud Firewall hole). Empty list = WG-only. +nftables_extra_ssh_cidrs: [] diff --git a/ansible/inventory/disposable.yml.example b/ansible/inventory/disposable.yml.example index 5c847fd..f1815fe 100644 --- a/ansible/inventory/disposable.yml.example +++ b/ansible/inventory/disposable.yml.example @@ -40,4 +40,5 @@ all: dnf_reboot_window: "04:00" dnf_reboot_dow: "Tue" - # Healthchecks ping URLs come from SOPS secrets, not from here. + # Healthchecks ping URLs come from the vaultwarden-healthchecks secret + # in Secret Manager, not from here. diff --git a/ansible/inventory/production.yml.example b/ansible/inventory/production.yml.example index 103c290..817de69 100644 --- a/ansible/inventory/production.yml.example +++ b/ansible/inventory/production.yml.example @@ -31,3 +31,6 @@ all: dnf_reboot_window: "04:00" dnf_reboot_dow: "Tue" + + # Healthchecks ping URLs come from the vaultwarden-healthchecks secret + # in Secret Manager, not from here. diff --git a/ansible/playbook.yml b/ansible/playbook.yml index f2d7740..742e1bf 100644 --- a/ansible/playbook.yml +++ b/ansible/playbook.yml @@ -2,10 +2,13 @@ # Site playbook. Apply with: # # ansible-galaxy install -r requirements.yml -# ansible-playbook -i inventory/disposable.yml playbook.yml +# ansible-playbook -i inventory/disposable.yml playbook.yml \ +# --extra-vars "caddy_image=$(cd ../pulumi && pulumi stack output imageRepo):latest" # -# Operator's age key must be readable at $SOPS_AGE_KEY_FILE -# (default ~/.config/sops/age/keys.txt) so the secrets role can decrypt. +# The controller (operator laptop OR Cloud Build) must be authed to GCP via +# ADC — `gcloud auth application-default login` locally, or the Cloud Build +# SA identity when running under cloudbuild.yaml. The secrets role uses +# `gcloud secrets versions access` to pull from Google Cloud Secret Manager. - name: Vaultwarden host configuration hosts: all diff --git a/ansible/roles/base/tasks/main.yml b/ansible/roles/base/tasks/main.yml index bcf5b58..4f0c17e 100644 --- a/ansible/roles/base/tasks/main.yml +++ b/ansible/roles/base/tasks/main.yml @@ -25,6 +25,10 @@ # Hetzner cloud-init mounts the attached volume at /mnt/HC_Volume_. We # don't care about the ugly path; we bind-mount it to /data so every other # role + every quadlet sees the same stable path. +# +# Retry: a Pulumi-fresh box may still be running cloud-init when the +# playbook reaches this task. 12 × 10s = 2 min covers the typical mkfs + +# mount window on a fresh ext4 volume. - name: Discover Hetzner volume mount ansible.builtin.shell: cmd: | @@ -32,7 +36,9 @@ findmnt -ln -o TARGET /mnt/HC_Volume_* | head -n1 register: hc_volume_mount changed_when: false - failed_when: hc_volume_mount.stdout == "" + retries: 12 + delay: 10 + until: hc_volume_mount.stdout != "" - name: Ensure /data exists ansible.builtin.file: diff --git a/ansible/roles/dnf_automatic/tasks/main.yml b/ansible/roles/dnf_automatic/tasks/main.yml index 46ff43f..8a498fb 100644 --- a/ansible/roles/dnf_automatic/tasks/main.yml +++ b/ansible/roles/dnf_automatic/tasks/main.yml @@ -12,6 +12,17 @@ group: root mode: "0644" +- name: dnf-automatic systemd drop-in dirs + ansible.builtin.file: + path: "{{ item }}" + state: directory + owner: root + group: root + mode: "0755" + loop: + - /etc/systemd/system/dnf-automatic-install.timer.d + - /etc/systemd/system/dnf-automatic-install.service.d + - name: Pin dnf-automatic-install.timer to maintenance window ansible.builtin.copy: dest: /etc/systemd/system/dnf-automatic-install.timer.d/schedule.conf diff --git a/ansible/roles/fail2ban/tasks/main.yml b/ansible/roles/fail2ban/tasks/main.yml index bf9b94e..063a62e 100644 --- a/ansible/roles/fail2ban/tasks/main.yml +++ b/ansible/roles/fail2ban/tasks/main.yml @@ -36,6 +36,14 @@ enabled: true state: started +- name: fail2ban systemd drop-in dir + ansible.builtin.file: + path: /etc/systemd/system/fail2ban.service.d + state: directory + owner: root + group: root + mode: "0755" + - name: OnFailure email for fail2ban ansible.builtin.copy: dest: /etc/systemd/system/fail2ban.service.d/onfailure.conf diff --git a/ansible/roles/nftables/templates/main.nft.j2 b/ansible/roles/nftables/templates/main.nft.j2 index 26bd9fe..f8ce165 100644 --- a/ansible/roles/nftables/templates/main.nft.j2 +++ b/ansible/roles/nftables/templates/main.nft.j2 @@ -26,9 +26,12 @@ table inet filter { # WireGuard — public. udp dport {{ wg_port }} accept - # SSH — WG subnet only. Public SSH is closed at this layer regardless - # of what Cloud Firewall allows (defence-in-depth). + # SSH — WG subnet always allowed. ip saddr {{ wg_subnet }} tcp dport 22 accept +{% for cidr in nftables_extra_ssh_cidrs | default([]) %} + # SSH from {{ cidr }} (extra, e.g. Cloud Build's outbound IP for the deploy). + ip saddr {{ cidr }} tcp dport 22 accept +{% endfor %} # Fail2Ban inserts a jump to f2b- chains at runtime; nothing to # declare here. diff --git a/ansible/roles/quadlets/tasks/main.yml b/ansible/roles/quadlets/tasks/main.yml index 213f675..53c5162 100644 --- a/ansible/roles/quadlets/tasks/main.yml +++ b/ansible/roles/quadlets/tasks/main.yml @@ -34,26 +34,38 @@ delete: true rsync_opts: - "--chown={{ vault_user }}:{{ vault_user }}" - - "--chmod=F640,D750" + - "--chmod=F644,D755" - "--exclude=*.j2" notify: reload user systemd +# 0644 here too — quadlet files are read by Podman's quadlet generator +# (running as the rootless user's systemd-user-generator), which crosses +# the same namespace boundary as the container mounts above. 0640 risks +# the generator not seeing the unit on first activation. - name: Template caddy.container with image from Pulumi ansible.builtin.template: src: "{{ playbook_dir }}/../quadlets/caddy.container.j2" dest: "/var/lib/{{ vault_user }}/.config/containers/systemd/caddy.container" owner: "{{ vault_user }}" group: "{{ vault_user }}" - mode: "0640" + mode: "0644" notify: reload user systemd +# Mode 0644 (not 0640) is deliberate. These files hold non-secret config +# (VAULT_FQDN, WG_SUBNET, DOMAIN — all public). Caddy runs in a distroless +# `:nonroot` container as uid 65532; with rootless Podman's default user +# namespace mapping, the container's nonroot uid is *outside* the host +# vault_user's subuid range, so a 0640 file owned by vault_user appears as +# 0:0 inside the container and is unreadable to Caddy. 0644 makes it +# readable as "other". On a single-purpose host, the wider mode has no +# practical exposure — no other UIDs are running. - name: Sync Caddyfile ansible.builtin.copy: src: "{{ playbook_dir }}/../caddy/Caddyfile" dest: "/var/lib/{{ vault_user }}/caddy/Caddyfile" owner: "{{ vault_user }}" group: "{{ vault_user }}" - mode: "0640" + mode: "0644" notify: reload user systemd - name: Render Caddyfile env vars @@ -64,7 +76,7 @@ WG_SUBNET={{ wg_subnet }} owner: "{{ vault_user }}" group: "{{ vault_user }}" - mode: "0640" + mode: "0644" notify: reload user systemd - name: Render Vaultwarden env vars @@ -74,7 +86,7 @@ DOMAIN=https://{{ vault_fqdn }} owner: "{{ vault_user }}" group: "{{ vault_user }}" - mode: "0640" + mode: "0644" notify: reload user systemd - name: Create caddy data + log directories @@ -87,3 +99,30 @@ loop: - data - logs + +# Coraza writes /var/log/caddy/coraza-audit.log inside the container, which +# bind-mounts out to /var/lib//caddy/logs/. SecAuditLog doesn't rotate +# itself; logrotate handles it on the host. copytruncate is required — +# Coraza holds the file open and we can't SIGHUP its writer. +- name: Install logrotate + ansible.builtin.package: + name: logrotate + state: present + +- name: Coraza audit logrotate config + ansible.builtin.copy: + dest: /etc/logrotate.d/coraza-audit + content: | + /var/lib/{{ vault_user }}/caddy/logs/coraza-audit.log { + daily + rotate 14 + compress + delaycompress + missingok + notifempty + copytruncate + su {{ vault_user }} {{ vault_user }} + } + owner: root + group: root + mode: "0644" diff --git a/ansible/roles/secrets/tasks/main.yml b/ansible/roles/secrets/tasks/main.yml index 224284c..c72e528 100644 --- a/ansible/roles/secrets/tasks/main.yml +++ b/ansible/roles/secrets/tasks/main.yml @@ -1,13 +1,13 @@ --- -# Materialize SOPS-encrypted material onto the host: +# Materialize secrets from Google Cloud Secret Manager onto the host: # # * Single-value secrets (ADMIN_TOKEN, GCP SA JSON) → Podman secrets, # consumed by quadlets via `Secret=name,type=...`. -# * Multi-value env files (restic) → mode-0600 env files owned by the -# service user, consumed by systemd EnvironmentFile=. +# * Multi-value env files (restic, healthchecks) → mode-0640 env files +# owned by root:vault_user, consumed by systemd EnvironmentFile=. # -# All decryption happens on the controller (operator's laptop). The host -# never holds the SOPS age key. +# All Secret Manager access happens on the controller (operator laptop or +# Cloud Build) via gcloud + ADC. The host has no GCP identity. - name: Stage directory for systemd env files ansible.builtin.file: @@ -25,6 +25,20 @@ | from_yaml }} no_log: true +# Fail fast on misconfigured Secret Manager content. An empty or malformed +# gcp_sa_dns_json silently breaks Caddy DNS-01; admin_token_argon2 can be +# empty (disables /admin, the documented default). +- name: Validate Vaultwarden secret contents + ansible.builtin.assert: + that: + - vw_secrets.gcp_sa_dns_json is defined + - vw_secrets.gcp_sa_dns_json | length > 0 + - vw_secrets.gcp_sa_dns_json | trim is match('^\\s*\\{') + fail_msg: >- + vw_secrets.gcp_sa_dns_json is missing, empty, or not JSON. Update the + vaultwarden-vaultwarden secret in Secret Manager with the full + service-account JSON (under gcp_sa_dns_json) and re-run. + - name: Load restic backup env from Secret Manager ansible.builtin.set_fact: restic_secrets: >- @@ -33,6 +47,15 @@ | from_yaml }} no_log: true +- name: Validate restic secret contents + ansible.builtin.assert: + that: + - restic_secrets.restic_repository | default('') | length > 0 + - restic_secrets.restic_password | default('') | length > 0 + - restic_secrets.r2_access_key_id | default('') | length > 0 + - restic_secrets.r2_secret_access_key | default('') | length > 0 + fail_msg: vaultwarden-restic is missing one of the required keys (restic_repository, restic_password, r2_access_key_id, r2_secret_access_key). + - name: Load Healthchecks URLs from Secret Manager ansible.builtin.set_fact: hc_secrets: >- @@ -41,6 +64,13 @@ | from_yaml }} no_log: true +- name: Validate Healthchecks secret contents + ansible.builtin.assert: + that: + - hc_secrets.hc_backup_url | default('') | length > 0 + - hc_secrets.hc_restic_check_url | default('') | length > 0 + fail_msg: vaultwarden-healthchecks is missing hc_backup_url or hc_restic_check_url. + - name: Write restic env file for the backup service ansible.builtin.copy: dest: /etc/vaultwarden/restic.env diff --git a/ansible/roles/wireguard/tasks/main.yml b/ansible/roles/wireguard/tasks/main.yml index aa2f1f8..4495ef7 100644 --- a/ansible/roles/wireguard/tasks/main.yml +++ b/ansible/roles/wireguard/tasks/main.yml @@ -54,6 +54,14 @@ state: started daemon_reload: true +- name: wg-quick systemd drop-in dir + ansible.builtin.file: + path: "/etc/systemd/system/wg-quick@{{ wg_interface }}.service.d" + state: directory + owner: root + group: root + mode: "0755" + - name: OnFailure email for wg-quick ansible.builtin.copy: dest: "/etc/systemd/system/wg-quick@{{ wg_interface }}.service.d/onfailure.conf" diff --git a/ansible/roles/wireguard/templates/wg0.conf.j2 b/ansible/roles/wireguard/templates/wg0.conf.j2 index 78e0f29..af70128 100644 --- a/ansible/roles/wireguard/templates/wg0.conf.j2 +++ b/ansible/roles/wireguard/templates/wg0.conf.j2 @@ -1,4 +1,5 @@ -# Managed by Ansible. Private key from SOPS; do not edit by hand. +# Managed by Ansible. Private key fetched from Secret Manager at deploy +# time (vaultwarden-wireguard); do not edit by hand. [Interface] Address = {{ wg_server_address }} ListenPort = {{ wg_port }} diff --git a/backup/vaultwarden-backup.sh b/backup/vaultwarden-backup.sh index 03ab4bc..996e4e4 100644 --- a/backup/vaultwarden-backup.sh +++ b/backup/vaultwarden-backup.sh @@ -38,15 +38,24 @@ sqlite3 "$DB" -cmd ".timeout 30000" ".backup $DB_SNAPSHOT" # Quick integrity check on the snapshot itself before we ship it. A corrupt # snapshot uploaded to R2 is a successful backup of garbage, which is worse # than a failed backup. -sqlite3 "$DB_SNAPSHOT" "PRAGMA integrity_check;" | grep -qx "ok" +# +# integrity_check prints exactly "ok" on a clean DB; on errors it prints one +# row per problem. Full-string equality is the correct guard — `grep -qx ok` +# would pass on a corrupt DB that happened to emit an "ok" line among +# errors. +integrity_result=$(sqlite3 "$DB_SNAPSHOT" "PRAGMA integrity_check;") +if [ "$integrity_result" != "ok" ]; then + printf 'snapshot integrity_check failed:\n%s\n' "$integrity_result" >&2 + exit 1 +fi restic backup \ --tag vaultwarden \ --tag daily \ --host "$TAG_HOST" \ --exclude "$DB" \ - --exclude "$DB.wal" \ - --exclude "$DB.shm" \ + --exclude "${DB}-wal" \ + --exclude "${DB}-shm" \ --exclude "$DATA_DIR/vaultwarden.log" \ --exclude "$DATA_DIR/vaultwarden.log.*" \ "$DATA_DIR" diff --git a/cloudbuild.yaml b/cloudbuild.yaml index df9645b..49413f0 100644 --- a/cloudbuild.yaml +++ b/cloudbuild.yaml @@ -135,13 +135,17 @@ steps: # Pulumi config flip, runs the playbook, then re-runs pulumi to close. # Both flips are in one bash step with `trap` so the close always runs. - id: ansible-deploy - name: docker.io/library/ubuntu:24.04 + # cloud-sdk:slim ships gcloud + python, which Ansible's gcloud-pipe + # secret lookups need. Plain ubuntu:24.04 lacked gcloud entirely. + name: gcr.io/google.com/cloudsdktool/cloud-sdk:slim entrypoint: bash secretEnv: [SSH_PRIVATE_KEY, HCLOUD_TOKEN] args: - -ceu - | - set -x + # No `set -x` here — this step touches SSH_PRIVATE_KEY and + # HCLOUD_TOKEN. Trace would dump the expanded `echo "$SSH_PRIVATE_KEY" + # > /root/.ssh/id_ed25519` into Cloud Build's persistent log. # Cleanup trap: close the firewall hole no matter how this step exits. close_hole() { @@ -155,32 +159,32 @@ steps: } trap close_hole EXIT - # Install ansible + pulumi (pulumi is needed inside this step too). + # Install ansible + pulumi on top of cloud-sdk:slim. export DEBIAN_FRONTEND=noninteractive apt-get update -qq - apt-get install -y -qq ansible-core curl openssh-client jq python3-pip - pip3 install --break-system-packages google-cloud-secret-manager + apt-get install -y -qq ansible-core openssh-client jq curl -fsSL https://get.pulumi.com | sh export PATH="$$HOME/.pulumi/bin:$$PATH" ansible-galaxy collection install -r /workspace/ansible/requirements.yml -p /workspace/ansible/.collections - # Open the firewall to this build's outbound IP. + # Discover this build's outbound IP. Used twice: the Pulumi-managed + # Cloud Firewall rule and the host nftables extra-ssh allow. MY_IP="$(curl -fsS https://ifconfig.me)/32" + + # Open the Cloud Firewall to MY_IP. pushd /workspace/pulumi >/dev/null pulumi login "${_PULUMI_STATE_BUCKET}" pulumi stack select "${_PULUMI_STACK}" pulumi config set vaultwarden:bootstrapSshCidr "$$MY_IP" pulumi up --yes --skip-preview - # Read pulumi outputs for the inventory. - HOST_IP="$$(pulumi stack output ipv4 --show-secrets)" - FQDN="$$(pulumi stack output fqdn --show-secrets)" + HOST_IP="$$(pulumi stack output ipv4)" popd >/dev/null # SSH key for Ansible. mkdir -p /root/.ssh - echo "$$SSH_PRIVATE_KEY" > /root/.ssh/id_ed25519 + printf '%s' "$$SSH_PRIVATE_KEY" > /root/.ssh/id_ed25519 chmod 600 /root/.ssh/id_ed25519 ssh-keyscan -H "$$HOST_IP" >> /root/.ssh/known_hosts @@ -196,9 +200,14 @@ steps: sed -i "s|REPLACE_WITH_IPV4_FROM_PULUMI|$$HOST_IP|" inventory/${_PULUMI_STACK}.yml IMAGE_REPO="$$(cat /workspace/.image-repo)" + # MY_IP gets a matching host-nftables accept rule via extra_ssh_cidrs; + # close_hole() removes the Cloud Firewall side after Ansible exits but + # the nftables rule will be removed on the next deploy when the var + # is no longer set (i.e. operator-local deploys revert to WG-only). ANSIBLE_HOST_KEY_CHECKING=False \ ansible-playbook -i inventory/${_PULUMI_STACK}.yml playbook.yml \ - --extra-vars "caddy_image=$$IMAGE_REPO:latest" + --extra-vars "caddy_image=$$IMAGE_REPO:latest" \ + --extra-vars "{\"nftables_extra_ssh_cidrs\":[\"$$MY_IP\"]}" options: machineType: E2_HIGHCPU_8 diff --git a/pulumi/Pulumi.disposable.yaml b/pulumi/Pulumi.disposable.yaml index d726205..0cf9558 100644 --- a/pulumi/Pulumi.disposable.yaml +++ b/pulumi/Pulumi.disposable.yaml @@ -1,10 +1,14 @@ # Example stack config for the disposable shakeout instance. # -# Copy this to your own non-tracked stack file or run `pulumi stack init disposable` -# and `pulumi config set` for each value below. +# This file is a TEMPLATE. Do not `pulumi up` against it as-is — every +# value with angle brackets must be replaced first. The recommended flow +# is `pulumi stack init disposable && pulumi config set …` for each line, +# rather than editing this file directly (which would risk committing +# secrets or environment-specific values). # -# Do not commit real values for fields that point at production DNS or your -# personal bootstrap IP — those are environment-specific. +# `vaultwarden:gcpArRegion` defaults to europe-west3 to match a typical +# Hetzner nbg1/fsn1 footprint — cross-region image pulls otherwise pay +# Atlantic latency on every `podman auto-update` tick. config: vaultwarden:location: nbg1 @@ -12,15 +16,18 @@ config: vaultwarden:image: alma-10 vaultwarden:hostname: vault-disposable vaultwarden:volumeSize: "20" - vaultwarden:sshPubkey: ssh-ed25519 AAAA...REPLACE... + vaultwarden:sshPubkey: "ssh-ed25519 " vaultwarden:wgPort: "51820" - # On first `pulumi up`, set this to your current public IP /32 so Ansible can - # SSH in. After WG is verified working from the host, set this to "" and - # re-run `pulumi up` to close public 22. - vaultwarden:bootstrapSshCidr: 203.0.113.1/32 - vaultwarden:gcpProject: REPLACE-with-gcp-project-id - vaultwarden:dnsManagedZone: REPLACE-with-zone-resource-name - vaultwarden:dnsRecordName: vault-disposable.example.com. + # On first `pulumi up`, set this to your current public IP /32 so Ansible + # can SSH in. After WG is verified working from the host, set this to "" + # and re-run `pulumi up` to close public 22. /0..7 are rejected. + vaultwarden:bootstrapSshCidr: "/32" + vaultwarden:gcpProject: "" + vaultwarden:dnsManagedZone: "" + vaultwarden:dnsRecordName: "" vaultwarden:dnsTtl: "300" - vaultwarden:gcpArRegion: us-central1 + vaultwarden:gcpArRegion: europe-west3 vaultwarden:arImageName: vaultwarden-caddy + # Leave false on disposable so destroy/up cycles work. Set true on + # production to refuse a stray `pulumi destroy` from wiping the volume. + vaultwarden:deleteProtection: "false" diff --git a/pulumi/Pulumi.yaml b/pulumi/Pulumi.yaml index edf568f..b65c602 100644 --- a/pulumi/Pulumi.yaml +++ b/pulumi/Pulumi.yaml @@ -47,3 +47,9 @@ config: vaultwarden:arImageName: description: Image name within the AR repo default: vaultwarden-caddy + vaultwarden:deleteProtection: + description: | + Enable Hetzner DeleteProtection on the data volume + Pulumi + protect-on-destroy. Required for production; leave false on disposable + so destroy/up cycles work. + default: false diff --git a/pulumi/main.go b/pulumi/main.go index 00c9277..90188bf 100644 --- a/pulumi/main.go +++ b/pulumi/main.go @@ -17,6 +17,9 @@ package main import ( "fmt" + "regexp" + "strconv" + "strings" "github.com/pulumi/pulumi-gcp/sdk/v9/go/gcp/artifactregistry" "github.com/pulumi/pulumi-gcp/sdk/v9/go/gcp/dns" @@ -25,6 +28,14 @@ import ( "github.com/pulumi/pulumi/sdk/v3/go/pulumi/config" ) +// sshPubkeyPattern accepts the standard OpenSSH single-line public-key +// formats: ssh-ed25519 / ssh-rsa / ssh-dss / ecdsa-sha2-* / sk-* (FIDO), +// followed by a base64 blob and an optional comment. Rejects multiline +// input — a stray newline would break cloud-init YAML rendering. +var sshPubkeyPattern = regexp.MustCompile( + `^(ssh-(?:ed25519|rsa|dss)|ecdsa-sha2-\S+|sk-(?:ssh-ed25519|ecdsa-sha2-nistp256)@openssh\.com) [A-Za-z0-9+/=]+( [^\r\n]*)?$`, +) + func main() { pulumi.Run(func(ctx *pulumi.Context) error { cfg := config.New(ctx, "vaultwarden") @@ -44,11 +55,17 @@ func main() { volumeSize = 20 } sshPubkey := cfg.Require("sshPubkey") + if !sshPubkeyPattern.MatchString(sshPubkey) { + return fmt.Errorf("sshPubkey doesn't look like a single-line OpenSSH public key (got %q)", sshPubkey) + } wgPort := cfg.Get("wgPort") if wgPort == "" { wgPort = "51820" } bootstrapSshCidr := cfg.Get("bootstrapSshCidr") + if err := validateBootstrapSshCidr(bootstrapSshCidr); err != nil { + return err + } gcpProject := cfg.Require("gcpProject") dnsManagedZone := cfg.Require("dnsManagedZone") @@ -62,6 +79,12 @@ func main() { if arImageName == "" { arImageName = "vaultwarden-caddy" } + // DeleteProtection on the data volume defaults to off (so disposable + // stacks can `pulumi destroy` cleanly) but should be enabled in + // production via `pulumi config set vaultwarden:deleteProtection true`. + // With it on, `pulumi destroy` will fail at the volume rather than + // silently dropping the live vault DB. + deleteProtection := cfg.GetBool("deleteProtection") // SSH key uploaded to Hetzner for first-boot user. sshKey, err := hcloud.NewSshKey(ctx, hostname+"-bootstrap", &hcloud.SshKeyArgs{ @@ -167,16 +190,17 @@ disable_root: true // Data volume — decoupled from instance lifecycle so the box can be // rebuilt without losing /data. _, err = hcloud.NewVolume(ctx, hostname+"-data", &hcloud.VolumeArgs{ - Name: pulumi.String(hostname + "-data"), - Size: pulumi.Int(volumeSize), - ServerId: server.ID().ApplyT(parsePulumiID).(pulumi.IntOutput), - Format: pulumi.String("ext4"), - Automount: pulumi.Bool(true), + Name: pulumi.String(hostname + "-data"), + Size: pulumi.Int(volumeSize), + ServerId: server.ID().ApplyT(parsePulumiID).(pulumi.IntOutput), + Format: pulumi.String("ext4"), + Automount: pulumi.Bool(true), + DeleteProtection: pulumi.Bool(deleteProtection), Labels: pulumi.StringMap{ "managed-by": pulumi.String("pulumi"), "stack": pulumi.String(ctx.Stack()), }, - }) + }, pulumi.Protect(deleteProtection)) if err != nil { return fmt.Errorf("create volume: %w", err) } @@ -270,3 +294,32 @@ func parsePulumiID(id pulumi.ID) (int, error) { } return n, nil } + +// validateBootstrapSshCidr rejects values that would obviously open public +// SSH wider than intended. Empty is fine (rule is omitted). Otherwise must +// be a single IPv4 /N where N >= 8 — a typo'd /0 or /4 isn't a real CIDR +// for a single operator endpoint. +func validateBootstrapSshCidr(v string) error { + if v == "" { + return nil + } + parts := strings.SplitN(v, "/", 2) + if len(parts) != 2 { + return fmt.Errorf("bootstrapSshCidr %q: must be IPv4/N", v) + } + octets := strings.Split(parts[0], ".") + if len(octets) != 4 { + return fmt.Errorf("bootstrapSshCidr %q: not IPv4", v) + } + for _, o := range octets { + n, err := strconv.Atoi(o) + if err != nil || n < 0 || n > 255 { + return fmt.Errorf("bootstrapSshCidr %q: bad octet %q", v, o) + } + } + mask, err := strconv.Atoi(parts[1]) + if err != nil || mask < 8 || mask > 32 { + return fmt.Errorf("bootstrapSshCidr %q: mask /%s rejected (require /8..32)", v, parts[1]) + } + return nil +} diff --git a/quadlets/caddy.container.j2 b/quadlets/caddy.container.j2 index f183c8c..b5c3ca1 100644 --- a/quadlets/caddy.container.j2 +++ b/quadlets/caddy.container.j2 @@ -1,8 +1,10 @@ [Unit] Description=Caddy reverse proxy (custom build with googleclouddns + Coraza) -Wants=network-online.target +Wants=network-online.target vaultwarden.service After=network-online.target vaultwarden.service -Requires=vaultwarden.service +# Wants= (not Requires=) on vaultwarden.service: a Vaultwarden restart +# returns a brief 502 from reverse_proxy rather than tearing down Caddy +# (which would drop TLS termination at 443). Client cache covers the gap. OnFailure=status-email-root@%n.service [Container] diff --git a/quadlets/vaultwarden.container b/quadlets/vaultwarden.container index 370487d..8ef614f 100644 --- a/quadlets/vaultwarden.container +++ b/quadlets/vaultwarden.container @@ -37,8 +37,9 @@ Environment=USE_SYSLOG=false Environment=WEBSOCKET_ENABLED=true # ADMIN_TOKEN is sourced from a Podman secret. Empty value = admin disabled -# (the plan's default). To enable /admin, populate admin_token_argon2 in -# secrets/vaultwarden.sops.yml; the secret rotates on next Ansible run. +# (the plan's default). To enable /admin, add admin_token_argon2 to the +# vaultwarden-vaultwarden secret in Secret Manager; the Podman secret +# rotates on next Ansible run. Secret=vaultwarden-admin-token,type=env,target=ADMIN_TOKEN # Health: a fixed 8080 is reached from Caddy on the shared bridge network.