Compare commits

21 Commits
Author SHA1 Message Date
luke-else a9d602f35e feat: Changed openvpn for wireguard 2026-07-21 17:17:33 +01:00
luke-else 7ebd02f6b8 chore: refined control script 2026-07-21 16:53:15 +01:00
luke-else aa3ea39ddf chore: refined control script 2026-07-21 15:30:51 +01:00
luke-else 9c0dc78a2c chore: Cleaned up ansible and added control script 2026-07-21 14:56:03 +01:00
luke-else 538d425d4a feat: Made some small updated to deploy 2026-07-20 23:42:11 +01:00
luke-else 1aa340f7ce feat: backup and restore 2026-07-20 22:45:43 +01:00
luke-else c12a507fc1 feat: Swapped tofu init scripting for ansible scripting 2026-07-13 21:23:31 +01:00
luke-else cf5504281c chore: Latest push 2026-07-13 20:24:47 +01:00
luke-else bab789a347 feat: Added automatic script copying functionality 2026-07-10 22:58:56 +01:00
luke-else 41d93563d0 chore: Removed password auth from traefik as not needed 2026-07-10 22:24:58 +01:00
luke-else ae34a78a1d feat: Added DNS control to tofu 2026-07-10 22:19:08 +01:00
luke-else e39e1637da feat: Added multiple SSH Key uploads 2026-07-10 22:01:48 +01:00
luke-else 547fee06fa chore: committing missing files from previous logic changes 2026-07-10 21:55:37 +01:00
luke-else 1ae71cbae3 feat: Added the ability to spin-up an abritrary number of runners at runtime 2026-07-10 21:54:42 +01:00
luke-else d4167d5d60 feat: Created post-install script 2026-07-10 21:28:32 +01:00
luke-else 242e8aa359 chore: Added missing changes from last commit 2026-07-10 20:59:55 +01:00
luke-else a7cfcc33ce chore: Restructured infra modules 2026-07-10 20:54:22 +01:00
luke-else d0d1385bf4 feat: Re-structured services and updated readme 2026-07-10 20:13:33 +01:00
luke-else 3bfa0e6f32 feat: Added terraform config... just a basic one at first 2026-07-09 21:55:50 +01:00
luke-else 5486716981 chore: Updated dependency on devcontainer 2026-07-09 20:53:59 +01:00
luke-else 40d3e20334 chore: Restructured repo 2026-07-09 20:52:25 +01:00
83 changed files with 2456 additions and 436 deletions
Submodule
+1
Submodule .devcontainer added at 7a54748c0e
+50 -5
View File
@@ -1,12 +1,57 @@
# Ignore all files
**
# Ignore all files related to services
**/services
# Allow docker-compose.yml
!**/docker-compose.yml
!/*
# Allow assets folder
!assets/**
# Don't allow .env files, just env-exmaple files
!.env-example
# Secrets generated at deploy time (e.g. services/dev/Runners/.env, holding the
# live Gitea Actions registration token) - never committed, regardless of the
# services/docker-compose.yml carve-out above.
**/.env
# Variables set through control.sh's "Set variables" menu (HCLOUD_TOKEN,
# BACKUP_S3_*, etc.) - never committed.
.control.env
### Ansible ###
*.retry
### Terraform ###
# Local .terraform directories
**/.terraform/*
# .tfstate files
*.tfstate
*.tfstate.*
# Crash log files
crash.log
crash.*.log
# Exclude all .tfvars files, which are likely to contain sensitive data, such as
# password, private keys, and other secrets. These should not be part of version
# control as they are data points which are potentially sensitive and subject
# to change depending on the environment.
*.tfvars
*.tfvars.json
# Ignore override files as they are usually used to override resources locally and so
# are not checked in
override.tf
override.tf.json
*_override.tf
*_override.tf.json
# Include override files you do wish to add to version control using negated pattern
# !example_override.tf
# Include tfplan files to ignore the plan output of command: terraform plan -out=tfplan
# example: *tfplan*
# Ignore CLI configuration files
.terraformrc
terraform.rc
+4
View File
@@ -0,0 +1,4 @@
[submodule ".devcontainer"]
path = .devcontainer
url = ssh://git@git.luke-else.co.uk:222/luke-else/.devcontainer.git
branch = tf
-24
View File
@@ -1,24 +0,0 @@
services:
#MongoDB (27017)
mongodb:
image: "mongo:latest"
container_name: mongoDB
volumes:
- ./mongo/:/data/db
networks:
- proxy
labels:
- "traefik.enable=true"
- "traefik.tcp.routers.mongodb.rule=HostSNI(`*`)"
- "traefik.tcp.routers.mongodb.entrypoints=mongo"
- "traefik.tcp.routers.mongodb.service=mongodb"
- "traefik.tcp.services.mongodb.loadbalancer.server.port=27017"
environment:
MONGO_INITDB_ROOT_USERNAME: root
MONGO_INITDB_ROOT_PASSWORD: rootpassword
restart: unless-stopped
networks:
proxy:
external: true
-33
View File
@@ -1,33 +0,0 @@
services:
# Runner
runner:
image: gitea/act_runner:latest
container_name: gitea_runner
volumes:
- ./config.yaml:/config.yaml
- ./gitea_runner:/data
- /var/run/docker.sock:/var/run/docker.sock
networks:
- proxy
environment:
CONFIG_FILE: /config.yaml
GITEA_INSTANCE_URL: "https://git.luke-else.co.uk"
GITEA_RUNNER_REGISTRATION_TOKEN: "INSERT REGISTRATION TOKEN"
GITEA_RUNNER_NAME: "CICD"
labels:
## Expose cicd Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.cicd-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.cicd-insecure.rule=Host(`cicd.luke-else.co.uk`)"
- "traefik.http.routers.cicd-insecure.entrypoints=web"
- "traefik.http.routers.cicd.rule=Host(`cicd.luke-else.co.uk`)"
- "traefik.http.routers.cicd.entrypoints=websecure"
- "traefik.http.routers.cicd.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
external: true
-51
View File
@@ -1,51 +0,0 @@
services:
status:
image: louislam/uptime-kuma:latest
container_name: status
volumes:
- ./uptime-kuma/data:/app/data
networks:
- proxy
labels:
## Expose uptime-kuma Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.status-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.status-insecure.rule=Host(`status.luke-else.co.uk`)"
- "traefik.http.routers.status-insecure.entrypoints=web"
- "traefik.http.routers.status.rule=Host(`status.luke-else.co.uk`)"
- "traefik.http.routers.status.entrypoints=websecure"
- "traefik.http.routers.status.tls.certresolver=myresolver"
restart: unless-stopped
portainer:
image: portainer/portainer-ce:latest
container_name: portainer
security_opt:
- no-new-privileges:true
volumes:
- /etc/localtime:/etc/localtime:ro
- /var/run/docker.sock:/var/run/docker.sock:ro
- ./portainer-data:/data
networks:
- proxy
labels:
## Expose portainer Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.portainer-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.portainer-insecure.rule=Host(`portainer.luke-else.co.uk`)"
- "traefik.http.routers.portainer-insecure.entrypoints=web"
- "traefik.http.routers.portainer.rule=Host(`portainer.luke-else.co.uk`)"
- "traefik.http.services.portainer.loadbalancer.server.port=9000"
- "traefik.http.routers.portainer.entrypoints=websecure"
- "traefik.http.routers.portainer.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
external: true
-52
View File
@@ -1,52 +0,0 @@
services:
ackee:
image: electerious/ackee
container_name: ackee
environment:
- WAIT_HOSTS=mongo:27017
- ACKEE_MONGODB=mongodb://mongo-ackee:27017/ackee
env_file:
- .env
expose:
- 3000
networks:
- tracking
- proxy
labels:
## Expose Ackee Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
# Enable CORS headers
- "traefik.http.middlewares.cors-tracking.headers.accesscontrolallowmethods=*"
- "traefik.http.middlewares.cors-tracking.headers.accesscontrolalloworiginlist=https://luke-else.co.uk"
- "traefik.http.middlewares.cors-tracking.headers.accesscontrolallowcredentials=true"
- "traefik.http.middlewares.cors-tracking.headers.accesscontrolallowheaders=Content-Type,Authorization"
- "traefik.http.middlewares.cors-tracking.headers.addvaryheader=true"
- "traefik.http.middlewares.cors-tracking.headers.accesscontrolmaxage=100"
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.ackee-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.ackee-insecure.rule=Host(`tracking.luke-else.co.uk`)"
- "traefik.http.routers.ackee-insecure.entrypoints=web"
- "traefik.http.routers.ackee.rule=Host(`tracking.luke-else.co.uk`)"
- "traefik.http.routers.ackee.entrypoints=websecure"
- "traefik.http.routers.ackee.tls.certresolver=myresolver"
- "traefik.http.routers.ackee.middlewares=cors-tracking"
depends_on:
- mongo
restart: unless-stopped
mongo:
image: mongo
container_name: mongo-ackee
volumes:
- ./data:/data/db
networks:
- tracking
restart: unless-stopped
networks:
proxy:
external: true
tracking:
-24
View File
@@ -1,24 +0,0 @@
services:
dockovpn:
image: alekslitvinenk/openvpn
cap_add:
- NET_ADMIN
ports:
- 1194:1194/udp # Expose tcp if you defined HOST_TUN_PROTOCOL=tcp
environment:
HOST_ADDR: vpn.luke-else.co.uk # Your VPN server address
volumes:
- ./openvpn_conf:/opt/Dockovpn_data
labels:
## Expose vpn Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.vpn-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.vpn-insecure.rule=Host(`vpn.luke-else.co.uk`)"
- "traefik.http.routers.vpn-insecure.entrypoints=web"
- "traefik.http.routers.vpn.rule=Host(`vpn.luke-else.co.uk`)"
- "traefik.http.routers.vpn.entrypoints=websecure"
- "traefik.http.routers.vpn.tls.certresolver=myresolver"
restart: always
+93
View File
@@ -0,0 +1,93 @@
# Ansible
Bootstraps, syncs, and starts/stops the Docker Compose stacks in `../services/`
on the three servers OpenTofu provisions (see `../infra/`). OpenTofu still owns
the actual cloud resources (servers, network, firewalls, DNS);
everything downstream of "the server exists" - installing Docker, creating the
`deploy` user, hardening SSH, copying `services/<host>/`, rendering `.env` /
the Gitea runner compose file, and running `spinup.sh`/`spindown.sh` - lives
here instead. Persistent service data lives in named Docker volumes backed up
to S3, not on a Hetzner volume - see `../readme.md`.
## Contents
- `inventory/hcloud.yml` - dynamic inventory, queries the Hetzner Cloud API
directly (not Terraform state) and groups servers by their `role` label
(`role_dev`, `role_prod`, `role_vpn` - set in `infra/modules/<host>/main.tf`).
- `roles/bootstrap/` - Docker install, `deploy` user creation, SSH hardening,
unattended-upgrades. Replaces `infra/scripts/bootstrap.sh.tftpl`.
- `roles/deploy/` - copies `services/<host>/` to the server and renders `.env`
(S3 backup credentials, see `../readme.md`) and, for `dev`,
`Runners/docker-compose.yml`.
- `playbooks/` - `bootstrap.yml`, `deploy.yml`, `spinup.yml`, `spindown.yml`,
`site.yml` (all three in order), and `group_vars/` (`all.yml` - shared:
`deploy_user`, SSH key path, `backup_s3_*` S3 backup credentials;
`role_dev.yml` / `role_prod.yml` / `role_vpn.yml` - per-role vars, e.g.
`service_group`). `group_vars/` lives here rather than at the `ansible/`
root because Ansible only auto-loads `group_vars/`/`host_vars/` from
directories adjacent to the playbook file or the inventory source - not
from an arbitrary sibling directory.
## Prerequisites
```sh
cd ansible
ansible-galaxy collection install -r requirements.yml
export HCLOUD_TOKEN=your-hetzner-api-token # never commit this
export BACKUP_S3_BUCKET=your-bucket-name
export BACKUP_S3_ACCESS_KEY_ID=your-access-key-id
export BACKUP_S3_SECRET_ACCESS_KEY=your-secret-access-key
export BACKUP_S3_ENDPOINT=your-s3-endpoint # omit for real AWS S3
```
`../control.sh`'s **Set variables** menu does the same thing interactively and
persists your answers to `../.control.env` (gitignored) so you don't have to
re-export them every session.
`deploy_user` (see `playbooks/group_vars/all.yml`) defaults to `deploy` - override with
the `DEPLOY_USER` env var if you want a different non-root user created on each
host. `ansible_ssh_private_key_file` (same file) defaults to
`~/.ssh/id_ed25519` - override with `ANSIBLE_SSH_PRIVATE_KEY_FILE` if you use a
different key. It must match one of the SSH keys OpenTofu installed on the
servers (`var.ssh_key_names`).
## Usage
Same chicken-and-egg as OpenTofu's own first-apply ordering (see
`../readme.md`): `dev`/`prod`'s firewalls only accept SSH from `vpn`'s own
public IP, so you must bring `vpn` up and connect to it before `dev`/`prod`
are reachable at all.
```sh
# 1. vpn first - its firewall accepts SSH from var.allowed_ssh_source_ips directly
ansible-playbook playbooks/bootstrap.yml -l role_vpn
ansible-playbook playbooks/deploy.yml -l role_vpn
ansible-playbook playbooks/spinup.yml -l role_vpn
# 2. visit https://vpn.luke-else.co.uk, complete wg-easy's first-run setup,
# add a client peer, and connect with a WireGuard client - now dev/prod
# are reachable, since your egress IP matches their firewall rule
ansible-playbook playbooks/bootstrap.yml -l role_dev,role_prod
ansible-playbook playbooks/deploy.yml -l role_dev,role_prod
ansible-playbook playbooks/spinup.yml -l role_dev,role_prod
# ...or once already bootstrapped and connected, do everything in one go:
ansible-playbook playbooks/site.yml -l role_dev,role_prod
```
Re-run `deploy.yml` + `spinup.yml` any time a compose file changes - `spinup.sh`
only starts/updates what's changed, same as before. `dev` always runs three
Gitea Actions runners (fixed in
`roles/deploy/templates/runners-docker-compose.yml.j2`).
`spindown.yml` runs `spindown.sh`, which does a full `docker system/volume
prune -a` on the host - it prompts for a `yes` confirmation before running,
since the dynamic inventory makes it trivial to target multiple hosts at once
(`-l role_dev,role_prod`).
Inspect what the dynamic inventory currently sees:
```sh
ansible-inventory --graph
ansible-inventory --list
```
+9
View File
@@ -0,0 +1,9 @@
[defaults]
inventory = inventory/hcloud.yml
roles_path = roles
host_key_checking = False
retry_files_enabled = False
interpreter_python = auto_silent
[inventory]
enable_plugins = hetzner.hcloud.hcloud, auto
+12
View File
@@ -0,0 +1,12 @@
---
# Dynamic inventory: queries the Hetzner Cloud API directly (HCLOUD_TOKEN),
# grouping hosts by the `role` label set on each hcloud_server resource in
# infra/modules/<host>/main.tf (role = dev/prod/vpn). Nothing here reads
# Terraform state - this always reflects whatever actually exists in Hetzner.
plugin: hetzner.hcloud.hcloud
token: "{{ lookup('env', 'HCLOUD_TOKEN') }}"
connect_with: public_ipv4
keyed_groups:
- key: hcloud_labels.role
prefix: role
separator: "_"
+9
View File
@@ -0,0 +1,9 @@
---
- name: Bootstrap newly created servers (idempotent, safe to re-run)
hosts: all
gather_facts: true
become: false
vars:
ansible_user: root
roles:
- bootstrap
+6
View File
@@ -0,0 +1,6 @@
---
- name: Render templates and sync service definitions to each host
hosts: all
gather_facts: false
roles:
- deploy
+22
View File
@@ -0,0 +1,22 @@
---
# Non-root sudo user the bootstrap role creates on each host - purely an
# Ansible-side concept, OpenTofu has no equivalent variable. Override with the
# DEPLOY_USER env var (control.sh's "Set variables" menu writes it) if you
# don't want the default.
deploy_user: "{{ lookup('env', 'DEPLOY_USER') | default('deploy', true) }}"
ansible_user: "{{ deploy_user }}"
ansible_ssh_private_key_file: "{{ lookup('env', 'ANSIBLE_SSH_PRIVATE_KEY_FILE') | default('~/.ssh/id_ed25519', true) }}"
# Repo root's services/ directory, relative to wherever a playbook lives under ansible/playbooks/.
services_root: "{{ playbook_dir }}/../../services"
# S3-compatible bucket that services/<host>/backup-docker-compose.yml backs up
# to and restores from - see readme.md. Not provisioned by OpenTofu (the
# hcloud provider has no Object Storage resource), so these must point at a
# bucket you've created yourself. AWS_ENDPOINT is only needed for a
# non-AWS S3-compatible provider (e.g. Hetzner Object Storage); leave it unset
# to use real AWS S3.
backup_s3_bucket: "{{ lookup('env', 'BACKUP_S3_BUCKET') }}"
backup_s3_access_key_id: "{{ lookup('env', 'BACKUP_S3_ACCESS_KEY_ID') }}"
backup_s3_secret_access_key: "{{ lookup('env', 'BACKUP_S3_SECRET_ACCESS_KEY') }}"
backup_s3_endpoint: "{{ lookup('env', 'BACKUP_S3_ENDPOINT') }}"
@@ -0,0 +1,2 @@
---
service_group: dev
@@ -0,0 +1,2 @@
---
service_group: prod
@@ -0,0 +1,2 @@
---
service_group: vpn
+4
View File
@@ -0,0 +1,4 @@
---
- ansible.builtin.import_playbook: bootstrap.yml
- ansible.builtin.import_playbook: deploy.yml
- ansible.builtin.import_playbook: spinup.yml
+27
View File
@@ -0,0 +1,27 @@
---
- name: Stop every compose stack on each host and prune Docker resources
hosts: all
gather_facts: false
vars_prompt:
- name: confirm
prompt: >-
This runs spindown.sh (docker compose down, then a full
docker image/volume prune -a) on the targeted host(s).
Type 'yes' to continue
private: false
tasks:
- name: Abort unless confirmed
ansible.builtin.fail:
msg: Aborted.
when: confirm != 'yes'
- name: Run spindown.sh
ansible.builtin.command: ./spindown.sh
args:
chdir: "/home/{{ deploy_user }}/services/{{ service_group }}"
register: _spindown
changed_when: true
- name: Show spindown output
ansible.builtin.debug:
var: _spindown.stdout_lines
+15
View File
@@ -0,0 +1,15 @@
---
- name: Start every compose stack on each host, in the order spinup.sh defines
hosts: all
gather_facts: false
tasks:
- name: Run spinup.sh
ansible.builtin.command: ./spinup.sh
args:
chdir: "/home/{{ deploy_user }}/services/{{ service_group }}"
register: _spinup
changed_when: true
- name: Show spinup output
ansible.builtin.debug:
var: _spinup.stdout_lines
+4
View File
@@ -0,0 +1,4 @@
---
collections:
- name: hetzner.hcloud
version: ">=4.0.0"
@@ -0,0 +1,5 @@
---
- name: Reload sshd
ansible.builtin.systemd:
name: ssh
state: reloaded
+95
View File
@@ -0,0 +1,95 @@
---
# Post-install bootstrap - replaces infra/scripts/bootstrap.sh.tftpl. Runs as
# root (see ansible_user override in playbooks/bootstrap.yml) since the deploy
# user doesn't exist yet on a freshly created server. Idempotent: safe to
# re-run against an already-bootstrapped host.
- name: Update apt cache and upgrade packages
ansible.builtin.apt:
update_cache: true
upgrade: dist
cache_valid_time: 3600
# Docker straight from Ubuntu's own repositories - no external apt repo, GPG
# key or architecture juggling. docker.io is the engine, docker-compose-v2 is
# the `docker compose` plugin. A little behind Docker's own channel, but far
# simpler and plenty for these hosts.
- name: Install Docker Engine and the Compose plugin
ansible.builtin.apt:
name:
- docker.io
- docker-compose-v2
state: present
- name: Ensure Docker is enabled and running
ansible.builtin.systemd:
name: docker
enabled: true
state: started
- name: Create non-root sudo user, given the same key root was provisioned with
ansible.builtin.user:
name: "{{ deploy_user }}"
shell: /bin/bash
groups: sudo,docker
append: true
create_home: true
- name: Ensure deploy user's .ssh directory exists
ansible.builtin.file:
path: "/home/{{ deploy_user }}/.ssh"
state: directory
owner: "{{ deploy_user }}"
group: "{{ deploy_user }}"
mode: "0700"
- name: Give deploy user the same authorized_keys as root
ansible.builtin.copy:
src: /root/.ssh/authorized_keys
dest: "/home/{{ deploy_user }}/.ssh/authorized_keys"
remote_src: true
owner: "{{ deploy_user }}"
group: "{{ deploy_user }}"
mode: "0600"
- name: Grant deploy user passwordless sudo
ansible.builtin.copy:
content: "{{ deploy_user }} ALL=(ALL) NOPASSWD:ALL\n"
dest: "/etc/sudoers.d/90-{{ deploy_user }}"
validate: "visudo -cf %s"
mode: "0440"
# Root keeps key-only login as a fallback rather than being fully locked out,
# in case deploy user setup above ever fails silently on a future image.
- name: Restrict root login to key-only
ansible.builtin.lineinfile:
path: /etc/ssh/sshd_config
regexp: "^#?PermitRootLogin"
line: "PermitRootLogin prohibit-password"
notify: Reload sshd
- name: Disable SSH password authentication
ansible.builtin.lineinfile:
path: /etc/ssh/sshd_config
regexp: "^#?PasswordAuthentication"
line: "PasswordAuthentication no"
notify: Reload sshd
- name: Install unattended-upgrades
ansible.builtin.apt:
name: unattended-upgrades
state: present
- name: Enable automatic security upgrades
ansible.builtin.copy:
dest: /etc/apt/apt.conf.d/20auto-upgrades
content: |
APT::Periodic::Update-Package-Lists "1";
APT::Periodic::Unattended-Upgrade "1";
mode: "0644"
- name: Enable and start unattended-upgrades
ansible.builtin.systemd:
name: unattended-upgrades
enabled: true
state: started
+43
View File
@@ -0,0 +1,43 @@
---
# Copies services/<service_group>/ to the server and renders the files that
# used to be generated by OpenTofu (services/<host>/.env and, for dev,
# Runners/docker-compose.yml) - replaces the null_resource.deploy_services +
# local_file provisioners previously in infra/modules/<host>/main.tf.
#
# Persistent data lives in named Docker volumes now, not on a Hetzner volume -
# see services/<host>/backup-docker-compose.yml. .env only carries the S3
# credentials that backup/restore need, rendered for every host.
- name: Ensure services directory exists on the host
ansible.builtin.file:
path: "/home/{{ deploy_user }}/services/{{ service_group }}"
state: directory
mode: "0755"
- name: Copy static service files (compose files, scripts, configs)
ansible.builtin.copy:
src: "{{ services_root }}/{{ service_group }}/"
dest: "/home/{{ deploy_user }}/services/{{ service_group }}/"
mode: preserve
- name: Render S3 backup credentials into .env (auto-loaded by docker compose)
ansible.builtin.template:
src: env.j2
dest: "/home/{{ deploy_user }}/services/{{ service_group }}/.env"
mode: "0600"
- name: Render Gitea Actions runner compose file
when: service_group == 'dev'
ansible.builtin.template:
src: runners-docker-compose.yml.j2
dest: "/home/{{ deploy_user }}/services/dev/Runners/docker-compose.yml"
mode: "0644"
- name: Make spinup/spindown/restore scripts executable
ansible.builtin.file:
path: "/home/{{ deploy_user }}/services/{{ service_group }}/{{ item }}"
mode: "0755"
loop:
- spinup.sh
- spindown.sh
- restore.sh
+5
View File
@@ -0,0 +1,5 @@
AWS_ACCESS_KEY_ID={{ backup_s3_access_key_id }}
AWS_SECRET_ACCESS_KEY={{ backup_s3_secret_access_key }}
AWS_S3_BUCKET_NAME={{ backup_s3_bucket }}
AWS_S3_PATH={{ service_group }}
AWS_ENDPOINT={{ backup_s3_endpoint }}
@@ -0,0 +1,40 @@
# Generated by Ansible - do not hand-edit. Three Gitea Actions runners, always.
# To change their shape (image, volumes, env), edit this template and re-run
# playbooks/deploy.yml.
#
# GITEA_RUNNER_REGISTRATION_TOKEN is intentionally left as a compose variable
# (not baked in here) - services/dev/spinup.sh generates a fresh token and
# writes it to Runners/.env immediately before starting these containers.
#
# /data is a disposable local cache (runner identity re-registers on every
# spinup.sh run anyway), so it's a plain named volume - not part of the S3
# backup/restore stack in services/dev/backup-docker-compose.yml.
{% set runner_count = 3 %}
services:
{% for i in range(1, runner_count + 1) %}
runner-{{ i }}:
image: gitea/act_runner:latest
container_name: gitea_runner_{{ i }}
volumes:
- ./config.yaml:/config.yaml
- gitea_runner_{{ i }}_data:/data
- /var/run/docker.sock:/var/run/docker.sock
networks:
- proxy
environment:
CONFIG_FILE: /config.yaml
GITEA_INSTANCE_URL: "https://git.luke-else.co.uk"
GITEA_RUNNER_REGISTRATION_TOKEN: "${GITEA_RUNNER_REGISTRATION_TOKEN}"
GITEA_RUNNER_NAME: "CICD#{{ i }}"
restart: unless-stopped
{% endfor %}
networks:
proxy:
external: true
volumes:
{% for i in range(1, runner_count + 1) %}
gitea_runner_{{ i }}_data:
name: gitea_runner_{{ i }}_data
{% endfor %}
-16
View File
@@ -1,16 +0,0 @@
#Script file for spinning down all CICD relevant docker-containers
cd ./Development/Runners/
docker compose down
cd ../..
cd ./TraefikRunner/
docker compose down
cd ..
cd ./Watchtower/
docker compose down
cd ..
docker rmi $(docker images -q)
docker system prune -f -a
docker volume prune -f -a
-13
View File
@@ -1,13 +0,0 @@
#Script file for spinning up all docker-containers
cd ./TraefikRunner/
docker compose pull && docker compose up -d
cd ..
cd ./Development/Runners/
docker compose pull && docker compose up -d
cd ../..
cd ./Watchtower/
docker compose pull && docker compose up -d
cd ..
Executable
+307
View File
@@ -0,0 +1,307 @@
#!/usr/bin/env bash
# control.sh - interactive control panel for this repo.
#
# Steps you through the pre-setup and day-to-day operations for both the
# OpenTofu infra (infra/) and the Ansible config (ansible/), in the order the
# firewall rules require (vpn first, then tunnel in, then dev/prod). It only
# ever runs `tofu`/`ansible-playbook` for you - it's a menu wrapper, nothing
# more, so anything it does you can also do by hand from infra/ or ansible/.
# Re-exec under bash if invoked as `sh control.sh` (or another POSIX shell) -
# this script relies on bash-only features (indirect expansion, arrays, [[ ]])
# that fail with cryptic errors like "bad substitution" under dash/sh.
if [ -z "${BASH_VERSION:-}" ]; then
exec bash "$0" "$@"
fi
set -uo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
INFRA_DIR="$REPO_ROOT/infra"
ANSIBLE_DIR="$REPO_ROOT/ansible"
CONTROL_ENV="$REPO_ROOT/.control.env"
# --- pretty output --------------------------------------------------------
if [ -t 1 ]; then
BOLD=$(tput bold); DIM=$(tput dim); RESET=$(tput sgr0)
RED=$(tput setaf 1); GREEN=$(tput setaf 2); YELLOW=$(tput setaf 3); CYAN=$(tput setaf 6)
else
BOLD=""; DIM=""; RESET=""; RED=""; GREEN=""; YELLOW=""; CYAN=""
fi
ok() { printf "%s ok %s %s\n" "$GREEN" "$RESET" "$1"; }
warn() { printf "%s warn%s %s\n" "$YELLOW" "$RESET" "$1"; }
bad() { printf "%s miss%s %s\n" "$RED" "$RESET" "$1"; }
title() { printf "\n%s%s%s\n" "$BOLD$CYAN" "$1" "$RESET"; }
# Pick the OpenTofu binary: prefer `tofu`, fall back to `terraform`.
TOFU=""
detect_tofu() {
if command -v tofu >/dev/null 2>&1; then TOFU="tofu"
elif command -v terraform >/dev/null 2>&1; then TOFU="terraform"
fi
}
detect_tofu
# --- variable persistence --------------------------------------------------
# Variables set via the "Set variables" menu (HCLOUD_TOKEN, BACKUP_S3_*,
# DEPLOY_USER, ANSIBLE_SSH_PRIVATE_KEY_FILE) live in .control.env, gitignored,
# so they survive between runs of this script. They are NOT exported into your
# normal shell - only into control.sh's own process and whatever it runs.
load_control_env() {
if [ -f "$CONTROL_ENV" ]; then
set -a
# shellcheck disable=SC1090
source "$CONTROL_ENV"
set +a
fi
}
load_control_env
# Upsert KEY=VALUE into .control.env and export it into this process.
save_var() {
local key="$1" val="$2"
touch "$CONTROL_ENV"; chmod 600 "$CONTROL_ENV"
local tmp; tmp="$(mktemp)"
grep -v "^${key}=" "$CONTROL_ENV" > "$tmp" 2>/dev/null || true
mv "$tmp" "$CONTROL_ENV"
printf '%s=%q\n' "$key" "$val" >> "$CONTROL_ENV"
export "${key?}=${val}"
}
# Run a command inside a directory, echoing it first. Returns the command's exit code.
run_in() {
local dir="$1"; shift
printf "%s\$ (cd %s && %s)%s\n" "$DIM" "${dir#$REPO_ROOT/}" "$*" "$RESET"
( cd "$dir" && "$@" )
}
pause() { read -rp $'\nPress Enter to return to the menu... ' _; }
confirm() {
local reply
read -rp "$1 [y/N] " reply
[[ "$reply" == "y" || "$reply" == "Y" ]]
}
# --- prerequisite checks --------------------------------------------------
check_prereqs() {
title "Tools"
if [ -n "$TOFU" ]; then ok "OpenTofu ($TOFU: $("$TOFU" version 2>/dev/null | head -n1))"
else bad "Neither 'tofu' nor 'terraform' found on PATH"; fi
if command -v ansible-playbook >/dev/null 2>&1; then ok "Ansible ($(ansible --version 2>/dev/null | head -n1))"
else bad "'ansible-playbook' not found on PATH"; fi
title "Environment variables"
[ -n "${HCLOUD_TOKEN:-}" ] && ok "HCLOUD_TOKEN set" || bad "HCLOUD_TOKEN not set (needed by both tofu and the ansible inventory)"
[ -n "${BACKUP_S3_BUCKET:-}" ] && ok "BACKUP_S3_BUCKET set" || warn "BACKUP_S3_BUCKET not set (needed for deploy/spinup - S3 backup .env)"
[ -n "${BACKUP_S3_ACCESS_KEY_ID:-}" ] && ok "BACKUP_S3_ACCESS_KEY_ID set" || warn "BACKUP_S3_ACCESS_KEY_ID not set"
[ -n "${BACKUP_S3_SECRET_ACCESS_KEY:-}" ] && ok "BACKUP_S3_SECRET_ACCESS_KEY set" || warn "BACKUP_S3_SECRET_ACCESS_KEY not set"
[ -n "${BACKUP_S3_ENDPOINT:-}" ] && ok "BACKUP_S3_ENDPOINT set" || warn "BACKUP_S3_ENDPOINT not set (omit only for real AWS S3)"
ok "DEPLOY_USER: ${DEPLOY_USER:-deploy} (default 'deploy' if unset)"
local key="${ANSIBLE_SSH_PRIVATE_KEY_FILE:-$HOME/.ssh/id_ed25519}"
[ -f "${key/#\~/$HOME}" ] && ok "SSH key present ($key)" || warn "SSH key not found at $key (set ANSIBLE_SSH_PRIVATE_KEY_FILE)"
[ -f "$CONTROL_ENV" ] && ok ".control.env present (loaded above)" || warn "No .control.env yet - use Configuration > Set variables to persist the above across runs"
title "Files"
[ -f "$INFRA_DIR/terraform.tfvars" ] && ok "infra/terraform.tfvars exists" \
|| bad "infra/terraform.tfvars missing (Configuration > Copy terraform.tfvars.example)"
[ -d "$INFRA_DIR/.terraform" ] && ok "infra/.terraform exists (tofu init has run)" \
|| warn "infra/.terraform missing - run OpenTofu > init"
if ansible-galaxy collection list hetzner.hcloud >/dev/null 2>&1; then ok "hetzner.hcloud collection installed"
else warn "hetzner.hcloud collection not installed - run Ansible > Install collections"; fi
}
# --- configuration actions -------------------------------------------------
# Prompts for one variable, keeping the current value if the user hits Enter.
# $3 = true masks input and the shown current value, for tokens/secrets.
prompt_var() {
local key="$1" desc="$2" secret="${3:-false}" current input shown
current="${!key:-}"
shown="$current"
[ "$secret" = "true" ] && [ -n "$current" ] && shown="(set, hidden)"
if [ "$secret" = "true" ]; then read -rsp "$desc [$shown]: " input; echo
else read -rp "$desc [$shown]: " input; fi
[ -n "$input" ] && save_var "$key" "$input"
}
set_variables() {
title "Set variables"
echo "Leave a prompt blank to keep its current value. Saved to .control.env" \
"(gitignored) and reloaded automatically next time you run control.sh -" \
"not exported to your normal shell outside of it."
echo
prompt_var HCLOUD_TOKEN "Hetzner Cloud API token" true
prompt_var DEPLOY_USER "Non-root deploy user Ansible creates (default: deploy)" false
prompt_var ANSIBLE_SSH_PRIVATE_KEY_FILE "SSH private key path (default: ~/.ssh/id_ed25519)" false
prompt_var BACKUP_S3_BUCKET "S3 backup bucket name" false
prompt_var BACKUP_S3_ACCESS_KEY_ID "S3 backup access key ID" true
prompt_var BACKUP_S3_SECRET_ACCESS_KEY "S3 backup secret access key" true
prompt_var BACKUP_S3_ENDPOINT "S3 backup endpoint (blank for real AWS S3)" false
ok "Variables saved."
}
copy_tfvars() {
local example="$INFRA_DIR/terraform.tfvars.example"
local target="$INFRA_DIR/terraform.tfvars"
if [ -f "$target" ]; then
warn "infra/terraform.tfvars already exists."
confirm "Overwrite it with a fresh copy of the example?" || return 0
fi
cp "$example" "$target"
ok "Copied terraform.tfvars.example -> terraform.tfvars"
local keys quoted k
read -rp "SSH key name(s) from Hetzner Console > Security > SSH Keys, comma-separated (blank = edit manually later): " keys
if [ -n "$keys" ]; then
quoted=""
IFS=',' read -ra _key_arr <<< "$keys"
for k in "${_key_arr[@]}"; do
k="$(echo "$k" | xargs)"
quoted+="\"$k\", "
done
quoted="${quoted%, }"
sed -i "s|^ssh_key_names = .*|ssh_key_names = [${quoted}]|" "$target"
ok "Set ssh_key_names = [${quoted}]"
else
warn "ssh_key_names still has the example's placeholder values - edit infra/terraform.tfvars before running tofu."
fi
}
# --- OpenTofu actions -----------------------------------------------------
require_tofu() {
if [ -z "$TOFU" ]; then bad "No tofu/terraform binary found."; return 1; fi
if [ ! -f "$INFRA_DIR/terraform.tfvars" ]; then
bad "infra/terraform.tfvars is missing - copy terraform.tfvars.example first."; return 1
fi
}
tofu_init() { require_tofu && run_in "$INFRA_DIR" "$TOFU" init; }
tofu_plan() { require_tofu && run_in "$INFRA_DIR" "$TOFU" plan; }
tofu_apply() { require_tofu && run_in "$INFRA_DIR" "$TOFU" apply; }
tofu_output() { require_tofu && run_in "$INFRA_DIR" "$TOFU" output; }
tofu_destroy() {
require_tofu || return 1
confirm "${RED}Destroy ALL infra managed by OpenTofu?${RESET}" && run_in "$INFRA_DIR" "$TOFU" destroy
}
# --- Ansible actions ------------------------------------------------------
ansible_collections() { run_in "$ANSIBLE_DIR" ansible-galaxy collection install -r requirements.yml; }
ansible_inventory() { run_in "$ANSIBLE_DIR" ansible-inventory --graph; }
# $1 = playbook, $2 = -l limit expression
play() { run_in "$ANSIBLE_DIR" ansible-playbook "playbooks/$1" -l "$2"; }
ansible_bootstrap_vpn() { play bootstrap.yml role_vpn; }
ansible_deploy_vpn() { play deploy.yml role_vpn && play spinup.yml role_vpn; }
ansible_bootstrap_dp() { play bootstrap.yml role_dev,role_prod; }
ansible_deploy_dp() { play deploy.yml role_dev,role_prod && play spinup.yml role_dev,role_prod; }
ansible_spindown() {
local target
read -rp "Spindown which hosts? (e.g. role_vpn, role_dev,role_prod, all): " target
[ -n "$target" ] && play spindown.yml "$target"
}
# --- guided walkthrough ---------------------------------------------------
guided() {
title "Guided full setup"
cat <<EOF
This walks the whole bring-up in the order the firewalls require. dev/prod only
accept SSH from the vpn server's public IP, so vpn must exist and you must be
connected to it (WireGuard, via wg-easy) before dev/prod are reachable.
Steps, in order:
0. Configuration: variables + terraform.tfvars (skip if already done)
1. OpenTofu: init -> plan -> apply (creates servers, network, firewalls, DNS)
2. Ansible: bootstrap + deploy + spinup for role_vpn
3. YOU: visit https://vpn.luke-else.co.uk, complete wg-easy's setup,
add a client peer, and connect with a WireGuard client
4. Ansible: bootstrap + deploy + spinup for role_dev,role_prod
EOF
confirm "Start the guided walkthrough?" || return 0
echo; title "0/4 Configuration"
if confirm "Set/update variables now (HCLOUD_TOKEN, DEPLOY_USER, BACKUP_S3_*, SSH key path)?"; then
set_variables
fi
if [ ! -f "$INFRA_DIR/terraform.tfvars" ] && confirm "infra/terraform.tfvars is missing - create it from the example now?"; then
copy_tfvars
fi
echo; title "1/4 OpenTofu"
if confirm "Run $TOFU init/plan/apply now?"; then
tofu_init && tofu_plan && tofu_apply || { bad "OpenTofu step failed - stopping."; return 1; }
fi
echo; title "2/4 Ansible - vpn"
if confirm "Bootstrap + deploy + spinup role_vpn now?"; then
ansible_bootstrap_vpn && ansible_deploy_vpn || { bad "vpn step failed - stopping."; return 1; }
fi
echo; title "3/4 Connect the VPN"
warn "Visit https://vpn.luke-else.co.uk, complete wg-easy's setup wizard, add a client peer, and connect with a WireGuard client now."
confirm "Are you connected to the VPN?" || { warn "Stopping - reconnect and re-run this step."; return 0; }
echo; title "4/4 Ansible - dev + prod"
if confirm "Bootstrap + deploy + spinup role_dev,role_prod now?"; then
ansible_bootstrap_dp && ansible_deploy_dp || { bad "dev/prod step failed."; return 1; }
fi
ok "Guided setup complete."
}
# --- menu -----------------------------------------------------------------
menu() {
clear
printf "%s==== server control panel ====%s\n" "$BOLD$CYAN" "$RESET"
printf "%srepo: %s tofu: %s%s\n" "$DIM" "$REPO_ROOT" "${TOFU:-none}" "$RESET"
title "Setup"
echo " 1) Check prerequisites"
echo " 2) Guided full setup (vpn -> connect -> dev/prod)"
title "Configuration"
echo " 3) Set variables (HCLOUD_TOKEN, DEPLOY_USER, BACKUP_S3_*, SSH key path)"
echo " 4) Copy terraform.tfvars.example -> terraform.tfvars"
title "OpenTofu (infra/)"
echo " 5) init 6) plan 7) apply"
echo " 8) output 9) destroy"
title "Ansible (ansible/)"
echo " 10) Install collections 11) Show inventory"
echo " 12) Bootstrap vpn 13) Deploy + spinup vpn"
echo " 14) Bootstrap dev+prod 15) Deploy + spinup dev+prod"
echo " 16) Spindown (choose target)"
echo
echo " 0) Quit"
echo
read -rp "Select an option: " choice
}
while true; do
menu
case "$choice" in
1) check_prereqs; pause ;;
2) guided; pause ;;
3) set_variables; pause ;;
4) copy_tfvars; pause ;;
5) tofu_init; pause ;;
6) tofu_plan; pause ;;
7) tofu_apply; pause ;;
8) tofu_output; pause ;;
9) tofu_destroy; pause ;;
10) ansible_collections; pause ;;
11) ansible_inventory; pause ;;
12) ansible_bootstrap_vpn; pause ;;
13) ansible_deploy_vpn; pause ;;
14) ansible_bootstrap_dp; pause ;;
15) ansible_deploy_dp; pause ;;
16) ansible_spindown; pause ;;
0) echo "Bye."; exit 0 ;;
"") ;;
*) warn "Unknown option: $choice"; pause ;;
esac
done
+67
View File
@@ -0,0 +1,67 @@
# This file is maintained automatically by "tofu init".
# Manual edits may be lost in future updates.
provider "registry.opentofu.org/hashicorp/hcloud" {
version = "1.66.0"
hashes = [
"h1:+R3t/5wqCb2ics5bNZ4JgtfaAa08Z6uRzXzNtwovVEY=",
"h1:9c1ECk3N9xhdQKz4NVCfYEFxebJVVeQZS3E6wrR/Hx8=",
"h1:EtYB/KjPN2cIJa6hpEaVc8q/3yf6e9IT6/uAybUGqps=",
"h1:GUupu6Fg5zFY9i4/3esiy8uHZnUrt+u2B6NaE4+8YqQ=",
"h1:JHehRJySEFRGYK+D6pFS0MTTDDRlGSQdSaKdEo0AemA=",
"h1:XF+SwyO1ttQJ0WXJ5O628O7cGk9U37sGn0o3LwWUnDQ=",
"h1:czuvCXAd8CcszXh3k3E0N6tFfQkYSNbKXo99LG+yOIc=",
"h1:dPIwO6zTxYs6bHn5yf/w/1AExaHPppCLyTrWrdzH4RE=",
"h1:gqrhnYuIBzpUvgWHITNRUMLaDcKACGlrNfrbtbtO7KY=",
"h1:iVAGP8gRbZK0kJF7SiYJRt61wz0D5AF9q+WMsrAiBI0=",
"h1:kLi29SbGCU/Z/Ch0zuNdYFSKP3mHp8zfcOpZsIN/KIQ=",
"h1:mq2+1Q/5gWJYI8XOXQCeteUg0AP7VezPOGOjkkHRQso=",
"h1:rnWTWAOlJhHtD3kkU4Qfw0bg9ko/dlH822inxYLBmfQ=",
"zh:1286cee6fb63dbcb18f53077bbb5e5d132a4e4d9f006af4e8d8edfc08d6bcdc8",
"zh:204460dacc044bda019a4a18b398e094289500c36913c7c9457f432adf31b8b2",
"zh:214175d50773481cbeaf9c9004e4121a3a1c9686c79424ebdc8ff189dd057d3e",
"zh:22b17bceff61cc13ad04a399ba87521356a3a134d4687273727473ae9eccf5f1",
"zh:368867dac5525c411de7e38f2e27de0a71854d1750867322ff2b9321128c88fb",
"zh:5289b75f8370bdbc4c6051d55cf33d0b1bd25dc6d71bfbd39b360249a37f1501",
"zh:81cb676aa50c5777df8fc80d4e69c9012330ae751f5e6f12bf6074bfd2e7c496",
"zh:ab08aead10643b21aa6b51af562b50492e12b9dd0ab7dca27a05aa63209b7d66",
"zh:af25c210d0570cf61ef767b2545bf9f3fb909178135f0e5e14bec0c1c9d07a63",
"zh:bcad66f4830c97118fa793723e53f8a4d27ddd34ea969ff259408842c2238331",
"zh:ce3ed323d75ae905d975925fa98c7054a7514c81276a485fc37da8232b53e39f",
"zh:d481bc0ef0c87ab1969c17777f526b2f59f823432d676145134c41a6d29bd98e",
"zh:ea7ef88df2c3ca154d86238920636d52a3c9066c7467543d3fa45f1e52ec2f7b",
]
}
provider "registry.opentofu.org/hetznercloud/hcloud" {
version = "1.66.0"
constraints = "~> 1.54"
hashes = [
"h1:+R3t/5wqCb2ics5bNZ4JgtfaAa08Z6uRzXzNtwovVEY=",
"h1:9c1ECk3N9xhdQKz4NVCfYEFxebJVVeQZS3E6wrR/Hx8=",
"h1:EtYB/KjPN2cIJa6hpEaVc8q/3yf6e9IT6/uAybUGqps=",
"h1:GUupu6Fg5zFY9i4/3esiy8uHZnUrt+u2B6NaE4+8YqQ=",
"h1:JHehRJySEFRGYK+D6pFS0MTTDDRlGSQdSaKdEo0AemA=",
"h1:XF+SwyO1ttQJ0WXJ5O628O7cGk9U37sGn0o3LwWUnDQ=",
"h1:czuvCXAd8CcszXh3k3E0N6tFfQkYSNbKXo99LG+yOIc=",
"h1:dPIwO6zTxYs6bHn5yf/w/1AExaHPppCLyTrWrdzH4RE=",
"h1:gqrhnYuIBzpUvgWHITNRUMLaDcKACGlrNfrbtbtO7KY=",
"h1:iVAGP8gRbZK0kJF7SiYJRt61wz0D5AF9q+WMsrAiBI0=",
"h1:kLi29SbGCU/Z/Ch0zuNdYFSKP3mHp8zfcOpZsIN/KIQ=",
"h1:mq2+1Q/5gWJYI8XOXQCeteUg0AP7VezPOGOjkkHRQso=",
"h1:rnWTWAOlJhHtD3kkU4Qfw0bg9ko/dlH822inxYLBmfQ=",
"zh:1286cee6fb63dbcb18f53077bbb5e5d132a4e4d9f006af4e8d8edfc08d6bcdc8",
"zh:204460dacc044bda019a4a18b398e094289500c36913c7c9457f432adf31b8b2",
"zh:214175d50773481cbeaf9c9004e4121a3a1c9686c79424ebdc8ff189dd057d3e",
"zh:22b17bceff61cc13ad04a399ba87521356a3a134d4687273727473ae9eccf5f1",
"zh:368867dac5525c411de7e38f2e27de0a71854d1750867322ff2b9321128c88fb",
"zh:5289b75f8370bdbc4c6051d55cf33d0b1bd25dc6d71bfbd39b360249a37f1501",
"zh:81cb676aa50c5777df8fc80d4e69c9012330ae751f5e6f12bf6074bfd2e7c496",
"zh:ab08aead10643b21aa6b51af562b50492e12b9dd0ab7dca27a05aa63209b7d66",
"zh:af25c210d0570cf61ef767b2545bf9f3fb909178135f0e5e14bec0c1c9d07a63",
"zh:bcad66f4830c97118fa793723e53f8a4d27ddd34ea969ff259408842c2238331",
"zh:ce3ed323d75ae905d975925fa98c7054a7514c81276a485fc37da8232b53e39f",
"zh:d481bc0ef0c87ab1969c17777f526b2f59f823432d676145134c41a6d29bd98e",
"zh:ea7ef88df2c3ca154d86238920636d52a3c9066c7467543d3fa45f1e52ec2f7b",
]
}
+102
View File
@@ -0,0 +1,102 @@
# Root module: wires together the shared network and the three host modules
# (one per server in docs/architecture.md). Each host module owns its own
# firewall - see modules/<host>/main.tf. Persistent data lives in named Docker
# volumes backed up to S3 (see services/<host>/backup-docker-compose.yml), not
# on Hetzner volumes.
locals {
# SSH to dev/prod is only permitted from the vpn server's public IP: admins must
# tunnel through the VPN first, whose egress traffic is then sourced from this IP.
# vpn itself can't require tunneling through itself, so it keeps
# var.allowed_ssh_source_ips for direct SSH access.
vpn_ssh_source_ips = ["${module.vpn.ipv4}/32"]
# IDs for every named key in var.ssh_key_names - installed on every server.
ssh_key_ids = [for k in data.hcloud_ssh_key.main : k.id]
# Every A record currently referenced by a Traefik Host() rule anywhere under
# services/ - kept in sync with the actual compose files, not aspirational.
# "dev.luke-else.co.uk" is a *website* served from prod (a dev-tagged build of
# luke-else.co.uk), not the dev server - don't confuse the two.
dns_records = [
{ zone = "luke-else.co.uk", name = "@", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "dev", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "metarius", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "traefik", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "status", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "bitwarden", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "rd", value = module.prod.ipv4 },
{ zone = "luke-else.co.uk", name = "git", value = module.dev.ipv4 },
{ zone = "luke-else.co.uk", name = "traefik.cicd", value = module.dev.ipv4 },
{ zone = "luke-else.co.uk", name = "vpn", value = module.vpn.ipv4 },
{ zone = "luke-else.co.uk", name = "traefik.vpn", value = module.vpn.ipv4 },
]
# Microsoft 365 mail records for luke-else.co.uk - carried over from the
# domain's old DNS provider (pre-Hetzner), not previously applied here, so
# these are genuinely new to the Hetzner zone rather than something to
# import.
mail_records = [
{ zone = "luke-else.co.uk", name = "@", type = "MX", value = "0 lukeelse-co-uk01b.mail.protection.outlook.com." },
{ zone = "luke-else.co.uk", name = "@", type = "TXT", value = "\"v=spf1 include:spf.protection.outlook.com -all\"" },
{ zone = "luke-else.co.uk", name = "@", type = "TXT", value = "\"google-site-verification=aKLk5BATowKUwNbFcU5zyfu-AIyCW9W9PBStePRjH3U\"" },
{ zone = "luke-else.co.uk", name = "autodiscover", type = "CNAME", value = "autodiscover.outlook.com." },
]
}
module "network" {
source = "./modules/network"
ip_range = var.network_ip_range
network_zone = var.network_zone
subnet_ip_range = var.subnet_ip_range
}
module "dev" {
source = "./modules/dev"
server_type = var.dev_server_type
image = var.server_image
location = var.location
ssh_key_ids = local.ssh_key_ids
network_id = module.network.id
private_ip = var.dev_private_ip
allowed_ssh_source_ips = local.vpn_ssh_source_ips
network_ip_range = var.network_ip_range
# module.network.id alone doesn't guarantee the subnet exists yet, and a server
# can't join a network before it has a subnet.
depends_on = [module.network]
}
module "prod" {
source = "./modules/prod"
server_type = var.prod_server_type
image = var.server_image
location = var.location
ssh_key_ids = local.ssh_key_ids
network_id = module.network.id
private_ip = var.prod_private_ip
allowed_ssh_source_ips = local.vpn_ssh_source_ips
network_ip_range = var.network_ip_range
depends_on = [module.network]
}
module "vpn" {
source = "./modules/vpn"
server_type = var.vpn_server_type
image = var.server_image
location = var.location
ssh_key_ids = local.ssh_key_ids
allowed_ssh_source_ips = var.allowed_ssh_source_ips
}
module "dns" {
source = "./modules/dns"
zone_names = var.dns_zones
records = concat(local.dns_records, local.mail_records)
}
+66
View File
@@ -0,0 +1,66 @@
# dev: Gitea + Runner + Traefik (git.luke-else.co.uk, cicd.luke-else.co.uk)
# Port sources: services/dev/*docker-compose.yml (published ports) and services/todo.md
# (the documented UFW allow-list). dev has no firewall in docs/architecture.md, but we
# add one anyway for baseline safety - see conversation history.
resource "hcloud_firewall" "this" {
name = "dev-firewall"
rule { # server ssh - wired to the vpn server's public IP only, see infra/main.tf
direction = "in"
protocol = "tcp"
port = "22"
source_ips = var.allowed_ssh_source_ips
}
rule { # gitea ssh (git.luke-else.co.uk, published as 222:22)
direction = "in"
protocol = "tcp"
port = "222"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # Traefik http/https (git.luke-else.co.uk, cicd.luke-else.co.uk)
direction = "in"
protocol = "tcp"
port = "80"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "tcp"
port = "443"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # traffic from prod over the private network
direction = "in"
protocol = "tcp"
port = "1-65535"
source_ips = [var.network_ip_range]
}
rule {
direction = "in"
protocol = "udp"
port = "1-65535"
source_ips = [var.network_ip_range]
}
}
resource "hcloud_server" "this" {
name = "dev"
server_type = var.server_type
image = var.image
location = var.location
ssh_keys = var.ssh_key_ids
firewall_ids = [hcloud_firewall.this.id]
labels = { role = "dev" } # picked up by ansible/inventory/hcloud.yml
network {
network_id = var.network_id
ip = var.private_ip
}
}
+7
View File
@@ -0,0 +1,7 @@
output "ipv4" {
value = hcloud_server.this.ipv4_address
}
output "private_ipv4" {
value = var.private_ip
}
+39
View File
@@ -0,0 +1,39 @@
variable "server_type" {
description = "Server type for dev (Gitea + Runner)."
type = string
}
variable "image" {
description = "OS image used for the server."
type = string
}
variable "location" {
description = "Hetzner Cloud datacenter location."
type = string
}
variable "ssh_key_ids" {
description = "IDs of the Hetzner Cloud SSH keys to install on the server."
type = list(string)
}
variable "network_id" {
description = "ID of the private network to attach dev to."
type = string
}
variable "private_ip" {
description = "Private network IP for dev."
type = string
}
variable "allowed_ssh_source_ips" {
description = "CIDRs allowed to reach port 22 on dev. Set by the root module to the vpn server's public IP."
type = list(string)
}
variable "network_ip_range" {
description = "CIDR of the private network, allowed through the firewall for traffic from prod."
type = string
}
+37
View File
@@ -0,0 +1,37 @@
# One Hetzner DNS zone per domain in var.zone_names. Hetzner becomes
# authoritative for these once the domain's registrar NS records are pointed at
# the nameservers in the `nameservers` output - that step happens at whichever
# registrar the domain was bought through, and can't be done from here.
resource "hcloud_zone" "this" {
for_each = toset(var.zone_names)
name = each.value
mode = "primary"
}
locals {
# Group flat var.records entries by zone/name/type into one set of values
# each, since a Hetzner RRSet is keyed on (zone, name, type) and can hold
# several values (e.g. an SPF and a google-site-verification TXT both live
# under the same "@" TXT RRSet).
rrsets = {
for key, items in { for r in var.records : "${r.zone}/${r.name}/${r.type}" => r... } : key => {
zone = items[0].zone
name = items[0].name
type = items[0].type
ttl = items[0].ttl
values = [for i in items : i.value]
}
}
}
resource "hcloud_zone_rrset" "this" {
for_each = local.rrsets
zone = hcloud_zone.this[each.value.zone].id
name = each.value.name
type = each.value.type
ttl = each.value.ttl
records = [for v in each.value.values : { value = v }]
}
+60
View File
@@ -0,0 +1,60 @@
# Generalizing hcloud_zone_rrset.a (for_each keyed "zone/name", A records
# only) into hcloud_zone_rrset.this (keyed "zone/name/type", any record type)
# changed every existing instance's resource address. Without these, `tofu
# apply` would destroy and recreate every A record already in state instead of
# just adding the new mail records - these moved blocks make it a no-op rename
# for anything that already existed. Safe to delete once applied and merged.
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/@"]
to = hcloud_zone_rrset.this["luke-else.co.uk/@/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/dev"]
to = hcloud_zone_rrset.this["luke-else.co.uk/dev/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/metarius"]
to = hcloud_zone_rrset.this["luke-else.co.uk/metarius/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/traefik"]
to = hcloud_zone_rrset.this["luke-else.co.uk/traefik/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/status"]
to = hcloud_zone_rrset.this["luke-else.co.uk/status/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/bitwarden"]
to = hcloud_zone_rrset.this["luke-else.co.uk/bitwarden/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/rd"]
to = hcloud_zone_rrset.this["luke-else.co.uk/rd/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/git"]
to = hcloud_zone_rrset.this["luke-else.co.uk/git/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/traefik.cicd"]
to = hcloud_zone_rrset.this["luke-else.co.uk/traefik.cicd/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/vpn"]
to = hcloud_zone_rrset.this["luke-else.co.uk/vpn/A"]
}
moved {
from = hcloud_zone_rrset.a["luke-else.co.uk/traefik.vpn"]
to = hcloud_zone_rrset.this["luke-else.co.uk/traefik.vpn/A"]
}
+9
View File
@@ -0,0 +1,9 @@
output "zone_ids" {
description = "Map of domain name to Hetzner zone ID."
value = { for name, z in hcloud_zone.this : name => z.id }
}
output "nameservers" {
description = "Map of domain name to the nameservers Hetzner assigned it - point your registrar's NS records at these to delegate DNS."
value = { for name, z in hcloud_zone.this : name => z.authoritative_nameservers.assigned }
}
+24
View File
@@ -0,0 +1,24 @@
variable "zone_names" {
description = "Domains to create as Hetzner DNS zones."
type = list(string)
}
variable "records" {
description = <<-EOT
Records to create across the zones in var.zone_names. Use name = "@" for
the zone apex. Entries sharing the same zone/name/type become a single
Hetzner RRSet with multiple values (e.g. two TXT strings under the same
name) - Hetzner has no concept of separate same-type records otherwise.
For MX, embed the priority in value using standard zonefile syntax (e.g.
"10 mail.example.com."); Hetzner's API has no separate priority field.
For TXT, wrap value in escaped double quotes (e.g. "\"v=spf1 ...\"").
EOT
type = list(object({
zone = string
name = string
value = string
type = optional(string, "A")
ttl = optional(number, 300)
}))
default = []
}
+13
View File
@@ -0,0 +1,13 @@
# Private network shared by dev and prod (see docs/architecture.md "network" group).
# vpn is intentionally not attached to this network - it sits outside it in the diagram.
resource "hcloud_network" "main" {
name = "server-network"
ip_range = var.ip_range
}
resource "hcloud_network_subnet" "main" {
network_id = hcloud_network.main.id
type = "cloud"
network_zone = var.network_zone
ip_range = var.subnet_ip_range
}
+9
View File
@@ -0,0 +1,9 @@
output "id" {
value = hcloud_network.main.id
}
# Exposed so dependents can depend_on the subnet existing, not just the network -
# attaching a server to a network fails if the network has no subnet yet.
output "subnet_id" {
value = hcloud_network_subnet.main.id
}
+14
View File
@@ -0,0 +1,14 @@
variable "ip_range" {
description = "IP range of the private network shared by dev and prod."
type = string
}
variable "network_zone" {
description = "Hetzner Cloud network zone matching var.location."
type = string
}
variable "subnet_ip_range" {
description = "IP range of the network subnet shared by dev and prod."
type = string
}
+93
View File
@@ -0,0 +1,93 @@
# prod: Traefik, Websites, Bitwarden, RustDesk, status page
# Port sources: services/prod/*docker-compose.yml (published ports) and services/todo.md
# (the documented UFW allow-list).
resource "hcloud_firewall" "this" {
name = "prod-firewall"
rule { # server ssh - wired to the vpn server's public IP only, see infra/main.tf
direction = "in"
protocol = "tcp"
port = "22"
source_ips = var.allowed_ssh_source_ips
}
rule { # Traefik http/https (Websites, Bitwarden, status page)
direction = "in"
protocol = "tcp"
port = "80"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "tcp"
port = "443"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # rustdesk hbbs
direction = "in"
protocol = "tcp"
port = "21115"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "tcp"
port = "21116"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "udp"
port = "21116"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # rustdesk hbbr
direction = "in"
protocol = "tcp"
port = "21117"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "tcp"
port = "21119"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # traffic from dev over the private network
direction = "in"
protocol = "tcp"
port = "1-65535"
source_ips = [var.network_ip_range]
}
rule {
direction = "in"
protocol = "udp"
port = "1-65535"
source_ips = [var.network_ip_range]
}
}
resource "hcloud_server" "this" {
name = "prod"
server_type = var.server_type
image = var.image
location = var.location
ssh_keys = var.ssh_key_ids
firewall_ids = [hcloud_firewall.this.id]
labels = { role = "prod" } # picked up by ansible/inventory/hcloud.yml
network {
network_id = var.network_id
ip = var.private_ip
}
}
+7
View File
@@ -0,0 +1,7 @@
output "ipv4" {
value = hcloud_server.this.ipv4_address
}
output "private_ipv4" {
value = var.private_ip
}
+39
View File
@@ -0,0 +1,39 @@
variable "server_type" {
description = "Server type for prod (Traefik, Websites, Bitwarden, RustDesk, status page)."
type = string
}
variable "image" {
description = "OS image used for the server."
type = string
}
variable "location" {
description = "Hetzner Cloud datacenter location."
type = string
}
variable "ssh_key_ids" {
description = "IDs of the Hetzner Cloud SSH keys to install on the server."
type = list(string)
}
variable "network_id" {
description = "ID of the private network to attach prod to."
type = string
}
variable "private_ip" {
description = "Private network IP for prod."
type = string
}
variable "allowed_ssh_source_ips" {
description = "CIDRs allowed to reach port 22 on prod. Set by the root module to the vpn server's public IP."
type = list(string)
}
variable "network_ip_range" {
description = "CIDR of the private network, allowed through the firewall for traffic from dev."
type = string
}
+42
View File
@@ -0,0 +1,42 @@
# vpn: WireGuard (wg-easy) + Traefik. Not attached to the private network (see docs/architecture.md).
resource "hcloud_firewall" "this" {
name = "vpn-firewall"
rule { # server ssh
direction = "in"
protocol = "tcp"
port = "22"
source_ips = var.allowed_ssh_source_ips
}
rule { # Traefik http/https (traefik.vpn.luke-else.co.uk)
direction = "in"
protocol = "tcp"
port = "80"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule {
direction = "in"
protocol = "tcp"
port = "443"
source_ips = ["0.0.0.0/0", "::/0"]
}
rule { # WireGuard tunnel - always direct to this server's public IP, never via a load balancer
direction = "in"
protocol = "udp"
port = "51820"
source_ips = ["0.0.0.0/0", "::/0"]
}
}
resource "hcloud_server" "this" {
name = "vpn"
server_type = var.server_type
image = var.image
location = var.location
ssh_keys = var.ssh_key_ids
firewall_ids = [hcloud_firewall.this.id]
labels = { role = "vpn" } # picked up by ansible/inventory/hcloud.yml
}
+3
View File
@@ -0,0 +1,3 @@
output "ipv4" {
value = hcloud_server.this.ipv4_address
}
+24
View File
@@ -0,0 +1,24 @@
variable "server_type" {
description = "Server type for vpn (WireGuard/wg-easy + Traefik)."
type = string
}
variable "image" {
description = "OS image used for the server."
type = string
}
variable "location" {
description = "Hetzner Cloud datacenter location."
type = string
}
variable "ssh_key_ids" {
description = "IDs of the Hetzner Cloud SSH keys to install on the server."
type = list(string)
}
variable "allowed_ssh_source_ips" {
description = "CIDRs allowed to reach port 22 on vpn."
type = list(string)
}
+24
View File
@@ -0,0 +1,24 @@
output "dev_ipv4" {
value = module.dev.ipv4
}
output "dev_private_ipv4" {
value = module.dev.private_ipv4
}
output "prod_ipv4" {
value = module.prod.ipv4
}
output "prod_private_ipv4" {
value = module.prod.private_ipv4
}
output "vpn_ipv4" {
value = module.vpn.ipv4
}
output "dns_nameservers" {
description = "Nameservers Hetzner assigned each zone - point each domain's registrar NS records at these to delegate DNS to Hetzner."
value = module.dns.nameservers
}
+4
View File
@@ -0,0 +1,4 @@
data "hcloud_ssh_key" "main" {
for_each = toset(var.ssh_key_names)
name = each.value
}
+18
View File
@@ -0,0 +1,18 @@
# Copy to terraform.tfvars and fill in - terraform.tfvars itself is gitignored,
# never commit real values there.
#
# The Hetzner API token is NOT set here: export it as HCLOUD_TOKEN in your shell
# before running tofu plan/apply. The same token and SSH key are used again by
# Ansible afterwards - see ../ansible/README.md.
# Names of SSH keys already uploaded to your Hetzner Cloud project
# (Console > Security > SSH Keys). All of them are installed on every server. Required.
ssh_key_names = ["luke-else@PC", "luke-else@laptop"]
# Optional overrides - defaults live in variables.tf
# location = "nbg1"
# dev_server_type = "cx23"
# prod_server_type = "cx23"
# vpn_server_type = "cx23"
# allowed_ssh_source_ips = ["0.0.0.0/0"]
# dns_zones = ["luke-else.co.uk"]
+76
View File
@@ -0,0 +1,76 @@
variable "location" {
description = "Hetzner Cloud datacenter location for all servers."
type = string
default = "nbg1"
}
variable "network_zone" {
description = "Hetzner Cloud network zone matching var.location."
type = string
default = "eu-central"
}
variable "server_image" {
description = "OS image used for all servers."
type = string
default = "ubuntu-26.04"
}
variable "dev_server_type" {
description = "Server type for dev"
type = string
default = "cx23"
}
variable "prod_server_type" {
description = "Server type for prod"
type = string
default = "cx23"
}
variable "vpn_server_type" {
description = "Server type for vpn"
type = string
default = "cx23"
}
variable "network_ip_range" {
description = "IP range of the private network shared by dev and prod."
type = string
default = "10.0.0.0/16"
}
variable "subnet_ip_range" {
description = "IP range of the network subnet shared by dev and prod."
type = string
default = "10.0.1.0/24"
}
variable "dev_private_ip" {
description = "Private network IP for dev."
type = string
default = "10.0.1.10"
}
variable "prod_private_ip" {
description = "Private network IP for prod."
type = string
default = "10.0.1.11"
}
variable "ssh_key_names" {
description = "Names of SSH keys already uploaded to your Hetzner Cloud project (Console > Security > SSH Keys). All are installed on every server."
type = list(string)
}
variable "allowed_ssh_source_ips" {
description = "CIDRs allowed to reach port 22 on the vpn server. Narrow this to your own IP(s) once known. dev and prod don't use this - their SSH is restricted to the vpn server's own public IP instead (see infra/main.tf)."
type = list(string)
default = ["0.0.0.0/0", "::/0"]
}
variable "dns_zones" {
description = "Domains to manage as Hetzner DNS zones. Point each domain's registrar NS records at tofu output dns_nameservers for Hetzner to actually become authoritative."
type = list(string)
default = ["luke-else.co.uk"]
}
+13
View File
@@ -0,0 +1,13 @@
terraform {
required_version = ">= 1.6.0"
required_providers {
hcloud = {
source = "hetznercloud/hcloud"
version = "~> 1.54" # hcloud_zone / hcloud_zone_rrset (DNS) require >= 1.54.0
}
}
}
# Reads the token from the HCLOUD_TOKEN environment variable.
provider "hcloud" {}
-42
View File
@@ -1,42 +0,0 @@
#Script file for spinning down all docker-containers
cd ./Websites/
docker compose down
cd ..
cd ./Tracking/
docker compose down
cd ..
cd ./Development/Gitea/
docker compose down
cd ../..
cd ./Database/
docker compose down
cd ..
cd ./Bitwarden/
docker compose down
cd ..
cd ./Rustdesk/
docker compose down
cd ..
cd ./Misc/
docker compose down
cd ..
cd ./Watchtower/
docker compose down
cd ..
cd ./Traefik/
docker compose down
cd ..
docker rmi $(docker images -q)
docker system prune -f -a
docker volume prune -f -a
-39
View File
@@ -1,39 +0,0 @@
#Script file for spinning up all docker-containers
cd ./Traefik/
docker compose pull && docker compose up -d
cd ..
cd ./Development/Gitea/
docker compose pull && docker compose up -d
cd ../..
sleep 20 # Allow Gitea + registry to start up before starting the rest of the services
cd ./Watchtower/
docker compose pull && docker compose up -d
cd ..
cd ./Tracking/
docker compose pull && docker compose up -d
cd ..
cd ./Websites/
docker compose pull && docker compose up -d
cd ..
cd ./Database/
docker compose pull && docker compose up -d
cd ..
cd ./Bitwarden/
docker compose pull && docker compose up -d
cd ..
cd ./Rustdesk/
docker compose pull && docker compose up -d
cd ..
cd ./Misc/
docker compose pull && docker compose up -d
cd ..
+311 -24
View File
@@ -1,48 +1,335 @@
# Server Repository
# Server
This repository contains various scripts and configurations for managing Docker containers and services.
Infrastructure-as-code and service definitions for luke-else.co.uk's self-hosted server estate: three [Hetzner Cloud](https://www.hetzner.com/cloud/) VPS instances provisioned with [OpenTofu](https://opentofu.org/), each bootstrapped and deployed with [Ansible](https://www.ansible.com/), running a set of Docker Compose stacks behind [Traefik](https://traefik.io/traefik/).
<p align="center">
<img src="assets/images/main.png" width="70%">
</p>
## Scripts
## Contents
- `cicd-spindown.sh`: Spins down all CICD relevant Docker containers.
- `cicd-spinup.sh`: Spins up all CICD relevant Docker containers.
- `main-spindown.sh`: Spins down all Docker containers.
- `main-spinup.sh`: Spins up all Docker containers.
- [Architecture](#architecture)
- [Repository layout](#repository-layout)
- [Prerequisites](#prerequisites)
- [Control panel (`control.sh`)](#control-panel-controlsh)
- [Provisioning the infrastructure](#provisioning-the-infrastructure-infra)
- [Bootstrapping and deploying with Ansible](#bootstrapping-and-deploying-with-ansible-ansible)
- [Service inventory](#service-inventory)
- [First-time setup](#first-time-setup)
- [Development container](#development-container)
- [Security notes](#security-notes)
## Services
## Architecture
### Bitwarden
Three servers, one shared private network:
Configuration for Bitwarden is located in [Bitwarden/docker-compose.yml](Bitwarden/docker-compose.yml).
```mermaid
architecture-beta
group cloud(cloud)[Hetzner]
### Database
group network(cloud)[network] in cloud
Configuration for MongoDB is located in [Database/docker-compose.yml](Database/docker-compose.yml).
service dev(mdi:server)[dev] in network
### Misc
service prod(mdi:server)[prod] in network
service prodfirewall(mdi:firewall)[firewall] in cloud
Configuration for miscellaneous services is located in [Misc/docker-compose.yml](Misc/docker-compose.yml).
service vpn(mdi:server)[vpn] in cloud
service vpnfirewall(mdi:firewall)[firewall] in cloud
### Tracking
service gateway(mdi:web)[gateway] in cloud
Configuration for tracking services is located in [Tracking/docker-compose.yml](Tracking/docker-compose.yml).
service backups(mdi:bucket)[S3 Backups]
### Traefik
dev:L -- R:prod
Configuration for Traefik is located in [Traefik/docker-compose.yml](Traefik/docker-compose.yml).
prod:B -- T:prodfirewall
vpn:B -- T:vpnfirewall
### TraefikRunner
prodfirewall: L -- R: gateway
vpnfirewall: B -- T: gateway
Configuration for TraefikRunner is located in [TraefikRunner/docker-compose.yml](TraefikRunner/docker-compose.yml).
dev:T -- B:backups
prod:T -- B:backups
vpn:R -- L:backups
```
### Websites
`gateway` represents the public internet, not a provisioned resource. `backups` is the S3-compatible bucket every host's `backup-docker-compose.yml` pushes to and restores from - not a Hetzner resource, see [Persistent data lives in named Docker volumes, backed up to S3](#persistent-data-lives-in-named-docker-volumes-backed-up-to-s3).
Configuration for websites is located in [Websites/docker-compose.yml](Websites/docker-compose.yml).
| Server | Purpose | Network |
|---|---|---|
| `dev` | Gitea, CI runner, dev-facing Traefik | Private network only (no public firewall exposure beyond CI/CD) |
| `prod` | Public-facing websites, Bitwarden, RustDesk, status page, prod Traefik | Private network + public firewall |
| `vpn` | WireGuard (wg-easy) + its own Traefik | **Not** attached to the private network — kept isolated so a compromised VPN endpoint can't pivot to `dev`/`prod` |
## ToDo
Each host's stateful data lives in local named Docker volumes, backed up nightly to a shared S3 bucket and restored automatically on first boot - see below. No host has a Hetzner Cloud Volume attached any more.
See [todo.md](todo.md) for a list of tasks and configurations that need to be completed.
`dev` and `prod` share a private Hetzner network (`10.0.1.0/24` by default) so CI/CD on `dev` can reach deployment targets on `prod` without exposing that traffic publicly. `vpn` is deliberately kept off this network. Each server has its own Hetzner Cloud Firewall (see `infra/modules/<host>/main.tf`) that only opens the ports actually used by the compose stacks running on it, plus SSH restricted to `var.allowed_ssh_source_ips`.
## Repository layout
```
.
├── infra/ # OpenTofu (Terraform-compatible) config — provisions the 3 servers, network, firewalls, DNS
│ ├── main.tf # root module: wires network + dev/prod/vpn/dns modules together
│ ├── variables.tf # shared inputs (sizes, locations, IP ranges, SSH key names)
│ ├── outputs.tf # pass-through outputs from each host module
│ ├── ssh.tf # looks up each SSH key already uploaded to Hetzner Cloud
│ ├── versions.tf # provider requirements
│ ├── terraform.tfvars.example
│ └── modules/
│ ├── network/ # shared private network + subnet (used by dev and prod)
│ ├── dev/ # dev server + firewall, labeled role=dev
│ ├── prod/ # prod server + firewall, labeled role=prod
│ ├── vpn/ # vpn server + firewall (no private network), labeled role=vpn
│ └── dns/ # Hetzner DNS zones + A records for every domain in var.dns_zones
├── ansible/ # Ansible: bootstraps each server and deploys/starts services/<host>/ onto it
│ ├── inventory/hcloud.yml # dynamic inventory — queries the Hetzner API, groups by the role label above
│ ├── roles/
│ │ ├── bootstrap/ # Docker, deploy user, sshd hardening, unattended-upgrades
│ │ └── deploy/ # copies services/<host>/, renders .env (S3 backup credentials) + Runners/docker-compose.yml
│ └── playbooks/ # bootstrap.yml, deploy.yml, spinup.yml, spindown.yml, site.yml,
│ └── group_vars/ # deploy_user, service_group, backup_s3_*, etc. - must sit next to
│ # the playbooks for Ansible to auto-load it (see ansible/README.md)
├── services/ # Docker Compose stacks, grouped by which server they run on
│ ├── dev/ # Gitea + CI runner + Traefik + backup/restore (git.luke-else.co.uk, cicd.luke-else.co.uk)
│ ├── prod/ # Websites, Bitwarden, RustDesk, status page + Traefik + backup/restore
│ ├── vpn/ # WireGuard (wg-easy) + Traefik + backup/restore
│ └── todo.md # Outstanding manual setup/hardening tasks
├── docs/
│ └── architecture.md # Source of the architecture diagram above
├── .devcontainer/ # Git submodule: shared devcontainer for working on this repo (OpenTofu tooling)
└── assets/
```
Each of `services/dev`, `services/prod`, `services/vpn` follows the same convention: one `*-docker-compose.yml` (or subdirectory, e.g. `Runners/`) per logical service, a `backup-docker-compose.yml` + `restore.sh` pair for that host's S3 backup/restore (see below), plus a `spinup.sh` / `spindown.sh` pair that brings up or tears down every stack on that host in the right order. `infra/modules/` and `ansible/roles/deploy` both mirror this same dev/prod/vpn split, so a given host's cloud resources, bootstrap/deploy logic, and compose stacks are easy to find side by side.
OpenTofu and Ansible have a clean split: OpenTofu only ever provisions cloud resources (servers, network, firewalls, DNS) and never touches anything over SSH. Everything from "the server exists" onward — installing Docker, creating the `deploy` user, hardening SSH, copying `services/<host>/`, and running `spinup.sh`/`spindown.sh` — is Ansible's job. See [`ansible/README.md`](ansible/README.md) for the full rundown.
## Prerequisites
- A [Hetzner Cloud](https://console.hetzner.cloud/) project and API token
- One or more SSH keys uploaded to that project (Console → Security → SSH Keys) - all are installed on every server - plus the private key matching one of them available locally (Ansible uses it to bootstrap and deploy — see below)
- [OpenTofu](https://opentofu.org/docs/intro/install/) `>= 1.6.0`
- [Ansible](https://docs.ansible.com/ansible/latest/installation_guide/index.html) `>= 2.15` and the `hetzner.hcloud` collection (`ansible-galaxy collection install -r ansible/requirements.yml`)
- Ownership of the domains in `var.dns_zones` at whatever registrar they're bought through, so you can point their NS records at Hetzner (see [Managing DNS](#managing-dns) — the zones and records themselves are created for you)
- An S3-compatible bucket (e.g. [Hetzner Object Storage](https://www.hetzner.com/storage/object-storage/), AWS S3, or any other S3-compatible provider) and an access key/secret pair with read/write access to it - not provisioned by OpenTofu (the `hcloud` provider has no Object Storage resource), so create this yourself and pass it to Ansible as `BACKUP_S3_BUCKET`/`BACKUP_S3_ACCESS_KEY_ID`/`BACKUP_S3_SECRET_ACCESS_KEY` (and `BACKUP_S3_ENDPOINT` for anything other than real AWS S3) - see [Persistent data lives in named Docker volumes, backed up to S3](#persistent-data-lives-in-named-docker-volumes-backed-up-to-s3)
Docker + the Compose plugin, the non-root `deploy` user, and SSH hardening no longer need doing by hand — Ansible's `bootstrap` role handles all of that (see below).
## Control panel (`control.sh`)
If you'd rather not remember the exact command order, [`control.sh`](control.sh) at the repo root is an interactive menu that steps you through the whole pre-setup and bring-up. Run it from anywhere:
```sh
./control.sh
```
It offers a **Check prerequisites** option (tools, `HCLOUD_TOKEN`/`DEPLOY_USER`/`BACKUP_S3_*` env vars, `terraform.tfvars`, SSH key, the `hetzner.hcloud` collection), a **Configuration** section, individual OpenTofu (`init`/`plan`/`apply`/`output`/`destroy`) and Ansible (bootstrap/deploy/spinup/spindown per role) actions, and a **Guided full setup** that runs everything in the firewall-imposed order — vpn first, pause for you to connect the VPN, then dev/prod. It's only a wrapper around the same `tofu` and `ansible-playbook` commands documented below, so you can always fall back to running them by hand.
The **Configuration** section covers every variable both tools need:
- **Set variables** prompts for `HCLOUD_TOKEN`, `DEPLOY_USER` (the non-root user Ansible creates — defaults to `deploy`), `ANSIBLE_SSH_PRIVATE_KEY_FILE`, and the `BACKUP_S3_*` credentials, one at a time (leave blank to keep the current value, secrets are input-masked). They're saved to `.control.env` at the repo root — gitignored, never committed — which is loaded automatically the next time you run `control.sh`, so you only enter them once. This is scoped to `control.sh` itself, not your regular shell.
- **Copy terraform.tfvars.example -> terraform.tfvars** creates `infra/terraform.tfvars` for you and prompts for `ssh_key_names` (the one required field with no default) so you don't leave the example's placeholder key names in place by accident. Everything else in that file (`location`, server types, `dns_zones`, etc.) still has sensible defaults you can edit by hand afterwards - see [Provisioning the infrastructure](#provisioning-the-infrastructure-infra) below.
## Provisioning the infrastructure (`infra/`)
```sh
cd infra
export HCLOUD_TOKEN=your-hetzner-api-token # never commit this
cp terraform.tfvars.example terraform.tfvars
$EDITOR terraform.tfvars # set ssh_key_names at minimum
tofu init
tofu plan
tofu apply
```
This creates, via `module.network` / `module.dev` / `module.prod` / `module.vpn` / `module.dns` in `infra/main.tf`:
- `hcloud_network` + subnet, shared by `dev` and `prod` (`modules/network`)
- one `hcloud_server` + `hcloud_firewall` per host, scoped to the ports each host actually uses, each server labeled `role = "dev"/"prod"/"vpn"` for Ansible's dynamic inventory (`modules/dev`, `modules/prod`, `modules/vpn`)
- one Hetzner DNS zone per domain in `var.dns_zones`, plus every A record in [Service inventory](#service-inventory) (`modules/dns` — see [Managing DNS](#managing-dns))
No servers have an attached volume — persistent data lives in named Docker volumes backed up to S3 instead, see below.
Useful outputs: `tofu output dev_ipv4`, `tofu output prod_ipv4`, `tofu output vpn_ipv4`, `tofu output dns_nameservers`.
`terraform.tfvars` and any `*.tfvars` file are gitignored — never commit real values there. Defaults for server sizes, locations, and IP ranges live in `infra/variables.tf` and are passed down into the modules from `infra/main.tf`; override them per-environment via `terraform.tfvars`.
OpenTofu never connects to the servers over SSH — no provisioners, no `bootstrap.sh`, no copying `services/` — that's all Ansible now. See below.
### Managing DNS
`module.dns` (in `infra/modules/dns`) creates one Hetzner DNS zone per domain in `var.dns_zones` (default: `luke-else.co.uk`, `divine-couture.co.uk`, `snexo.co.uk`) and an `hcloud_zone_rrset` per record in `var.records` — kept in sync with `local.dns_records` (A records, one per hostname currently referenced by a Traefik `Host()` rule anywhere under `services/`, pointed at whichever of `dev`/`prod`/`vpn` actually serves it) and `local.mail_records` (see below) in `infra/main.tf`. Zones have `prevent_destroy = true`, matching the volumes — losing one deletes every record in it.
Creating the zone doesn't make Hetzner authoritative for the domain by itself: you still need to point that domain's NS records at Hetzner's nameservers at whichever registrar it's registered through. Run `tofu output dns_nameservers` after applying to get the exact nameservers per domain, and set those as the domain's NS records at the registrar. Propagation can take a while depending on the registrar and the domain's previous NS TTL.
Adding a new subdomain: add an entry to `local.dns_records` in `infra/main.tf` (zone, name, and the target `module.<host>.ipv4`) and re-run `tofu apply` — don't hand-create records in the Hetzner console, they'll drift from state. Note `name = "@"` is Hetzner's convention for a zone's apex record (e.g. bare `snexo.co.uk`), not an empty string.
Only A records for IPv4 are managed for hosts — none of the modules currently track servers' IPv6 addresses, so AAAA records aren't generated even though the firewalls already allow IPv6 traffic.
`var.records` isn't limited to A records: each entry takes an optional `type` (`A` by default, or `MX`/`TXT`/`CNAME`/etc.), and entries sharing the same `zone`/`name`/`type` are merged into a single Hetzner RRSet with multiple values (needed for e.g. two TXT strings under the same name) - see `infra/modules/dns/variables.tf` for the exact value format each type expects (MX embeds the priority in the value string, TXT values need escaped double quotes).
#### Mail records (Microsoft 365)
`local.mail_records` in `infra/main.tf` tracks the DNS records Microsoft 365 needs for `luke-else.co.uk` - the MX record, the SPF and Google-site-verification TXT records (merged into one `@` TXT RRSet), and the `autodiscover` CNAME. These carried over from the domain's DNS provider prior to Hetzner and were never applied to the Hetzner zone, so they're genuinely new records here (not something to `tofu import` - nothing to collide with).
**Not yet included:** a DMARC policy (`_dmarc` TXT) and DKIM signing (two `selector1._domainkey`/`selector2._domainkey` CNAMEs). DKIM's targets are generated per-tenant by Microsoft 365 (Defender portal → Email & collaboration → Policies & rules → DKIM) and can't be guessed - enable DKIM there first, then add the two CNAMEs it gives you to `local.mail_records`.
`infra/modules/dns/moved.tf` re-points every pre-existing A record at its new resource address, since generalizing the module to support non-A record types changed how they're keyed internally - without it, `tofu apply` would destroy and recreate every A record for no functional reason. It's a one-time migration aid: run `tofu plan` first to confirm the only real changes are the new mail records being created, then the file can be deleted after a successful apply.
## Bootstrapping and deploying with Ansible (`ansible/`)
Once `tofu apply` has created the servers, [`ansible/`](ansible/README.md) takes over everything else: installing Docker, creating the `deploy` user, hardening SSH, copying `services/<host>/` to each server, rendering `.env` (S3 backup credentials) and, for `dev`, `Runners/docker-compose.yml`, and running `spinup.sh`/`spindown.sh`. Full detail lives in [`ansible/README.md`](ansible/README.md); the short version:
```sh
cd ansible
ansible-galaxy collection install -r requirements.yml
export HCLOUD_TOKEN=your-hetzner-api-token # never commit this
# vpn first - its firewall accepts SSH from var.allowed_ssh_source_ips directly
ansible-playbook playbooks/site.yml -l role_vpn
# SSH to vpn as deploy and connect to the WireGuard tunnel it just started, then:
ansible-playbook playbooks/site.yml -l role_dev,role_prod
```
Hosts are discovered dynamically from the Hetzner API (`ansible/inventory/hcloud.yml`), grouped into `role_dev`/`role_prod`/`role_vpn` by the `role` label OpenTofu sets on each server — there's no static inventory file to keep in sync, and nothing here reads Terraform state.
### Persistent data lives in named Docker volumes, backed up to S3
Every stateful service's compose file mounts a plain named Docker volume rather than a bind mount to a Hetzner volume - `gitea_data:/data`, `bitwarden_data:/data/`, `letsencrypt_data:/letsencrypt`, and so on across `services/dev/*.yml`, `services/prod/*.yml`, and `services/vpn/*.yml`. Each volume is declared with an explicit `name:` in every compose file that mounts it (the same pattern already used for the shared `proxy` network), so Compose creates it once and every stack that references it reuses the same underlying volume regardless of which compose file happens to run first.
Each host's `backup-docker-compose.yml` defines two services against those same named volumes:
- **`backup`** — [`offen/docker-volume-backup`](https://github.com/offen/docker-volume-backup), running on a nightly cron (`0 3 * * *`), tars up every mounted volume and uploads it to `s3://$AWS_S3_BUCKET_NAME/$AWS_S3_PATH/<host>-<timestamp>.tar.gz`, keeping 14 days of history (`BACKUP_RETENTION_DAYS`). Containers labeled `docker-volume-backup.stop-during-backup=true` (Gitea, Bitwarden, RustDesk, Uptime Kuma, snexo, wg-easy) are stopped for the duration of the backup so their data is captured in a consistent state, then restarted automatically. Started by `spinup.sh` alongside everything else.
- **`restore`** — a one-shot container (`amazon/aws-cli` running `restore.sh`) that `spinup.sh` runs *before* anything else starts. It checks each of the host's named volumes; any that are already empty are populated from the most recent object under that host's S3 prefix, any that already contain data are left completely untouched. On a fresh server with no prior backup (or a bucket that doesn't exist yet), it's a no-op - services just start with empty volumes as before.
`AWS_S3_BUCKET_NAME`, `AWS_S3_PATH` (set to the host name, e.g. `dev`), `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, and `AWS_ENDPOINT` are rendered into `services/<host>/.env` by Ansible's `deploy` role from `BACKUP_S3_BUCKET`/`BACKUP_S3_ACCESS_KEY_ID`/`BACKUP_S3_SECRET_ACCESS_KEY`/`BACKUP_S3_ENDPOINT` in your shell environment (see [Prerequisites](#prerequisites)) — `docker compose` auto-loads `.env` from its working directory. `spinup.sh` on every host refuses to start anything if `.env` is missing.
The Gitea Actions runners are the one exception: their `/data` is a disposable local cache (runner identity re-registers with Gitea on every `spinup.sh` run anyway), so it's a plain named volume per runner (`gitea_runner_N_data`, baked directly into the rendered `Runners/docker-compose.yml`) that's deliberately left out of the backup/restore stack.
To trigger a backup or restore manually rather than waiting for cron or a restart:
```sh
cd services/<host>
docker compose -f backup-docker-compose.yml exec backup backup # force an immediate backup
docker compose -f backup-docker-compose.yml run --rm restore # restore any currently-empty volumes
```
### Gitea Actions runners
`dev` always runs three Gitea Actions runner containers, configured automatically on deploy. Re-running `ansible-playbook playbooks/deploy.yml -l role_dev` renders [`roles/deploy/templates/runners-docker-compose.yml.j2`](ansible/roles/deploy/templates/runners-docker-compose.yml.j2) straight onto the server as `services/dev/Runners/docker-compose.yml` — three `runner-N` services, each with its own container name and `/data` volume so their registrations don't collide. That file is generated on the remote host, so edit the template rather than hand-editing the rendered file (the count lives in a `{% raw %}{% set runner_count = 3 %}{% endraw %}` line at the top).
Rendering it doesn't start anything by itself — re-run `ansible-playbook playbooks/spinup.yml -l role_dev` afterwards to apply a template change.
Registration tokens are handled automatically, not baked into the generated file: `services/dev/spinup.sh` waits for Gitea to come up, runs `gitea actions generate-runner-token` inside the Gitea container, and writes the result to `Runners/.env`, which Compose loads automatically. There's no manual admin-UI step for this anymore.
### First deploy: bootstrapping order matters
`dev` and `prod`'s firewalls only accept SSH from `vpn`'s public IP (see [Security notes](#security-notes)), but `vpn`'s own WireGuard service isn't running until you deploy it — so on a from-scratch estate, Ansible can't reach `dev`/`prod` yet. Bring it up in this order:
1. `ansible-playbook playbooks/site.yml -l role_vpn` — bootstraps, deploys, and starts `vpn` only; its firewall allows SSH from `var.allowed_ssh_source_ips` directly.
2. Visit `https://vpn.luke-else.co.uk` and complete wg-easy's first-run setup wizard (admin username/password, then add a client peer and download its config/QR code) - see [Setting up WireGuard (wg-easy)](#setting-up-wireguard-wg-easy). Import that config into a WireGuard client and connect.
3. `ansible-playbook playbooks/site.yml -l role_dev,role_prod` — now that your machine is tunneled through `vpn`, its NATed egress IP matches the firewall rule and `dev`/`prod` become reachable.
If step 3 is run before you're connected to the VPN, Ansible will simply fail to connect — re-run it once connected.
## Deploying the services (`services/`)
After `ansible-playbook playbooks/deploy.yml` has copied `services/<host>/` to `/home/deploy/services/<host>` on the matching server, SSH in as `deploy` and run the matching script from inside that directory — or just use `ansible-playbook playbooks/spinup.yml` / `spindown.yml`, which do exactly this remotely (see [`ansible/README.md`](ansible/README.md)):
```sh
# on dev
./spinup.sh # Traefik → Gitea → (waits, generates a runner token) → Runners → Watchtower
./spindown.sh # reverse order, then prunes images/volumes
# on prod
./spinup.sh # Traefik, → Watchtower → status → websites → Bitwarden → RustDesk
./spindown.sh
# on vpn
./spinup.sh # Traefik → wg-easy → Watchtower
./spindown.sh
```
### Setting up WireGuard (wg-easy)
`vpn` runs [wg-easy](https://github.com/wg-easy/wg-easy) - WireGuard plus a web UI for managing peers - fronted by Traefik like everything else. `spinup.sh` only starts the container; wg-easy itself has no automated first-run setup here (see `services/vpn/vpn-docker-compose.yml`), so once it's up:
1. Visit `https://vpn.luke-else.co.uk` and complete the setup wizard - pick an admin username and a strong password (this account can view/rotate every client's private key, and it's reachable from the public internet). Enabling 2FA afterwards (wg-easy supports it) is worth doing since this isn't just a status page like the old OpenVPN setup was.
2. Set the host clients should connect to (`vpn.luke-else.co.uk`) and the port (`51820`, matching the firewall rule in `infra/modules/vpn/main.tf`) when prompted.
3. Add a client per device from the UI, then download its config file or scan the QR code directly into the WireGuard app.
wg-easy's own state (admin account, peers, keys) lives in the `wg_easy_data` named volume, backed up to S3 like everything else - see [Persistent data lives in named Docker volumes, backed up to S3](#persistent-data-lives-in-named-docker-volumes-backed-up-to-s3). Client configs themselves are only ever downloaded through the UI, not stored anywhere else.
If you ever need to force a re-sync without going through Ansible (e.g. testing a local edit before committing it), a manual `scp -r services/prod deploy@<prod_ipv4>:~/services` still works fine - `ansible-playbook playbooks/deploy.yml` will just overwrite it again next time it's run.
Each stack can also be managed individually with plain Compose, e.g.:
```sh
cd services/prod
docker compose -f bitwarden-docker-compose.yml up -d
docker compose -f bitwarden-docker-compose.yml down
```
All three hosts run [Watchtower](https://containrrr.dev/watchtower/) polling every 60s with cleanup enabled, so images are kept current automatically once deployed — the spinup scripts only need to be re-run after adding/removing a service or changing compose files.
Every public-facing service is fronted by its host's own Traefik instance, terminating TLS via Let's Encrypt (`tlschallenge`, port 80/443). Each stack joins an external `proxy` Docker network and opts in via `traefik.enable=true` labels rather than publishing ports directly (RustDesk and the Gitea SSH port are the deliberate exceptions, since they aren't HTTP).
## Service inventory
### `dev`
| Service | Compose file | Domain(s) |
|---|---|---|
| Traefik | `traefik-docker-compose.yml` | `traefik.cicd.luke-else.co.uk` |
| Gitea | `gitea-docker-compose.yml` | `git.luke-else.co.uk` (HTTP), SSH on `222` |
| Gitea Actions runner(s) | `Runners/docker-compose.yml` (generated — see [Scaling Gitea Actions runners](#scaling-gitea-actions-runners)) | N/A |
| Watchtower | `watchtower-docker-compose.yml` | — |
| Backup/restore | `backup-docker-compose.yml` | — |
### `prod`
| Service | Compose file | Domain(s) |
|---|---|---|
| Traefik | `traefik-docker-compose.yml` | `traefik.luke-else.co.uk` |
| Websites | `web-docker-compose.yml` | `luke-else.co.uk`, `dev.luke-else.co.uk`, `metarius.luke-else.co.uk`, `www.divine-couture.co.uk`, `snexo.co.uk` |
| Status page (Uptime Kuma) | `status-docker-compose.yml` | `status.luke-else.co.uk` |
| Bitwarden (Vaultwarden) | `bitwarden-docker-compose.yml` | `bitwarden.luke-else.co.uk` |
| RustDesk relay (hbbs/hbbr) | `rd-docker-compose.yml` | `rd.luke-else.co.uk`, ports `21115-21119` |
| Watchtower | `watchtower-docker-compose.yml` | — |
| Backup/restore | `backup-docker-compose.yml` | — |
### `vpn`
| Service | Compose file | Domain(s) |
|---|---|---|
| Traefik | `traefik-docker-compose.yml` | `traefik.vpn.luke-else.co.uk` |
| WireGuard (wg-easy) | `vpn-docker-compose.yml` | `vpn.luke-else.co.uk`, UDP `51820` |
| Watchtower | `watchtower-docker-compose.yml` | — |
| Backup/restore | `backup-docker-compose.yml` | — |
## First-time setup
A few things need manual attention before a stack is fully live — tracked in [`services/todo.md`](services/todo.md), summarized here:
- **General host hardening**: non-root user, Docker, and unattended-upgrades are now handled automatically by [Ansible's `bootstrap` role](ansible/roles/bootstrap/tasks/main.yml); UFW is the remaining manual item in `services/todo.md` (the Hetzner Cloud Firewalls already allowlist per-host ports — see [Security notes](#security-notes)).
## Development container
`.devcontainer` is a git submodule providing a ready-to-use OpenTofu development environment (VS Code + OpenTofu/Docker/Mermaid extensions). After cloning:
```sh
git submodule update --init --recursive
```
Then reopen the repo in VS Code with the Dev Containers extension.
## Security notes
- Real secrets (`HCLOUD_TOKEN`, `BACKUP_S3_ACCESS_KEY_ID`/`BACKUP_S3_SECRET_ACCESS_KEY`, `*.tfvars`, your SSH private key) must never be committed — see `.gitignore`. The S3 credentials end up in each host's `services/<host>/.env` (rendered by Ansible, mode `0600`, gitignored) - anyone with root on a host can read them, same trust boundary as everything else `deploy` can already do there.
- SSH to `dev` and `prod` is restricted to the `vpn` server's own public IP — you must be tunneled into the VPN to reach them over SSH. SSH to `vpn` itself is gated by `var.allowed_ssh_source_ips`; narrow this from the default `0.0.0.0/0` once you know your own egress IP(s).
- `vpn` is intentionally excluded from the private network so that a compromised VPN endpoint cannot reach `dev` or `prod` directly over it — SSH access still works because the VPN's egress traffic is NATed through its own public IP.
- Firewalls are allowlists scoped per host in `infra/modules/<host>/main.tf` — only ports actually used by that host's compose stacks (plus the cross-host private network range) are open.
- Every host's [Ansible bootstrap role](ansible/roles/bootstrap/tasks/main.yml) disables SSH password authentication, restricts root login to key-only, and creates a separate sudo user (`deploy_user`, see `ansible/playbooks/group_vars/all.yml`) for day-to-day access.
- DNS zones (`modules/dns`) carry both Hetzner's own `delete_protection` and Terraform's `prevent_destroy` — losing a zone takes every record in it with it, including for `divine-couture.co.uk` and `snexo.co.uk`, not just `luke-else.co.uk`.
+48
View File
@@ -0,0 +1,48 @@
services:
# Nightly backup of every named volume on this host to S3. Containers
# labeled docker-volume-backup.stop-during-backup=true (see
# gitea-docker-compose.yml) are stopped for the duration of the backup so
# their data is captured in a consistent state, then restarted.
backup:
image: offen/docker-volume-backup:v2
container_name: backup
environment:
BACKUP_FILENAME: "dev-%Y-%m-%dT%H-%M-%S.{{ .Extension }}"
BACKUP_CRON_EXPRESSION: "0 3 * * *"
BACKUP_RETENTION_DAYS: "14"
BACKUP_PRUNING_LEEWAY: "5m"
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- gitea_data:/backup/gitea_data:ro
- letsencrypt_data:/backup/letsencrypt_data:ro
- /var/run/docker.sock:/var/run/docker.sock:ro
restart: unless-stopped
# One-shot: restores empty volumes from the latest S3 backup, then exits.
# Run explicitly by spinup.sh before the services that own these volumes
# start - not started by `docker compose up`.
restore:
image: amazon/aws-cli:latest
container_name: backup-restore
entrypoint: ["/bin/bash", "/restore.sh"]
environment:
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- ./restore.sh:/restore.sh:ro
- gitea_data:/restore/gitea_data
- letsencrypt_data:/restore/letsencrypt_data
restart: "no"
volumes:
gitea_data:
name: gitea_data
letsencrypt_data:
name: letsencrypt_data
@@ -4,7 +4,7 @@ services:
image: gitea/gitea:latest
container_name: gitea
volumes:
- ./gitea:/data
- gitea_data:/data
- /etc/timezone:/etc/timezone:ro
- /etc/localtime:/etc/localtime:ro
networks:
@@ -29,7 +29,11 @@ services:
labels:
## Expose Gitea Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
# Stopped for the duration of each backup run so the sqlite db/repos are
# captured in a consistent state - see backup-docker-compose.yml.
- "docker-volume-backup.stop-during-backup=true"
- "traefik.http.middlewares.cors-gitea.headers.accesscontrolallowmethods=*"
- "traefik.http.middlewares.cors-gitea.headers.accesscontrolalloworiginlist=*"
- "traefik.http.middlewares.cors-gitea.headers.addvaryheader=true"
@@ -51,4 +55,8 @@ services:
networks:
proxy:
external: true
external: true
volumes:
gitea_data:
name: gitea_data
+54
View File
@@ -0,0 +1,54 @@
#!/bin/bash
# Restores this host's named volumes from the latest S3 backup, but only into
# volumes that are currently empty - never overwrites data that's already
# there. Run by spinup.sh via `docker compose -f backup-docker-compose.yml
# run --rm restore` before any service that owns one of the mounted volumes
# starts. Safe to run on a bucket/prefix that doesn't exist yet (fresh
# estate): it just leaves the empty volumes alone.
set -euo pipefail
S3_URI="s3://${AWS_S3_BUCKET_NAME}/${AWS_S3_PATH}"
ENDPOINT_ARGS=()
if [ -n "${AWS_ENDPOINT:-}" ]; then
ENDPOINT_ARGS=(--endpoint-url "https://${AWS_ENDPOINT}")
fi
empty_volumes=()
for dir in /restore/*/; do
name="$(basename "$dir")"
if [ -z "$(ls -A "$dir" 2>/dev/null)" ]; then
empty_volumes+=("$name")
fi
done
if [ "${#empty_volumes[@]}" -eq 0 ]; then
echo "All volumes already contain data - skipping restore."
exit 0
fi
echo "Empty volumes: ${empty_volumes[*]}"
latest="$(aws "${ENDPOINT_ARGS[@]}" s3 ls "${S3_URI}/" 2>/dev/null | awk '{print $4}' | sort | tail -n1 || true)"
if [ -z "$latest" ]; then
echo "No existing backup found at ${S3_URI} - starting with empty volumes."
exit 0
fi
echo "Restoring ${S3_URI}/${latest} into empty volumes only..."
tmp="$(mktemp -d)"
trap 'rm -rf "$tmp"' EXIT
aws "${ENDPOINT_ARGS[@]}" s3 cp "${S3_URI}/${latest}" "$tmp/backup.tar.gz"
mkdir "$tmp/extracted"
tar -xzf "$tmp/backup.tar.gz" -C "$tmp/extracted"
for name in "${empty_volumes[@]}"; do
if [ -d "$tmp/extracted/$name" ]; then
echo "Restoring $name..."
cp -a "$tmp/extracted/$name/." "/restore/$name/"
else
echo "No $name/ in backup - leaving volume empty."
fi
done
echo "Restore complete."
+17
View File
@@ -0,0 +1,17 @@
#!/bin/bash
# Script file for spinning down all dev docker-containers
cd "$(dirname "$0")"
cd Runners
docker compose down
cd ..
docker compose -f gitea-docker-compose.yml down
docker compose -f traefik-docker-compose.yml down
docker compose -f watchtower-docker-compose.yml down
docker compose -f backup-docker-compose.yml down
docker rmi $(docker images -q)
docker system prune -f -a
# Deliberately no `docker volume prune` - gitea_data/letsencrypt_data/etc. are
# named volumes holding real data, backed up to S3 but not disposable.
+40
View File
@@ -0,0 +1,40 @@
#!/bin/bash
# Script file for spinning up all dev docker-containers
set -e
cd "$(dirname "$0")"
if [ ! -f .env ]; then
echo "Missing .env (expected AWS_S3_BUCKET_NAME/AWS_ACCESS_KEY_ID/...) - run 'ansible-playbook playbooks/deploy.yml' in ansible/ first." >&2
exit 1
fi
echo "Restoring named volumes from S3 backup if any are empty..."
docker compose -f backup-docker-compose.yml pull restore
docker compose -f backup-docker-compose.yml run --rm restore
docker compose -f traefik-docker-compose.yml pull && docker compose -f traefik-docker-compose.yml up -d
docker compose -f gitea-docker-compose.yml pull && docker compose -f gitea-docker-compose.yml up -d
echo "Waiting for Gitea to become ready..."
for i in $(seq 1 30); do
if docker exec gitea gitea admin user list >/dev/null 2>&1; then
break
fi
sleep 2
done
# Generate a fresh Actions runner registration token straight from Gitea and
# hand it to the runner containers via .env (docker compose auto-loads this) -
# no manual copy/paste from the admin UI needed.
RUNNER_TOKEN=$(docker exec gitea gitea actions generate-runner-token)
echo "GITEA_RUNNER_REGISTRATION_TOKEN=$RUNNER_TOKEN" > Runners/.env
cd Runners
docker compose pull && docker compose up -d
cd ..
docker compose -f watchtower-docker-compose.yml pull && docker compose -f watchtower-docker-compose.yml up -d
docker compose -f backup-docker-compose.yml pull backup
docker compose -f backup-docker-compose.yml up -d backup
@@ -15,7 +15,7 @@ services:
- "80:80"
- "443:443"
volumes:
- "./letsencrypt:/letsencrypt"
- "letsencrypt_data:/letsencrypt"
- "/var/run/docker.sock:/var/run/docker.sock:ro"
networks:
- proxy
@@ -30,11 +30,13 @@ services:
- "traefik.http.routers.traefik.rule=Host(`traefik.cicd.luke-else.co.uk`)"
- "traefik.http.routers.traefik.entrypoints=websecure"
- "traefik.http.routers.traefik.service=api@internal"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
- "traefik.http.routers.traefik.middlewares=traefik-auth"
- "traefik.http.middlewares.traefik-auth.basicauth.users=user:$$2y$$05$$s/vPphFtSO2fWJR7SYkEb.90UwPDRM3aOKqgOF/rme/3fUQ5tvpTS"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
name: proxy
name: proxy
volumes:
letsencrypt_data:
name: letsencrypt_data
+59
View File
@@ -0,0 +1,59 @@
services:
# Nightly backup of every named volume on this host to S3. Containers
# labeled docker-volume-backup.stop-during-backup=true (see
# bitwarden-docker-compose.yml, rd-docker-compose.yml,
# status-docker-compose.yml, web-docker-compose.yml) are stopped for the
# duration of the backup so their data is captured in a consistent state,
# then restarted.
backup:
image: offen/docker-volume-backup:v2
container_name: backup
environment:
BACKUP_FILENAME: "prod-%Y-%m-%dT%H-%M-%S.{{ .Extension }}"
BACKUP_CRON_EXPRESSION: "0 3 * * *"
BACKUP_RETENTION_DAYS: "14"
BACKUP_PRUNING_LEEWAY: "5m"
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- bitwarden_data:/backup/bitwarden_data:ro
- rustdesk_data:/backup/rustdesk_data:ro
- uptime_kuma_data:/backup/uptime_kuma_data:ro
- snexo_data:/backup/snexo_data:ro
- letsencrypt_data:/backup/letsencrypt_data:ro
- /var/run/docker.sock:/var/run/docker.sock:ro
restart: unless-stopped
# One-shot: restores empty volumes from the latest S3 backup, then exits.
# Run explicitly by spinup.sh before the services that own these volumes
# start - not started by `docker compose up`.
restore:
image: amazon/aws-cli:latest
container_name: backup-restore
entrypoint: [ "/bin/bash", "/restore.sh" ]
environment:
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- ./restore.sh:/restore.sh:ro
- bitwarden_data:/restore/bitwarden_data
- rustdesk_data:/restore/rustdesk_data
- uptime_kuma_data:/restore/uptime_kuma_data
- letsencrypt_data:/restore/letsencrypt_data
restart: "no"
volumes:
bitwarden_data:
name: bitwarden_data
rustdesk_data:
name: rustdesk_data
uptime_kuma_data:
name: uptime_kuma_data
letsencrypt_data:
name: letsencrypt_data
@@ -5,13 +5,16 @@ services:
image: "vaultwarden/server:latest"
container_name: vaultwarden
volumes:
- ./bitwarden/:/data/
- bitwarden_data:/data/
networks:
- proxy
labels:
## Expose Bitwarden Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
# Stopped for the duration of each backup run - see backup-docker-compose.yml.
- "docker-volume-backup.stop-during-backup=true"
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.bitwarden-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.bitwarden-insecure.rule=Host(`bitwarden.luke-else.co.uk`)"
@@ -24,4 +27,8 @@ services:
networks:
proxy:
external: true
external: true
volumes:
bitwarden_data:
name: bitwarden_data
@@ -6,7 +6,7 @@ services:
restart: unless-stopped
volumes:
- ./data:/root
- rustdesk_data:/root
networks:
- proxy
@@ -20,6 +20,9 @@ services:
labels:
- "traefik.enable=true"
# Stopped for the duration of each backup run - see backup-docker-compose.yml.
- "docker-volume-backup.stop-during-backup=true"
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.rustdesk-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.rustdesk-insecure.rule=Host(`rd.luke-else.co.uk`)"
@@ -39,7 +42,7 @@ services:
restart: unless-stopped
volumes:
- ./data:/root
- rustdesk_data:/root
networks:
- proxy
@@ -50,4 +53,8 @@ services:
networks:
proxy:
external: true
external: true
volumes:
rustdesk_data:
name: rustdesk_data
+54
View File
@@ -0,0 +1,54 @@
#!/bin/bash
# Restores this host's named volumes from the latest S3 backup, but only into
# volumes that are currently empty - never overwrites data that's already
# there. Run by spinup.sh via `docker compose -f backup-docker-compose.yml
# run --rm restore` before any service that owns one of the mounted volumes
# starts. Safe to run on a bucket/prefix that doesn't exist yet (fresh
# estate): it just leaves the empty volumes alone.
set -euo pipefail
S3_URI="s3://${AWS_S3_BUCKET_NAME}/${AWS_S3_PATH}"
ENDPOINT_ARGS=()
if [ -n "${AWS_ENDPOINT:-}" ]; then
ENDPOINT_ARGS=(--endpoint-url "https://${AWS_ENDPOINT}")
fi
empty_volumes=()
for dir in /restore/*/; do
name="$(basename "$dir")"
if [ -z "$(ls -A "$dir" 2>/dev/null)" ]; then
empty_volumes+=("$name")
fi
done
if [ "${#empty_volumes[@]}" -eq 0 ]; then
echo "All volumes already contain data - skipping restore."
exit 0
fi
echo "Empty volumes: ${empty_volumes[*]}"
latest="$(aws "${ENDPOINT_ARGS[@]}" s3 ls "${S3_URI}/" 2>/dev/null | awk '{print $4}' | sort | tail -n1 || true)"
if [ -z "$latest" ]; then
echo "No existing backup found at ${S3_URI} - starting with empty volumes."
exit 0
fi
echo "Restoring ${S3_URI}/${latest} into empty volumes only..."
tmp="$(mktemp -d)"
trap 'rm -rf "$tmp"' EXIT
aws "${ENDPOINT_ARGS[@]}" s3 cp "${S3_URI}/${latest}" "$tmp/backup.tar.gz"
mkdir "$tmp/extracted"
tar -xzf "$tmp/backup.tar.gz" -C "$tmp/extracted"
for name in "${empty_volumes[@]}"; do
if [ -d "$tmp/extracted/$name" ]; then
echo "Restoring $name..."
cp -a "$tmp/extracted/$name/." "/restore/$name/"
else
echo "No $name/ in backup - leaving volume empty."
fi
done
echo "Restore complete."
+16
View File
@@ -0,0 +1,16 @@
#!/bin/bash
# Script file for spinning down all prod docker-containers
cd "$(dirname "$0")"
docker compose -f web-docker-compose.yml down
docker compose -f status-docker-compose.yml down
docker compose -f bitwarden-docker-compose.yml down
docker compose -f rd-docker-compose.yml down
docker compose -f watchtower-docker-compose.yml down
docker compose -f traefik-docker-compose.yml down
docker compose -f backup-docker-compose.yml down
docker rmi $(docker images -q)
docker system prune -f -a
# Deliberately no `docker volume prune` - bitwarden_data/rustdesk_data/etc. are
# named volumes holding real data, backed up to S3 but not disposable.
+26
View File
@@ -0,0 +1,26 @@
#!/bin/bash
# Script file for spinning up all prod docker-containers
set -e
cd "$(dirname "$0")"
if [ ! -f .env ]; then
echo "Missing .env (expected AWS_S3_BUCKET_NAME/AWS_ACCESS_KEY_ID/...) - run 'ansible-playbook playbooks/deploy.yml' in ansible/ first." >&2
exit 1
fi
echo "Restoring named volumes from S3 backup if any are empty..."
docker compose -f backup-docker-compose.yml pull restore
docker compose -f backup-docker-compose.yml run --rm restore
docker compose -f traefik-docker-compose.yml pull && docker compose -f traefik-docker-compose.yml up -d
sleep 20 # Allow Traefik + registry auth to settle before starting the rest of the services
docker compose -f watchtower-docker-compose.yml pull && docker compose -f watchtower-docker-compose.yml up -d
docker compose -f status-docker-compose.yml pull && docker compose -f status-docker-compose.yml up -d
docker compose -f web-docker-compose.yml pull && docker compose -f web-docker-compose.yml up -d
docker compose -f bitwarden-docker-compose.yml pull && docker compose -f bitwarden-docker-compose.yml up -d
docker compose -f rd-docker-compose.yml pull && docker compose -f rd-docker-compose.yml up -d
docker compose -f backup-docker-compose.yml pull backup
docker compose -f backup-docker-compose.yml up -d backup
+32
View File
@@ -0,0 +1,32 @@
services:
status:
image: louislam/uptime-kuma:latest
container_name: status
volumes:
- uptime_kuma_data:/app/data
networks:
- proxy
labels:
## Expose uptime-kuma Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
# Stopped for the duration of each backup run - see backup-docker-compose.yml.
- "docker-volume-backup.stop-during-backup=true"
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.status-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.status-insecure.rule=Host(`status.luke-else.co.uk`)"
- "traefik.http.routers.status-insecure.entrypoints=web"
- "traefik.http.routers.status.rule=Host(`status.luke-else.co.uk`)"
- "traefik.http.routers.status.entrypoints=websecure"
- "traefik.http.routers.status.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
external: true
volumes:
uptime_kuma_data:
name: uptime_kuma_data
@@ -8,7 +8,6 @@ services:
- "--providers.docker.exposedbydefault=false"
- "--entrypoints.web.address=:80"
- "--entrypoints.websecure.address=:443"
- "--entrypoints.mongo.address=:27017"
- "--entrypoints.web.transport.respondingTimeouts.readTimeout=120s"
- "--entrypoints.websecure.transport.respondingTimeouts.readTimeout=120s"
- "--entrypoints.web.transport.respondingTimeouts.writeTimeout=120s"
@@ -21,7 +20,7 @@ services:
- "443:443"
- "27017:27017"
volumes:
- "./letsencrypt:/letsencrypt"
- "letsencrypt_data:/letsencrypt"
- "/var/run/docker.sock:/var/run/docker.sock:ro"
networks:
- proxy
@@ -36,11 +35,13 @@ services:
- "traefik.http.routers.traefik.rule=Host(`traefik.luke-else.co.uk`)"
- "traefik.http.routers.traefik.entrypoints=websecure"
- "traefik.http.routers.traefik.service=api@internal"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
- "traefik.http.routers.traefik.middlewares=traefik-auth"
- "traefik.http.middlewares.traefik-auth.basicauth.users=user:$$2y$$05$$s/vPphFtSO2fWJR7SYkEb.90UwPDRM3aOKqgOF/rme/3fUQ5tvpTS"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
name: proxy
name: proxy
volumes:
letsencrypt_data:
name: letsencrypt_data
@@ -0,0 +1,8 @@
services:
watchtower:
image: nickfedor/watchtower
volumes:
- /var/run/docker.sock:/var/run/docker.sock
environment:
- WATCHTOWER_CLEANUP=true
- WATCHTOWER_POLL_INTERVAL=60
@@ -1,6 +1,6 @@
services:
#Websites luke-else.co.uk (8000) snexo.co.uk (8001) divine-couture.co.uk (80) wmgzon.luke-else.co.uk (8080)
#Websites luke-else.co.uk (8000) snexo.co.uk (8001) divine-couture.co.uk (80) wmgzon.luke-else.co.uk (8080)
luke-else:
image: git.luke-else.co.uk/luke-else/luke-else.co.uk
container_name: luke-else
@@ -9,7 +9,7 @@ services:
labels:
## Expose luke-else Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.personal-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.personal-insecure.rule=Host(`luke-else.co.uk`)"
@@ -28,7 +28,7 @@ services:
labels:
## Expose luke-else Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.personal-dev-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.personal-dev-insecure.rule=Host(`dev.luke-else.co.uk`)"
@@ -47,7 +47,7 @@ services:
labels:
## Expose metarius Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.metarius-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.metarius-insecure.rule=Host(`metarius.luke-else.co.uk`)"
@@ -66,7 +66,7 @@ services:
labels:
## Expose divine-couture Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.divine-couture-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.divine-couture-insecure.rule=Host(`www.divine-couture.co.uk`)"
@@ -77,27 +77,6 @@ services:
- "traefik.http.routers.divine-couture.tls.certresolver=myresolver"
restart: unless-stopped
snexo:
image: "php:apache"
container_name: snexo
volumes:
- ./snexo.co.uk/:/var/www/html
networks:
- proxy
labels:
## Expose Snexo Through Trefik ##
- "traefik.enable=true" # <== Enable traefik to proxy this container
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.snexo-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.snexo-insecure.rule=Host(`snexo.co.uk`)"
- "traefik.http.routers.snexo-insecure.entrypoints=web"
- "traefik.http.routers.snexo.rule=Host(`snexo.co.uk`)"
- "traefik.http.routers.snexo.entrypoints=websecure"
- "traefik.http.routers.snexo.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
external: true
-32
View File
@@ -3,13 +3,6 @@
## General
- Setup non root user
- UFW should be setup to keep VPS secure and only allow for:
- https
- http
- ssh
- ftp
- 27017
- "21115:21115/tcp"
- "21116:21116/tcp"
@@ -17,8 +10,6 @@
- "21117:21117/tcp"
- "21119:21119/tcp"
- Install SSH keys
- Setup unattended upgrades
- Install docker, docker-compose and apache utils.
## Traefik + TraefikRunner + Traefik VPN
@@ -30,19 +21,6 @@
- Ensure that ports are assigned correctly for the system
# Gitea Runner
- Ensure that a registration token has been setup before continuing
## Tracking
Create a .env file with the following content:
```sh
ACKEE_USERNAME=luke-else
ACKEE_PASSWORD=XXX
```
## Websites
- Ensure website files are copied over
@@ -52,13 +30,3 @@ ACKEE_PASSWORD=XXX
- Ensure that all data is fully encrypted during transfer.
- Ensure that ports are assigned correctly for the system
## Rustdesk
- No additional setup required bar the ports
## Database
- Ensure that mysql root password, user and default database are updated.
- Ensure that mongo root password, and user are updated.
- Ensure database ports are correctly assigned and do not have to pass through traefik.
+48
View File
@@ -0,0 +1,48 @@
services:
# Nightly backup of every named volume on this host to S3. Containers
# labeled docker-volume-backup.stop-during-backup=true (see
# vpn-docker-compose.yml) are stopped for the duration of the backup so
# their data is captured in a consistent state, then restarted.
backup:
image: offen/docker-volume-backup:v2
container_name: backup
environment:
BACKUP_FILENAME: "vpn-%Y-%m-%dT%H-%M-%S.{{ .Extension }}"
BACKUP_CRON_EXPRESSION: "0 3 * * *"
BACKUP_RETENTION_DAYS: "14"
BACKUP_PRUNING_LEEWAY: "5m"
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- wg_easy_data:/backup/wg_easy_data:ro
- letsencrypt_data:/backup/letsencrypt_data:ro
- /var/run/docker.sock:/var/run/docker.sock:ro
restart: unless-stopped
# One-shot: restores empty volumes from the latest S3 backup, then exits.
# Run explicitly by spinup.sh before the services that own these volumes
# start - not started by `docker compose up`.
restore:
image: amazon/aws-cli:latest
container_name: backup-restore
entrypoint: ["/bin/bash", "/restore.sh"]
environment:
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME}
AWS_S3_PATH: ${AWS_S3_PATH}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY}
AWS_ENDPOINT: ${AWS_ENDPOINT}
volumes:
- ./restore.sh:/restore.sh:ro
- wg_easy_data:/restore/wg_easy_data
- letsencrypt_data:/restore/letsencrypt_data
restart: "no"
volumes:
wg_easy_data:
name: wg_easy_data
letsencrypt_data:
name: letsencrypt_data
+54
View File
@@ -0,0 +1,54 @@
#!/bin/bash
# Restores this host's named volumes from the latest S3 backup, but only into
# volumes that are currently empty - never overwrites data that's already
# there. Run by spinup.sh via `docker compose -f backup-docker-compose.yml
# run --rm restore` before any service that owns one of the mounted volumes
# starts. Safe to run on a bucket/prefix that doesn't exist yet (fresh
# estate): it just leaves the empty volumes alone.
set -euo pipefail
S3_URI="s3://${AWS_S3_BUCKET_NAME}/${AWS_S3_PATH}"
ENDPOINT_ARGS=()
if [ -n "${AWS_ENDPOINT:-}" ]; then
ENDPOINT_ARGS=(--endpoint-url "https://${AWS_ENDPOINT}")
fi
empty_volumes=()
for dir in /restore/*/; do
name="$(basename "$dir")"
if [ -z "$(ls -A "$dir" 2>/dev/null)" ]; then
empty_volumes+=("$name")
fi
done
if [ "${#empty_volumes[@]}" -eq 0 ]; then
echo "All volumes already contain data - skipping restore."
exit 0
fi
echo "Empty volumes: ${empty_volumes[*]}"
latest="$(aws "${ENDPOINT_ARGS[@]}" s3 ls "${S3_URI}/" 2>/dev/null | awk '{print $4}' | sort | tail -n1 || true)"
if [ -z "$latest" ]; then
echo "No existing backup found at ${S3_URI} - starting with empty volumes."
exit 0
fi
echo "Restoring ${S3_URI}/${latest} into empty volumes only..."
tmp="$(mktemp -d)"
trap 'rm -rf "$tmp"' EXIT
aws "${ENDPOINT_ARGS[@]}" s3 cp "${S3_URI}/${latest}" "$tmp/backup.tar.gz"
mkdir "$tmp/extracted"
tar -xzf "$tmp/backup.tar.gz" -C "$tmp/extracted"
for name in "${empty_volumes[@]}"; do
if [ -d "$tmp/extracted/$name" ]; then
echo "Restoring $name..."
cp -a "$tmp/extracted/$name/." "/restore/$name/"
else
echo "No $name/ in backup - leaving volume empty."
fi
done
echo "Restore complete."
+13
View File
@@ -0,0 +1,13 @@
#!/bin/bash
# Script file for spinning down all vpn docker-containers
cd "$(dirname "$0")"
docker compose -f vpn-docker-compose.yml down
docker compose -f traefik-docker-compose.yml down
docker compose -f watchtower-docker-compose.yml down
docker compose -f backup-docker-compose.yml down
docker rmi $(docker images -q)
docker system prune -f -a
# Deliberately no `docker volume prune` - wg_easy_data/letsencrypt_data are
# named volumes holding real data, backed up to S3 but not disposable.
+20
View File
@@ -0,0 +1,20 @@
#!/bin/bash
# Script file for spinning up all vpn docker-containers
set -e
cd "$(dirname "$0")"
if [ ! -f .env ]; then
echo "Missing .env (expected AWS_S3_BUCKET_NAME/AWS_ACCESS_KEY_ID/...) - run 'ansible-playbook playbooks/deploy.yml' in ansible/ first." >&2
exit 1
fi
echo "Restoring named volumes from S3 backup if any are empty..."
docker compose -f backup-docker-compose.yml pull restore
docker compose -f backup-docker-compose.yml run --rm restore
docker compose -f traefik-docker-compose.yml pull && docker compose -f traefik-docker-compose.yml up -d
docker compose -f vpn-docker-compose.yml pull && docker compose -f vpn-docker-compose.yml up -d
docker compose -f watchtower-docker-compose.yml pull && docker compose -f watchtower-docker-compose.yml up -d
docker compose -f backup-docker-compose.yml pull backup
docker compose -f backup-docker-compose.yml up -d backup
@@ -15,7 +15,7 @@ services:
- "80:80"
- "443:443"
volumes:
- "./letsencrypt:/letsencrypt"
- "letsencrypt_data:/letsencrypt"
- "/var/run/docker.sock:/var/run/docker.sock:ro"
networks:
- proxy
@@ -30,11 +30,13 @@ services:
- "traefik.http.routers.traefik.rule=Host(`traefik.vpn.luke-else.co.uk`)"
- "traefik.http.routers.traefik.entrypoints=websecure"
- "traefik.http.routers.traefik.service=api@internal"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
- "traefik.http.routers.traefik.middlewares=traefik-auth"
- "traefik.http.middlewares.traefik-auth.basicauth.users=user:$$2y$$05$$s/vPphFtSO2fWJR7SYkEb.90UwPDRM3aOKqgOF/rme/3fUQ5tvpTS"
- "traefik.http.routers.traefik.tls.certresolver=myresolver"
restart: unless-stopped
networks:
proxy:
name: proxy
name: proxy
volumes:
letsencrypt_data:
name: letsencrypt_data
+44
View File
@@ -0,0 +1,44 @@
services:
wg-easy:
image: ghcr.io/wg-easy/wg-easy:15
cap_add:
- NET_ADMIN
- SYS_MODULE # lets the container modprobe wireguard if the host kernel doesn't already have it loaded
sysctls:
- net.ipv4.ip_forward=1
- net.ipv4.conf.all.src_valid_mark=1
- net.ipv6.conf.all.disable_ipv6=0
- net.ipv6.conf.all.forwarding=1
- net.ipv6.conf.default.forwarding=1
ports:
- 51820:51820/udp # WireGuard tunnel - the web UI (51821) only needs to be reachable via Traefik below, not published directly
networks:
- proxy
volumes:
- /lib/modules:/lib/modules:ro
- wg_easy_data:/etc/wireguard
labels:
## Expose the wg-easy web UI through Traefik ##
- "traefik.enable=true"
# Stopped for the duration of each backup run - see backup-docker-compose.yml.
- "docker-volume-backup.stop-during-backup=true"
- "traefik.http.middlewares.redirect-web-secure.redirectscheme.scheme=https"
- "traefik.http.routers.vpn-insecure.middlewares=redirect-web-secure"
- "traefik.http.routers.vpn-insecure.rule=Host(`vpn.luke-else.co.uk`)"
- "traefik.http.routers.vpn-insecure.entrypoints=web"
- "traefik.http.routers.vpn.rule=Host(`vpn.luke-else.co.uk`)"
- "traefik.http.routers.vpn.entrypoints=websecure"
- "traefik.http.routers.vpn.tls.certresolver=myresolver"
- "traefik.http.services.vpn.loadbalancer.server.port=51821"
restart: unless-stopped
networks:
proxy:
external: true
volumes:
wg_easy_data:
name: wg_easy_data
@@ -0,0 +1,8 @@
services:
watchtower:
image: nickfedor/watchtower
volumes:
- /var/run/docker.sock:/var/run/docker.sock
environment:
- WATCHTOWER_CLEANUP=true
- WATCHTOWER_POLL_INTERVAL=60
-17
View File
@@ -1,17 +0,0 @@
#Script file for spinning down all CICD relevant docker-containers
cd ./VPN/
docker compose down
cd ..
cd ./TraefikVPN/
docker compose down
cd ..
cd ./Watchtower/
docker compose down
cd ..
docker rmi $(docker images -q)
docker system prune -f -a
docker volume prune -f -a
-13
View File
@@ -1,13 +0,0 @@
#Script file for spinning up all docker-containers
cd ./TraefikRunner/
docker compose pull && docker compose up -d
cd ..
cd ./VPN/
docker compose pull && docker compose up -d
cd ..
cd ./Watchtower/
docker compose pull && docker compose up -d
cd ..