From 85c35258959d3ef0706bc19b6c75f3bed4c9e32c Mon Sep 17 00:00:00 2001 From: Richard Date: Tue, 28 Jul 2026 17:41:48 +0200 Subject: [PATCH] initial(main): Add files for basic gridscale deployment of 2 hosts with config --- .env.example | 2 + .gitignore | 8 ++ README.md | 210 +++++++++++++++++++++++++++++++++++ inventory.tf | 30 +++++ main.tf | 13 +++ outputs.tf | 24 ++++ playbook.yml | 117 +++++++++++++++++++ resources.tf | 165 +++++++++++++++++++++++++++ templates/Caddyfile.j2 | 13 +++ templates/alert_rules.yml.j2 | 20 ++++ templates/prometheus.yml.j2 | 10 ++ variables.tf | 22 ++++ 12 files changed, 634 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 README.md create mode 100644 inventory.tf create mode 100644 main.tf create mode 100644 outputs.tf create mode 100644 playbook.yml create mode 100644 resources.tf create mode 100644 templates/Caddyfile.j2 create mode 100644 templates/alert_rules.yml.j2 create mode 100644 templates/prometheus.yml.j2 create mode 100644 variables.tf diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..a044abd --- /dev/null +++ b/.env.example @@ -0,0 +1,2 @@ +export TF_VAR_gridscale_uuid="your-gridscale-uuid-here" +export TF_VAR_gridscale_token="your-gridscale-token-here" diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..824e280 --- /dev/null +++ b/.gitignore @@ -0,0 +1,8 @@ +.env +.terraform* +terraform.tfstate* +inventory.ini +.python-version +.venv +pyproject.toml +uv.lock \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..d672c30 --- /dev/null +++ b/README.md @@ -0,0 +1,210 @@ +# Technical Writeup: OVH Gridscale, Caddy & Prometheus Infrastructure + +## Overview + +This document provides a comprehensive technical writeup of the infrastructure deployment and configuration management workflow. The project provisions cloud resources on **OVH / Gridscale** using **Terraform** and automates host configuration using **Ansible**. + +--- + +## 1. Initial Infrastructure Setup & Technical Caveats + +### 1.1 Terraform Provider & Resource Foundation +The initial configuration (`main.tf`, `variables.tf`, `resources.tf`) established authentication with the official Gridscale provider (`gridscale/gridscale`). Infrastructure resources were defined using the **Debian 13** image template and SSH key authentication. + +### 1.2 Key Failures & Solutions + +During initial testing and deployment, several technical hurdles were encountered and resolved: + +1. **SSH Connection Timeouts (Public Network Requirement)**: + * **Issue**: Initial attempts to connect via SSH to the server's public IPv4 address timed out (`ssh: connect to host ... port 22: Operation timed out`). + * **Root Cause & Fix**: Gridscale instances require explicit attachment to the public network. Because no dynamic network UUID data lookup was provided in the task specification, the public network UUID (`5557a73b-31ee-4b1f-aa15-7789ad6ae04c`) was hardcoded directly into the server's network block in `resources.tf`. + +2. **Unpowered Server Instances**: + * **Issue**: Newly created instances remained offline and unreachable after `terraform apply`. + * **Root Cause & Fix**: Gridscale creates servers in an unpowered state by default. Setting `power = true` explicitly in `gridscale_server` resources ensured servers automatically boot upon creation. + +3. **Ansible SSH Host Key Verification Failures**: + * **Issue**: Ansible fact-gathering failed with `Host key verification failed`. + * **Root Cause & Fix**: Cloud IPs change dynamically on fresh deployments. Added `ansible_ssh_extra_args='-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null'` to the `[all:vars]` block in `inventory.tf` to bypass interactive host key verification prompts seamlessly. + +--- + +## 2. Network Topology & Firewall Security + +### 2.1 Multi-Server Topology +The architecture consists of two dedicated nodes: +* **`caddy01`**: Web server hosting Caddy. +* **`monitoring01`**: Telemetry node hosting Prometheus. + +### 2.2 Dual-Interface Network Isolation & Inbound Firewall Rules +To secure management endpoints and telemetry data, instances were configured with dual network interfaces: + +1. **Public Network**: + * Attached to Gridscale public network UUID `5557a73b-31ee-4b1f-aa15-7789ad6ae04c`. + * **Firewall Rules (`rules_v4_in`, `rules_v6_in`)**: + * Allow TCP port `22` (SSH). + * Allow TCP port `80` (HTTP). + * Block all other inbound public traffic. + +2. **Private Network (`gridscale_network.network_internal`)**: + * Internal DHCP subnet: `192.168.121.0/27` (Gateway: `192.168.121.1`, DNS: `192.168.121.2`, Reserved: `192.168.121.0/31`). + * **Firewall Rules**: + * On `caddy01`, allows inbound TCP port `2019` (Caddy Admin API / Metrics) **only** from the internal CIDR `192.168.121.0/27`. Public access to port 2019 is strictly denied at the firewall layer. + +--- + +## 3. Dynamic Ansible Inventory Generation + +### 3.1 Automated Inventory Build (`inventory.tf`) +To eliminate manual IP tracking, Terraform's `local_file` resource generates `inventory.ini` dynamically upon `terraform apply`. + +### 3.2 Dynamic IP Interpolation +- Extracts auto-assigned private IP addresses from resource state: `${gridscale_server.server_caddy.network[1].auto_assigned_ip}`. +- Automatically exports host groups (`[caddy]`, `[monitoring]`) and internal IP variables (`privat_ip_caddy`, `privat_ip_monitoring`) for Ansible playbooks. +- Exposes outputs `caddy_public_ip`, `monitoring_public_ip`, `caddy_private_ip`, and `monitoring_private_ip` in `outputs.tf`. + +--- + +## 4. Ansible Playbook Orchestration (`playbook.yml`) + +### 4.1 Base System Maintenance & Idempotent Reboot (`all`) +* **Package Updates**: Executes `apt update` and `upgrade: dist`. +* **Idempotent Reboot Check**: Uses `ansible.builtin.stat` to check for `/var/run/reboot-required`. The reboot task triggers *only* when required (`when: reboot_required_file.stat.exists == true`), avoiding unnecessary reboot delays during repeat runs. +* **Prerequisites**: Installs `debian-keyring`, `debian-archive-keyring`, `apt-transport-https`, and `curl`. + +### 4.2 Caddy Web Server Configuration (`caddy`) +* **Official Deb822 Repository**: Per Caddy official documentation, configured Caddy's Debian repository using `ansible.builtin.deb822_repository` (`https://dl.cloudsmith.io/public/caddy/stable/deb/debian` with official GPG key verification). +* **Metrics & Admin API Security**: + * Configured `Caddyfile` with `/metrics` and bound the Admin API to `:2019`. + * Set `origins http://{{ privat_ip_caddy }}:2019` so Caddy validates origin headers and only processes authorized internal requests. +* **Service Handlers**: Uses Ansible handlers to restart the `caddy` service automatically upon `Caddyfile` updates. + +### 4.3 Prometheus Monitoring Setup (`monitoring`) +* **Installation**: Installs the `prometheus` package on `monitoring01`. +* **Private Scrape Target**: Configured `prometheus.yml` to scrape Caddy metrics target `{{ privat_ip_caddy }}:2019` strictly over the internal private network (`192.168.121.0/27`). +* **Service Handlers**: Triggers `Restart Prometheus service` automatically when `prometheus.yml` changes. + +--- + +## 5. Execution Guide + +### 5.1 Provision Infrastructure (Terraform) +```bash +# Initialize Terraform providers +terraform init + +# Apply infrastructure changes +terraform apply -auto-approve +``` + +### 5.2 Run Ansible Provisioning + +```bash +# Provision Caddy Web Server +uv run ansible-playbook -i inventory.ini -l caddy playbook.yml + +# Provision Monitoring Server +uv run ansible-playbook -i inventory.ini -l monitoring playbook.yml + +# Provision All Servers +uv run ansible-playbook -i inventory.ini playbook.yml +``` + +--- + +## 6. Modular Jinja2 Templating & Prometheus Alerting Rules + +### 6.1 Refactoring Inline Configurations to Jinja2 Templates +To improve maintainability and separate configuration data from playbook logic, inline file content blocks in `playbook.yml` were refactored into dedicated Jinja2 template files located in the `templates/` directory: + +* **[templates/Caddyfile.j2](file:///Users/doctor/Git/ovh_gridscale_test_task/templates/Caddyfile.j2)**: Contains Caddy server block definitions and metrics setup. +* **[templates/prometheus.yml.j2](file:///Users/doctor/Git/ovh_gridscale_test_task/templates/prometheus.yml.j2)**: Contains global scrape configuration and imports rule files via `rule_files: [/etc/prometheus/alert_rules.yml]`. +* **[templates/alert_rules.yml.j2](file:///Users/doctor/Git/ovh_gridscale_test_task/templates/alert_rules.yml.j2)**: Defines Prometheus alert rules for monitoring Caddy service availability and error rates. + +The playbook tasks were updated to use `ansible.builtin.template` instead of `ansible.builtin.copy`. + +### 6.2 Variable Naming Refactoring +Standardized variable naming from `privat_ip_*` to `private_ip_*` across: +* `inventory.tf` (Terraform dynamic inventory builder) +* `inventory.ini` (Generated inventory) +* `playbook.yml` & Jinja2 templates (`{{ private_ip_caddy }}`) + +### 6.3 Prometheus Alert Rules & Raw Jinja Escaping +Configured alerting rules in `templates/alert_rules.yml.j2` using Jinja `{% raw %}` ... `{% endraw %}` blocks to prevent Ansible from misinterpreting Prometheus template variables (such as `{{ $labels.instance }}` and `{{ $value }}`): + +1. **`Caddy Down`**: Triggers a critical alert if `up{job="caddy"} == 0` for 1 minute. +2. **`CaddyHighHTTP4xxErrorRateService`**: Triggers a critical alert if the 4xx HTTP error rate exceeds 5% over a 3-minute window (with total request count > 0). + * *Source*: Rule specification adapted from [Awesome Prometheus Alerts (Caddy Rules)](https://samber.github.io/awesome-prometheus-alerts/rules/proxies-load-balancers-and-service-meshes/caddy/). + +### 6.4 Verification +Verified playbook execution and rule rendering: +```bash +uv run ansible-playbook -i inventory.ini -l monitoring playbook.yml +``` + +### 6.5 Automated Playbook Execution via Terraform (`null_resource`) +To fully automate the end-to-end deployment, a `null_resource` provisioner was integrated into `inventory.tf`. It executes automatically as the final step of `terraform apply` after server creation and inventory file generation: + +```hcl +resource "null_resource" "ansible_provisioner" { + depends_on = [ + local_file.ansible_inventory, + gridscale_server.server_caddy, + gridscale_server.server_monitoring + ] + + provisioner "local-exec" { + command = "uv run ansible-playbook -i ${local_file.ansible_inventory.filename} playbook.yml" + } +} +``` + +--- + +## 7. End-to-End Environment Setup & Deployment Workflow + +### 7.1 Environment Prerequisites & Credentials (`.env`) +Copy the `.env.example` template and export your Gridscale API credentials into your shell session: + +```bash +# Copy template environment file +cp .env.example .env + +# Edit .env with your credentials, then export variables +source .env +``` + +### 7.2 Python Environment Initialization & Dependency Setup (`uv`) +Initialize Python environment and install Ansible using `uv`: + +```bash +# Initialize uv virtual environment and lockfile +uv init --bare + +# Add Ansible dependency +uv add "ansible>=14.2.0" +``` + +### 7.3 Infrastructure Provisioning & Automated Execution +Initialize Terraform and apply the configuration. Terraform will provision the servers, dynamically construct `inventory.ini`, and automatically trigger the Ansible playbook via the `null_resource`: + +```bash +# Initialize Terraform providers +terraform init + +# Apply infrastructure changes (automatically provisions servers & runs Ansible) +terraform apply -auto-approve +``` + +*(Optional)* Execute Ansible playbooks manually via `uv`: +```bash +# Run Ansible playbook for all hosts +uv run ansible-playbook -i inventory.ini playbook.yml + +# Target specific host groups +uv run ansible-playbook -i inventory.ini -l caddy playbook.yml +uv run ansible-playbook -i inventory.ini -l monitoring playbook.yml +``` + + + diff --git a/inventory.tf b/inventory.tf new file mode 100644 index 0000000..5d40761 --- /dev/null +++ b/inventory.tf @@ -0,0 +1,30 @@ +resource "local_file" "ansible_inventory" { + filename = "${path.module}/inventory.ini" + file_permission = "0644" + content = <<-EOT + [all:vars] + ansible_ssh_extra_args='-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null' + private_ip_caddy='${gridscale_server.server_caddy.network[1].auto_assigned_ip}' + private_ip_monitoring='${gridscale_server.server_monitoring.network[1].auto_assigned_ip}' + + [caddy] + ${var.caddy_hostname} ansible_host=${gridscale_ipv4.public_ipv4.ip} ansible_user=root + + [monitoring] + ${var.monitoring_hostname} ansible_host=${gridscale_ipv4.public_ipv4_monitoring.ip} ansible_user=root + EOT +} + +resource "null_resource" "ansible_provisioner" { + depends_on = [ + local_file.ansible_inventory, + gridscale_server.server_caddy, + gridscale_server.server_monitoring + ] + + provisioner "local-exec" { + # Added 60 seconds delay for the servers to boot up + command = "sleep 60 && uv run ansible-playbook -i ${local_file.ansible_inventory.filename} playbook.yml" + } +} + diff --git a/main.tf b/main.tf new file mode 100644 index 0000000..9c12022 --- /dev/null +++ b/main.tf @@ -0,0 +1,13 @@ +terraform { + required_providers { + gridscale = { + source = "gridscale/gridscale" + } + } +} + + +provider "gridscale" { + uuid = var.gridscale_uuid + token = var.gridscale_token +} diff --git a/outputs.tf b/outputs.tf new file mode 100644 index 0000000..77926cf --- /dev/null +++ b/outputs.tf @@ -0,0 +1,24 @@ +output "caddy_public_ip" { + description = "Public IPv4 address of the Caddy server" + value = gridscale_ipv4.public_ipv4.ip +} + +output "monitoring_public_ip" { + description = "Public IPv4 address of the Monitoring server" + value = gridscale_ipv4.public_ipv4_monitoring.ip +} + +output "caddy_private_ip" { + description = "Private IPv4 address of the Caddy server" + value = gridscale_server.server_caddy.network[1].auto_assigned_ip +} + +output "monitoring_private_ip" { + description = "Private IPv4 address of the Monitoring server" + value = gridscale_server.server_monitoring.network[1].auto_assigned_ip +} + +output "ansible_inventory" { + description = "Generated Ansible inventory content" + value = local_file.ansible_inventory.content +} diff --git a/playbook.yml b/playbook.yml new file mode 100644 index 0000000..8f1f7da --- /dev/null +++ b/playbook.yml @@ -0,0 +1,117 @@ +--- +- name: Configure Servers + hosts: all + become: true + + tasks: + - name: Ping host to verify connection + ansible.builtin.ping: + + - name: Update apt cache and upgrade system packages + ansible.builtin.apt: + update_cache: true + upgrade: dist + cache_valid_time: 3600 + + - name: Check if a reboot is required. + ansible.builtin.stat: + path: /var/run/reboot-required + get_checksum: no + register: reboot_required_file + + - name: Reboot server and wait for it to finish + when: reboot_required_file.stat.exists == true + ansible.builtin.reboot: + msg: "Rebooting post system update" + connect_timeout: 5 + reboot_timeout: 300 + pre_reboot_delay: 0 + post_reboot_delay: 15 + test_command: uptime + + - name: Install prerequisites + ansible.builtin.apt: + pkg: + - debian-keyring + - debian-archive-keyring + - apt-transport-https + - curl + state: present + +- name: Configure Caddy Server + hosts: caddy + become: true + tags: + - caddy + + handlers: + - name: Restart Caddy service + ansible.builtin.service: + name: caddy + state: restarted + enabled: true + + tasks: + - name: Add Caddy's official Repository + ansible.builtin.deb822_repository: + name: caddy + uris: https://dl.cloudsmith.io/public/caddy/stable/deb/debian + signed_by: https://dl.cloudsmith.io/public/caddy/stable/gpg.key + suites: + - any-version + components: + - main + + - name: Update apt cache to ensure we have the latest version of Caddy + ansible.builtin.apt: + update_cache: true + + - name: Install Caddy + ansible.builtin.apt: + pkg: + - caddy + state: present + + - name: Create Caddyfile from template + notify: Restart Caddy service + ansible.builtin.template: + src: Caddyfile.j2 + dest: /etc/caddy/Caddyfile + mode: "0644" + + + + +- name: Configure Monitoring Server + hosts: monitoring + become: true + tags: + - monitoring + + handlers: + - name: Restart Prometheus service + ansible.builtin.service: + name: prometheus + state: restarted + enabled: true + + tasks: + - name: Install Prometheus + ansible.builtin.apt: + pkg: + - prometheus + state: present + + - name: Configure Prometheus Alert Rules from template + notify: Restart Prometheus service + ansible.builtin.template: + src: alert_rules.yml.j2 + dest: /etc/prometheus/alert_rules.yml + mode: "0644" + + - name: Configure Prometheus from template + notify: Restart Prometheus service + ansible.builtin.template: + src: prometheus.yml.j2 + dest: /etc/prometheus/prometheus.yml + mode: "0644" \ No newline at end of file diff --git a/resources.tf b/resources.tf new file mode 100644 index 0000000..7e793f0 --- /dev/null +++ b/resources.tf @@ -0,0 +1,165 @@ +resource "gridscale_network" "network_internal"{ + name = "network_internal" + dhcp_active = true + dhcp_gateway = "192.168.121.1" + dhcp_dns = "192.168.121.2" + dhcp_range = "192.168.121.0/27" + dhcp_reserved_subnet = ["192.168.121.0/31"] +} + +resource "gridscale_ipv4" "public_ipv4" { + name = "public_ipv4" +} + +resource "gridscale_ipv6" "public_ipv6" { + name = "public_ipv6" +} + +resource "gridscale_ipv4" "public_ipv4_monitoring" { + name = "public_ipv4_monitoring" +} + +resource "gridscale_ipv6" "public_ipv6_monitoring" { + name = "public_ipv6_monitoring" +} + +data "gridscale_template" "template_debian_13" { + name = "Debian 13" +} + +resource "gridscale_sshkey" "sshkey_richard" { + name = "Richard SSH Key" + sshkey = file("~/.ssh/id_ed25519.pub") +} + +resource "gridscale_storage" "storage_caddy" { + name = "storage_caddy" + storage_type = "storage" + capacity = 20 + template { + sshkeys = [gridscale_sshkey.sshkey_richard.id] + template_uuid = data.gridscale_template.template_debian_13.id + hostname = var.caddy_hostname + } +} + +resource "gridscale_server" "server_caddy" { + name = var.caddy_hostname + cores = 2 + memory = 1 + power = true + storage { + object_uuid = gridscale_storage.storage_caddy.id + } + network { + # Public network + object_uuid = "5557a73b-31ee-4b1f-aa15-7789ad6ae04c" + rules_v4_in { + order = 0 + protocol = "tcp" + action = "accept" + dst_port = 22 + comment = "Allow SSH access" + } + rules_v4_in { + order = 10 + protocol = "tcp" + action = "accept" + dst_port = 80 + comment = "Allow HTTP access" + } + rules_v6_in { + order = 0 + protocol = "tcp" + action = "accept" + dst_port = 22 + comment = "Allow SSHv6 access" + } + rules_v6_in { + order = 10 + protocol = "tcp" + action = "accept" + dst_port = 80 + comment = "Allow HTTPv6 access" + } + } + network { + # Private network + object_uuid = gridscale_network.network_internal.id + rules_v4_in { + order = 0 + protocol = "tcp" + action = "accept" + dst_port = 2019 + comment = "Allow Prometheus access" + src_cidr = "192.168.121.0/27" + } + } + ipv4 = gridscale_ipv4.public_ipv4.id + ipv6 = gridscale_ipv6.public_ipv6.id + timeouts { + create = "10m" + } +} + +resource "gridscale_storage" "storage_monitoring" { + name = "storage_monitoring" + storage_type = "storage" + capacity = 20 + template { + sshkeys = [gridscale_sshkey.sshkey_richard.id] + template_uuid = data.gridscale_template.template_debian_13.id + hostname = var.monitoring_hostname + } +} + +resource "gridscale_server" "server_monitoring" { + name = var.monitoring_hostname + cores = 2 + memory = 1 + power = true + storage { + object_uuid = gridscale_storage.storage_monitoring.id + } + network { + # Public Network + object_uuid = "5557a73b-31ee-4b1f-aa15-7789ad6ae04c" + rules_v4_in { + order = 0 + protocol = "tcp" + action = "accept" + dst_port = 22 + comment = "Allow SSH access" + } + rules_v4_in { + order = 10 + protocol = "tcp" + action = "accept" + dst_port = 80 + comment = "Allow HTTP access" + } + rules_v6_in { + order = 0 + protocol = "tcp" + action = "accept" + dst_port = 22 + comment = "Allow SSHv6 access" + } + rules_v6_in { + order = 10 + protocol = "tcp" + action = "accept" + dst_port = 80 + comment = "Allow HTTPv6 access" + } + } + network { + # Private Network + object_uuid = gridscale_network.network_internal.id + } + ipv4 = gridscale_ipv4.public_ipv4_monitoring.id + ipv6 = gridscale_ipv6.public_ipv6_monitoring.id + timeouts { + create = "10m" + } +} \ No newline at end of file diff --git a/templates/Caddyfile.j2 b/templates/Caddyfile.j2 new file mode 100644 index 0000000..b1cd3fd --- /dev/null +++ b/templates/Caddyfile.j2 @@ -0,0 +1,13 @@ +{ + metrics /metrics { + per_host + observe_catchall_hosts + } + admin :2019 { + origins http://{{ private_ip_caddy }}:2019 + } +} +:80 { + root * /usr/share/caddy + file_server +} diff --git a/templates/alert_rules.yml.j2 b/templates/alert_rules.yml.j2 new file mode 100644 index 0000000..9e5e79b --- /dev/null +++ b/templates/alert_rules.yml.j2 @@ -0,0 +1,20 @@ +{% raw %} +groups: + - name: demo + rules: + - alert: Caddy Down + expr: up{job="caddy"} == 0 + for: 1m + labels: + severity: critical + + # 5% 4xx rate is a rough default. Client-error rates vary widely by application (bots, API misuse, rate limits, short-lived token expiry, malformed clients) — adjust based on your baseline. + - alert: CaddyHighHTTP4xxErrorRateService + expr: sum(rate(caddy_http_request_duration_seconds_count{code=~"4.."}[3m])) by (instance) / sum(rate(caddy_http_request_duration_seconds_count[3m])) by (instance) * 100 > 5 and sum(rate(caddy_http_request_duration_seconds_count[3m])) by (instance) > 0 + for: 1m + labels: + severity: critical + annotations: + summary: Caddy high HTTP 4xx error rate service (instance {{ $labels.instance }}) + description: "Caddy service 4xx error rate is above 5%\n VALUE = {{ $value }}\n LABELS = {{ $labels }}" +{% endraw %} \ No newline at end of file diff --git a/templates/prometheus.yml.j2 b/templates/prometheus.yml.j2 new file mode 100644 index 0000000..e6a9ff8 --- /dev/null +++ b/templates/prometheus.yml.j2 @@ -0,0 +1,10 @@ +global: + scrape_interval: 15s # default is 1 minute + +rule_files: + - /etc/prometheus/alert_rules.yml + +scrape_configs: + - job_name: caddy + static_configs: + - targets: ['{{ private_ip_caddy }}:2019'] \ No newline at end of file diff --git a/variables.tf b/variables.tf new file mode 100644 index 0000000..de97e23 --- /dev/null +++ b/variables.tf @@ -0,0 +1,22 @@ +variable "gridscale_uuid" { + type = string + description = "OVH Gridscale UUID" +} + +variable "gridscale_token" { + type = string + description = "OVH Gridscale API Token" + sensitive = true +} + +variable "caddy_hostname" { + type = string + description = "Caddy hostname" + default = "caddy01" +} + +variable "monitoring_hostname" { + type = string + description = "Monitoring hostname" + default = "monitoring01" +} \ No newline at end of file