Skip to content

Latest commit

 

History

2 Commits

Folders and files

NameName
Last commit message
Last commit date
 
 
 
 

Repository files navigation

Awesome Cloud & DevOps Engineering Roadmap 2026 ☁️

Production-grade cloud architecture patterns, Infrastructure-as-Code templates, CI/CD pipelines, containerization guides, and curated learning paths for AWS, Azure, GCP, and multi-cloud engineers.

License AWS Docker Terraform Platform PRs Welcome


📑 Table of Contents

  1. Cloud & DevOps Engineer Roadmap
  2. Docker & Containerization
  3. Kubernetes Production Patterns
  4. Infrastructure as Code with Terraform
  5. CI/CD Pipeline Templates
  6. AWS Architecture Patterns
  7. Observability & Monitoring
  8. Project Blueprints
  9. Curated Learning Resources
  10. Contributing

1. Cloud & DevOps Engineer Roadmap

flowchart TD
    subgraph Foundation["Foundation"]
        A1["Linux, Networking, DNS"] --> A2["Bash/Python Scripting"]
        A2 --> A3["Git, Branching Strategies"]
    end

    subgraph Containers["Containerization"]
        B1["Docker: Images, Volumes, Networks"] --> B2["Docker Compose & Multi-Stage Builds"]
        B2 --> B3["Container Security Scanning"]
    end

    subgraph Orchestration["Orchestration"]
        C1["Kubernetes: Pods, Services, Ingress"] --> C2["Helm Charts & Kustomize"]
        C2 --> C3["Auto-scaling: HPA, VPA, Cluster Autoscaler"]
    end

    subgraph IaC["Infrastructure as Code"]
        D1["Terraform: Providers, State, Modules"] --> D2["AWS CDK / Pulumi"]
        D2 --> D3["Policy as Code: OPA, Sentinel"]
    end

    subgraph CICD["CI/CD & GitOps"]
        E1["GitHub Actions / GitLab CI"] --> E2["ArgoCD / Flux GitOps"]
        E2 --> E3["Blue/Green & Canary Deployments"]
    end

    subgraph Observe["Observability"]
        F1["Prometheus + Grafana"] --> F2["Distributed Tracing: Jaeger/OpenTelemetry"]
        F2 --> F3["Log Aggregation: ELK / Loki"]
    end

    Foundation --> Containers
    Containers --> Orchestration
    Orchestration --> IaC
    IaC --> CICD
    CICD --> Observe
Loading

2. Docker & Containerization

Production Multi-Stage Dockerfile for Node.js

# Stage 1: Dependencies
FROM node:20-alpine AS deps
WORKDIR /app
COPY package.json package-lock.json ./
RUN npm ci --only=production && npm cache clean --force

# Stage 2: Build
FROM node:20-alpine AS builder
WORKDIR /app
COPY --from=deps /app/node_modules ./node_modules
COPY . .
RUN npm run build

# Stage 3: Production Runtime
FROM node:20-alpine AS runner
WORKDIR /app
ENV NODE_ENV=production

RUN addgroup --system --gid 1001 nodejs \
    && adduser --system --uid 1001 nextjs

COPY --from=builder --chown=nextjs:nodejs /app/.next/standalone ./
COPY --from=builder --chown=nextjs:nodejs /app/.next/static ./.next/static
COPY --from=builder --chown=nextjs:nodejs /app/public ./public

USER nextjs
EXPOSE 3000
CMD ["node", "server.js"]

Docker Compose for Development Stack

version: '3.9'

services:
  app:
    build:
      context: .
      target: builder
    ports:
      - "3000:3000"
    environment:
      - DATABASE_URL=postgresql://user:pass@db:5432/appdb
      - REDIS_URL=redis://cache:6379
    depends_on:
      db:
        condition: service_healthy
      cache:
        condition: service_started

  db:
    image: postgres:16-alpine
    environment:
      POSTGRES_USER: user
      POSTGRES_PASSWORD: pass
      POSTGRES_DB: appdb
    volumes:
      - pgdata:/var/lib/postgresql/data
    healthcheck:
      test: ["CMD-SHELL", "pg_isready -U user -d appdb"]
      interval: 5s
      timeout: 3s
      retries: 5

  cache:
    image: redis:7-alpine
    command: redis-server --maxmemory 256mb --maxmemory-policy allkeys-lru

volumes:
  pgdata:

3. Kubernetes Production Patterns

Deployment with Resource Limits & Health Checks

apiVersion: apps/v1
kind: Deployment
metadata:
  name: course-api
  labels:
    app: course-api
spec:
  replicas: 3
  strategy:
    type: RollingUpdate
    rollingUpdate:
      maxSurge: 1
      maxUnavailable: 0
  selector:
    matchLabels:
      app: course-api
  template:
    metadata:
      labels:
        app: course-api
    spec:
      containers:
        - name: api
          image: lucebra/course-api:v1.2.0
          ports:
            - containerPort: 8000
          resources:
            requests:
              cpu: "250m"
              memory: "256Mi"
            limits:
              cpu: "500m"
              memory: "512Mi"
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 15
            periodSeconds: 10
          readinessProbe:
            httpGet:
              path: /ready
              port: 8000
            initialDelaySeconds: 5
            periodSeconds: 5
          env:
            - name: DATABASE_URL
              valueFrom:
                secretKeyRef:
                  name: db-credentials
                  key: connection-string

4. Infrastructure as Code with Terraform

AWS VPC + ECS Fargate Module

terraform {
  required_providers {
    aws = {
      source  = "hashicorp/aws"
      version = "~> 5.0"
    }
  }
}

module "vpc" {
  source  = "terraform-aws-modules/vpc/aws"
  version = "5.0.0"

  name = "lucebra-production"
  cidr = "10.0.0.0/16"

  azs             = ["us-east-1a", "us-east-1b", "us-east-1c"]
  private_subnets = ["10.0.1.0/24", "10.0.2.0/24", "10.0.3.0/24"]
  public_subnets  = ["10.0.101.0/24", "10.0.102.0/24", "10.0.103.0/24"]

  enable_nat_gateway = true
  single_nat_gateway = false

  tags = {
    Environment = "production"
    ManagedBy   = "terraform"
  }
}

resource "aws_ecs_cluster" "main" {
  name = "lucebra-cluster"

  setting {
    name  = "containerInsights"
    value = "enabled"
  }
}

resource "aws_ecs_service" "api" {
  name            = "course-api"
  cluster         = aws_ecs_cluster.main.id
  task_definition = aws_ecs_task_definition.api.arn
  desired_count   = 3
  launch_type     = "FARGATE"

  network_configuration {
    subnets         = module.vpc.private_subnets
    security_groups = [aws_security_group.api.id]
  }
}

5. CI/CD Pipeline Templates

GitHub Actions: Build, Test, Deploy

name: CI/CD Pipeline

on:
  push:
    branches: [main]
  pull_request:
    branches: [main]

jobs:
  test:
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@v4
      - uses: actions/setup-node@v4
        with:
          node-version: 20
          cache: 'npm'
      - run: npm ci
      - run: npm run lint
      - run: npm run test -- --coverage
      - run: npm run build

  deploy:
    needs: test
    if: github.ref == 'refs/heads/main'
    runs-on: ubuntu-latest
    permissions:
      id-token: write
      contents: read
    steps:
      - uses: actions/checkout@v4
      - uses: aws-actions/configure-aws-credentials@v4
        with:
          role-to-assume: ${{ secrets.AWS_ROLE_ARN }}
          aws-region: us-east-1
      - run: |
          docker build -t $ECR_REGISTRY/course-api:${{ github.sha }} .
          docker push $ECR_REGISTRY/course-api:${{ github.sha }}
          aws ecs update-service --cluster lucebra --service api \
            --force-new-deployment

6. AWS Architecture Patterns

Serverless Event-Driven Architecture

flowchart LR
    A["API Gateway"] --> B["Lambda Functions"]
    B --> C["DynamoDB"]
    B --> D["SQS Queue"]
    D --> E["Lambda Workers"]
    E --> F["S3 Storage"]
    F --> G["CloudFront CDN"]

    H["EventBridge Rules"] --> B
    B --> I["SNS Notifications"]
Loading

Cost-Optimized Architecture Decision Matrix

Workload Type Recommended Service Estimated Cost (1M req/mo)
Stateless API Lambda + API Gateway ~$3.50
Long-running API ECS Fargate (Spot) ~$35
Static Content S3 + CloudFront ~$1.50
Database (relational) Aurora Serverless v2 ~$50
Cache ElastiCache (Redis) ~$25
Background Jobs SQS + Lambda ~$0.40

7. Observability & Monitoring

Prometheus Alert Rules

groups:
  - name: application-alerts
    rules:
      - alert: HighErrorRate
        expr: |
          sum(rate(http_requests_total{status=~"5.."}[5m]))
          / sum(rate(http_requests_total[5m])) > 0.05
        for: 5m
        labels:
          severity: critical
        annotations:
          summary: "Error rate exceeds 5% for 5 minutes"

      - alert: HighLatency
        expr: |
          histogram_quantile(0.99, rate(http_request_duration_seconds_bucket[5m])) > 2
        for: 10m
        labels:
          severity: warning
        annotations:
          summary: "P99 latency exceeds 2 seconds"

8. Project Blueprints

Level Project Stack Deliverable
Beginner Static Site on S3 + CloudFront AWS CLI, S3, Route53 HTTPS-enabled global website
Intermediate Containerized API on ECS Fargate Docker, Terraform, ECS Auto-scaling microservice with ALB
Advanced GitOps Kubernetes Platform K8s, ArgoCD, Helm, Prometheus Full cluster with GitOps-managed deployments
Expert Multi-Region Active-Active Architecture Terraform, Route53, DynamoDB Global Tables Cross-region failover with < 1min RTO

9. Curated Learning Resources

Open-Source References

Accredited Courses with Verifiable Certificates


10. Contributing

We welcome contributions from cloud and DevOps engineers:

  1. Fork this repository.
  2. Create a feature branch (git checkout -b feature/add-terraform-module).
  3. Validate all IaC with terraform validate and tflint.
  4. Submit a Pull Request with architecture diagrams where applicable.

Distributed under CC0-1.0 by Lucebra Global Education (www.lucebra.com)

About

Production-grade Docker, Kubernetes, Terraform, CI/CD, and AWS architecture patterns with cost-optimization guides.

Topics

Resources

Stars

2 stars

Watchers

0 watching

Forks

Releases

Packages

Contributors