HardПрактика11 min

Управление окружениями: dev/stage/prod

Подходы к окружениям, CI/CD для Terraform, plan review в PR, secrets management, cost estimation и blast radius

Зачем разделять окружения

Каждое окружение имеет разные требования к инфраструктуре:

┌─────────────────────────────────────────────────────────────┐
│                    Environments                              │
├──────────────┬──────────────────┬───────────────────────────┤
│     DEV      │    STAGING       │      PRODUCTION           │
├──────────────┼──────────────────┼───────────────────────────┤
│ t3.micro     │ t3.small         │ t3.large                  │
│ 1 instance   │ 2 instances      │ 3+ instances (auto-scale) │
│ db.t3.micro  │ db.t3.small      │ db.r6g.large              │
│ Single AZ    │ Single AZ        │ Multi-AZ                  │
│ No backups   │ 7-day backups    │ 30-day backups            │
│ No monitoring│ Basic monitoring │ Full monitoring + alerts   │
│ No SSL       │ Self-signed      │ ACM certificate           │
│ $50/month    │ $200/month       │ $2000/month               │
├──────────────┼──────────────────┼───────────────────────────┤
│ Fast iterate │ Test releases    │ Zero downtime             │
│ Break freely │ Catch bugs       │ Never break               │
└──────────────┴──────────────────┴───────────────────────────┘

Подходы к управлению окружениями

Подход 1: Terraform Workspaces

Один набор конфигурации, разные workspace для каждого окружения:

infrastructure/
├── main.tf
├── variables.tf
├── outputs.tf
├── networking.tf
├── compute.tf
├── database.tf
└── env/
    ├── dev.tfvars
    ├── staging.tfvars
    └── prod.tfvars
# main.tf
locals {
  environment = terraform.workspace

  config = {
    dev = {
      instance_type  = "t3.micro"
      instance_count = 1
      db_class       = "db.t3.micro"
      multi_az       = false
      backup_days    = 0
    }
    staging = {
      instance_type  = "t3.small"
      instance_count = 2
      db_class       = "db.t3.small"
      multi_az       = false
      backup_days    = 7
    }
    prod = {
      instance_type  = "t3.large"
      instance_count = 3
      db_class       = "db.r6g.large"
      multi_az       = true
      backup_days    = 30
    }
  }

  env = local.config[local.environment]
}
# Usage
terraform workspace select dev
terraform apply -var-file="env/dev.tfvars"

terraform workspace select prod
terraform apply -var-file="env/prod.tfvars"

Плюсы: Минимум дублирования, простота. Минусы: Одна ошибка может затронуть все окружения, нет отдельного PR для prod.

Подход 2: Directory per Environment (рекомендуется)

Отдельная директория для каждого окружения, общие модули:

infrastructure/
├── modules/                    # Shared modules
│   ├── networking/
│   │   ├── main.tf
│   │   ├── variables.tf
│   │   └── outputs.tf
│   ├── compute/
│   │   ├── main.tf
│   │   ├── variables.tf
│   │   └── outputs.tf
│   ├── database/
│   │   ├── main.tf
│   │   ├── variables.tf
│   │   └── outputs.tf
│   └── monitoring/
│       ├── main.tf
│       ├── variables.tf
│       └── outputs.tf
├── environments/
│   ├── dev/
│   │   ├── main.tf             # Module calls
│   │   ├── variables.tf
│   │   ├── outputs.tf
│   │   ├── backend.tf          # S3 backend with unique key
│   │   └── terraform.tfvars    # Dev-specific values
│   ├── staging/
│   │   ├── main.tf
│   │   ├── variables.tf
│   │   ├── outputs.tf
│   │   ├── backend.tf
│   │   └── terraform.tfvars
│   └── prod/
│       ├── main.tf
│       ├── variables.tf
│       ├── outputs.tf
│       ├── backend.tf
│       └── terraform.tfvars
└── global/                     # Shared infrastructure
    ├── iam/
    │   └── main.tf
    └── dns/
        └── main.tf

environments/dev/backend.tf:

terraform {
  backend "s3" {
    bucket         = "myapp-terraform-state"
    key            = "environments/dev/terraform.tfstate"
    region         = "eu-central-1"
    dynamodb_table = "terraform-locks"
    encrypt        = true
  }
}

environments/dev/main.tf:

terraform {
  required_version = ">= 1.9"

  required_providers {
    aws = {
      source  = "hashicorp/aws"
      version = "~> 5.80"
    }
  }
}

provider "aws" {
  region = var.aws_region

  default_tags {
    tags = {
      Project     = "myapp"
      Environment = "dev"
      ManagedBy   = "terraform"
    }
  }
}

module "networking" {
  source = "../../modules/networking"

  environment = "dev"
  vpc_cidr    = var.vpc_cidr
}

module "database" {
  source = "../../modules/database"

  name        = "myapp"
  environment = "dev"

  vpc_id     = module.networking.vpc_id
  subnet_ids = module.networking.private_subnet_ids

  instance_class          = var.db_instance_class
  allocated_storage       = var.db_storage
  multi_az                = false
  backup_retention_period = 0
  deletion_protection     = false

  allowed_security_group_ids = [module.compute.security_group_id]
}

module "compute" {
  source = "../../modules/compute"

  name        = "myapp"
  environment = "dev"

  vpc_id     = module.networking.vpc_id
  subnet_ids = module.networking.private_subnet_ids

  instance_type  = var.instance_type
  instance_count = var.instance_count
  min_capacity   = 1
  max_capacity   = 2

  db_endpoint = module.database.endpoint
  db_secret   = module.database.secret_arn
}

environments/dev/terraform.tfvars:

# Dev environment values
aws_region       = "eu-central-1"
vpc_cidr         = "10.0.0.0/16"
instance_type    = "t3.micro"
instance_count   = 1
db_instance_class = "db.t3.micro"
db_storage        = 20

environments/prod/terraform.tfvars:

# Production environment values
aws_region       = "eu-central-1"
vpc_cidr         = "10.1.0.0/16"
instance_type    = "t3.large"
instance_count   = 3
db_instance_class = "db.r6g.large"
db_storage        = 200

Подход 3: Terragrunt

Terragrunt -- обёртка над Terraform от Gruntwork, которая устраняет дублирование:

infrastructure/
├── terragrunt.hcl              # Root config
├── modules/
│   └── ...
├── environments/
│   ├── _env/
│   │   └── common.hcl          # Shared config
│   ├── dev/
│   │   ├── terragrunt.hcl
│   │   ├── networking/
│   │   │   └── terragrunt.hcl
│   │   ├── database/
│   │   │   └── terragrunt.hcl
│   │   └── compute/
│   │       └── terragrunt.hcl
│   └── prod/
│       ├── terragrunt.hcl
│       ├── networking/
│       │   └── terragrunt.hcl
│       └── ...
# environments/dev/database/terragrunt.hcl
include "root" {
  path = find_in_parent_folders()
}

include "env" {
  path = "${dirname(find_in_parent_folders())}/_env/common.hcl"
}

terraform {
  source = "../../../modules/database"
}

dependency "networking" {
  config_path = "../networking"
}

inputs = {
  name        = "myapp"
  environment = "dev"

  vpc_id     = dependency.networking.outputs.vpc_id
  subnet_ids = dependency.networking.outputs.private_subnet_ids

  instance_class    = "db.t3.micro"
  allocated_storage = 20
  multi_az          = false
}

Сравнение подходов

Критерий Workspaces Directory Terragrunt
Дублирование кода Минимум Среднее Минимум
Изоляция state Автоматическая Ручная Автоматическая
Blast radius Высокий Низкий Низкий
Сложность Низкая Средняя Высокая
CI/CD Простой Гибкий Гибкий
Зависимости между модулями Вручную Вручную Автоматические
Learning curve Низкая Низкая Средняя
Рекомендация Sandbox Production Production+

CI/CD Pipeline для Terraform

GitHub Actions

# .github/workflows/terraform.yml
name: Terraform

on:
  pull_request:
    paths:
      - 'infrastructure/**'
  push:
    branches:
      - main
    paths:
      - 'infrastructure/**'

permissions:
  contents: read
  pull-requests: write
  id-token: write  # For OIDC authentication

env:
  TF_VERSION: "1.9.8"
  AWS_REGION: "eu-central-1"

jobs:
  # Detect which environments changed
  detect-changes:
    runs-on: ubuntu-latest
    outputs:
      environments: ${{ steps.changes.outputs.environments }}
    steps:
      - uses: actions/checkout@v4

      - id: changes
        name: Detect changed environments
        run: |
          ENVS=$(git diff --name-only ${{ github.event.before }} ${{ github.sha }} \
            | grep '^infrastructure/environments/' \
            | cut -d'/' -f3 \
            | sort -u \
            | jq -R -s -c 'split("\n")[:-1]')
          echo "environments=$ENVS" >> "$GITHUB_OUTPUT"

  # Run plan for each changed environment
  plan:
    needs: detect-changes
    if: github.event_name == 'pull_request'
    runs-on: ubuntu-latest
    strategy:
      matrix:
        environment: ${{ fromJson(needs.detect-changes.outputs.environments) }}
    steps:
      - uses: actions/checkout@v4

      - uses: hashicorp/setup-terraform@v3
        with:
          terraform_version: ${{ env.TF_VERSION }}

      - name: Configure AWS Credentials
        uses: aws-actions/configure-aws-credentials@v4
        with:
          role-to-assume: arn:aws:iam::${{ secrets.AWS_ACCOUNT_ID }}:role/terraform-ci
          aws-region: ${{ env.AWS_REGION }}

      - name: Terraform Init
        working-directory: infrastructure/environments/${{ matrix.environment }}
        run: terraform init

      - name: Terraform Format Check
        working-directory: infrastructure/environments/${{ matrix.environment }}
        run: terraform fmt -check -recursive

      - name: Terraform Validate
        working-directory: infrastructure/environments/${{ matrix.environment }}
        run: terraform validate

      - name: Terraform Plan
        id: plan
        working-directory: infrastructure/environments/${{ matrix.environment }}
        run: |
          terraform plan -no-color -out=tfplan 2>&1 | tee plan_output.txt
        continue-on-error: true

      - name: Post Plan to PR
        uses: actions/github-script@v7
        with:
          script: |
            const fs = require('fs');
            const plan = fs.readFileSync(
              'infrastructure/environments/${{ matrix.environment }}/plan_output.txt',
              'utf8'
            );

            const truncated = plan.length > 60000
              ? plan.substring(0, 60000) + '\n... (truncated)'
              : plan;

            const body = `### Terraform Plan: \`${{ matrix.environment }}\`

            \`\`\`
            ${truncated}
            \`\`\`

            **Status:** ${{ steps.plan.outcome }}
            `;

            github.rest.issues.createComment({
              issue_number: context.issue.number,
              owner: context.repo.owner,
              repo: context.repo.repo,
              body: body
            });

      - name: Plan Status
        if: steps.plan.outcome == 'failure'
        run: exit 1

  # Apply after merge to main
  apply-dev:
    needs: detect-changes
    if: |
      github.ref == 'refs/heads/main' &&
      github.event_name == 'push' &&
      contains(fromJson(needs.detect-changes.outputs.environments), 'dev')
    runs-on: ubuntu-latest
    environment: dev  # No approval needed
    steps:
      - uses: actions/checkout@v4
      - uses: hashicorp/setup-terraform@v3
        with:
          terraform_version: ${{ env.TF_VERSION }}

      - name: Configure AWS Credentials
        uses: aws-actions/configure-aws-credentials@v4
        with:
          role-to-assume: arn:aws:iam::${{ secrets.AWS_ACCOUNT_ID }}:role/terraform-ci
          aws-region: ${{ env.AWS_REGION }}

      - name: Terraform Apply
        working-directory: infrastructure/environments/dev
        run: |
          terraform init
          terraform apply -auto-approve

  apply-prod:
    needs: [detect-changes, apply-dev]
    if: |
      github.ref == 'refs/heads/main' &&
      github.event_name == 'push' &&
      contains(fromJson(needs.detect-changes.outputs.environments), 'prod')
    runs-on: ubuntu-latest
    environment: production  # Requires manual approval!
    steps:
      - uses: actions/checkout@v4
      - uses: hashicorp/setup-terraform@v3
        with:
          terraform_version: ${{ env.TF_VERSION }}

      - name: Configure AWS Credentials
        uses: aws-actions/configure-aws-credentials@v4
        with:
          role-to-assume: arn:aws:iam::${{ secrets.AWS_ACCOUNT_ID }}:role/terraform-ci
          aws-region: ${{ env.AWS_REGION }}

      - name: Terraform Apply
        working-directory: infrastructure/environments/prod
        run: |
          terraform init
          terraform apply -auto-approve

GitLab CI

# .gitlab-ci.yml
stages:
  - validate
  - plan
  - apply

variables:
  TF_VERSION: "1.9.8"

.terraform:
  image: hashicorp/terraform:${TF_VERSION}
  before_script:
    - cd infrastructure/environments/${ENVIRONMENT}
    - terraform init

validate:
  extends: .terraform
  stage: validate
  variables:
    ENVIRONMENT: dev
  script:
    - terraform fmt -check
    - terraform validate

plan:dev:
  extends: .terraform
  stage: plan
  variables:
    ENVIRONMENT: dev
  script:
    - terraform plan -out=tfplan
  artifacts:
    paths:
      - infrastructure/environments/dev/tfplan
  rules:
    - if: $CI_MERGE_REQUEST_IID
      changes:
        - infrastructure/**

plan:prod:
  extends: .terraform
  stage: plan
  variables:
    ENVIRONMENT: prod
  script:
    - terraform plan -out=tfplan
  artifacts:
    paths:
      - infrastructure/environments/prod/tfplan
  rules:
    - if: $CI_MERGE_REQUEST_IID
      changes:
        - infrastructure/**

apply:dev:
  extends: .terraform
  stage: apply
  variables:
    ENVIRONMENT: dev
  script:
    - terraform apply -auto-approve tfplan
  dependencies:
    - plan:dev
  rules:
    - if: $CI_COMMIT_BRANCH == "main"
      changes:
        - infrastructure/**

apply:prod:
  extends: .terraform
  stage: apply
  variables:
    ENVIRONMENT: prod
  script:
    - terraform apply -auto-approve tfplan
  dependencies:
    - plan:prod
  when: manual  # Manual approval!
  rules:
    - if: $CI_COMMIT_BRANCH == "main"
      changes:
        - infrastructure/**

Cost Estimation с Infracost

Infracost показывает стоимость инфраструктурных изменений прямо в PR:

# .github/workflows/infracost.yml
name: Infracost

on:
  pull_request:
    paths:
      - 'infrastructure/**'

jobs:
  infracost:
    runs-on: ubuntu-latest
    permissions:
      pull-requests: write
    steps:
      - uses: actions/checkout@v4

      - name: Setup Infracost
        uses: infracost/actions/setup@v3
        with:
          api-key: ${{ secrets.INFRACOST_API_KEY }}

      - name: Generate Infracost diff
        run: |
          infracost diff \
            --path=infrastructure/environments/prod \
            --format=json \
            --out-file=/tmp/infracost.json

      - name: Post comment
        run: |
          infracost comment github \
            --path=/tmp/infracost.json \
            --repo=$GITHUB_REPOSITORY \
            --pull-request=${{ github.event.pull_request.number }} \
            --github-token=${{ secrets.GITHUB_TOKEN }} \
            --behavior=update

Результат в PR:

💰 Monthly cost will increase by $142.50 (12%)

Project                          Monthly Cost   Change
────────────────────────────────────────────────────────
myapp-prod
├─ aws_instance.web (x3)            +$103.50   t3.medium → t3.large
├─ aws_db_instance.main              +$39.00   Added Multi-AZ
└─ aws_s3_bucket.backups              +$0.00   Storage only

Total                             $1,332.50    +$142.50

Secrets Management

AWS Systems Manager Parameter Store

# Store secret
resource "aws_ssm_parameter" "db_password" {
  name        = "/${var.project}/${var.environment}/database/password"
  description = "Database master password"
  type        = "SecureString"
  value       = random_password.db.result

  tags = local.common_tags
}

# Read secret in application
data "aws_ssm_parameter" "api_key" {
  name = "/${var.project}/${var.environment}/external/api-key"
}

resource "aws_ecs_task_definition" "app" {
  container_definitions = jsonencode([{
    name = "app"
    secrets = [
      {
        name      = "DB_PASSWORD"
        valueFrom = aws_ssm_parameter.db_password.arn
      },
      {
        name      = "API_KEY"
        valueFrom = data.aws_ssm_parameter.api_key.arn
      }
    ]
  }])
}

AWS Secrets Manager

# Create secret with rotation
resource "aws_secretsmanager_secret" "db_credentials" {
  name = "${var.project}/${var.environment}/db-credentials"

  tags = local.common_tags
}

resource "aws_secretsmanager_secret_version" "db_credentials" {
  secret_id = aws_secretsmanager_secret.db_credentials.id
  secret_string = jsonencode({
    username = "app_user"
    password = random_password.db.result
    engine   = "postgres"
    host     = aws_db_instance.main.address
    port     = 5432
    dbname   = "app_db"
  })
}

# Read secret in Terraform
data "aws_secretsmanager_secret_version" "external_api" {
  secret_id = "external-service/api-key"
}

locals {
  external_api_key = jsondecode(data.aws_secretsmanager_secret_version.external_api.secret_string)["api_key"]
}

HashiCorp Vault Provider

provider "vault" {
  address = "https://vault.mycompany.com"
  # Token from VAULT_TOKEN environment variable
}

# Read secret from Vault
data "vault_generic_secret" "db" {
  path = "secret/data/${var.environment}/database"
}

resource "aws_db_instance" "main" {
  username = data.vault_generic_secret.db.data["username"]
  password = data.vault_generic_secret.db.data["password"]
}

# Write generated secret to Vault
resource "vault_generic_secret" "app_credentials" {
  path = "secret/data/${var.environment}/app"

  data_json = jsonencode({
    db_host     = aws_db_instance.main.address
    db_password = random_password.db.result
    redis_url   = aws_elasticache_replication_group.main.primary_endpoint_address
  })
}

Сравнение инструментов для секретов

Критерий SSM Parameter Store Secrets Manager Vault
Стоимость Бесплатно (Standard) $0.40/secret/month Self-hosted / Enterprise
Ротация Нет Автоматическая Автоматическая
Cross-account Нет Да Да
Multi-cloud Нет Нет Да
Terraform provider AWS AWS Vault
Рекомендация Простые секреты DB credentials Enterprise multi-cloud

Blast Radius и ограничение scope

Blast radius -- объём инфраструктуры, который может быть затронут одной ошибкой.

Blast Radius:

Монолитный state (ПЛОХО):
┌────────────────────────────────────────────────┐
│              Один terraform apply               │
│  VPC + EC2 + RDS + S3 + IAM + CloudFront +     │
│  Route53 + ACM + ECS + ALB + ...               │
│                                                 │
│  Ошибка здесь → ВСЁ может быть затронуто       │
│  Blast radius: МАКСИМАЛЬНЫЙ                     │
└────────────────────────────────────────────────┘

Разделённый state (ХОРОШО):
┌──────────┐  ┌──────────┐  ┌──────────┐  ┌──────────┐
│ Network  │  │ Database │  │ Compute  │  │  CDN     │
│ state    │  │ state    │  │ state    │  │  state   │
│          │  │          │  │          │  │          │
│ VPC      │  │ RDS      │  │ ECS      │  │ CF + S3  │
│ Subnets  │  │ ElastiC. │  │ ALB      │  │ Route53  │
│ SGs      │  │          │  │ ASG      │  │ ACM      │
└──────────┘  └──────────┘  └──────────┘  └──────────┘

Ошибка в Compute → только ECS/ALB затронуты
Blast radius: МИНИМАЛЬНЫЙ

Стратегии ограничения blast radius

# 1. Separate state per component
# infrastructure/environments/prod/networking/backend.tf
terraform {
  backend "s3" {
    key = "prod/networking/terraform.tfstate"
  }
}

# infrastructure/environments/prod/database/backend.tf
terraform {
  backend "s3" {
    key = "prod/database/terraform.tfstate"
  }
}

# 2. Read other state via data source
data "terraform_remote_state" "networking" {
  backend = "s3"
  config = {
    bucket = "myapp-terraform-state"
    key    = "prod/networking/terraform.tfstate"
    region = "eu-central-1"
  }
}

resource "aws_db_instance" "main" {
  db_subnet_group_name = data.terraform_remote_state.networking.outputs.db_subnet_group
}

# 3. Use -target for surgical changes
# terraform apply -target=aws_instance.web[2]

# 4. prevent_destroy for critical resources
resource "aws_db_instance" "main" {
  lifecycle {
    prevent_destroy = true
  }
}

resource "aws_s3_bucket" "data" {
  lifecycle {
    prevent_destroy = true
  }
}

Promotion Flow: dev -> staging -> prod

┌──────────────────────────────────────────────────────────┐
│                  Promotion Flow                           │
├──────────────────────────────────────────────────────────┤
│                                                           │
│  1. Developer changes infrastructure code                 │
│     └── Creates PR with changes                          │
│                                                           │
│  2. CI runs for all affected environments:                │
│     ├── terraform fmt -check                             │
│     ├── terraform validate                               │
│     ├── tfsec / checkov (security)                       │
│     ├── infracost (cost)                                 │
│     └── terraform plan (per environment)                 │
│                                                           │
│  3. PR Review:                                            │
│     ├── Code review by team                              │
│     ├── Review plan output                               │
│     └── Review cost impact                               │
│                                                           │
│  4. Merge to main:                                        │
│     ├── Auto-apply to DEV          (no approval)         │
│     ├── Auto-apply to STAGING      (no approval)         │
│     └── Manual approval for PROD   (requires approval)   │
│                                                           │
│  5. Post-apply:                                           │
│     ├── Smoke tests                                      │
│     ├── Monitoring check                                 │
│     └── Rollback if needed                               │
│                                                           │
└──────────────────────────────────────────────────────────┘

Rollback стратегия

# Option 1: Revert commit and re-apply
# git revert <commit-hash>  (выполняется пользователем)
# CI auto-applies the reverted config

# Option 2: Apply previous state manually
# terraform apply -target=... with previous values

# Option 3: Use terraform version pinning
# Keep previous .tfvars version in git history

Полный пример: структура проекта с CI/CD

my-project/
├── .github/
│   └── workflows/
│       ├── terraform-plan.yml     # PR: lint + plan
│       ├── terraform-apply.yml    # Main: apply
│       └── infracost.yml          # PR: cost estimation
├── infrastructure/
│   ├── modules/
│   │   ├── networking/
│   │   │   ├── main.tf
│   │   │   ├── variables.tf
│   │   │   └── outputs.tf
│   │   ├── compute/
│   │   │   ├── main.tf
│   │   │   ├── variables.tf
│   │   │   └── outputs.tf
│   │   ├── database/
│   │   │   ├── main.tf
│   │   │   ├── variables.tf
│   │   │   └── outputs.tf
│   │   └── monitoring/
│   │       ├── main.tf
│   │       ├── variables.tf
│   │       └── outputs.tf
│   ├── environments/
│   │   ├── dev/
│   │   │   ├── main.tf
│   │   │   ├── variables.tf
│   │   │   ├── outputs.tf
│   │   │   ├── backend.tf
│   │   │   └── terraform.tfvars
│   │   ├── staging/
│   │   │   └── ...
│   │   └── prod/
│   │       └── ...
│   └── global/
│       └── iam/
│           └── main.tf
├── .gitignore
└── Makefile

Makefile для удобства

# Makefile
ENV ?= dev
DIR = infrastructure/environments/$(ENV)

.PHONY: init plan apply destroy fmt validate

init:
	cd $(DIR) && terraform init

plan:
	cd $(DIR) && terraform plan

apply:
	cd $(DIR) && terraform apply

destroy:
	cd $(DIR) && terraform destroy

fmt:
	terraform fmt -recursive infrastructure/

validate:
	cd $(DIR) && terraform validate

# Usage:
# make plan ENV=dev
# make plan ENV=prod
# make apply ENV=staging

Проверь себя

5 из 7

Какой инструмент лучше всего подходит для хранения секретов БД в multi-cloud среде?

Зачем нужен manual approval для apply в production?

Что даёт разделение state по компонентам (networking, database, compute)?

Что делает Infracost в CI/CD pipeline?

Какой подход к управлению окружениями рекомендуется для production-проектов?