Capacity planning — процесс определения необходимых ресурсов для обеспечения текущей и будущей нагрузки. Цель: иметь достаточно ресурсов для обслуживания пользователей без избыточных затрат.
Зачем планировать ёмкость
Проблема
Следствие
Недостаток ресурсов
Деградация, потеря пользователей
Избыток ресурсов
Лишние затраты на инфраструктуру
Внезапный рост
Сервис падает при вирусном контенте
Без прогноза
Reactive вместо proactive масштабирования
Процесс Capacity Planning
1. Сбор данных → Текущее использование, рост трафика
2. Моделирование → Проекция нагрузки на 3-12 месяцев
3. Тестирование → Определение пределов текущей инфраструктуры
4. Планирование → Когда и что масштабировать
5. Мониторинг → Проверка прогнозов vs реальность
(Max capacity - Current load) / Max capacity * 100%
Growth rate
(Current - Previous) / Previous * 100%
Time to saturation
Headroom / Monthly growth rate
Required capacity
Peak load * (1 + Safety margin)
Калькулятор ёмкости
<?php
declare(strict_types=1);
namespace App\Capacity;
final readonly class CapacityCalculator
{
/**
* Calculate time until resource saturation.
*
* @param float $currentUsage Current usage (e.g., 65%)
* @param float $maxCapacity Maximum capacity (e.g., 100%)
* @param float $monthlyGrowthRate Monthly growth rate (e.g., 0.08 for 8%)
* @return array{months_to_saturation: float, saturation_date: string, headroom_percent: float}
*/
public function timeToSaturation(
float $currentUsage,
float $maxCapacity,
float $monthlyGrowthRate,
): array {
if ($monthlyGrowthRate <= 0) {
return [
'months_to_saturation' => INF,
'saturation_date' => 'never',
'headroom_percent' => round(($maxCapacity - $currentUsage) / $maxCapacity * 100, 1),
];
}
// Using exponential growth: current * (1 + rate)^months = max
$months = log($maxCapacity / $currentUsage) / log(1 + $monthlyGrowthRate);
$saturationDate = (new \DateTimeImmutable())->modify(sprintf('+%d months', (int) ceil($months)));
return [
'months_to_saturation' => round($months, 1),
'saturation_date' => $saturationDate->format('Y-m'),
'headroom_percent' => round(($maxCapacity - $currentUsage) / $maxCapacity * 100, 1),
];
}
/**
* Calculate required infrastructure for target RPS.
*
* @param float $targetRps Target requests per second
* @param float $avgLatencyMs Average latency per request in ms
* @param int $workersPerInstance Number of PHP-FPM workers per instance
* @param float $safetyMargin Safety margin (0.3 = 30%)
* @return array{instances_needed: int, total_workers: int, max_rps: float}
*/
public function calculateInstances(
float $targetRps,
float $avgLatencyMs,
int $workersPerInstance = 50,
float $safetyMargin = 0.3,
): array {
// Each worker handles: 1000ms / avgLatencyMs requests per second
$rpsPerWorker = 1000 / $avgLatencyMs;
$rpsPerInstance = $rpsPerWorker * $workersPerInstance;
// Add safety margin
$requiredCapacity = $targetRps * (1 + $safetyMargin);
$instancesNeeded = (int) ceil($requiredCapacity / $rpsPerInstance);
return [
'instances_needed' => $instancesNeeded,
'total_workers' => $instancesNeeded * $workersPerInstance,
'max_rps' => round($instancesNeeded * $rpsPerInstance, 0),
];
}
/**
* Estimate PHP-FPM worker count based on available memory.
*
* @param int $availableMemoryMb Available memory in MB
* @param int $avgWorkerMemoryMb Average memory per PHP-FPM worker
* @param float $memoryReserve Reserve for OS and other processes (0.2 = 20%)
* @return array{max_workers: int, memory_per_worker_mb: int, total_usable_mb: int}
*/
public function estimateWorkers(
int $availableMemoryMb,
int $avgWorkerMemoryMb = 64,
float $memoryReserve = 0.2,
): array {
$usableMemory = (int) ($availableMemoryMb * (1 - $memoryReserve));
$maxWorkers = (int) floor($usableMemory / $avgWorkerMemoryMb);
return [
'max_workers' => $maxWorkers,
'memory_per_worker_mb' => $avgWorkerMemoryMb,
'total_usable_mb' => $usableMemory,
];
}
}
// Example usage
$calc = new CapacityCalculator();
// When will we run out of capacity?
$saturation = $calc->timeToSaturation(
currentUsage: 65,
maxCapacity: 100,
monthlyGrowthRate: 0.08, // 8% monthly growth
);
// Result: ~5.4 months to saturation
// How many servers for 10,000 RPS?
$infra = $calc->calculateInstances(
targetRps: 10_000,
avgLatencyMs: 50,
workersPerInstance: 50,
safetyMargin: 0.3,
);
// Each worker: 20 RPS, each instance: 1000 RPS
// Need: 13 instances (with 30% margin)
package capacity
import (
"fmt"
"math"
"time"
)
// SaturationResult holds time-to-saturation analysis.
type SaturationResult struct {
MonthsToSaturation float64 `json:"months_to_saturation"`
SaturationDate string `json:"saturation_date"`
HeadroomPercent float64 `json:"headroom_percent"`
}
// TimeToSaturation calculates when resources will be exhausted.
func TimeToSaturation(currentUsage, maxCapacity, monthlyGrowthRate float64) SaturationResult {
headroom := math.Round((maxCapacity-currentUsage)/maxCapacity*1000) / 10
if monthlyGrowthRate <= 0 {
return SaturationResult{
MonthsToSaturation: math.Inf(1),
SaturationDate: "never",
HeadroomPercent: headroom,
}
}
months := math.Log(maxCapacity/currentUsage) / math.Log(1+monthlyGrowthRate)
satDate := time.Now().AddDate(0, int(math.Ceil(months)), 0)
return SaturationResult{
MonthsToSaturation: math.Round(months*10) / 10,
SaturationDate: satDate.Format("2006-01"),
HeadroomPercent: headroom,
}
}
// InfraRequirements holds the result of an instance calculation.
type InfraRequirements struct {
InstancesNeeded int `json:"instances_needed"`
TotalWorkers int `json:"total_workers"`
MaxRPS float64 `json:"max_rps"`
}
// CalculateInstances determines how many servers are needed for target RPS.
func CalculateInstances(targetRPS, avgLatencyMs float64, workersPerInstance int, safetyMargin float64) InfraRequirements {
rpsPerWorker := 1000 / avgLatencyMs
rpsPerInstance := rpsPerWorker * float64(workersPerInstance)
requiredCapacity := targetRPS * (1 + safetyMargin)
needed := int(math.Ceil(requiredCapacity / rpsPerInstance))
return InfraRequirements{
InstancesNeeded: needed,
TotalWorkers: needed * workersPerInstance,
MaxRPS: math.Round(float64(needed) * rpsPerInstance),
}
}
// WorkerEstimate holds the result of worker estimation.
type WorkerEstimate struct {
MaxWorkers int `json:"max_workers"`
MemoryPerWorker int `json:"memory_per_worker_mb"`
TotalUsableMB int `json:"total_usable_mb"`
}
// EstimateWorkers calculates max goroutines/workers based on available memory.
func EstimateWorkers(availableMemoryMB, avgWorkerMemoryMB int, memoryReserve float64) WorkerEstimate {
usable := int(float64(availableMemoryMB) * (1 - memoryReserve))
maxWorkers := usable / avgWorkerMemoryMB
return WorkerEstimate{
MaxWorkers: maxWorkers,
MemoryPerWorker: avgWorkerMemoryMB,
TotalUsableMB: usable,
}
}
// Usage:
// sat := TimeToSaturation(65, 100, 0.08) // ~5.4 months
// infra := CalculateInstances(10000, 50, 50, 0.3) // 13 instances
namespace App.Capacity;
// Time-to-saturation analysis for a resource.
public readonly record struct SaturationResult(
double MonthsToSaturation,
string SaturationDate,
double HeadroomPercent);
// Infrastructure required to serve a target RPS.
public readonly record struct InfraRequirements(
int InstancesNeeded,
int TotalWorkers,
double MaxRps);
// Worker count that fits into the available memory.
public readonly record struct WorkerEstimate(
int MaxWorkers,
int MemoryPerWorkerMb,
int TotalUsableMb);
public sealed class CapacityCalculator
{
// Calculate time until resource saturation.
public SaturationResult TimeToSaturation(double currentUsage, double maxCapacity, double monthlyGrowthRate)
{
var headroom = Math.Round((maxCapacity - currentUsage) / maxCapacity * 100, 1);
if (monthlyGrowthRate <= 0)
{
return new SaturationResult(double.PositiveInfinity, "never", headroom);
}
// Using exponential growth: current * (1 + rate)^months = max
var months = Math.Log(maxCapacity / currentUsage) / Math.Log(1 + monthlyGrowthRate);
var saturationDate = DateTime.UtcNow.AddMonths((int)Math.Ceiling(months));
return new SaturationResult(
Math.Round(months, 1),
saturationDate.ToString("yyyy-MM"),
headroom);
}
// Calculate the infrastructure required for a target RPS.
// Kestrel has no PHP-FPM pool: a "worker" here is the concurrency level
// one instance sustains on the thread pool, not an OS process.
public InfraRequirements CalculateInstances(
double targetRps,
double avgLatencyMs,
int workersPerInstance = 50,
double safetyMargin = 0.3)
{
// Each worker handles 1000ms / avgLatencyMs requests per second
var rpsPerWorker = 1000 / avgLatencyMs;
var rpsPerInstance = rpsPerWorker * workersPerInstance;
// Add safety margin
var requiredCapacity = targetRps * (1 + safetyMargin);
var instancesNeeded = (int)Math.Ceiling(requiredCapacity / rpsPerInstance);
return new InfraRequirements(
instancesNeeded,
instancesNeeded * workersPerInstance,
Math.Round(instancesNeeded * rpsPerInstance));
}
// Estimate the concurrent worker count based on available memory.
public WorkerEstimate EstimateWorkers(
int availableMemoryMb,
int avgWorkerMemoryMb = 64,
double memoryReserve = 0.2)
{
var usableMemory = (int)(availableMemoryMb * (1 - memoryReserve));
return new WorkerEstimate(usableMemory / avgWorkerMemoryMb, avgWorkerMemoryMb, usableMemory);
}
}
// Usage:
// var calc = new CapacityCalculator();
// // When will we run out of capacity?
// var saturation = calc.TimeToSaturation(currentUsage: 65, maxCapacity: 100, monthlyGrowthRate: 0.08);
// // Result: ~5.4 months to saturation
// // How many servers for 10,000 RPS?
// var infra = calc.CalculateInstances(targetRps: 10_000, avgLatencyMs: 50);
// // Each worker: 20 RPS, each instance: 1000 RPS -> 13 instances (with 30% margin)
import math
from dataclasses import dataclass
from datetime import UTC, datetime
from dateutil.relativedelta import relativedelta
@dataclass(frozen=True, slots=True)
class SaturationResult:
"""Time-to-saturation analysis for a resource."""
months_to_saturation: float
saturation_date: str
headroom_percent: float
@dataclass(frozen=True, slots=True)
class InfraRequirements:
"""Infrastructure required to serve a target RPS."""
instances_needed: int
total_workers: int
max_rps: float
@dataclass(frozen=True, slots=True)
class WorkerEstimate:
"""Worker count that fits into the available memory."""
max_workers: int
memory_per_worker_mb: int
total_usable_mb: int
class CapacityCalculator:
def time_to_saturation(
self,
current_usage: float,
max_capacity: float,
monthly_growth_rate: float,
) -> SaturationResult:
"""Calculate time until resource saturation."""
headroom = round((max_capacity - current_usage) / max_capacity * 100, 1)
if monthly_growth_rate <= 0:
return SaturationResult(math.inf, "never", headroom)
# Using exponential growth: current * (1 + rate)^months = max
months = math.log(max_capacity / current_usage) / math.log(1 + monthly_growth_rate)
saturation_date = datetime.now(UTC) + relativedelta(months=math.ceil(months))
return SaturationResult(round(months, 1), saturation_date.strftime("%Y-%m"), headroom)
def calculate_instances(
self,
target_rps: float,
avg_latency_ms: float,
workers_per_instance: int = 50,
safety_margin: float = 0.3,
) -> InfraRequirements:
"""Calculate the infrastructure required for a target RPS.
For a sync WSGI stack a worker is a Gunicorn process, the direct analogue
of a PHP-FPM worker; for async ASGI it is the event-loop concurrency level.
"""
# Each worker handles 1000ms / avg_latency_ms requests per second
rps_per_worker = 1000 / avg_latency_ms
rps_per_instance = rps_per_worker * workers_per_instance
# Add safety margin
required_capacity = target_rps * (1 + safety_margin)
instances_needed = math.ceil(required_capacity / rps_per_instance)
return InfraRequirements(
instances_needed=instances_needed,
total_workers=instances_needed * workers_per_instance,
max_rps=round(instances_needed * rps_per_instance),
)
def estimate_workers(
self,
available_memory_mb: int,
avg_worker_memory_mb: int = 64,
memory_reserve: float = 0.2,
) -> WorkerEstimate:
"""Estimate the worker count based on available memory."""
usable_memory = int(available_memory_mb * (1 - memory_reserve))
return WorkerEstimate(
max_workers=usable_memory // avg_worker_memory_mb,
memory_per_worker_mb=avg_worker_memory_mb,
total_usable_mb=usable_memory,
)
# Example usage
calc = CapacityCalculator()
# When will we run out of capacity?
saturation = calc.time_to_saturation(
current_usage=65,
max_capacity=100,
monthly_growth_rate=0.08, # 8% monthly growth
)
# Result: ~5.4 months to saturation
# How many servers for 10,000 RPS?
infra = calc.calculate_instances(
target_rps=10_000,
avg_latency_ms=50,
workers_per_instance=50,
safety_margin=0.3,
)
# Each worker: 20 RPS, each instance: 1000 RPS
# Need: 13 instances (with 30% margin)
## Load Testing в PHP
Подготовка к нагрузочному тестированию
Шаг
Действие
1
Определить сценарии (top-5 API endpoints)
2
Определить целевую нагрузку (peak * 1.5)
3
Подготовить тестовые данные
4
Настроить мониторинг (метрики, дашборды)
5
Изолировать тестовую среду от production
6
Провести тест, собрать результаты
Генератор нагрузки
<?php
declare(strict_types=1);
namespace App\LoadTest;
/**
* Simple HTTP load generator for testing.
* In production use k6, Locust, or Gatling.
*/
final class LoadGenerator
{
/** @var array<RequestResult> */
private array $results = [];
public function __construct(
private readonly int $concurrency = 10,
private readonly int $totalRequests = 1000,
) {}
/**
* Run load test against a URL.
*/
public function run(string $url, string $method = 'GET', ?string $body = null): LoadTestReport
{
$this->results = [];
$requestsPerWorker = (int) ceil($this->totalRequests / $this->concurrency);
// Sequential execution (for single-process PHP)
// In real world, use pcntl_fork or async HTTP client
for ($i = 0; $i < $this->totalRequests; $i++) {
$this->results[] = $this->sendRequest($url, $method, $body);
}
return $this->generateReport();
}
private function sendRequest(string $url, string $method, ?string $body): RequestResult
{
$ch = curl_init($url);
curl_setopt_array($ch, [
CURLOPT_RETURNTRANSFER => true,
CURLOPT_CUSTOMREQUEST => $method,
CURLOPT_TIMEOUT => 30,
CURLOPT_CONNECTTIMEOUT => 5,
]);
if ($body !== null) {
curl_setopt($ch, CURLOPT_POSTFIELDS, $body);
curl_setopt($ch, CURLOPT_HTTPHEADER, ['Content-Type: application/json']);
}
$startTime = hrtime(true);
$response = curl_exec($ch);
$durationMs = (hrtime(true) - $startTime) / 1_000_000;
$statusCode = (int) curl_getinfo($ch, CURLINFO_HTTP_CODE);
$error = curl_error($ch);
curl_close($ch);
return new RequestResult(
statusCode: $statusCode,
durationMs: $durationMs,
error: $error ?: null,
);
}
private function generateReport(): LoadTestReport
{
$latencies = array_map(
static fn(RequestResult $r) => $r->durationMs,
$this->results,
);
sort($latencies);
$successCount = count(array_filter(
$this->results,
static fn(RequestResult $r) => $r->statusCode >= 200 && $r->statusCode < 400,
));
$totalDuration = array_sum($latencies);
$count = count($latencies);
return new LoadTestReport(
totalRequests: $count,
successfulRequests: $successCount,
failedRequests: $count - $successCount,
avgLatencyMs: $totalDuration / $count,
p50Ms: $latencies[(int) ($count * 0.50)] ?? 0,
p95Ms: $latencies[(int) ($count * 0.95)] ?? 0,
p99Ms: $latencies[(int) ($count * 0.99)] ?? 0,
minMs: $latencies[0] ?? 0,
maxMs: $latencies[$count - 1] ?? 0,
rps: $count / ($totalDuration / 1000),
);
}
}
final readonly class RequestResult
{
public function __construct(
public int $statusCode,
public float $durationMs,
public ?string $error = null,
) {}
}
final readonly class LoadTestReport
{
public function __construct(
public int $totalRequests,
public int $successfulRequests,
public int $failedRequests,
public float $avgLatencyMs,
public float $p50Ms,
public float $p95Ms,
public float $p99Ms,
public float $minMs,
public float $maxMs,
public float $rps,
) {}
public function toArray(): array
{
return [
'total_requests' => $this->totalRequests,
'successful' => $this->successfulRequests,
'failed' => $this->failedRequests,
'error_rate' => round($this->failedRequests / $this->totalRequests * 100, 2) . '%',
'latency' => [
'avg' => round($this->avgLatencyMs, 2) . 'ms',
'p50' => round($this->p50Ms, 2) . 'ms',
'p95' => round($this->p95Ms, 2) . 'ms',
'p99' => round($this->p99Ms, 2) . 'ms',
'min' => round($this->minMs, 2) . 'ms',
'max' => round($this->maxMs, 2) . 'ms',
],
'throughput' => round($this->rps, 1) . ' RPS',
];
}
}
package loadtest
import (
"context"
"fmt"
"io"
"math"
"net/http"
"sort"
"strings"
"sync"
"time"
)
// RequestResult records the outcome of a single HTTP request.
type RequestResult struct {
StatusCode int
DurationMs float64
Err error
}
// Report summarizes load test results.
type Report struct {
TotalRequests int `json:"total_requests"`
SuccessfulRequests int `json:"successful"`
FailedRequests int `json:"failed"`
AvgLatencyMs float64 `json:"avg_ms"`
P50Ms float64 `json:"p50_ms"`
P95Ms float64 `json:"p95_ms"`
P99Ms float64 `json:"p99_ms"`
MinMs float64 `json:"min_ms"`
MaxMs float64 `json:"max_ms"`
RPS float64 `json:"rps"`
}
// LoadGenerator sends concurrent HTTP requests for load testing.
type LoadGenerator struct {
Concurrency int
TotalRequests int
}
// Run executes the load test against the given URL.
func (g *LoadGenerator) Run(ctx context.Context, url, method string, body string) Report {
results := make([]RequestResult, g.TotalRequests)
var wg sync.WaitGroup
sem := make(chan struct{}, g.Concurrency)
for i := range g.TotalRequests {
wg.Add(1)
sem <- struct{}{}
go func(idx int) {
defer wg.Done()
defer func() { <-sem }()
results[idx] = sendRequest(ctx, url, method, body)
}(i)
}
wg.Wait()
return generateReport(results)
}
func sendRequest(ctx context.Context, url, method, body string) RequestResult {
var bodyReader io.Reader
if body != "" {
bodyReader = strings.NewReader(body)
}
req, err := http.NewRequestWithContext(ctx, method, url, bodyReader)
if err != nil {
return RequestResult{Err: err}
}
if body != "" {
req.Header.Set("Content-Type", "application/json")
}
start := time.Now()
resp, err := http.DefaultClient.Do(req)
duration := float64(time.Since(start).Microseconds()) / 1000.0
if err != nil {
return RequestResult{DurationMs: duration, Err: err}
}
defer resp.Body.Close()
io.Copy(io.Discard, resp.Body)
return RequestResult{StatusCode: resp.StatusCode, DurationMs: duration}
}
func generateReport(results []RequestResult) Report {
var latencies []float64
successCount := 0
for _, r := range results {
latencies = append(latencies, r.DurationMs)
if r.StatusCode >= 200 && r.StatusCode < 400 {
successCount++
}
}
sort.Float64s(latencies)
n := len(latencies)
var total float64
for _, l := range latencies {
total += l
}
return Report{
TotalRequests: n,
SuccessfulRequests: successCount,
FailedRequests: n - successCount,
AvgLatencyMs: total / float64(n),
P50Ms: latencies[int(float64(n)*0.50)],
P95Ms: latencies[int(float64(n)*0.95)],
P99Ms: latencies[int(float64(n)*0.99)],
MinMs: latencies[0],
MaxMs: latencies[n-1],
RPS: float64(n) / (total / 1000),
}
}
namespace App.LoadTest;
using System.Diagnostics;
using System.Text;
// RequestResult records the outcome of a single HTTP request.
public readonly record struct RequestResult(int StatusCode, double DurationMs, string? Error);
// LoadTestReport summarizes load test results.
public readonly record struct LoadTestReport(
int TotalRequests,
int SuccessfulRequests,
int FailedRequests,
double AvgLatencyMs,
double P50Ms,
double P95Ms,
double P99Ms,
double MinMs,
double MaxMs,
double Rps)
{
public IReadOnlyDictionary<string, object> ToDictionary() => new Dictionary<string, object>
{
["total_requests"] = TotalRequests,
["successful"] = SuccessfulRequests,
["failed"] = FailedRequests,
["error_rate"] = $"{(double)FailedRequests / TotalRequests * 100:F2}%",
["latency"] = new Dictionary<string, string>
{
["avg"] = $"{AvgLatencyMs:F2}ms",
["p50"] = $"{P50Ms:F2}ms",
["p95"] = $"{P95Ms:F2}ms",
["p99"] = $"{P99Ms:F2}ms",
["min"] = $"{MinMs:F2}ms",
["max"] = $"{MaxMs:F2}ms",
},
["throughput"] = $"{Rps:F1} RPS",
};
}
// LoadGenerator sends concurrent HTTP requests for load testing.
// In production use k6, NBomber, or Gatling.
public sealed class LoadGenerator(HttpClient http, int concurrency = 10, int totalRequests = 1000)
{
// Run a load test against a URL.
public async Task<LoadTestReport> RunAsync(
string url,
HttpMethod? method = null,
string? body = null,
CancellationToken ct = default)
{
var results = new RequestResult[totalRequests];
var options = new ParallelOptions { MaxDegreeOfParallelism = concurrency, CancellationToken = ct };
await Parallel.ForAsync(0, totalRequests, options, async (index, token) =>
results[index] = await SendRequestAsync(url, method ?? HttpMethod.Get, body, token));
return GenerateReport(results);
}
private async Task<RequestResult> SendRequestAsync(
string url,
HttpMethod method,
string? body,
CancellationToken ct)
{
using var request = new HttpRequestMessage(method, url);
if (body is not null)
{
request.Content = new StringContent(body, Encoding.UTF8, "application/json");
}
var startTimestamp = Stopwatch.GetTimestamp();
try
{
using var response = await http.SendAsync(request, ct);
await response.Content.CopyToAsync(Stream.Null, ct);
return new RequestResult(
(int)response.StatusCode,
Stopwatch.GetElapsedTime(startTimestamp).TotalMilliseconds,
null);
}
catch (HttpRequestException ex)
{
return new RequestResult(0, Stopwatch.GetElapsedTime(startTimestamp).TotalMilliseconds, ex.Message);
}
}
private static LoadTestReport GenerateReport(RequestResult[] results)
{
var latencies = results.Select(result => result.DurationMs).Order().ToArray();
var successCount = results.Count(result => result.StatusCode is >= 200 and < 400);
var totalDuration = latencies.Sum();
var count = latencies.Length;
return new LoadTestReport(
TotalRequests: count,
SuccessfulRequests: successCount,
FailedRequests: count - successCount,
AvgLatencyMs: totalDuration / count,
P50Ms: latencies[(int)(count * 0.50)],
P95Ms: latencies[(int)(count * 0.95)],
P99Ms: latencies[(int)(count * 0.99)],
MinMs: latencies[0],
MaxMs: latencies[count - 1],
Rps: count / (totalDuration / 1000));
}
}
import asyncio
import time
from collections.abc import Sequence
from dataclasses import dataclass
import httpx
@dataclass(frozen=True, slots=True)
class RequestResult:
"""Records the outcome of a single HTTP request."""
status_code: int
duration_ms: float
error: str | None = None
@dataclass(frozen=True, slots=True)
class LoadTestReport:
"""Summarizes load test results."""
total_requests: int
successful_requests: int
failed_requests: int
avg_latency_ms: float
p50_ms: float
p95_ms: float
p99_ms: float
min_ms: float
max_ms: float
rps: float
def to_dict(self) -> dict[str, object]:
return {
"total_requests": self.total_requests,
"successful": self.successful_requests,
"failed": self.failed_requests,
"error_rate": f"{self.failed_requests / self.total_requests * 100:.2f}%",
"latency": {
"avg": f"{self.avg_latency_ms:.2f}ms",
"p50": f"{self.p50_ms:.2f}ms",
"p95": f"{self.p95_ms:.2f}ms",
"p99": f"{self.p99_ms:.2f}ms",
"min": f"{self.min_ms:.2f}ms",
"max": f"{self.max_ms:.2f}ms",
},
"throughput": f"{self.rps:.1f} RPS",
}
class LoadGenerator:
"""Sends concurrent HTTP requests for load testing.
In production use k6, Locust, or Gatling.
"""
def __init__(self, concurrency: int = 10, total_requests: int = 1000) -> None:
self._concurrency = concurrency
self._total_requests = total_requests
async def run(self, url: str, method: str = "GET", body: str | None = None) -> LoadTestReport:
"""Run a load test against a URL.
asyncio gives real concurrency in a single process, unlike the
sequential PHP version that needs pcntl_fork to parallelize.
"""
semaphore = asyncio.Semaphore(self._concurrency)
async with httpx.AsyncClient(timeout=30.0) as client:
async def worker() -> RequestResult:
async with semaphore:
return await self._send_request(client, url, method, body)
results = await asyncio.gather(*(worker() for _ in range(self._total_requests)))
return self._generate_report(results)
@staticmethod
async def _send_request(
client: httpx.AsyncClient,
url: str,
method: str,
body: str | None,
) -> RequestResult:
headers = {"Content-Type": "application/json"} if body is not None else None
start = time.perf_counter()
try:
response = await client.request(method, url, content=body, headers=headers)
except httpx.HTTPError as exc:
return RequestResult(0, (time.perf_counter() - start) * 1000, str(exc))
return RequestResult(response.status_code, (time.perf_counter() - start) * 1000)
@staticmethod
def _generate_report(results: Sequence[RequestResult]) -> LoadTestReport:
latencies = sorted(result.duration_ms for result in results)
success_count = sum(1 for result in results if 200 <= result.status_code < 400)
total_duration = sum(latencies)
count = len(latencies)
return LoadTestReport(
total_requests=count,
successful_requests=success_count,
failed_requests=count - success_count,
avg_latency_ms=total_duration / count,
p50_ms=latencies[int(count * 0.50)],
p95_ms=latencies[int(count * 0.95)],
p99_ms=latencies[int(count * 0.99)],
min_ms=latencies[0],
max_ms=latencies[count - 1],
rps=count / (total_duration / 1000),
)
package capacity
import "math"
import "time"
// Projection holds a projected future value.
type Projection struct {
Month string `json:"month"`
ProjectedValue float64 `json:"projected_value"`
Confidence string `json:"confidence"`
}
// LinearProjection projects future load using simple linear regression.
func LinearProjection(data []struct{ Month string; Value float64 }, monthsAhead int) []Projection {
n := len(data)
if n < 2 {
return nil
}
// Simple linear regression: y = mx + b
var sumX, sumY, sumXY, sumXX float64
for i, p := range data {
x := float64(i)
sumX += x
sumY += p.Value
sumXY += x * p.Value
sumXX += x * x
}
nf := float64(n)
m := (nf*sumXY - sumX*sumY) / (nf*sumXX - sumX*sumX)
b := (sumY - m*sumX) / nf
lastDate, _ := time.Parse("2006-01", data[n-1].Month)
var projections []Projection
for i := 1; i <= monthsAhead; i++ {
x := float64(n + i - 1)
projected := m*x + b
date := lastDate.AddDate(0, i, 0)
confidence := "low"
if i <= 3 {
confidence = "high"
} else if i <= 6 {
confidence = "medium"
}
projections = append(projections, Projection{
Month: date.Format("2006-01"),
ProjectedValue: math.Max(0, math.Round(projected*10)/10),
Confidence: confidence,
})
}
return projections
}
namespace App.Capacity;
using System.Globalization;
// A single observed data point in the historical series.
public readonly record struct DataPoint(string Month, double Value);
// A projected future value with its confidence level.
public readonly record struct Projection(string Month, double ProjectedValue, string Confidence);
public sealed class GrowthProjection
{
// Project future load based on historical data.
public IReadOnlyList<Projection> LinearProjection(IReadOnlyList<DataPoint> historicalData, int monthsAhead)
{
var n = historicalData.Count;
if (n < 2)
{
return [];
}
// Simple linear regression: y = mx + b
double sumX = 0, sumY = 0, sumXy = 0, sumXx = 0;
for (var i = 0; i < n; i++)
{
sumX += i;
sumY += historicalData[i].Value;
sumXy += i * historicalData[i].Value;
sumXx += (double)i * i;
}
var m = (n * sumXy - sumX * sumY) / (n * sumXx - sumX * sumX);
var b = (sumY - m * sumX) / n;
var lastDate = DateTime.ParseExact(historicalData[^1].Month, "yyyy-MM", CultureInfo.InvariantCulture);
var projections = new List<Projection>(monthsAhead);
for (var i = 1; i <= monthsAhead; i++)
{
var x = n + i - 1;
var projected = m * x + b;
var confidence = i switch
{
<= 3 => "high",
<= 6 => "medium",
_ => "low",
};
projections.Add(new Projection(
lastDate.AddMonths(i).ToString("yyyy-MM"),
Math.Round(Math.Max(0, projected), 1),
confidence));
}
return projections;
}
}
from collections.abc import Sequence
from dataclasses import dataclass
from datetime import datetime
from dateutil.relativedelta import relativedelta
@dataclass(frozen=True, slots=True)
class DataPoint:
"""A single observed data point in the historical series."""
month: str # "YYYY-MM"
value: float
@dataclass(frozen=True, slots=True)
class Projection:
"""A projected future value with its confidence level."""
month: str
projected_value: float
confidence: str
class GrowthProjection:
def linear_projection(
self,
historical_data: Sequence[DataPoint],
months_ahead: int,
) -> list[Projection]:
"""Project future load based on historical data."""
n = len(historical_data)
if n < 2:
return []
# Simple linear regression: y = mx + b
sum_x = sum_y = sum_xy = sum_xx = 0.0
for i, point in enumerate(historical_data):
sum_x += i
sum_y += point.value
sum_xy += i * point.value
sum_xx += i * i
m = (n * sum_xy - sum_x * sum_y) / (n * sum_xx - sum_x * sum_x)
b = (sum_y - m * sum_x) / n
last_date = datetime.strptime(historical_data[-1].month, "%Y-%m")
projections: list[Projection] = []
for i in range(1, months_ahead + 1):
x = n + i - 1
projected = m * x + b
if i <= 3:
confidence = "high"
elif i <= 6:
confidence = "medium"
else:
confidence = "low"
projections.append(
Projection(
month=(last_date + relativedelta(months=i)).strftime("%Y-%m"),
projected_value=round(max(0.0, projected), 1),
confidence=confidence,
)
)
return projections
## Стратегии масштабирования
Стратегия
Когда
Плюсы
Минусы
Vertical
Нужно быстро
Просто, без изменения кода
Предел hardware
Horizontal
Долгосрочный рост
Нет предела, fault tolerance
Сложность архитектуры
Auto-scaling
Переменная нагрузка
Оптимальные затраты
Задержка масштабирования
Pre-scaling
Предсказуемые пики
Готов к пику
Лишние ресурсы в обычное время
Правило: Планируйте ёмкость на 6-12 месяцев вперёд. Используйте данные, а не интуицию. Проводите load testing регулярно, а не перед запуском.