MidТеория8 min

Dataclasses и современные паттерны

@dataclass (frozen, slots, kw_only), NamedTuple, __slots__, Pydantic и attrs

dataclass -- декоратор, который автоматически генерирует __init__, __repr__, __eq__ и другие методы. Это современный способ создания классов-контейнеров данных в Python.

Основы dataclass

from dataclasses import dataclass

# Without dataclass - verbose and repetitive
class UserOld:
    def __init__(self, name: str, age: int, email: str) -> None:
        self.name = name
        self.age = age
        self.email = email

    def __repr__(self) -> str:
        return f"User({self.name!r}, {self.age!r}, {self.email!r})"

    def __eq__(self, other) -> bool:
        if not isinstance(other, UserOld):
            return NotImplemented
        return (self.name, self.age, self.email) == (other.name, other.age, other.email)

# With dataclass - concise and complete
@dataclass
class User:
    name: str
    age: int
    email: str

# Auto-generated: __init__, __repr__, __eq__
user = User("Иван", 25, "[email protected]")
print(user)              # User(name='Иван', age=25, email='[email protected]')
print(user.name)         # Иван

# __eq__ compares all fields
user2 = User("Иван", 25, "[email protected]")
print(user == user2)     # True

Значения по умолчанию

from dataclasses import dataclass, field
from datetime import datetime

@dataclass
class Article:
    title: str
    author: str
    content: str = ""
    tags: list[str] = field(default_factory=list)  # mutable defaults need field()
    created_at: datetime = field(default_factory=datetime.now)
    views: int = 0

    # Fields with defaults MUST come after fields without defaults

article = Article("Python 3.14", "Иван")
print(article)
# Article(title='Python 3.14', author='Иван', content='', tags=[], ...)

article.tags.append("python")
print(article.tags)  # ['python']

# Each instance gets its own list (no shared mutable default bug!)
article2 = Article("Go 1.25", "Мария")
print(article2.tags)  # [] (independent)

Параметры @dataclass

from dataclasses import dataclass, field

# frozen=True - immutable dataclass (hashable)
@dataclass(frozen=True)
class Point:
    x: float
    y: float

p = Point(1.0, 2.0)
# p.x = 3.0  # FrozenInstanceError!

# Can be used in sets and as dict keys
points = {Point(0, 0), Point(1, 1), Point(0, 0)}
print(points)  # {Point(x=0, y=0), Point(x=1, y=1)}

# slots=True - use __slots__ for memory efficiency (Python 3.10+)
@dataclass(slots=True)
class Vector:
    x: float
    y: float
    z: float

v = Vector(1.0, 2.0, 3.0)
# v.w = 4.0  # AttributeError! (no __dict__, can't add attributes)

# kw_only=True - all fields are keyword-only (Python 3.10+)
@dataclass(kw_only=True)
class Config:
    host: str
    port: int = 8080
    debug: bool = False

config = Config(host="localhost", debug=True)
# Config("localhost")  # TypeError: positional arguments not allowed

# order=True - generate __lt__, __le__, __gt__, __ge__
@dataclass(order=True)
class Version:
    major: int
    minor: int
    patch: int

versions = [Version(3, 14, 0), Version(3, 13, 1), Version(3, 14, 1)]
print(sorted(versions))
# [Version(major=3, minor=13, patch=1), Version(major=3, minor=14, patch=0), Version(major=3, minor=14, patch=1)]

Поле с исключением из сравнения

from dataclasses import dataclass, field

@dataclass
class CachedResult:
    query: str
    result: list[dict] = field(default_factory=list)
    # Exclude from repr, comparison, and hash
    cache_key: str = field(repr=False, compare=False, hash=False, default="")
    _internal: int = field(repr=False, init=False, default=0)

r1 = CachedResult("SELECT *", [{"id": 1}], "key1")
r2 = CachedResult("SELECT *", [{"id": 1}], "key2")
print(r1 == r2)  # True (cache_key excluded from comparison)
print(r1)        # CachedResult(query='SELECT *', result=[{'id': 1}])

post_init -- пост-инициализация

from dataclasses import dataclass, field
import re

@dataclass
class Email:
    address: str
    domain: str = field(init=False)  # computed, not in __init__

    def __post_init__(self) -> None:
        """Called after auto-generated __init__."""
        # Validation
        if not re.match(r"^[^@]+@[^@]+\.[^@]+$", self.address):
            raise ValueError(f"Невалидный email: {self.address}")
        # Compute derived field
        self.domain = self.address.split("@")[1]

email = Email("[email protected]")
print(email.domain)  # mail.ru
# Email("invalid")   # ValueError: Невалидный email: invalid

@dataclass
class Rectangle:
    width: float
    height: float
    area: float = field(init=False)
    perimeter: float = field(init=False)

    def __post_init__(self) -> None:
        if self.width <= 0 or self.height <= 0:
            raise ValueError("Dimensions must be positive")
        self.area = self.width * self.height
        self.perimeter = 2 * (self.width + self.height)

rect = Rectangle(3, 4)
print(f"Площадь: {rect.area}, Периметр: {rect.perimeter}")
# Площадь: 12, Периметр: 14

Наследование dataclass

from dataclasses import dataclass, field
from datetime import datetime

@dataclass
class BaseModel:
    id: int
    created_at: datetime = field(default_factory=datetime.now)

@dataclass
class User(BaseModel):
    name: str = ""
    email: str = ""

@dataclass
class Admin(User):
    permissions: list[str] = field(default_factory=list)

admin = Admin(id=1, name="Суперадмин", email="[email protected]", permissions=["all"])
print(admin)
# Admin(id=1, created_at=..., name='Суперадмин', email='[email protected]', permissions=['all'])

# All fields from parent classes are included
print(admin.id)          # 1
print(admin.created_at)  # datetime.now()

slots для оптимизации памяти

import sys

# Regular class - uses __dict__ (flexible but memory-heavy)
class RegularPoint:
    def __init__(self, x: float, y: float) -> None:
        self.x = x
        self.y = y

# Slots class - fixed attributes (memory-efficient)
class SlottedPoint:
    __slots__ = ("x", "y")

    def __init__(self, x: float, y: float) -> None:
        self.x = x
        self.y = y

# Memory comparison
regular = RegularPoint(1.0, 2.0)
slotted = SlottedPoint(1.0, 2.0)

print(sys.getsizeof(regular.__dict__))  # ~104 bytes
# slotted has no __dict__

# Cannot add arbitrary attributes to slotted class
# slotted.z = 3.0  # AttributeError!

# With dataclass (Python 3.10+)
from dataclasses import dataclass

@dataclass(slots=True)
class OptimizedPoint:
    x: float
    y: float
    z: float

# Best of both worlds: auto-generated methods + slots optimization

# Performance impact: slots are ~20-30% faster for attribute access
import timeit

def bench_regular():
    p = RegularPoint(1.0, 2.0)
    for _ in range(1000):
        _ = p.x
        _ = p.y

def bench_slotted():
    p = SlottedPoint(1.0, 2.0)
    for _ in range(1000):
        _ = p.x
        _ = p.y

print(f"Regular: {timeit.timeit(bench_regular, number=10000):.3f}s")
print(f"Slotted: {timeit.timeit(bench_slotted, number=10000):.3f}s")

NamedTuple vs dataclass

from typing import NamedTuple
from dataclasses import dataclass

# NamedTuple - immutable, tuple subclass
class PointNT(NamedTuple):
    x: float
    y: float

# Dataclass (frozen) - immutable, regular class
@dataclass(frozen=True)
class PointDC:
    x: float
    y: float

# Key differences:
p_nt = PointNT(1.0, 2.0)
p_dc = PointDC(1.0, 2.0)

# NamedTuple is a tuple - supports unpacking and indexing
x, y = p_nt              # unpacking works
print(p_nt[0])           # 1.0 (index access)
print(isinstance(p_nt, tuple))  # True

# Dataclass is a regular class
# x, y = p_dc            # won't work (not iterable by default)
# print(p_dc[0])         # won't work (no __getitem__)

# NamedTuple is more memory-efficient
import sys
print(sys.getsizeof(p_nt))  # ~64 bytes
print(sys.getsizeof(p_dc))  # ~48 bytes (with slots even less)

# When to use which:
# NamedTuple: immutable records, function return values, tuple compatibility
# dataclass:  mutable or immutable, complex logic, inheritance, methods

Практический пример: система конфигурации

from dataclasses import dataclass, field, asdict, astuple
from typing import Self
import json

@dataclass(frozen=True, slots=True)
class DatabaseConfig:
    host: str = "localhost"
    port: int = 5432
    name: str = "mydb"
    user: str = "postgres"
    password: str = field(default="", repr=False)  # hide in repr

    @property
    def connection_string(self) -> str:
        auth = f"{self.user}:{self.password}@" if self.password else f"{self.user}@"
        return f"postgresql://{auth}{self.host}:{self.port}/{self.name}"

@dataclass(frozen=True, slots=True)
class RedisConfig:
    host: str = "localhost"
    port: int = 6379
    db: int = 0

@dataclass(frozen=True, slots=True)
class AppConfig:
    debug: bool = False
    secret_key: str = "change-me"
    database: DatabaseConfig = field(default_factory=DatabaseConfig)
    redis: RedisConfig = field(default_factory=RedisConfig)
    allowed_hosts: tuple[str, ...] = ("localhost",)

    @classmethod
    def from_json(cls, path: str) -> Self:
        """Load configuration from a JSON file."""
        with open(path) as f:
            data = json.load(f)
        db_data = data.pop("database", {})
        redis_data = data.pop("redis", {})
        return cls(
            database=DatabaseConfig(**db_data),
            redis=RedisConfig(**redis_data),
            **data,
        )

    def to_dict(self) -> dict:
        """Convert to dictionary (for serialization)."""
        return asdict(self)

# Usage
config = AppConfig(
    debug=True,
    database=DatabaseConfig(host="db.prod.com", password="s3cret"),
    allowed_hosts=("example.com", "*.example.com"),
)

print(config.database.connection_string)
# postgresql://postgres:[email protected]:5432/mydb

print(config)
# AppConfig(debug=True, secret_key='change-me',
#   database=DatabaseConfig(host='db.prod.com', port=5432, name='mydb', user='postgres'),
#   redis=RedisConfig(host='localhost', port=6379, db=0),
#   allowed_hosts=('example.com', '*.example.com'))

# Serialize
print(json.dumps(config.to_dict(), indent=2))

asdict, astuple, replace

from dataclasses import dataclass, asdict, astuple, replace

@dataclass
class Product:
    name: str
    price: float
    quantity: int

p = Product("Python Book", 1500.0, 10)

# Convert to dict
d = asdict(p)
print(d)  # {'name': 'Python Book', 'price': 1500.0, 'quantity': 10}

# Convert to tuple
t = astuple(p)
print(t)  # ('Python Book', 1500.0, 10)

# Create a modified copy (like _replace for NamedTuple)
p2 = replace(p, price=1200.0, quantity=15)
print(p2)  # Product(name='Python Book', price=1200.0, quantity=15)
print(p)   # Original unchanged

Сравнение: dataclass vs attrs vs Pydantic

# 1. dataclass (stdlib) - lightweight, good for most cases
from dataclasses import dataclass

@dataclass
class UserDC:
    name: str
    age: int

# 2. attrs - more features, external library
# pip install attrs
# import attrs
# @attrs.define
# class UserAttrs:
#     name: str
#     age: int = attrs.field(validator=attrs.validators.gt(0))

# 3. Pydantic - validation-first, for APIs
# pip install pydantic
# from pydantic import BaseModel, Field
# class UserPydantic(BaseModel):
#     name: str = Field(min_length=1)
#     age: int = Field(gt=0, lt=150)

# When to use which:
# dataclass: simple data containers, internal code
# attrs: complex validation, converters, more control
# Pydantic: API models, config parsing, data validation from external sources

Итоги

  • @dataclass автоматически генерирует __init__, __repr__, __eq__
  • field(default_factory=...) -- для изменяемых значений по умолчанию
  • frozen=True -- неизменяемый и хешируемый dataclass
  • slots=True (Python 3.10+) -- экономия памяти и ускорение доступа
  • kw_only=True -- все поля только через именованные аргументы
  • __post_init__ -- валидация и вычисление производных полей
  • asdict(), astuple(), replace() -- конвертация и создание копий
  • NamedTuple -- для неизменяемых записей с tuple-совместимостью
  • slots -- до 30% экономии памяти для множества объектов
  • Выбор: dataclass (stdlib), attrs (расширенная валидация), Pydantic (API/внешние данные)

Проверь себя

Почему для изменяемых значений по умолчанию в dataclass используется field(default_factory=list)?

Когда вызывается __post_init__ в dataclass?

Какой параметр @dataclass делает экземпляры неизменяемыми?

Какая функция создаёт изменённую копию dataclass-экземпляра?