dataclass -- декоратор, который автоматически генерирует __init__, __repr__, __eq__ и другие методы. Это современный способ создания классов-контейнеров данных в Python.
Основы dataclass
from dataclasses import dataclass
# Without dataclass - verbose and repetitive
class UserOld:
def __init__(self, name: str, age: int, email: str) -> None:
self.name = name
self.age = age
self.email = email
def __repr__(self) -> str:
return f"User({self.name!r}, {self.age!r}, {self.email!r})"
def __eq__(self, other) -> bool:
if not isinstance(other, UserOld):
return NotImplemented
return (self.name, self.age, self.email) == (other.name, other.age, other.email)
# With dataclass - concise and complete
@dataclass
class User:
name: str
age: int
email: str
# Auto-generated: __init__, __repr__, __eq__
user = User("Иван", 25, "[email protected]")
print(user) # User(name='Иван', age=25, email='[email protected]')
print(user.name) # Иван
# __eq__ compares all fields
user2 = User("Иван", 25, "[email protected]")
print(user == user2) # True
Значения по умолчанию
from dataclasses import dataclass, field
from datetime import datetime
@dataclass
class Article:
title: str
author: str
content: str = ""
tags: list[str] = field(default_factory=list) # mutable defaults need field()
created_at: datetime = field(default_factory=datetime.now)
views: int = 0
# Fields with defaults MUST come after fields without defaults
article = Article("Python 3.14", "Иван")
print(article)
# Article(title='Python 3.14', author='Иван', content='', tags=[], ...)
article.tags.append("python")
print(article.tags) # ['python']
# Each instance gets its own list (no shared mutable default bug!)
article2 = Article("Go 1.25", "Мария")
print(article2.tags) # [] (independent)
Параметры @dataclass
from dataclasses import dataclass, field
# frozen=True - immutable dataclass (hashable)
@dataclass(frozen=True)
class Point:
x: float
y: float
p = Point(1.0, 2.0)
# p.x = 3.0 # FrozenInstanceError!
# Can be used in sets and as dict keys
points = {Point(0, 0), Point(1, 1), Point(0, 0)}
print(points) # {Point(x=0, y=0), Point(x=1, y=1)}
# slots=True - use __slots__ for memory efficiency (Python 3.10+)
@dataclass(slots=True)
class Vector:
x: float
y: float
z: float
v = Vector(1.0, 2.0, 3.0)
# v.w = 4.0 # AttributeError! (no __dict__, can't add attributes)
# kw_only=True - all fields are keyword-only (Python 3.10+)
@dataclass(kw_only=True)
class Config:
host: str
port: int = 8080
debug: bool = False
config = Config(host="localhost", debug=True)
# Config("localhost") # TypeError: positional arguments not allowed
# order=True - generate __lt__, __le__, __gt__, __ge__
@dataclass(order=True)
class Version:
major: int
minor: int
patch: int
versions = [Version(3, 14, 0), Version(3, 13, 1), Version(3, 14, 1)]
print(sorted(versions))
# [Version(major=3, minor=13, patch=1), Version(major=3, minor=14, patch=0), Version(major=3, minor=14, patch=1)]
Поле с исключением из сравнения
from dataclasses import dataclass, field
@dataclass
class CachedResult:
query: str
result: list[dict] = field(default_factory=list)
# Exclude from repr, comparison, and hash
cache_key: str = field(repr=False, compare=False, hash=False, default="")
_internal: int = field(repr=False, init=False, default=0)
r1 = CachedResult("SELECT *", [{"id": 1}], "key1")
r2 = CachedResult("SELECT *", [{"id": 1}], "key2")
print(r1 == r2) # True (cache_key excluded from comparison)
print(r1) # CachedResult(query='SELECT *', result=[{'id': 1}])
post_init -- пост-инициализация
from dataclasses import dataclass, field
import re
@dataclass
class Email:
address: str
domain: str = field(init=False) # computed, not in __init__
def __post_init__(self) -> None:
"""Called after auto-generated __init__."""
# Validation
if not re.match(r"^[^@]+@[^@]+\.[^@]+$", self.address):
raise ValueError(f"Невалидный email: {self.address}")
# Compute derived field
self.domain = self.address.split("@")[1]
email = Email("[email protected]")
print(email.domain) # mail.ru
# Email("invalid") # ValueError: Невалидный email: invalid
@dataclass
class Rectangle:
width: float
height: float
area: float = field(init=False)
perimeter: float = field(init=False)
def __post_init__(self) -> None:
if self.width <= 0 or self.height <= 0:
raise ValueError("Dimensions must be positive")
self.area = self.width * self.height
self.perimeter = 2 * (self.width + self.height)
rect = Rectangle(3, 4)
print(f"Площадь: {rect.area}, Периметр: {rect.perimeter}")
# Площадь: 12, Периметр: 14
Наследование dataclass
from dataclasses import dataclass, field
from datetime import datetime
@dataclass
class BaseModel:
id: int
created_at: datetime = field(default_factory=datetime.now)
@dataclass
class User(BaseModel):
name: str = ""
email: str = ""
@dataclass
class Admin(User):
permissions: list[str] = field(default_factory=list)
admin = Admin(id=1, name="Суперадмин", email="[email protected]", permissions=["all"])
print(admin)
# Admin(id=1, created_at=..., name='Суперадмин', email='[email protected]', permissions=['all'])
# All fields from parent classes are included
print(admin.id) # 1
print(admin.created_at) # datetime.now()
slots для оптимизации памяти
import sys
# Regular class - uses __dict__ (flexible but memory-heavy)
class RegularPoint:
def __init__(self, x: float, y: float) -> None:
self.x = x
self.y = y
# Slots class - fixed attributes (memory-efficient)
class SlottedPoint:
__slots__ = ("x", "y")
def __init__(self, x: float, y: float) -> None:
self.x = x
self.y = y
# Memory comparison
regular = RegularPoint(1.0, 2.0)
slotted = SlottedPoint(1.0, 2.0)
print(sys.getsizeof(regular.__dict__)) # ~104 bytes
# slotted has no __dict__
# Cannot add arbitrary attributes to slotted class
# slotted.z = 3.0 # AttributeError!
# With dataclass (Python 3.10+)
from dataclasses import dataclass
@dataclass(slots=True)
class OptimizedPoint:
x: float
y: float
z: float
# Best of both worlds: auto-generated methods + slots optimization
# Performance impact: slots are ~20-30% faster for attribute access
import timeit
def bench_regular():
p = RegularPoint(1.0, 2.0)
for _ in range(1000):
_ = p.x
_ = p.y
def bench_slotted():
p = SlottedPoint(1.0, 2.0)
for _ in range(1000):
_ = p.x
_ = p.y
print(f"Regular: {timeit.timeit(bench_regular, number=10000):.3f}s")
print(f"Slotted: {timeit.timeit(bench_slotted, number=10000):.3f}s")
NamedTuple vs dataclass
from typing import NamedTuple
from dataclasses import dataclass
# NamedTuple - immutable, tuple subclass
class PointNT(NamedTuple):
x: float
y: float
# Dataclass (frozen) - immutable, regular class
@dataclass(frozen=True)
class PointDC:
x: float
y: float
# Key differences:
p_nt = PointNT(1.0, 2.0)
p_dc = PointDC(1.0, 2.0)
# NamedTuple is a tuple - supports unpacking and indexing
x, y = p_nt # unpacking works
print(p_nt[0]) # 1.0 (index access)
print(isinstance(p_nt, tuple)) # True
# Dataclass is a regular class
# x, y = p_dc # won't work (not iterable by default)
# print(p_dc[0]) # won't work (no __getitem__)
# NamedTuple is more memory-efficient
import sys
print(sys.getsizeof(p_nt)) # ~64 bytes
print(sys.getsizeof(p_dc)) # ~48 bytes (with slots even less)
# When to use which:
# NamedTuple: immutable records, function return values, tuple compatibility
# dataclass: mutable or immutable, complex logic, inheritance, methods
Практический пример: система конфигурации
from dataclasses import dataclass, field, asdict, astuple
from typing import Self
import json
@dataclass(frozen=True, slots=True)
class DatabaseConfig:
host: str = "localhost"
port: int = 5432
name: str = "mydb"
user: str = "postgres"
password: str = field(default="", repr=False) # hide in repr
@property
def connection_string(self) -> str:
auth = f"{self.user}:{self.password}@" if self.password else f"{self.user}@"
return f"postgresql://{auth}{self.host}:{self.port}/{self.name}"
@dataclass(frozen=True, slots=True)
class RedisConfig:
host: str = "localhost"
port: int = 6379
db: int = 0
@dataclass(frozen=True, slots=True)
class AppConfig:
debug: bool = False
secret_key: str = "change-me"
database: DatabaseConfig = field(default_factory=DatabaseConfig)
redis: RedisConfig = field(default_factory=RedisConfig)
allowed_hosts: tuple[str, ...] = ("localhost",)
@classmethod
def from_json(cls, path: str) -> Self:
"""Load configuration from a JSON file."""
with open(path) as f:
data = json.load(f)
db_data = data.pop("database", {})
redis_data = data.pop("redis", {})
return cls(
database=DatabaseConfig(**db_data),
redis=RedisConfig(**redis_data),
**data,
)
def to_dict(self) -> dict:
"""Convert to dictionary (for serialization)."""
return asdict(self)
# Usage
config = AppConfig(
debug=True,
database=DatabaseConfig(host="db.prod.com", password="s3cret"),
allowed_hosts=("example.com", "*.example.com"),
)
print(config.database.connection_string)
# postgresql://postgres:[email protected]:5432/mydb
print(config)
# AppConfig(debug=True, secret_key='change-me',
# database=DatabaseConfig(host='db.prod.com', port=5432, name='mydb', user='postgres'),
# redis=RedisConfig(host='localhost', port=6379, db=0),
# allowed_hosts=('example.com', '*.example.com'))
# Serialize
print(json.dumps(config.to_dict(), indent=2))
asdict, astuple, replace
from dataclasses import dataclass, asdict, astuple, replace
@dataclass
class Product:
name: str
price: float
quantity: int
p = Product("Python Book", 1500.0, 10)
# Convert to dict
d = asdict(p)
print(d) # {'name': 'Python Book', 'price': 1500.0, 'quantity': 10}
# Convert to tuple
t = astuple(p)
print(t) # ('Python Book', 1500.0, 10)
# Create a modified copy (like _replace for NamedTuple)
p2 = replace(p, price=1200.0, quantity=15)
print(p2) # Product(name='Python Book', price=1200.0, quantity=15)
print(p) # Original unchanged
Сравнение: dataclass vs attrs vs Pydantic
# 1. dataclass (stdlib) - lightweight, good for most cases
from dataclasses import dataclass
@dataclass
class UserDC:
name: str
age: int
# 2. attrs - more features, external library
# pip install attrs
# import attrs
# @attrs.define
# class UserAttrs:
# name: str
# age: int = attrs.field(validator=attrs.validators.gt(0))
# 3. Pydantic - validation-first, for APIs
# pip install pydantic
# from pydantic import BaseModel, Field
# class UserPydantic(BaseModel):
# name: str = Field(min_length=1)
# age: int = Field(gt=0, lt=150)
# When to use which:
# dataclass: simple data containers, internal code
# attrs: complex validation, converters, more control
# Pydantic: API models, config parsing, data validation from external sources
Итоги
@dataclassавтоматически генерирует__init__,__repr__,__eq__field(default_factory=...)-- для изменяемых значений по умолчаниюfrozen=True-- неизменяемый и хешируемый dataclassslots=True(Python 3.10+) -- экономия памяти и ускорение доступаkw_only=True-- все поля только через именованные аргументы__post_init__-- валидация и вычисление производных полейasdict(),astuple(),replace()-- конвертация и создание копий- NamedTuple -- для неизменяемых записей с tuple-совместимостью
- slots -- до 30% экономии памяти для множества объектов
- Выбор:
dataclass(stdlib),attrs(расширенная валидация),Pydantic(API/внешние данные)