MidТеория6 min

JSON, CSV, TOML

Сериализация данных: модули json, csv, tomllib, pickle и интеграция с dataclasses

Сериализация -- это преобразование объектов Python в формат, пригодный для хранения или передачи. Python предоставляет встроенные модули для работы с популярными форматами данных: JSON, CSV, TOML и pickle.

JSON

JSON (JavaScript Object Notation) -- самый распространенный формат обмена данными. Модуль json входит в стандартную библиотеку.

Сериализация (Python -> JSON)

import json

data = {
    "name": "Alice",
    "age": 30,
    "skills": ["Python", "SQL", "Docker"],
    "address": {
        "city": "Moscow",
        "country": "Russia"
    },
    "active": True,
    "score": None
}

# Convert to JSON string
json_str = json.dumps(data)
print(json_str)
# {"name": "Alice", "age": 30, "skills": ["Python", "SQL", "Docker"], ...}

# Pretty-printed JSON
pretty = json.dumps(data, indent=2, ensure_ascii=False)
print(pretty)
# {
#   "name": "Alice",
#   "age": 30,
#   ...
# }

# Write to file
with open("user.json", "w", encoding="utf-8") as f:
    json.dump(data, f, indent=2, ensure_ascii=False)

Десериализация (JSON -> Python)

import json

# From string
json_str = '{"name": "Alice", "age": 30}'
data = json.loads(json_str)
print(data["name"])  # Alice
print(type(data))    # <class 'dict'>

# From file
with open("user.json", encoding="utf-8") as f:
    data = json.load(f)
    print(data["skills"])  # ['Python', 'SQL', 'Docker']

Соответствие типов

Python JSON
dict object
list, tuple array
str string
int, float number
True / False true / false
None null

Кастомная сериализация

JSON не поддерживает все типы Python. Для кастомных объектов нужен encoder:

import json
from datetime import datetime, date
from dataclasses import dataclass, asdict
from decimal import Decimal
from pathlib import Path

class CustomEncoder(json.JSONEncoder):
    """JSON encoder that handles common Python types."""

    def default(self, obj):
        if isinstance(obj, (datetime, date)):
            return obj.isoformat()
        if isinstance(obj, Decimal):
            return str(obj)
        if isinstance(obj, Path):
            return str(obj)
        if isinstance(obj, set):
            return sorted(obj)
        if isinstance(obj, bytes):
            return obj.decode("utf-8", errors="replace")
        return super().default(obj)

data = {
    "timestamp": datetime.now(),
    "price": Decimal("19.99"),
    "config_path": Path("/etc/app/config.toml"),
    "tags": {"python", "json", "tutorial"},
}

result = json.dumps(data, cls=CustomEncoder, indent=2)
print(result)

dataclasses и JSON

from dataclasses import dataclass, asdict, field
import json

@dataclass
class Address:
    city: str
    country: str
    zip_code: str = ""

@dataclass
class User:
    name: str
    email: str
    age: int
    address: Address
    tags: list[str] = field(default_factory=list)

# Serialize dataclass to JSON
user = User(
    name="Bob",
    email="[email protected]",
    age=25,
    address=Address(city="Saint Petersburg", country="Russia"),
    tags=["developer", "python"]
)

json_str = json.dumps(asdict(user), indent=2, ensure_ascii=False)
print(json_str)

# Deserialize JSON to dataclass
data = json.loads(json_str)
restored = User(
    name=data["name"],
    email=data["email"],
    age=data["age"],
    address=Address(**data["address"]),
    tags=data["tags"]
)
print(restored)

CSV

Модуль csv работает с табличными данными в формате CSV (Comma-Separated Values).

Чтение CSV

import csv

# Read as lists
with open("data.csv", encoding="utf-8") as f:
    reader = csv.reader(f)
    header = next(reader)  # First row is header
    for row in reader:
        print(row)  # ['Alice', '30', 'Python']

# Read as dictionaries (recommended)
with open("data.csv", encoding="utf-8") as f:
    reader = csv.DictReader(f)
    for row in reader:
        print(row["name"], row["age"])  # Alice 30

Запись CSV

import csv

# Write from lists
header = ["name", "age", "language"]
rows = [
    ["Alice", 30, "Python"],
    ["Bob", 25, "Go"],
    ["Charlie", 35, "Rust"],
]

with open("output.csv", "w", encoding="utf-8", newline="") as f:
    writer = csv.writer(f)
    writer.writerow(header)
    writer.writerows(rows)

# Write from dictionaries
users = [
    {"name": "Alice", "age": 30, "language": "Python"},
    {"name": "Bob", "age": 25, "language": "Go"},
]

with open("users.csv", "w", encoding="utf-8", newline="") as f:
    writer = csv.DictWriter(f, fieldnames=["name", "age", "language"])
    writer.writeheader()
    writer.writerows(users)

Диалекты и разделители

import csv

# Semicolon-separated (common in Europe)
with open("european.csv", encoding="utf-8") as f:
    reader = csv.reader(f, delimiter=";")
    for row in reader:
        print(row)

# Tab-separated (TSV)
with open("data.tsv", encoding="utf-8") as f:
    reader = csv.reader(f, delimiter="\t")
    for row in reader:
        print(row)

# Custom dialect
csv.register_dialect("pipes", delimiter="|", quoting=csv.QUOTE_MINIMAL)
with open("piped.csv", encoding="utf-8") as f:
    reader = csv.reader(f, dialect="pipes")
    for row in reader:
        print(row)

TOML

TOML (Tom's Obvious Minimal Language) -- формат конфигурационных файлов. Модуль tomllib для чтения добавлен в Python 3.11.

Чтение TOML (tomllib)

import tomllib

# Read TOML file
with open("pyproject.toml", "rb") as f:  # Note: binary mode!
    config = tomllib.load(f)

print(config["project"]["name"])
print(config["project"]["version"])

# Parse TOML string
toml_str = """
[database]
host = "localhost"
port = 5432
name = "mydb"
ssl = true

[database.pool]
min_size = 5
max_size = 20
"""

config = tomllib.loads(toml_str)
print(config["database"]["host"])         # localhost
print(config["database"]["pool"]["max_size"])  # 20

TOML типы данных

import tomllib

toml_data = """
# Strings
title = "TOML Example"
multiline = '''
Multiple
lines
'''

# Numbers
integer = 42
float_val = 3.14
hex_val = 0xff
infinity = inf

# Boolean
enabled = true

# Dates
created = 2024-01-15
updated = 2024-01-15T10:30:00

# Arrays
tags = ["python", "toml", "config"]
nested = [[1, 2], [3, 4]]

# Inline table
point = { x = 1, y = 2 }
"""

config = tomllib.loads(toml_data)
print(type(config["created"]))   # <class 'datetime.date'>
print(type(config["updated"]))   # <class 'datetime.datetime'>
print(config["tags"])            # ['python', 'toml', 'config']

Запись TOML

tomllib поддерживает только чтение. Для записи используйте tomli-w:

# pip install tomli-w
import tomli_w

config = {
    "project": {
        "name": "myapp",
        "version": "1.0.0",
        "dependencies": ["requests", "pydantic"],
    },
    "tool": {
        "ruff": {
            "line-length": 88,
            "target-version": "py312",
        }
    }
}

with open("config.toml", "wb") as f:
    tomli_w.dump(config, f)

pickle -- бинарная сериализация

pickle сериализует произвольные объекты Python в бинарный формат:

import pickle

# Serialize any Python object
data = {
    "users": [{"name": "Alice", "scores": [95, 87, 92]}],
    "metadata": {"version": 2, "tags": {"important", "reviewed"}},
}

# Write to file
with open("data.pkl", "wb") as f:
    pickle.dump(data, f)

# Read from file
with open("data.pkl", "rb") as f:
    restored = pickle.load(f)
    print(restored["metadata"]["tags"])  # {'important', 'reviewed'}

Предупреждение о безопасности

import pickle

# WARNING: pickle can execute arbitrary code!
# NEVER unpickle data from untrusted sources

# This is DANGEROUS:
# pickle.loads(data_from_internet)  # Could execute malicious code

# Safe alternatives for data exchange:
# - JSON for simple data structures
# - Protocol Buffers / MessagePack for performance
# - Pydantic for validated data

pickle vs JSON

Аспект pickle JSON
Типы Любые Python-объекты Ограниченные (dict, list, str, int, float, bool, None)
Читаемость Бинарный Текстовый
Безопасность Опасен (code execution) Безопасен
Язык Только Python Межъязыковой
Скорость Быстрее Медленнее
Применение Кэш, ML-модели API, конфиги, обмен данными

Практический пример: конвертер форматов

import json
import csv
import tomllib
from pathlib import Path
from dataclasses import dataclass, asdict

@dataclass
class Config:
    host: str
    port: int
    debug: bool
    allowed_origins: list[str]

def load_config(path: str) -> Config:
    """Load config from JSON, TOML, or CSV."""
    file_path = Path(path)
    suffix = file_path.suffix.lower()

    if suffix == ".json":
        with file_path.open(encoding="utf-8") as f:
            data = json.load(f)
    elif suffix == ".toml":
        with file_path.open("rb") as f:
            data = tomllib.load(f)
    else:
        raise ValueError(f"Unsupported format: {suffix}")

    return Config(
        host=data["host"],
        port=data["port"],
        debug=data.get("debug", False),
        allowed_origins=data.get("allowed_origins", []),
    )

def export_as_json(config: Config, path: str) -> None:
    """Export config as JSON."""
    with open(path, "w", encoding="utf-8") as f:
        json.dump(asdict(config), f, indent=2)

def export_as_csv(configs: list[Config], path: str) -> None:
    """Export multiple configs as CSV."""
    with open(path, "w", encoding="utf-8", newline="") as f:
        writer = csv.DictWriter(
            f,
            fieldnames=["host", "port", "debug", "allowed_origins"]
        )
        writer.writeheader()
        for config in configs:
            row = asdict(config)
            row["allowed_origins"] = ";".join(row["allowed_origins"])
            writer.writerow(row)

Проверь себя

Какая функция преобразует dataclass в словарь для сериализации в JSON?

В каком режиме нужно открывать файл для tomllib.load()?

Какой класс модуля csv позволяет читать строки как словари?

Почему pickle опасен для данных из недоверенных источников?

Какой метод модуля json записывает данные в файл?