EasyТеория5 min

Чтение и запись файлов

Функция open(), режимы открытия, кодировки, конструкция with и работа с бинарными файлами

Работа с файлами -- одна из базовых операций в любом языке программирования. Python предоставляет простой и мощный API для чтения, записи и обработки файлов через встроенную функцию open().

Функция open()

open() -- основной способ работы с файлами. Она возвращает файловый объект:

# Basic signature
# open(file, mode='r', encoding=None, errors=None, newline=None)

# Open a text file for reading (default mode)
file = open("data.txt")
content = file.read()
file.close()  # Don't forget to close!

Всегда используйте with

Контекстный менеджер гарантирует закрытие файла:

# CORRECT — file is always closed
with open("data.txt") as f:
    content = f.read()
# File is closed here, even if an exception occurred

# WRONG — file may not be closed on error
f = open("data.txt")
content = f.read()  # If this raises, file stays open
f.close()

Режимы открытия файлов

Режим Описание Создает файл Стирает содержимое
'r' Чтение (по умолчанию) Нет Нет
'w' Запись Да Да
'a' Дозапись Да Нет
'x' Эксклюзивное создание Да (ошибка если существует) --
'r+' Чтение и запись Нет Нет
'w+' Запись и чтение Да Да
'b' Бинарный режим -- --
't' Текстовый режим (по умолчанию) -- --
# Write mode — creates or overwrites
with open("output.txt", "w") as f:
    f.write("Hello, World!\n")

# Append mode — adds to the end
with open("log.txt", "a") as f:
    f.write("New log entry\n")

# Exclusive creation — fails if file exists
try:
    with open("new_file.txt", "x") as f:
        f.write("This file must not exist before")
except FileExistsError:
    print("File already exists!")

# Read and write
with open("data.txt", "r+") as f:
    content = f.read()
    f.write("\nAppended text")

Чтение файлов

read() -- все содержимое

with open("data.txt") as f:
    content = f.read()  # Entire file as a string
    print(len(content))

read(n) -- n символов

with open("data.txt") as f:
    chunk = f.read(100)  # First 100 characters
    print(chunk)

readline() -- одна строка

with open("data.txt") as f:
    first_line = f.readline()    # Includes '\n'
    second_line = f.readline()
    print(first_line.strip())    # Remove trailing newline

readlines() -- все строки в список

with open("data.txt") as f:
    lines = f.readlines()  # List of strings with '\n'
    print(f"Total lines: {len(lines)}")

# Strip newlines
with open("data.txt") as f:
    lines = [line.strip() for line in f.readlines()]

Итерация по строкам (рекомендуемый способ)

Самый эффективный способ для больших файлов -- итерация по файловому объекту:

# Memory-efficient — reads line by line
with open("large_file.txt") as f:
    for line_number, line in enumerate(f, start=1):
        if "ERROR" in line:
            print(f"Line {line_number}: {line.strip()}")

Запись файлов

write() -- запись строки

with open("output.txt", "w") as f:
    f.write("First line\n")
    f.write("Second line\n")
    # write() returns the number of characters written
    chars_written = f.write("Third line\n")
    print(f"Wrote {chars_written} characters")

writelines() -- запись списка строк

lines = ["Line 1\n", "Line 2\n", "Line 3\n"]

with open("output.txt", "w") as f:
    f.writelines(lines)  # Does NOT add newlines automatically!

# With generator expression
data = ["Alice", "Bob", "Charlie"]
with open("names.txt", "w") as f:
    f.writelines(f"{name}\n" for name in data)
with open("report.txt", "w") as f:
    print("Report Title", file=f)
    print("=" * 40, file=f)
    print(f"Total items: {42}", file=f)
    print("Item 1", "Item 2", sep=" | ", file=f)

Кодировки

Всегда указывайте кодировку явно, особенно для не-ASCII текста:

# UTF-8 — default on most systems, but be explicit
with open("data.txt", encoding="utf-8") as f:
    content = f.read()

# Write with encoding
with open("russian.txt", "w", encoding="utf-8") as f:
    f.write("Привет, мир!\n")

# Read Windows-encoded file
with open("legacy.txt", encoding="cp1251") as f:
    content = f.read()

# Handle encoding errors
with open("mixed.txt", encoding="utf-8", errors="replace") as f:
    content = f.read()  # Invalid bytes replaced with '?'

with open("mixed.txt", encoding="utf-8", errors="ignore") as f:
    content = f.read()  # Invalid bytes silently skipped

Python 3.14: UTF-8 по умолчанию

Начиная с Python 3.15, UTF-8 станет кодировкой по умолчанию. В Python 3.14 можно включить предупреждения:

import sys

# Check current default encoding
print(sys.getdefaultencoding())  # 'utf-8'

# Best practice: always specify encoding explicitly
with open("data.txt", encoding="utf-8") as f:
    content = f.read()

Бинарные файлы

Для работы с изображениями, аудио и другими бинарными данными используйте режим 'b':

# Read binary file
with open("image.png", "rb") as f:
    data = f.read()  # Returns bytes, not str
    print(type(data))   # <class 'bytes'>
    print(data[:8])     # First 8 bytes (PNG header)

# Write binary file
with open("copy.png", "wb") as f:
    f.write(data)

# Copy a file efficiently
def copy_file(src: str, dst: str, chunk_size: int = 8192) -> int:
    """Copy a file in chunks — memory efficient."""
    total_bytes = 0
    with open(src, "rb") as source, open(dst, "wb") as target:
        while chunk := source.read(chunk_size):
            target.write(chunk)
            total_bytes += len(chunk)
    return total_bytes

bytes_copied = copy_file("large_video.mp4", "backup.mp4")
print(f"Copied {bytes_copied:,} bytes")

Walrus-оператор для чтения чанками

# Read large file in chunks using walrus operator (:=)
with open("huge_file.bin", "rb") as f:
    while chunk := f.read(4096):
        process_chunk(chunk)

Позиция в файле

with open("data.txt", "r+") as f:
    # Read first 10 characters
    start = f.read(10)
    print(f"Position after read: {f.tell()}")  # 10

    # Move to beginning
    f.seek(0)
    print(f"Position after seek(0): {f.tell()}")  # 0

    # Move to position 5
    f.seek(5)
    rest = f.read()  # Read from position 5 to end

Временные файлы

Модуль tempfile создает временные файлы, которые автоматически удаляются:

import tempfile

# Temporary file — deleted when closed
with tempfile.NamedTemporaryFile(mode="w", suffix=".txt", delete=False) as f:
    f.write("Temporary data\n")
    print(f"Temp file: {f.name}")

# Temporary directory
with tempfile.TemporaryDirectory() as tmpdir:
    filepath = f"{tmpdir}/data.txt"
    with open(filepath, "w") as f:
        f.write("Will be deleted with the directory")
# Directory and all contents are deleted here

Практический пример: обработка лог-файла

from collections import Counter
from datetime import datetime

def analyze_log(path: str) -> dict:
    """Analyze a log file and return statistics."""
    levels = Counter()
    errors: list[str] = []
    total_lines = 0

    with open(path, encoding="utf-8") as f:
        for line in f:
            total_lines += 1
            line = line.strip()

            # Count log levels
            for level in ("DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"):
                if level in line:
                    levels[level] += 1
                    break

            # Collect error messages
            if "ERROR" in line or "CRITICAL" in line:
                errors.append(line)

    return {
        "total_lines": total_lines,
        "levels": dict(levels),
        "errors": errors[:10],  # First 10 errors
        "error_rate": levels.get("ERROR", 0) / max(total_lines, 1),
    }

# Usage
stats = analyze_log("app.log")
print(f"Total lines: {stats['total_lines']}")
print(f"Error rate: {stats['error_rate']:.2%}")

Проверь себя

Чем отличается режим 'w' от 'a'?

Почему рекомендуется использовать with open() вместо простого open()?

Что возвращает read() при открытии файла в бинарном режиме ('rb')?

Какой самый эффективный по памяти способ чтения большого файла построчно?

Какой режим открытия файла используется по умолчанию в open()?