Работа с файлами -- одна из базовых операций в любом языке программирования. Python предоставляет простой и мощный API для чтения, записи и обработки файлов через встроенную функцию open().
Функция open()
open() -- основной способ работы с файлами. Она возвращает файловый объект:
# Basic signature
# open(file, mode='r', encoding=None, errors=None, newline=None)
# Open a text file for reading (default mode)
file = open("data.txt")
content = file.read()
file.close() # Don't forget to close!
Всегда используйте with
Контекстный менеджер гарантирует закрытие файла:
# CORRECT — file is always closed
with open("data.txt") as f:
content = f.read()
# File is closed here, even if an exception occurred
# WRONG — file may not be closed on error
f = open("data.txt")
content = f.read() # If this raises, file stays open
f.close()
Режимы открытия файлов
| Режим | Описание | Создает файл | Стирает содержимое |
|---|---|---|---|
'r' |
Чтение (по умолчанию) | Нет | Нет |
'w' |
Запись | Да | Да |
'a' |
Дозапись | Да | Нет |
'x' |
Эксклюзивное создание | Да (ошибка если существует) | -- |
'r+' |
Чтение и запись | Нет | Нет |
'w+' |
Запись и чтение | Да | Да |
'b' |
Бинарный режим | -- | -- |
't' |
Текстовый режим (по умолчанию) | -- | -- |
# Write mode — creates or overwrites
with open("output.txt", "w") as f:
f.write("Hello, World!\n")
# Append mode — adds to the end
with open("log.txt", "a") as f:
f.write("New log entry\n")
# Exclusive creation — fails if file exists
try:
with open("new_file.txt", "x") as f:
f.write("This file must not exist before")
except FileExistsError:
print("File already exists!")
# Read and write
with open("data.txt", "r+") as f:
content = f.read()
f.write("\nAppended text")
Чтение файлов
read() -- все содержимое
with open("data.txt") as f:
content = f.read() # Entire file as a string
print(len(content))
read(n) -- n символов
with open("data.txt") as f:
chunk = f.read(100) # First 100 characters
print(chunk)
readline() -- одна строка
with open("data.txt") as f:
first_line = f.readline() # Includes '\n'
second_line = f.readline()
print(first_line.strip()) # Remove trailing newline
readlines() -- все строки в список
with open("data.txt") as f:
lines = f.readlines() # List of strings with '\n'
print(f"Total lines: {len(lines)}")
# Strip newlines
with open("data.txt") as f:
lines = [line.strip() for line in f.readlines()]
Итерация по строкам (рекомендуемый способ)
Самый эффективный способ для больших файлов -- итерация по файловому объекту:
# Memory-efficient — reads line by line
with open("large_file.txt") as f:
for line_number, line in enumerate(f, start=1):
if "ERROR" in line:
print(f"Line {line_number}: {line.strip()}")
Запись файлов
write() -- запись строки
with open("output.txt", "w") as f:
f.write("First line\n")
f.write("Second line\n")
# write() returns the number of characters written
chars_written = f.write("Third line\n")
print(f"Wrote {chars_written} characters")
writelines() -- запись списка строк
lines = ["Line 1\n", "Line 2\n", "Line 3\n"]
with open("output.txt", "w") as f:
f.writelines(lines) # Does NOT add newlines automatically!
# With generator expression
data = ["Alice", "Bob", "Charlie"]
with open("names.txt", "w") as f:
f.writelines(f"{name}\n" for name in data)
print() в файл
with open("report.txt", "w") as f:
print("Report Title", file=f)
print("=" * 40, file=f)
print(f"Total items: {42}", file=f)
print("Item 1", "Item 2", sep=" | ", file=f)
Кодировки
Всегда указывайте кодировку явно, особенно для не-ASCII текста:
# UTF-8 — default on most systems, but be explicit
with open("data.txt", encoding="utf-8") as f:
content = f.read()
# Write with encoding
with open("russian.txt", "w", encoding="utf-8") as f:
f.write("Привет, мир!\n")
# Read Windows-encoded file
with open("legacy.txt", encoding="cp1251") as f:
content = f.read()
# Handle encoding errors
with open("mixed.txt", encoding="utf-8", errors="replace") as f:
content = f.read() # Invalid bytes replaced with '?'
with open("mixed.txt", encoding="utf-8", errors="ignore") as f:
content = f.read() # Invalid bytes silently skipped
Python 3.14: UTF-8 по умолчанию
Начиная с Python 3.15, UTF-8 станет кодировкой по умолчанию. В Python 3.14 можно включить предупреждения:
import sys
# Check current default encoding
print(sys.getdefaultencoding()) # 'utf-8'
# Best practice: always specify encoding explicitly
with open("data.txt", encoding="utf-8") as f:
content = f.read()
Бинарные файлы
Для работы с изображениями, аудио и другими бинарными данными используйте режим 'b':
# Read binary file
with open("image.png", "rb") as f:
data = f.read() # Returns bytes, not str
print(type(data)) # <class 'bytes'>
print(data[:8]) # First 8 bytes (PNG header)
# Write binary file
with open("copy.png", "wb") as f:
f.write(data)
# Copy a file efficiently
def copy_file(src: str, dst: str, chunk_size: int = 8192) -> int:
"""Copy a file in chunks — memory efficient."""
total_bytes = 0
with open(src, "rb") as source, open(dst, "wb") as target:
while chunk := source.read(chunk_size):
target.write(chunk)
total_bytes += len(chunk)
return total_bytes
bytes_copied = copy_file("large_video.mp4", "backup.mp4")
print(f"Copied {bytes_copied:,} bytes")
Walrus-оператор для чтения чанками
# Read large file in chunks using walrus operator (:=)
with open("huge_file.bin", "rb") as f:
while chunk := f.read(4096):
process_chunk(chunk)
Позиция в файле
with open("data.txt", "r+") as f:
# Read first 10 characters
start = f.read(10)
print(f"Position after read: {f.tell()}") # 10
# Move to beginning
f.seek(0)
print(f"Position after seek(0): {f.tell()}") # 0
# Move to position 5
f.seek(5)
rest = f.read() # Read from position 5 to end
Временные файлы
Модуль tempfile создает временные файлы, которые автоматически удаляются:
import tempfile
# Temporary file — deleted when closed
with tempfile.NamedTemporaryFile(mode="w", suffix=".txt", delete=False) as f:
f.write("Temporary data\n")
print(f"Temp file: {f.name}")
# Temporary directory
with tempfile.TemporaryDirectory() as tmpdir:
filepath = f"{tmpdir}/data.txt"
with open(filepath, "w") as f:
f.write("Will be deleted with the directory")
# Directory and all contents are deleted here
Практический пример: обработка лог-файла
from collections import Counter
from datetime import datetime
def analyze_log(path: str) -> dict:
"""Analyze a log file and return statistics."""
levels = Counter()
errors: list[str] = []
total_lines = 0
with open(path, encoding="utf-8") as f:
for line in f:
total_lines += 1
line = line.strip()
# Count log levels
for level in ("DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"):
if level in line:
levels[level] += 1
break
# Collect error messages
if "ERROR" in line or "CRITICAL" in line:
errors.append(line)
return {
"total_lines": total_lines,
"levels": dict(levels),
"errors": errors[:10], # First 10 errors
"error_rate": levels.get("ERROR", 0) / max(total_lines, 1),
}
# Usage
stats = analyze_log("app.log")
print(f"Total lines: {stats['total_lines']}")
print(f"Error rate: {stats['error_rate']:.2%}")