Модуль itertools -- это набор быстрых и эффективных по памяти инструментов для работы с итераторами. Написанный на C, он значительно быстрее эквивалентного Python-кода. Это один из самых полезных модулей стандартной библиотеки.
Бесконечные итераторы
count -- бесконечный счетчик
from itertools import count
# Count from 0 with step 1 (default)
for i in count():
if i >= 5:
break
print(i) # 0, 1, 2, 3, 4
# Count with start and step
for i in count(start=10, step=3):
if i > 25:
break
print(i) # 10, 13, 16, 19, 22, 25
# Useful for numbering
data = ["Alice", "Bob", "Charlie"]
numbered = list(zip(count(1), data))
print(numbered) # [(1, 'Alice'), (2, 'Bob'), (3, 'Charlie')]
cycle -- бесконечный цикл
from itertools import cycle, islice
# Cycle through elements forever
colors = cycle(["red", "green", "blue"])
print(list(islice(colors, 7)))
# ['red', 'green', 'blue', 'red', 'green', 'blue', 'red']
# Assign colors to items
items = ["item_1", "item_2", "item_3", "item_4", "item_5"]
colored = list(zip(items, cycle(["primary", "secondary"])))
print(colored)
# [('item_1', 'primary'), ('item_2', 'secondary'), ('item_3', 'primary'), ...]
repeat -- повторение значения
from itertools import repeat
# Repeat a value n times
zeros = list(repeat(0, 5))
print(zeros) # [0, 0, 0, 0, 0]
# Useful with map for constant operations
from operator import mul
result = list(map(mul, range(1, 6), repeat(3)))
print(result) # [3, 6, 9, 12, 15] — each multiplied by 3
Комбинирование итераторов
chain -- объединение последовательностей
from itertools import chain
# Concatenate multiple iterables
combined = list(chain([1, 2], [3, 4], [5, 6]))
print(combined) # [1, 2, 3, 4, 5, 6]
# chain.from_iterable — for a single iterable of iterables
lists = [[1, 2], [3, 4], [5, 6]]
flat = list(chain.from_iterable(lists))
print(flat) # [1, 2, 3, 4, 5, 6]
# Flatten one level of nesting
nested = [["a", "b"], ["c"], ["d", "e", "f"]]
flat = list(chain.from_iterable(nested))
print(flat) # ['a', 'b', 'c', 'd', 'e', 'f']
islice -- срез итератора
from itertools import islice
# Like slicing but for any iterator
def infinite_squares():
i = 0
while True:
yield i ** 2
i += 1
# Take first 5 elements
first_five = list(islice(infinite_squares(), 5))
print(first_five) # [0, 1, 4, 9, 16]
# Skip first 3, take next 4
skipped = list(islice(infinite_squares(), 3, 7))
print(skipped) # [9, 16, 25, 36]
# Every other element from first 10
every_other = list(islice(infinite_squares(), 0, 10, 2))
print(every_other) # [0, 4, 16, 36, 64]
zip_longest -- zip с заполнением
from itertools import zip_longest
names = ["Alice", "Bob", "Charlie"]
scores = [95, 87]
# Regular zip stops at shortest
print(list(zip(names, scores)))
# [('Alice', 95), ('Bob', 87)]
# zip_longest fills missing values
print(list(zip_longest(names, scores, fillvalue=0)))
# [('Alice', 95), ('Bob', 87), ('Charlie', 0)]
Фильтрация
takewhile и dropwhile
from itertools import takewhile, dropwhile
data = [1, 3, 5, 7, 2, 4, 6, 8]
# Take elements WHILE condition is true
small = list(takewhile(lambda x: x < 6, data))
print(small) # [1, 3, 5]
# Drop elements WHILE condition is true, then take the rest
big = list(dropwhile(lambda x: x < 6, data))
print(big) # [7, 2, 4, 6, 8]
filterfalse -- обратный filter
from itertools import filterfalse
numbers = range(10)
# Keep elements where predicate is False
odd = list(filterfalse(lambda x: x % 2 == 0, numbers))
print(odd) # [1, 3, 5, 7, 9]
compress -- фильтрация по маске
from itertools import compress
data = ["a", "b", "c", "d", "e"]
mask = [True, False, True, False, True]
result = list(compress(data, mask))
print(result) # ['a', 'c', 'e']
Группировка
groupby -- группировка последовательных элементов
from itertools import groupby
# IMPORTANT: data must be sorted by the grouping key!
data = [
{"name": "Alice", "dept": "Engineering"},
{"name": "Bob", "dept": "Engineering"},
{"name": "Charlie", "dept": "Marketing"},
{"name": "Diana", "dept": "Marketing"},
{"name": "Eve", "dept": "Sales"},
]
# Group by department
for dept, members in groupby(data, key=lambda x: x["dept"]):
member_list = list(members)
print(f"{dept}: {[m['name'] for m in member_list]}")
# Engineering: ['Alice', 'Bob']
# Marketing: ['Charlie', 'Diana']
# Sales: ['Eve']
Группировка чисел
from itertools import groupby
# Consecutive identical elements
text = "aaabbccdddeee"
groups = [(char, len(list(group))) for char, group in groupby(text)]
print(groups) # [('a', 3), ('b', 2), ('c', 2), ('d', 3), ('e', 3)]
# Run-length encoding
def rle_encode(data: str) -> list[tuple[str, int]]:
"""Run-length encode a string."""
return [(char, len(list(group))) for char, group in groupby(data)]
def rle_decode(encoded: list[tuple[str, int]]) -> str:
"""Decode run-length encoded data."""
return "".join(char * count for char, count in encoded)
encoded = rle_encode("aaabbccdddeee")
print(encoded) # [('a', 3), ('b', 2), ...]
print(rle_decode(encoded)) # aaabbccdddeee
batched (Python 3.12+)
Разбивает итерируемый объект на группы фиксированного размера:
from itertools import batched
data = range(10)
# Split into batches of 3
for batch in batched(data, 3):
print(batch)
# (0, 1, 2)
# (3, 4, 5)
# (6, 7, 8)
# (9,) — last batch may be shorter
# Process items in batches (e.g., batch API calls)
users = ["user_1", "user_2", "user_3", "user_4", "user_5"]
for batch in batched(users, 2):
print(f"Processing batch: {batch}")
# Processing batch: ('user_1', 'user_2')
# Processing batch: ('user_3', 'user_4')
# Processing batch: ('user_5',)
pairwise (Python 3.10+)
Возвращает пары последовательных элементов:
from itertools import pairwise
# Consecutive pairs
data = [1, 2, 3, 4, 5]
pairs = list(pairwise(data))
print(pairs) # [(1, 2), (2, 3), (3, 4), (4, 5)]
# Calculate differences between consecutive elements
values = [10, 15, 12, 20, 18]
diffs = [b - a for a, b in pairwise(values)]
print(diffs) # [5, -3, 8, -2]
# Check if sequence is sorted
numbers = [1, 3, 5, 7, 9]
is_sorted = all(a <= b for a, b in pairwise(numbers))
print(is_sorted) # True
Комбинаторные функции
product -- декартово произведение
from itertools import product
# All combinations of two sequences
colors = ["red", "blue"]
sizes = ["S", "M", "L"]
for combo in product(colors, sizes):
print(combo)
# ('red', 'S'), ('red', 'M'), ('red', 'L'),
# ('blue', 'S'), ('blue', 'M'), ('blue', 'L')
# Repeat parameter — like nested loops
digits = [0, 1]
for combo in product(digits, repeat=3):
print(combo)
# (0, 0, 0), (0, 0, 1), (0, 1, 0), ..., (1, 1, 1)
permutations -- перестановки
from itertools import permutations
# All orderings of elements
perms = list(permutations([1, 2, 3]))
print(perms)
# [(1, 2, 3), (1, 3, 2), (2, 1, 3), (2, 3, 1), (3, 1, 2), (3, 2, 1)]
# Permutations of specific length
perms_2 = list(permutations("ABC", 2))
print(perms_2)
# [('A', 'B'), ('A', 'C'), ('B', 'A'), ('B', 'C'), ('C', 'A'), ('C', 'B')]
combinations -- сочетания (без повторений)
from itertools import combinations, combinations_with_replacement
# Choose 2 from 4 (order doesn't matter)
combos = list(combinations([1, 2, 3, 4], 2))
print(combos)
# [(1, 2), (1, 3), (1, 4), (2, 3), (2, 4), (3, 4)]
# With replacement — elements can repeat
combos_wr = list(combinations_with_replacement("AB", 3))
print(combos_wr)
# [('A', 'A', 'A'), ('A', 'A', 'B'), ('A', 'B', 'B'), ('B', 'B', 'B')]
Аккумуляция
accumulate -- накопительная сумма и другие операции
from itertools import accumulate
import operator
numbers = [1, 2, 3, 4, 5]
# Running sum (default)
running_sum = list(accumulate(numbers))
print(running_sum) # [1, 3, 6, 10, 15]
# Running product
running_prod = list(accumulate(numbers, operator.mul))
print(running_prod) # [1, 2, 6, 24, 120]
# Running maximum
data = [3, 1, 4, 1, 5, 9, 2, 6]
running_max = list(accumulate(data, max))
print(running_max) # [3, 3, 4, 4, 5, 9, 9, 9]
# With initial value (Python 3.8+)
running_sum_init = list(accumulate(numbers, initial=100))
print(running_sum_init) # [100, 101, 103, 106, 110, 115]
Практический пример: обработка данных
from itertools import chain, groupby, batched
from operator import itemgetter
def process_logs(log_files: list[str]) -> dict[str, int]:
"""Count log levels across multiple files."""
def parse_line(line: str) -> tuple[str, str]:
"""Extract level and message from log line."""
# Format: "2024-01-15 ERROR: Something went wrong"
parts = line.split(" ", 2)
level = parts[1].rstrip(":")
message = parts[2] if len(parts) > 2 else ""
return level, message
def read_lines(path: str):
with open(path, encoding="utf-8") as f:
for line in f:
yield line.strip()
# Chain all files into a single stream
all_lines = chain.from_iterable(read_lines(f) for f in log_files)
# Parse and count levels
levels: dict[str, int] = {}
for line in all_lines:
if not line:
continue
try:
level, _ = parse_line(line)
levels[level] = levels.get(level, 0) + 1
except (IndexError, ValueError):
continue
return levels
def batch_api_calls(items: list[dict], batch_size: int = 50) -> list[dict]:
"""Send items to API in batches."""
results = []
for batch in batched(items, batch_size):
# Simulate batch API call
response = {"processed": len(batch), "items": batch}
results.append(response)
print(f"Sent batch of {len(batch)} items")
return results