MidТеория6 min

itertools

Модуль itertools: chain, islice, groupby, batched, product, permutations, pairwise и другие комбинаторные функции

Модуль itertools -- это набор быстрых и эффективных по памяти инструментов для работы с итераторами. Написанный на C, он значительно быстрее эквивалентного Python-кода. Это один из самых полезных модулей стандартной библиотеки.

Бесконечные итераторы

count -- бесконечный счетчик

from itertools import count

# Count from 0 with step 1 (default)
for i in count():
    if i >= 5:
        break
    print(i)  # 0, 1, 2, 3, 4

# Count with start and step
for i in count(start=10, step=3):
    if i > 25:
        break
    print(i)  # 10, 13, 16, 19, 22, 25

# Useful for numbering
data = ["Alice", "Bob", "Charlie"]
numbered = list(zip(count(1), data))
print(numbered)  # [(1, 'Alice'), (2, 'Bob'), (3, 'Charlie')]

cycle -- бесконечный цикл

from itertools import cycle, islice

# Cycle through elements forever
colors = cycle(["red", "green", "blue"])
print(list(islice(colors, 7)))
# ['red', 'green', 'blue', 'red', 'green', 'blue', 'red']

# Assign colors to items
items = ["item_1", "item_2", "item_3", "item_4", "item_5"]
colored = list(zip(items, cycle(["primary", "secondary"])))
print(colored)
# [('item_1', 'primary'), ('item_2', 'secondary'), ('item_3', 'primary'), ...]

repeat -- повторение значения

from itertools import repeat

# Repeat a value n times
zeros = list(repeat(0, 5))
print(zeros)  # [0, 0, 0, 0, 0]

# Useful with map for constant operations
from operator import mul
result = list(map(mul, range(1, 6), repeat(3)))
print(result)  # [3, 6, 9, 12, 15] — each multiplied by 3

Комбинирование итераторов

chain -- объединение последовательностей

from itertools import chain

# Concatenate multiple iterables
combined = list(chain([1, 2], [3, 4], [5, 6]))
print(combined)  # [1, 2, 3, 4, 5, 6]

# chain.from_iterable — for a single iterable of iterables
lists = [[1, 2], [3, 4], [5, 6]]
flat = list(chain.from_iterable(lists))
print(flat)  # [1, 2, 3, 4, 5, 6]

# Flatten one level of nesting
nested = [["a", "b"], ["c"], ["d", "e", "f"]]
flat = list(chain.from_iterable(nested))
print(flat)  # ['a', 'b', 'c', 'd', 'e', 'f']

islice -- срез итератора

from itertools import islice

# Like slicing but for any iterator
def infinite_squares():
    i = 0
    while True:
        yield i ** 2
        i += 1

# Take first 5 elements
first_five = list(islice(infinite_squares(), 5))
print(first_five)  # [0, 1, 4, 9, 16]

# Skip first 3, take next 4
skipped = list(islice(infinite_squares(), 3, 7))
print(skipped)  # [9, 16, 25, 36]

# Every other element from first 10
every_other = list(islice(infinite_squares(), 0, 10, 2))
print(every_other)  # [0, 4, 16, 36, 64]

zip_longest -- zip с заполнением

from itertools import zip_longest

names = ["Alice", "Bob", "Charlie"]
scores = [95, 87]

# Regular zip stops at shortest
print(list(zip(names, scores)))
# [('Alice', 95), ('Bob', 87)]

# zip_longest fills missing values
print(list(zip_longest(names, scores, fillvalue=0)))
# [('Alice', 95), ('Bob', 87), ('Charlie', 0)]

Фильтрация

takewhile и dropwhile

from itertools import takewhile, dropwhile

data = [1, 3, 5, 7, 2, 4, 6, 8]

# Take elements WHILE condition is true
small = list(takewhile(lambda x: x < 6, data))
print(small)  # [1, 3, 5]

# Drop elements WHILE condition is true, then take the rest
big = list(dropwhile(lambda x: x < 6, data))
print(big)  # [7, 2, 4, 6, 8]

filterfalse -- обратный filter

from itertools import filterfalse

numbers = range(10)

# Keep elements where predicate is False
odd = list(filterfalse(lambda x: x % 2 == 0, numbers))
print(odd)  # [1, 3, 5, 7, 9]

compress -- фильтрация по маске

from itertools import compress

data = ["a", "b", "c", "d", "e"]
mask = [True, False, True, False, True]

result = list(compress(data, mask))
print(result)  # ['a', 'c', 'e']

Группировка

groupby -- группировка последовательных элементов

from itertools import groupby

# IMPORTANT: data must be sorted by the grouping key!
data = [
    {"name": "Alice", "dept": "Engineering"},
    {"name": "Bob", "dept": "Engineering"},
    {"name": "Charlie", "dept": "Marketing"},
    {"name": "Diana", "dept": "Marketing"},
    {"name": "Eve", "dept": "Sales"},
]

# Group by department
for dept, members in groupby(data, key=lambda x: x["dept"]):
    member_list = list(members)
    print(f"{dept}: {[m['name'] for m in member_list]}")
# Engineering: ['Alice', 'Bob']
# Marketing: ['Charlie', 'Diana']
# Sales: ['Eve']

Группировка чисел

from itertools import groupby

# Consecutive identical elements
text = "aaabbccdddeee"
groups = [(char, len(list(group))) for char, group in groupby(text)]
print(groups)  # [('a', 3), ('b', 2), ('c', 2), ('d', 3), ('e', 3)]

# Run-length encoding
def rle_encode(data: str) -> list[tuple[str, int]]:
    """Run-length encode a string."""
    return [(char, len(list(group))) for char, group in groupby(data)]

def rle_decode(encoded: list[tuple[str, int]]) -> str:
    """Decode run-length encoded data."""
    return "".join(char * count for char, count in encoded)

encoded = rle_encode("aaabbccdddeee")
print(encoded)          # [('a', 3), ('b', 2), ...]
print(rle_decode(encoded))  # aaabbccdddeee

batched (Python 3.12+)

Разбивает итерируемый объект на группы фиксированного размера:

from itertools import batched

data = range(10)

# Split into batches of 3
for batch in batched(data, 3):
    print(batch)
# (0, 1, 2)
# (3, 4, 5)
# (6, 7, 8)
# (9,)        — last batch may be shorter

# Process items in batches (e.g., batch API calls)
users = ["user_1", "user_2", "user_3", "user_4", "user_5"]
for batch in batched(users, 2):
    print(f"Processing batch: {batch}")
# Processing batch: ('user_1', 'user_2')
# Processing batch: ('user_3', 'user_4')
# Processing batch: ('user_5',)

pairwise (Python 3.10+)

Возвращает пары последовательных элементов:

from itertools import pairwise

# Consecutive pairs
data = [1, 2, 3, 4, 5]
pairs = list(pairwise(data))
print(pairs)  # [(1, 2), (2, 3), (3, 4), (4, 5)]

# Calculate differences between consecutive elements
values = [10, 15, 12, 20, 18]
diffs = [b - a for a, b in pairwise(values)]
print(diffs)  # [5, -3, 8, -2]

# Check if sequence is sorted
numbers = [1, 3, 5, 7, 9]
is_sorted = all(a <= b for a, b in pairwise(numbers))
print(is_sorted)  # True

Комбинаторные функции

product -- декартово произведение

from itertools import product

# All combinations of two sequences
colors = ["red", "blue"]
sizes = ["S", "M", "L"]

for combo in product(colors, sizes):
    print(combo)
# ('red', 'S'), ('red', 'M'), ('red', 'L'),
# ('blue', 'S'), ('blue', 'M'), ('blue', 'L')

# Repeat parameter — like nested loops
digits = [0, 1]
for combo in product(digits, repeat=3):
    print(combo)
# (0, 0, 0), (0, 0, 1), (0, 1, 0), ..., (1, 1, 1)

permutations -- перестановки

from itertools import permutations

# All orderings of elements
perms = list(permutations([1, 2, 3]))
print(perms)
# [(1, 2, 3), (1, 3, 2), (2, 1, 3), (2, 3, 1), (3, 1, 2), (3, 2, 1)]

# Permutations of specific length
perms_2 = list(permutations("ABC", 2))
print(perms_2)
# [('A', 'B'), ('A', 'C'), ('B', 'A'), ('B', 'C'), ('C', 'A'), ('C', 'B')]

combinations -- сочетания (без повторений)

from itertools import combinations, combinations_with_replacement

# Choose 2 from 4 (order doesn't matter)
combos = list(combinations([1, 2, 3, 4], 2))
print(combos)
# [(1, 2), (1, 3), (1, 4), (2, 3), (2, 4), (3, 4)]

# With replacement — elements can repeat
combos_wr = list(combinations_with_replacement("AB", 3))
print(combos_wr)
# [('A', 'A', 'A'), ('A', 'A', 'B'), ('A', 'B', 'B'), ('B', 'B', 'B')]

Аккумуляция

accumulate -- накопительная сумма и другие операции

from itertools import accumulate
import operator

numbers = [1, 2, 3, 4, 5]

# Running sum (default)
running_sum = list(accumulate(numbers))
print(running_sum)  # [1, 3, 6, 10, 15]

# Running product
running_prod = list(accumulate(numbers, operator.mul))
print(running_prod)  # [1, 2, 6, 24, 120]

# Running maximum
data = [3, 1, 4, 1, 5, 9, 2, 6]
running_max = list(accumulate(data, max))
print(running_max)  # [3, 3, 4, 4, 5, 9, 9, 9]

# With initial value (Python 3.8+)
running_sum_init = list(accumulate(numbers, initial=100))
print(running_sum_init)  # [100, 101, 103, 106, 110, 115]

Практический пример: обработка данных

from itertools import chain, groupby, batched
from operator import itemgetter

def process_logs(log_files: list[str]) -> dict[str, int]:
    """Count log levels across multiple files."""

    def parse_line(line: str) -> tuple[str, str]:
        """Extract level and message from log line."""
        # Format: "2024-01-15 ERROR: Something went wrong"
        parts = line.split(" ", 2)
        level = parts[1].rstrip(":")
        message = parts[2] if len(parts) > 2 else ""
        return level, message

    def read_lines(path: str):
        with open(path, encoding="utf-8") as f:
            for line in f:
                yield line.strip()

    # Chain all files into a single stream
    all_lines = chain.from_iterable(read_lines(f) for f in log_files)

    # Parse and count levels
    levels: dict[str, int] = {}
    for line in all_lines:
        if not line:
            continue
        try:
            level, _ = parse_line(line)
            levels[level] = levels.get(level, 0) + 1
        except (IndexError, ValueError):
            continue

    return levels

def batch_api_calls(items: list[dict], batch_size: int = 50) -> list[dict]:
    """Send items to API in batches."""
    results = []
    for batch in batched(items, batch_size):
        # Simulate batch API call
        response = {"processed": len(batch), "items": batch}
        results.append(response)
        print(f"Sent batch of {len(batch)} items")
    return results

Проверь себя

Чем permutations отличается от combinations?

Какое требование предъявляет groupby к входным данным?

Что возвращает list(pairwise([1, 2, 3, 4]))?

Что делает itertools.batched([1,2,3,4,5], 2)?

Что возвращает itertools.chain([1,2], [3,4])?