MidТеория5 min

Генераторы коллекций

List, dict, set comprehensions, вложенные конструкции и фильтрация данных

Генераторы коллекций (Comprehensions)

Comprehensions -- одна из самых узнаваемых и мощных особенностей Python. Они позволяют создавать коллекции в одну строку, сочетая читаемость и производительность.

List Comprehensions

# Basic syntax: [expression for item in iterable]
squares = [x**2 for x in range(10)]
print(squares)  # [0, 1, 4, 9, 16, 25, 36, 49, 64, 81]

# Equivalent loop
squares_loop = []
for x in range(10):
    squares_loop.append(x**2)

# With filtering: [expression for item in iterable if condition]
evens = [x for x in range(20) if x % 2 == 0]
print(evens)  # [0, 2, 4, 6, 8, 10, 12, 14, 16, 18]

# Transform and filter
words = ["hello", "world", "python", "is", "great"]
long_upper = [w.upper() for w in words if len(w) > 3]
print(long_upper)  # ['HELLO', 'WORLD', 'PYTHON', 'GREAT']

# Conditional expression (if-else in expression part)
labels = ["чёт" if x % 2 == 0 else "нечёт" for x in range(6)]
print(labels)  # ['чёт', 'нечёт', 'чёт', 'нечёт', 'чёт', 'нечёт']

# Multiple conditions
fizzbuzz = [
    "FizzBuzz" if x % 15 == 0
    else "Fizz" if x % 3 == 0
    else "Buzz" if x % 5 == 0
    else str(x)
    for x in range(1, 16)
]
print(fizzbuzz)
# ['1', '2', 'Fizz', '4', 'Buzz', 'Fizz', '7', '8', 'Fizz', 'Buzz',
#  '11', 'Fizz', '13', '14', 'FizzBuzz']

Dict Comprehensions

# Basic syntax: {key_expr: value_expr for item in iterable}
squares = {x: x**2 for x in range(6)}
print(squares)  # {0: 0, 1: 1, 2: 4, 3: 9, 4: 16, 5: 25}

# With filtering
adults = {name: age for name, age in [
    ("Иван", 25), ("Мария", 17), ("Пётр", 30), ("Анна", 15)
] if age >= 18}
print(adults)  # {'Иван': 25, 'Пётр': 30}

# Swap keys and values
original = {"a": 1, "b": 2, "c": 3}
swapped = {v: k for k, v in original.items()}
print(swapped)  # {1: 'a', 2: 'b', 3: 'c'}

# Word length mapping
words = ["Python", "is", "awesome"]
lengths = {word: len(word) for word in words}
print(lengths)  # {'Python': 6, 'is': 2, 'awesome': 7}

# Group by first character
from itertools import groupby

names = sorted(["Анна", "Алексей", "Борис", "Анастасия", "Борислав"])
groups = {
    key: list(group)
    for key, group in groupby(names, key=lambda n: n[0])
}
print(groups)
# {'А': ['Алексей', 'Анастасия', 'Анна'], 'Б': ['Борис', 'Борислав']}

Set Comprehensions

# Basic syntax: {expression for item in iterable}
unique_lengths = {len(word) for word in ["hello", "world", "hi", "python", "ok"]}
print(unique_lengths)  # {2, 5, 6}

# Remove duplicates with transformation
words = ["Hello", "HELLO", "hello", "World", "WORLD"]
unique_lower = {w.lower() for w in words}
print(unique_lower)  # {'hello', 'world'}

# Vowels in a text
text = "Python programming is fun"
vowels = {c.lower() for c in text if c.lower() in "aeiou"}
print(vowels)  # {'a', 'i', 'o', 'u'}

Вложенные Comprehensions

# Flatten a 2D list
matrix = [[1, 2, 3], [4, 5, 6], [7, 8, 9]]
flat = [x for row in matrix for x in row]
print(flat)  # [1, 2, 3, 4, 5, 6, 7, 8, 9]

# Equivalent nested loop (order matches!)
flat_loop = []
for row in matrix:
    for x in row:
        flat_loop.append(x)

# Create a matrix
matrix = [[i * 3 + j + 1 for j in range(3)] for i in range(3)]
print(matrix)  # [[1, 2, 3], [4, 5, 6], [7, 8, 9]]

# Transpose a matrix
transposed = [[row[i] for row in matrix] for i in range(3)]
print(transposed)  # [[1, 4, 7], [2, 5, 8], [3, 6, 9]]

# Cartesian product
colors = ["red", "green"]
sizes = ["S", "M", "L"]
products = [(c, s) for c in colors for s in sizes]
print(products)
# [('red', 'S'), ('red', 'M'), ('red', 'L'),
#  ('green', 'S'), ('green', 'M'), ('green', 'L')]

# Nested with filtering
# Find all pairs where sum is even
pairs = [(x, y) for x in range(5) for y in range(5) if (x + y) % 2 == 0]
print(pairs[:5])  # [(0, 0), (0, 2), (0, 4), (1, 1), (1, 3)]

Многоуровневая вложенность

# Deep nesting (use with caution - readability suffers)
data = {
    "users": [
        {"name": "Иван", "scores": [85, 92, 78]},
        {"name": "Мария", "scores": [90, 88, 95]},
    ]
}

# Extract all scores
all_scores = [
    score
    for user in data["users"]
    for score in user["scores"]
]
print(all_scores)  # [85, 92, 78, 90, 88, 95]

# Better: break into steps for readability
users = data["users"]
scores_by_user = {
    user["name"]: sum(user["scores"]) / len(user["scores"])
    for user in users
}
print(scores_by_user)  # {'Иван': 85.0, 'Мария': 91.0}

Generator Expressions

Generator expression имеет синтаксис как list comprehension, но в круглых скобках. Он ленивый -- не создаёт список в памяти:

# Generator expression - lazy evaluation
gen = (x**2 for x in range(1_000_000))
print(type(gen))  # <class 'generator'>

# Uses almost no memory (compared to list)
import sys
list_size = sys.getsizeof([x**2 for x in range(1000)])
gen_size = sys.getsizeof(x**2 for x in range(1000))
print(f"List: {list_size} bytes")   # ~8856 bytes
print(f"Generator: {gen_size} bytes")  # ~200 bytes

# Pass directly to functions (no extra parentheses needed)
total = sum(x**2 for x in range(100))
print(total)  # 328350

max_len = max(len(word) for word in ["Python", "is", "awesome"])
print(max_len)  # 7

has_negative = any(x < 0 for x in [1, -2, 3])
print(has_negative)  # True

all_positive = all(x > 0 for x in [1, 2, 3])
print(all_positive)  # True

# Generator is exhausted after one pass
gen = (x for x in range(3))
print(list(gen))  # [0, 1, 2]
print(list(gen))  # [] (already exhausted!)

Walrus Operator в Comprehensions

# Walrus operator (:=) avoids redundant computation
import math

# Without walrus - sqrt computed twice
results = [(x, math.sqrt(x)) for x in range(100) if math.sqrt(x) > 5]

# With walrus - sqrt computed once
results = [(x, root) for x in range(100) if (root := math.sqrt(x)) > 5]
print(results[:3])  # [(26, 5.0990...), (27, 5.1961...), (28, 5.2915...)]

# Filter and transform with walrus
raw_data = ["  hello  ", "", "  world  ", "  ", "  python  "]
cleaned = [stripped for s in raw_data if (stripped := s.strip())]
print(cleaned)  # ['hello', 'world', 'python']

Когда использовать Comprehensions

# GOOD: simple, readable comprehension
squares = [x**2 for x in range(10)]
adults = {n: a for n, a in people if a >= 18}

# BAD: too complex, hard to read
# result = [func(x) for x in data if pred(x) for y in other if cond(x, y)]

# Better: use regular loops for complex logic
result = []
for x in data:
    if pred(x):
        for y in other:
            if cond(x, y):
                result.append(func(x))

# Rule of thumb:
# 1 for + 0-1 if: comprehension
# 2 for OR complex logic: consider regular loop
# Side effects needed: always use loop

# Performance: comprehensions are ~10-30% faster than equivalent loops
import timeit

# Comprehension
t1 = timeit.timeit('[x**2 for x in range(1000)]', number=10000)

# Loop
t2 = timeit.timeit('''
result = []
for x in range(1000):
    result.append(x**2)
''', number=10000)

print(f"Comprehension: {t1:.3f}s")  # faster
print(f"Loop: {t2:.3f}s")

Итоги

  • List comprehension: [expr for x in iterable if cond]
  • Dict comprehension: {key: value for x in iterable if cond}
  • Set comprehension: {expr for x in iterable if cond}
  • Generator expression: (expr for x in iterable if cond) -- ленивый, экономит память
  • Фильтрация if -- после for, условное выражение if-else -- перед for
  • Вложенные циклы идут слева направо (как в обычных вложенных for)
  • Walrus operator := исключает повторные вычисления в comprehensions
  • Используйте comprehension для простой логики, обычные циклы -- для сложной
  • Comprehensions на 10-30% быстрее эквивалентных циклов

Проверь себя

Где в list comprehension размещается условие фильтрации (if без else)?

Что создаёт выражение {x**2 for x in [1,2,2,3,3,3]}?

В чём главное отличие generator expression от list comprehension?

Когда НЕ стоит использовать comprehension?