Генераторы -- это элегантный способ создания итераторов без написания классов с __iter__/__next__. Ключевое слово yield превращает обычную функцию в генератор, который лениво производит значения по запросу.
Генераторные функции
Функция, содержащая yield, становится генераторной. При вызове она не выполняется сразу, а возвращает объект-генератор:
def count_up_to(n: int):
"""Generate numbers from 1 to n."""
i = 1
while i <= n:
yield i # Pause here, return i
i += 1 # Resume here on next call
# Calling the function returns a generator object
gen = count_up_to(3)
print(type(gen)) # <class 'generator'>
# Get values one by one
print(next(gen)) # 1
print(next(gen)) # 2
print(next(gen)) # 3
# next(gen) would raise StopIteration
# Or use in a for loop
for num in count_up_to(5):
print(num) # 1, 2, 3, 4, 5
Как работает yield
yield приостанавливает выполнение функции и возвращает значение. При следующем вызове next() выполнение продолжается с того же места:
def demo_yield():
"""Demonstrate yield execution flow."""
print("Step 1: before first yield")
yield "first"
print("Step 2: between yields")
yield "second"
print("Step 3: after last yield")
# Function ends — StopIteration raised automatically
gen = demo_yield()
value = next(gen)
# Output: Step 1: before first yield
print(f"Got: {value}") # Got: first
value = next(gen)
# Output: Step 2: between yields
print(f"Got: {value}") # Got: second
try:
next(gen)
# Output: Step 3: after last yield
except StopIteration:
print("Generator exhausted")
Генераторы vs списки: эффективность памяти
import sys
# List: stores ALL values in memory
numbers_list = [i ** 2 for i in range(1_000_000)]
print(f"List size: {sys.getsizeof(numbers_list):,} bytes") # ~8 MB
# Generator: computes values on-the-fly
numbers_gen = (i ** 2 for i in range(1_000_000))
print(f"Generator size: {sys.getsizeof(numbers_gen):,} bytes") # ~200 bytes
# Processing a huge file — generator is essential
def read_large_csv(path: str):
"""Read CSV line by line — constant memory usage."""
with open(path, encoding="utf-8") as f:
header = next(f).strip().split(",")
for line in f:
values = line.strip().split(",")
yield dict(zip(header, values))
# Processes millions of rows without loading all into memory
for row in read_large_csv("huge_data.csv"):
if row["status"] == "active":
process(row)
Генераторные выражения
Компактный синтаксис для создания генераторов (аналог list comprehension, но с круглыми скобками):
# List comprehension — creates a list
squares_list = [x ** 2 for x in range(10)]
# Generator expression — creates a generator
squares_gen = (x ** 2 for x in range(10))
# Use directly in functions
total = sum(x ** 2 for x in range(1000)) # No extra brackets needed
print(total)
# With conditions
evens = (x for x in range(100) if x % 2 == 0)
print(list(evens)) # [0, 2, 4, ..., 98]
# Chaining
result = sum(
len(word)
for word in ["hello", "world", "python"]
if len(word) > 4
)
print(result) # 11 (hello=5, world=5, python=6... wait: >4 means 5+5+6=16? Let's check)
# "hello"(5 > 4 ✓), "world"(5 > 4 ✓), "python"(6 > 4 ✓) → 5+5+6 = 16
Метод send()
send() позволяет отправить значение внутрь генератора. Это делает генераторы двусторонними:
def accumulator():
"""Generator that accumulates sent values."""
total = 0
while True:
value = yield total # Yield current total, receive new value
if value is None:
break
total += value
gen = accumulator()
next(gen) # Prime the generator — advance to first yield (returns 0)
print(gen.send(10)) # Send 10, get total: 10
print(gen.send(20)) # Send 20, get total: 30
print(gen.send(5)) # Send 5, get total: 35
Практический пример: скользящее среднее
def running_average():
"""Calculate running average of sent values."""
total = 0.0
count = 0
average = 0.0
while True:
value = yield average
if value is None:
return average
total += value
count += 1
average = total / count
avg = running_average()
next(avg) # Prime the generator
print(avg.send(10)) # 10.0
print(avg.send(20)) # 15.0
print(avg.send(30)) # 20.0
print(avg.send(40)) # 25.0
Методы throw() и close()
def careful_generator():
"""Generator with error handling."""
try:
while True:
value = yield
print(f"Received: {value}")
except ValueError as e:
print(f"Error handled: {e}")
yield "error_handled"
finally:
print("Generator cleanup")
gen = careful_generator()
next(gen) # Prime
gen.send("hello") # Received: hello
gen.send("world") # Received: world
# Throw an exception into the generator
result = gen.throw(ValueError, "bad input")
print(result) # error_handled
# Close the generator (triggers finally)
gen.close()
# Output: Generator cleanup
yield from -- делегирование
yield from делегирует итерацию подгенератору, избавляя от ручного цикла:
def flatten(nested: list) -> list:
"""Flatten a nested list using yield from."""
for item in nested:
if isinstance(item, list):
yield from flatten(item) # Delegate to recursive call
else:
yield item
data = [1, [2, 3], [4, [5, 6]], 7]
print(list(flatten(data))) # [1, 2, 3, 4, 5, 6, 7]
yield from с другими итерируемыми
def chain_iterables(*iterables):
"""Yield all items from multiple iterables."""
for iterable in iterables:
yield from iterable # Much cleaner than: for item in iterable: yield item
result = list(chain_iterables([1, 2], "abc", range(3)))
print(result) # [1, 2, 'a', 'b', 'c', 0, 1, 2]
yield from пробрасывает send() и throw()
def inner():
"""Sub-generator that receives values."""
total = 0
while True:
value = yield total
if value is None:
return total # Return value goes to yield from
total += value
def outer():
"""Delegating generator."""
result = yield from inner() # Gets return value of inner()
print(f"Inner returned: {result}")
yield result
gen = outer()
next(gen) # Prime: advances to inner's first yield
print(gen.send(10)) # 10 — sent directly to inner
print(gen.send(20)) # 30
print(gen.send(30)) # 60
try:
gen.send(None) # Triggers inner's return
except StopIteration:
pass
# Output: Inner returned: 60
Практические паттерны
Pipeline (конвейер обработки данных)
from typing import Iterator
def read_lines(path: str) -> Iterator[str]:
"""Stage 1: Read lines from file."""
with open(path, encoding="utf-8") as f:
for line in f:
yield line.strip()
def filter_non_empty(lines: Iterator[str]) -> Iterator[str]:
"""Stage 2: Filter out empty lines."""
for line in lines:
if line:
yield line
def parse_records(lines: Iterator[str]) -> Iterator[dict]:
"""Stage 3: Parse CSV-like lines into dicts."""
header = next(lines).split(",")
for line in lines:
values = line.split(",")
yield dict(zip(header, values))
def filter_active(records: Iterator[dict]) -> Iterator[dict]:
"""Stage 4: Keep only active records."""
for record in records:
if record.get("status") == "active":
yield record
# Build pipeline — nothing executes until we iterate!
pipeline = filter_active(
parse_records(
filter_non_empty(
read_lines("users.csv")
)
)
)
# Process lazily — memory-efficient for huge files
for user in pipeline:
print(user["name"])
Infinite sequence with state
def fibonacci() -> Iterator[int]:
"""Generate Fibonacci numbers infinitely."""
a, b = 0, 1
while True:
yield a
a, b = b, a + b
# Take first 10 Fibonacci numbers
from itertools import islice
fibs = list(islice(fibonacci(), 10))
print(fibs) # [0, 1, 1, 2, 3, 5, 8, 13, 21, 34]
# Find first Fibonacci number > 1000
for fib in fibonacci():
if fib > 1000:
print(f"First Fibonacci > 1000: {fib}") # 1597
break