Comprehensions and generators are Python's most elegant features. They let you create and process collections in a concise, readable, and memory-efficient way.
List Comprehensions
Python
# Basic syntax: [expression for item in iterable if condition]
# Square numbers
squares = [x**2 for x in range(10)]
# [0, 1, 4, 9, 16, 25, 36, 49, 64, 81]
# Filter: only even numbers
evens = [x for x in range(20) if x % 2 == 0]
# [0, 2, 4, 6, 8, 10, 12, 14, 16, 18]
# Transform: uppercase names
names = ["alice", "bob", "charlie"]
upper = [name.upper() for name in names]
# ["ALICE", "BOB", "CHARLIE"]
# Nested: flatten 2D list
matrix = [[1, 2, 3], [4, 5, 6], [7, 8, 9]]
flat = [num for row in matrix for num in row]
# [1, 2, 3, 4, 5, 6, 7, 8, 9]
# Conditional expression: pass/fail
scores = [85, 92, 78, 95, 88]
results = ["PASS" if s >= 80 else "FAIL" for s in scores]
# ["PASS", "PASS", "FAIL", "PASS", "PASS"]
# Find common elements
list1 = [1, 2, 3, 4, 5]
list2 = [4, 5, 6, 7, 8]
common = [x for x in list1 if x in list2]
# [4, 5]Dictionary Comprehensions
Python
# Create dict from two lists
names = ["Alice", "Bob", "Charlie"]
scores = [95, 87, 92]
grade_book = {name: score for name, score in zip(names, scores)}
# {'Alice': 95, 'Bob': 87, 'Charlie': 92}
# Invert dictionary
swapped = {v: k for k, v in grade_book.items()}
# {95: 'Alice', 87: 'Bob', 92: 'Charlie'}
# Filter dictionary
high_scores = {k: v for k, v in grade_book.items() if v >= 90}
# {'Alice': 95, 'Charlie': 92}
# Word lengths
sentence = "the quick brown fox jumps over the lazy dog"
lengths = {word: len(word) for word in sentence.split()}
# {'the': 3, 'quick': 5, 'brown': 5, ...}Set Comprehensions
Python
# Unique characters in a string
chars = {c.lower() for c in "Hello World" if c.isalpha()}
# {'h', 'e', 'l', 'o', 'w', 'r', 'd'}
# Unique file extensions
files = ["doc.pdf", "image.jpg", "data.csv", "photo.jpg", "report.pdf"]
extensions = {f.split('.')[-1] for f in files}
# {'pdf', 'jpg', 'csv'}Generator Expressions
Python
# Generators are LAZY — they produce values one at a time
# Memory efficient for large datasets
# List comprehension (stores ALL values in memory)
squares_list = [x**2 for x in range(1000000)] # ~8MB memory
# Generator expression (stores NOTHING in memory)
squares_gen = (x**2 for x in range(1000000)) # ~100 bytes!
# Use generators with: sum, min, max, any, all, list
total = sum(x**2 for x in range(1000000)) # Computed lazily
first_large = next(x for x in range(1000000) if x > 999990) # 999991
# Practical: process large file line by line
def read_large_file(path):
"""Memory-efficient file reading."""
with open(path, 'r') as f:
for line in f:
yield line.strip()
# Only loads one line at a time!
for line in read_large_file("huge_log.txt"):
if "ERROR" in line:
print(line)Generator Functions (yield)
Python
# Generator function — uses yield instead of return
def fibonacci():
"""Infinite Fibonacci sequence."""
a, b = 0, 1
while True:
yield a # Produces value and pauses
a, b = b, a + b
# Use with itertools or manual iteration
import itertools
first_10_fibs = list(itertools.islice(fibonacci(), 10))
# [0, 1, 1, 2, 3, 5, 8, 13, 21, 34]
# Practical: pagination generator
def paginate(items, page_size):
"""Yield pages of items."""
for i in range(0, len(items), page_size):
yield items[i:i + page_size]
users = list(range(100))
for page_num, page in enumerate(paginate(users, 10), 1):
print(f"Page {page_num}: {page}")
# Generator pipeline (Unix pipe style)
def read_words(filename):
with open(filename) as f:
for line in f:
yield from line.split()
def filter_long(words, min_length=5):
return (w for w in words if len(w) >= min_length)
def uppercase(words):
return (w.upper() for w in words)
# Compose generators into a pipeline
pipeline = uppercase(filter_long(read_words("book.txt")))
unique_words = set(pipeline) # Lazy evaluation until this pointWhen to Use What
Python
# List comprehension: when you need the actual list
squares = [x**2 for x in range(100)] # Need the list for indexing, slicing
# Generator expression: when iterating once is enough
total = sum(x**2 for x in range(1000000)) # Don't need to store all squares
# Generator function: when logic is complex
def process_data(filename):
with open(filename) as f:
for line in f:
parsed = parse_line(line)
if validate(parsed):
yield transform(parsed)Knowledge Check
4 questions — test your understanding
1
What is the main advantage of a generator over a list comprehension?
2
What does `yield` do in a function?
3
How do you create a dictionary comprehension?
4
What does `itertools.islice(fibonacci(), 10)` do?