Python Comprehensions and Generators

Comprehensions and generators are Python's most elegant features. They let you create and process collections in a concise, readable, and memory-efficient way.

List Comprehensions

Python
# Basic syntax: [expression for item in iterable if condition]

# Square numbers
squares = [x**2 for x in range(10)]
# [0, 1, 4, 9, 16, 25, 36, 49, 64, 81]

# Filter: only even numbers
evens = [x for x in range(20) if x % 2 == 0]
# [0, 2, 4, 6, 8, 10, 12, 14, 16, 18]

# Transform: uppercase names
names = ["alice", "bob", "charlie"]
upper = [name.upper() for name in names]
# ["ALICE", "BOB", "CHARLIE"]

# Nested: flatten 2D list
matrix = [[1, 2, 3], [4, 5, 6], [7, 8, 9]]
flat = [num for row in matrix for num in row]
# [1, 2, 3, 4, 5, 6, 7, 8, 9]

# Conditional expression: pass/fail
scores = [85, 92, 78, 95, 88]
results = ["PASS" if s >= 80 else "FAIL" for s in scores]
# ["PASS", "PASS", "FAIL", "PASS", "PASS"]

# Find common elements
list1 = [1, 2, 3, 4, 5]
list2 = [4, 5, 6, 7, 8]
common = [x for x in list1 if x in list2]
# [4, 5]

Dictionary Comprehensions

Python
# Create dict from two lists
names = ["Alice", "Bob", "Charlie"]
scores = [95, 87, 92]
grade_book = {name: score for name, score in zip(names, scores)}
# {'Alice': 95, 'Bob': 87, 'Charlie': 92}

# Invert dictionary
swapped = {v: k for k, v in grade_book.items()}
# {95: 'Alice', 87: 'Bob', 92: 'Charlie'}

# Filter dictionary
high_scores = {k: v for k, v in grade_book.items() if v >= 90}
# {'Alice': 95, 'Charlie': 92}

# Word lengths
sentence = "the quick brown fox jumps over the lazy dog"
lengths = {word: len(word) for word in sentence.split()}
# {'the': 3, 'quick': 5, 'brown': 5, ...}

Set Comprehensions

Python
# Unique characters in a string
chars = {c.lower() for c in "Hello World" if c.isalpha()}
# {'h', 'e', 'l', 'o', 'w', 'r', 'd'}

# Unique file extensions
files = ["doc.pdf", "image.jpg", "data.csv", "photo.jpg", "report.pdf"]
extensions = {f.split('.')[-1] for f in files}
# {'pdf', 'jpg', 'csv'}

Generator Expressions

Python
# Generators are LAZY — they produce values one at a time
# Memory efficient for large datasets

# List comprehension (stores ALL values in memory)
squares_list = [x**2 for x in range(1000000)]  # ~8MB memory

# Generator expression (stores NOTHING in memory)
squares_gen = (x**2 for x in range(1000000))   # ~100 bytes!

# Use generators with: sum, min, max, any, all, list
total = sum(x**2 for x in range(1000000))  # Computed lazily
first_large = next(x for x in range(1000000) if x > 999990)  # 999991

# Practical: process large file line by line
def read_large_file(path):
    """Memory-efficient file reading."""
    with open(path, 'r') as f:
        for line in f:
            yield line.strip()

# Only loads one line at a time!
for line in read_large_file("huge_log.txt"):
    if "ERROR" in line:
        print(line)

Generator Functions (yield)

Python
# Generator function — uses yield instead of return
def fibonacci():
    """Infinite Fibonacci sequence."""
    a, b = 0, 1
    while True:
        yield a       # Produces value and pauses
        a, b = b, a + b

# Use with itertools or manual iteration
import itertools
first_10_fibs = list(itertools.islice(fibonacci(), 10))
# [0, 1, 1, 2, 3, 5, 8, 13, 21, 34]

# Practical: pagination generator
def paginate(items, page_size):
    """Yield pages of items."""
    for i in range(0, len(items), page_size):
        yield items[i:i + page_size]

users = list(range(100))
for page_num, page in enumerate(paginate(users, 10), 1):
    print(f"Page {page_num}: {page}")

# Generator pipeline (Unix pipe style)
def read_words(filename):
    with open(filename) as f:
        for line in f:
            yield from line.split()

def filter_long(words, min_length=5):
    return (w for w in words if len(w) >= min_length)

def uppercase(words):
    return (w.upper() for w in words)

# Compose generators into a pipeline
pipeline = uppercase(filter_long(read_words("book.txt")))
unique_words = set(pipeline)  # Lazy evaluation until this point

When to Use What

Python
# List comprehension: when you need the actual list
squares = [x**2 for x in range(100)]  # Need the list for indexing, slicing

# Generator expression: when iterating once is enough
total = sum(x**2 for x in range(1000000))  # Don't need to store all squares

# Generator function: when logic is complex
def process_data(filename):
    with open(filename) as f:
        for line in f:
            parsed = parse_line(line)
            if validate(parsed):
                yield transform(parsed)

Knowledge Check

4 questions — test your understanding

1

What is the main advantage of a generator over a list comprehension?

2

What does `yield` do in a function?

3

How do you create a dictionary comprehension?

4

What does `itertools.islice(fibonacci(), 10)` do?