Python Bytecode Fundamentals¶
Overview¶
Python source code is compiled to bytecode before execution: - Bytecode: Low-level instructions executed by Python VM - Code objects: Container for bytecode and metadata - Stack-based: Python VM uses a stack to execute instructions - Compiled to .pyc: Bytecode cached on disk for faster loading
Understanding bytecode reveals why certain code patterns are fast or slow.
From Source to Bytecode¶
Compilation Process¶
Source Code (.py)
↓
Parser (creates parse tree)
↓
AST (Abstract Syntax Tree)
↓
Compiler (generates bytecode)
↓
Code Object (contains bytecode)
↓
Python VM (executes bytecode)
Viewing Bytecode with dis¶
import dis
def add(a, b):
"""Simple function to disassemble."""
return a + b
# Disassemble to see bytecode
dis.dis(add)
# Output:
# 2 0 LOAD_FAST 0 (a)
# 2 LOAD_FAST 1 (b)
# 4 BINARY_ADD
# 6 RETURN_VALUE
Understanding Bytecode Instructions¶
import dis
# LOAD_FAST: Load local variable onto stack
# LOAD_GLOBAL: Load global variable onto stack
# BINARY_ADD: Pop two values, add them, push result
# CALL_FUNCTION: Call function with arguments
# RETURN_VALUE: Return value from function
def example():
x = 5 # LOAD_CONST, STORE_FAST
y = 10 # LOAD_CONST, STORE_FAST
z = x + y # LOAD_FAST (x), LOAD_FAST (y), BINARY_ADD, STORE_FAST (z)
return z # LOAD_FAST (z), RETURN_VALUE
dis.dis(example)
Code Objects¶
Inspecting Code Objects¶
def fibonacci(n):
"""Calculate Fibonacci number."""
if n <= 1:
return n
return fibonacci(n-1) + fibonacci(n-2)
code = fibonacci.__code__
print(f"Name: {code.co_name}") # 'fibonacci'
print(f"Arguments: {code.co_varnames[:code.co_argcount]}") # ('n',)
print(f"Bytecode length: {len(code.co_code)}") # Number of bytes
print(f"Constants: {code.co_consts}") # (1, None)
print(f"Names: {code.co_names}") # (global references)
print(f"Local variables: {code.co_varnames}") # ('n', ...)
print(f"Bytecode: {code.co_code.hex()}") # Hex representation
Frame Objects¶
import sys
def inspect_frame():
"""Inspect current frame."""
frame = sys._getframe()
print(f"Function: {frame.f_code.co_name}")
print(f"Line number: {frame.f_lineno}")
print(f"Locals: {frame.f_locals}")
print(f"Globals keys: {list(frame.f_globals.keys())[:5]}")
inspect_frame()
Stack-Based Execution¶
Understanding the Stack¶
import dis
def stack_example():
"""Show stack operations."""
a = 5
b = 3
c = a + b + 2
return c
# Bytecode (simplified):
# LOAD_CONST 1 (5) # Push 5 onto stack: [5]
# STORE_FAST 0 (a) # Pop 5, store in 'a': []
#
# LOAD_CONST 2 (3) # Push 3: [3]
# STORE_FAST 1 (b) # Pop 3, store in 'b': []
#
# LOAD_FAST 0 (a) # Push a (5): [5]
# LOAD_FAST 1 (b) # Push b (3): [5, 3]
# BINARY_ADD # Pop 3,5 -> 8: [8]
# LOAD_CONST 3 (2) # Push 2: [8, 2]
# BINARY_ADD # Pop 2,8 -> 10: [10]
# STORE_FAST 2 (c) # Pop 10, store in 'c': []
#
# LOAD_FAST 2 (c) # Push c (10): [10]
# RETURN_VALUE # Return top of stack: []
dis.dis(stack_example)
Common Bytecode Patterns¶
Function Call Bytecode¶
import dis
def call_function(x):
"""Call another function."""
return len([x, x, x])
dis.dis(call_function)
# Key instructions:
# LOAD_FAST 0 (x) # Load argument
# BUILD_LIST 3 # Create list with 3 items
# LOAD_GLOBAL len # Load len function
# ROT_TWO # Swap arguments
# CALL_FUNCTION 1 # Call function with 1 argument
# RETURN_VALUE # Return result
Loop Bytecode¶
import dis
def loop_example():
"""Show loop bytecode."""
total = 0
for i in range(10):
total += i
return total
dis.dis(loop_example)
# Key instructions:
# LOAD_GLOBAL range # Load range function
# LOAD_CONST 1 (10) # Load 10
# CALL_FUNCTION 1 # Call range(10)
# GET_ITER # Get iterator
#
# loop_start:
# FOR_ITER # Get next item, or jump to end
# STORE_FAST 1 (i) # Store in 'i'
# LOAD_FAST 0 (total) # Load total
# LOAD_FAST 1 (i) # Load i
# INPLACE_ADD # Add in place
# STORE_FAST 0 (total) # Store back
# JUMP_ABSOLUTE loop_start # Loop back
Bytecode Optimization Patterns¶
Pattern 1: Local vs Global Access¶
import dis
# SLOW: Global access
def slow_sum():
total = 0
for i in range(1000000):
total += sum([i, i, i]) # sum is global lookup each time!
return total
# FAST: Local access
def fast_sum():
total = 0
local_sum = sum # Cache global in local
for i in range(1000000):
total += local_sum([i, i, i]) # Local access is faster
return total
print("Slow version:")
dis.dis(slow_sum)
print("\nFast version:")
dis.dis(fast_sum)
# Difference: LOAD_FAST vs LOAD_GLOBAL
# LOAD_FAST: 1 bytecode instruction
# LOAD_GLOBAL: 2 bytecode instructions + name lookup
Pattern 2: Direct Attribute Access vs Function Call¶
import dis
import math
# SLOW: Function calls
def slow_math():
total = 0
for i in range(1000000):
total += math.sqrt(i)
return total
# FAST: Pre-fetch function
def fast_math():
total = 0
sqrt = math.sqrt # Cache function reference
for i in range(1000000):
total += sqrt(i)
return total
print("Slow (function call):")
dis.dis(slow_math)
print("\nFast (cached function):")
dis.dis(fast_math)
Pattern 3: List Comprehension vs Loop¶
import dis
# Loop version
def loop_version():
result = []
for i in range(10):
result.append(i * 2)
return result
# List comprehension (optimized)
def comprehension_version():
return [i * 2 for i in range(10)]
print("Loop version:")
dis.dis(loop_version)
print("\nList comprehension (more efficient):")
dis.dis(comprehension_version)
# List comprehensions are optimized to use BUILD_LIST_UNPACK
# and avoid repeated APPEND calls
Measuring Bytecode Efficiency¶
Bytecode Size as Proxy for Speed¶
import dis
import sys
def measure_bytecode_efficiency(func):
"""Estimate bytecode efficiency."""
code = func.__code__
bytecode_size = len(code.co_code)
print(f"Function: {func.__name__}")
print(f"Bytecode size: {bytecode_size} bytes")
print(f"Instructions: {bytecode_size // 2}") # Each instruction is 2 bytes
print(f"Constants: {len(code.co_consts)}")
print(f"Names: {len(code.co_names)}")
def simple():
return 1 + 1
def complex_add():
a = 1
b = 1
c = a + b
d = c + 1
return d
print("Simple function:")
measure_bytecode_efficiency(simple)
print("\nComplex function:")
measure_bytecode_efficiency(complex_add)
Real-World Example: PyTorch Execution¶
import torch
import dis
# How PyTorch forward pass looks at bytecode level
def simple_forward():
x = torch.randn(1000, 1000)
w = torch.randn(1000, 100)
# Matrix multiply
y = x @ w # This is BINARY_MATRIX_MULTIPLY (custom PyTorch opcode)
return y
dis.dis(simple_forward)
# Key insight: PyTorch overloads operators (@, +, *, etc.)
# to intercept bytecode execution and run optimized CUDA kernels
# instead of Python operations
Summary: Bytecode Insights for ML¶
| Pattern | Bytecode Impact | Speed Impact | Recommendation |
|---|---|---|---|
| Global lookup | +1 inst | Slower | Cache in local |
| Function call | +2 inst | Slower | Pre-fetch function |
| Loop vs comprehension | More inst | Slower | Use comprehension |
| Direct attribute | +1 inst | Fast | Preferred |
| Operator overload | Custom | Variable | Use framework ops |
Related Topics¶
- 02 Execution Model & Compilation - How bytecode is generated
- 03 Jit Compilation & Optimization - Making bytecode execution faster
- 04 Profiling & Performance Analysis - Measuring bytecode performance
- 00 Readme - Memory during bytecode execution