mirror of
https://github.com/MLSysBook/TinyTorch.git
synced 2026-07-21 12:50:46 -05:00
✅ Renamed modules for clearer pedagogical flow: - 05_networks → 05_dense (multi-layer dense/fully connected networks) - 06_cnn → 06_spatial (convolutional networks for spatial patterns) - 06_attention → 07_attention (attention mechanisms for sequences) ✅ Shifted remaining modules down by 1: - 07_dataloader → 08_dataloader - 08_autograd → 09_autograd - 09_optimizers → 10_optimizers - 10_training → 11_training - 11_compression → 12_compression - 12_kernels → 13_kernels - 13_benchmarking → 14_benchmarking - 14_mlops → 15_mlops - 15_capstone → 16_capstone ✅ Updated module metadata (module.yaml files): - Updated names, descriptions, dependencies - Fixed prerequisite chains and enables relationships - Updated export paths to match new names New learner progression: Foundation → Individual Layers → Dense Networks → Spatial Networks → Attention Networks → Training Pipeline Perfect pedagogical flow: Build one layer → Stack dense layers → Add spatial patterns → Add attention mechanisms → Learn to train them all.
69 KiB
69 KiB
In [ ]:
#| default_exp core.optimizers
#| export
import math
import numpy as np
import sys
import os
from typing import List, Dict, Any, Optional, Union
from collections import defaultdict
# Helper function to set up import paths
def setup_import_paths():
"""Set up import paths for development modules."""
import sys
import os
# Add module directories to path
base_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
tensor_dir = os.path.join(base_dir, '01_tensor')
autograd_dir = os.path.join(base_dir, '07_autograd')
if tensor_dir not in sys.path:
sys.path.append(tensor_dir)
if autograd_dir not in sys.path:
sys.path.append(autograd_dir)
# Import our existing components
try:
from tinytorch.core.tensor import Tensor
from tinytorch.core.autograd import Variable
except ImportError:
# For development, try local imports
try:
setup_import_paths()
from tensor_dev import Tensor
from autograd_dev import Variable
except ImportError:
# Create minimal fallback classes for testing
print("Warning: Using fallback classes for testing")
class Tensor:
def __init__(self, data):
self.data = np.array(data)
self.shape = self.data.shape
def __str__(self):
return f"Tensor({self.data})"
class Variable:
def __init__(self, data, requires_grad=True):
if isinstance(data, (int, float)):
self.data = Tensor([data])
else:
self.data = Tensor(data)
self.requires_grad = requires_grad
self.grad = None
def zero_grad(self):
self.grad = None
def __str__(self):
return f"Variable({self.data.data})"In [ ]:
print("🔥 TinyTorch Optimizers Module")
print(f"NumPy version: {np.__version__}")
print(f"Python version: {sys.version_info.major}.{sys.version_info.minor}")
print("Ready to build optimization algorithms!")In [ ]:
#| export
def gradient_descent_step(parameter: Variable, learning_rate: float) -> None:
"""
Perform one step of gradient descent on a parameter.
Args:
parameter: Variable with gradient information
learning_rate: How much to update parameter
TODO: Implement basic gradient descent parameter update.
STEP-BY-STEP IMPLEMENTATION:
1. Check if parameter has a gradient
2. Get current parameter value and gradient
3. Update parameter: new_value = old_value - learning_rate * gradient
4. Update parameter data with new value
5. Handle edge cases (no gradient, invalid values)
EXAMPLE USAGE:
```python
# Parameter with gradient
w = Variable(2.0, requires_grad=True)
w.grad = Variable(0.5) # Gradient from loss
# Update parameter
gradient_descent_step(w, learning_rate=0.1)
# w.data now contains: 2.0 - 0.1 * 0.5 = 1.95
```
IMPLEMENTATION HINTS:
- Check if parameter.grad is not None
- Use parameter.grad.data.data to get gradient value
- Update parameter.data with new Tensor
- Don't modify gradient (it's used for logging)
LEARNING CONNECTIONS:
- This is the foundation of all neural network training
- PyTorch's optimizer.step() does exactly this
- The learning rate determines convergence speed
"""
### BEGIN SOLUTION
if parameter.grad is not None:
# Get current parameter value and gradient
current_value = parameter.data.data
gradient_value = parameter.grad.data.data
# Update parameter: new_value = old_value - learning_rate * gradient
new_value = current_value - learning_rate * gradient_value
# Update parameter data
parameter.data = Tensor(new_value)
### END SOLUTIONIn [ ]:
def test_gradient_descent_step():
"""Test basic gradient descent parameter update"""
print("🔬 Unit Test: Gradient Descent Step...")
# Test basic parameter update
try:
w = Variable(2.0, requires_grad=True)
w.grad = Variable(0.5) # Positive gradient
original_value = w.data.data.item()
gradient_descent_step(w, learning_rate=0.1)
new_value = w.data.data.item()
expected_value = original_value - 0.1 * 0.5 # 2.0 - 0.05 = 1.95
assert abs(new_value - expected_value) < 1e-6, f"Expected {expected_value}, got {new_value}"
print("✅ Basic parameter update works")
except Exception as e:
print(f"❌ Basic parameter update failed: {e}")
raise
# Test with negative gradient
try:
w2 = Variable(1.0, requires_grad=True)
w2.grad = Variable(-0.2) # Negative gradient
gradient_descent_step(w2, learning_rate=0.1)
expected_value2 = 1.0 - 0.1 * (-0.2) # 1.0 + 0.02 = 1.02
assert abs(w2.data.data.item() - expected_value2) < 1e-6, "Negative gradient test failed"
print("✅ Negative gradient handling works")
except Exception as e:
print(f"❌ Negative gradient handling failed: {e}")
raise
# Test with no gradient (should not update)
try:
w3 = Variable(3.0, requires_grad=True)
w3.grad = None
original_value3 = w3.data.data.item()
gradient_descent_step(w3, learning_rate=0.1)
assert w3.data.data.item() == original_value3, "Parameter with no gradient should not update"
print("✅ No gradient case works")
except Exception as e:
print(f"❌ No gradient case failed: {e}")
raise
print("🎯 Gradient descent step behavior:")
print(" Updates parameters in negative gradient direction")
print(" Uses learning rate to control step size")
print(" Skips updates when gradient is None")
print("📈 Progress: Gradient Descent Step ✓")
# Test function is called by auto-discovery systemIn [ ]:
#| export
class SGD:
"""
SGD Optimizer with Momentum
Implements stochastic gradient descent with momentum:
v_t = momentum * v_{t-1} + gradient
parameter = parameter - learning_rate * v_t
"""
def __init__(self, parameters: List[Variable], learning_rate: float = 0.01,
momentum: float = 0.0, weight_decay: float = 0.0):
"""
Initialize SGD optimizer.
Args:
parameters: List of Variables to optimize
learning_rate: Learning rate (default: 0.01)
momentum: Momentum coefficient (default: 0.0)
weight_decay: L2 regularization coefficient (default: 0.0)
TODO: Implement SGD optimizer initialization.
APPROACH:
1. Store parameters and hyperparameters
2. Initialize momentum buffers for each parameter
3. Set up state tracking for optimization
4. Prepare for step() and zero_grad() methods
EXAMPLE:
```python
# Create optimizer
optimizer = SGD([w1, w2, b1, b2], learning_rate=0.01, momentum=0.9)
# In training loop:
optimizer.zero_grad()
loss.backward()
optimizer.step()
```
HINTS:
- Store parameters as a list
- Initialize momentum buffers as empty dict
- Use parameter id() as key for momentum tracking
- Momentum buffers will be created lazily in step()
"""
### BEGIN SOLUTION
self.parameters = parameters
self.learning_rate = learning_rate
self.momentum = momentum
self.weight_decay = weight_decay
# Initialize momentum buffers (created lazily)
self.momentum_buffers = {}
# Track optimization steps
self.step_count = 0
### END SOLUTION
def step(self) -> None:
"""
Perform one optimization step.
TODO: Implement SGD parameter update with momentum.
APPROACH:
1. Iterate through all parameters
2. For each parameter with gradient:
a. Get current gradient
b. Apply weight decay if specified
c. Update momentum buffer (or create if first time)
d. Update parameter using momentum
3. Increment step count
MATHEMATICAL FORMULATION:
- If weight_decay > 0: gradient = gradient + weight_decay * parameter
- momentum_buffer = momentum * momentum_buffer + gradient
- parameter = parameter - learning_rate * momentum_buffer
IMPLEMENTATION HINTS:
- Use id(param) as key for momentum buffers
- Initialize buffer with zeros if not exists
- Handle case where momentum = 0 (no momentum)
- Update parameter.data with new Tensor
"""
### BEGIN SOLUTION
for param in self.parameters:
if param.grad is not None:
# Get gradient
gradient = param.grad.data.data
# Apply weight decay (L2 regularization)
if self.weight_decay > 0:
gradient = gradient + self.weight_decay * param.data.data
# Get or create momentum buffer
param_id = id(param)
if param_id not in self.momentum_buffers:
self.momentum_buffers[param_id] = np.zeros_like(param.data.data)
# Update momentum buffer
self.momentum_buffers[param_id] = (
self.momentum * self.momentum_buffers[param_id] + gradient
)
# Update parameter
param.data = Tensor(
param.data.data - self.learning_rate * self.momentum_buffers[param_id]
)
self.step_count += 1
### END SOLUTION
def zero_grad(self) -> None:
"""
Zero out gradients for all parameters.
TODO: Implement gradient zeroing.
APPROACH:
1. Iterate through all parameters
2. Set gradient to None for each parameter
3. This prepares for next backward pass
IMPLEMENTATION HINTS:
- Simply set param.grad = None
- This is called before loss.backward()
- Essential for proper gradient accumulation
"""
### BEGIN SOLUTION
for param in self.parameters:
param.grad = None
### END SOLUTIONIn [ ]:
def test_sgd_optimizer():
"""Test SGD optimizer implementation"""
print("🔬 Unit Test: SGD Optimizer...")
# Create test parameters
w1 = Variable(1.0, requires_grad=True)
w2 = Variable(2.0, requires_grad=True)
b = Variable(0.5, requires_grad=True)
# Create optimizer
optimizer = SGD([w1, w2, b], learning_rate=0.1, momentum=0.9)
# Test zero_grad
try:
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
optimizer.zero_grad()
assert w1.grad is None, "Gradient should be None after zero_grad"
assert w2.grad is None, "Gradient should be None after zero_grad"
assert b.grad is None, "Gradient should be None after zero_grad"
print("✅ zero_grad() works correctly")
except Exception as e:
print(f"❌ zero_grad() failed: {e}")
raise
# Test step with gradients
try:
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
# First step (no momentum yet)
original_w1 = w1.data.data.item()
original_w2 = w2.data.data.item()
original_b = b.data.data.item()
optimizer.step()
# Check parameter updates
expected_w1 = original_w1 - 0.1 * 0.1 # 1.0 - 0.01 = 0.99
expected_w2 = original_w2 - 0.1 * 0.2 # 2.0 - 0.02 = 1.98
expected_b = original_b - 0.1 * 0.05 # 0.5 - 0.005 = 0.495
assert abs(w1.data.data.item() - expected_w1) < 1e-6, f"w1 update failed: expected {expected_w1}, got {w1.data.data.item()}"
assert abs(w2.data.data.item() - expected_w2) < 1e-6, f"w2 update failed: expected {expected_w2}, got {w2.data.data.item()}"
assert abs(b.data.data.item() - expected_b) < 1e-6, f"b update failed: expected {expected_b}, got {b.data.data.item()}"
print("✅ Parameter updates work correctly")
except Exception as e:
print(f"❌ Parameter updates failed: {e}")
raise
# Test momentum buffers
try:
assert len(optimizer.momentum_buffers) == 3, f"Should have 3 momentum buffers, got {len(optimizer.momentum_buffers)}"
assert optimizer.step_count == 1, f"Step count should be 1, got {optimizer.step_count}"
print("✅ Momentum buffers created correctly")
except Exception as e:
print(f"❌ Momentum buffers failed: {e}")
raise
# Test step counting
try:
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
optimizer.step()
assert optimizer.step_count == 2, f"Step count should be 2, got {optimizer.step_count}"
print("✅ Step counting works correctly")
except Exception as e:
print(f"❌ Step counting failed: {e}")
raise
print("🎯 SGD optimizer behavior:")
print(" Maintains momentum buffers for accelerated updates")
print(" Tracks step count for learning rate scheduling")
print(" Supports weight decay for regularization")
print("📈 Progress: SGD Optimizer ✓")
# Run the test
test_sgd_optimizer()In [ ]:
#| export
class Adam:
"""
Adam Optimizer
Implements Adam algorithm with adaptive learning rates:
- First moment: exponential moving average of gradients
- Second moment: exponential moving average of squared gradients
- Bias correction: accounts for initialization bias
- Adaptive updates: different learning rate per parameter
"""
def __init__(self, parameters: List[Variable], learning_rate: float = 0.001,
beta1: float = 0.9, beta2: float = 0.999, epsilon: float = 1e-8,
weight_decay: float = 0.0):
"""
Initialize Adam optimizer.
Args:
parameters: List of Variables to optimize
learning_rate: Learning rate (default: 0.001)
beta1: Exponential decay rate for first moment (default: 0.9)
beta2: Exponential decay rate for second moment (default: 0.999)
epsilon: Small constant for numerical stability (default: 1e-8)
weight_decay: L2 regularization coefficient (default: 0.0)
TODO: Implement Adam optimizer initialization.
APPROACH:
1. Store parameters and hyperparameters
2. Initialize first moment buffers (m_t)
3. Initialize second moment buffers (v_t)
4. Set up step counter for bias correction
EXAMPLE:
```python
# Create Adam optimizer
optimizer = Adam([w1, w2, b1, b2], learning_rate=0.001)
# In training loop:
optimizer.zero_grad()
loss.backward()
optimizer.step()
```
HINTS:
- Store all hyperparameters
- Initialize moment buffers as empty dicts
- Use parameter id() as key for tracking
- Buffers will be created lazily in step()
"""
### BEGIN SOLUTION
self.parameters = parameters
self.learning_rate = learning_rate
self.beta1 = beta1
self.beta2 = beta2
self.epsilon = epsilon
self.weight_decay = weight_decay
# Initialize moment buffers (created lazily)
self.first_moment = {} # m_t
self.second_moment = {} # v_t
# Track optimization steps for bias correction
self.step_count = 0
### END SOLUTION
def step(self) -> None:
"""
Perform one optimization step using Adam algorithm.
TODO: Implement Adam parameter update.
APPROACH:
1. Increment step count
2. For each parameter with gradient:
a. Get current gradient
b. Apply weight decay if specified
c. Update first moment (momentum)
d. Update second moment (variance)
e. Apply bias correction
f. Update parameter with adaptive learning rate
MATHEMATICAL FORMULATION:
- m_t = beta1 * m_{t-1} + (1 - beta1) * gradient
- v_t = beta2 * v_{t-1} + (1 - beta2) * gradient^2
- m_hat = m_t / (1 - beta1^t)
- v_hat = v_t / (1 - beta2^t)
- parameter = parameter - learning_rate * m_hat / (sqrt(v_hat) + epsilon)
IMPLEMENTATION HINTS:
- Use id(param) as key for moment buffers
- Initialize buffers with zeros if not exists
- Use np.sqrt() for square root
- Handle numerical stability with epsilon
"""
### BEGIN SOLUTION
self.step_count += 1
for param in self.parameters:
if param.grad is not None:
# Get gradient
gradient = param.grad.data.data
# Apply weight decay (L2 regularization)
if self.weight_decay > 0:
gradient = gradient + self.weight_decay * param.data.data
# Get or create moment buffers
param_id = id(param)
if param_id not in self.first_moment:
self.first_moment[param_id] = np.zeros_like(param.data.data)
self.second_moment[param_id] = np.zeros_like(param.data.data)
# Update first moment (momentum)
self.first_moment[param_id] = (
self.beta1 * self.first_moment[param_id] +
(1 - self.beta1) * gradient
)
# Update second moment (variance)
self.second_moment[param_id] = (
self.beta2 * self.second_moment[param_id] +
(1 - self.beta2) * gradient * gradient
)
# Bias correction
first_moment_corrected = (
self.first_moment[param_id] / (1 - self.beta1 ** self.step_count)
)
second_moment_corrected = (
self.second_moment[param_id] / (1 - self.beta2 ** self.step_count)
)
# Update parameter with adaptive learning rate
param.data = Tensor(
param.data.data - self.learning_rate * first_moment_corrected /
(np.sqrt(second_moment_corrected) + self.epsilon)
)
### END SOLUTION
def zero_grad(self) -> None:
"""
Zero out gradients for all parameters.
TODO: Implement gradient zeroing (same as SGD).
IMPLEMENTATION HINTS:
- Set param.grad = None for all parameters
- This is identical to SGD implementation
"""
### BEGIN SOLUTION
for param in self.parameters:
param.grad = None
### END SOLUTIONIn [ ]:
def test_adam_optimizer():
"""Test Adam optimizer implementation"""
print("🔬 Unit Test: Adam Optimizer...")
# Create test parameters
w1 = Variable(1.0, requires_grad=True)
w2 = Variable(2.0, requires_grad=True)
b = Variable(0.5, requires_grad=True)
# Create optimizer
optimizer = Adam([w1, w2, b], learning_rate=0.01, beta1=0.9, beta2=0.999, epsilon=1e-8)
# Test zero_grad
try:
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
optimizer.zero_grad()
assert w1.grad is None, "Gradient should be None after zero_grad"
assert w2.grad is None, "Gradient should be None after zero_grad"
assert b.grad is None, "Gradient should be None after zero_grad"
print("✅ zero_grad() works correctly")
except Exception as e:
print(f"❌ zero_grad() failed: {e}")
raise
# Test step with gradients
try:
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
# First step
original_w1 = w1.data.data.item()
original_w2 = w2.data.data.item()
original_b = b.data.data.item()
optimizer.step()
# Check that parameters were updated (Adam uses adaptive learning rates)
assert w1.data.data.item() != original_w1, "w1 should have been updated"
assert w2.data.data.item() != original_w2, "w2 should have been updated"
assert b.data.data.item() != original_b, "b should have been updated"
print("✅ Parameter updates work correctly")
except Exception as e:
print(f"❌ Parameter updates failed: {e}")
raise
# Test moment buffers
try:
assert len(optimizer.first_moment) == 3, f"Should have 3 first moment buffers, got {len(optimizer.first_moment)}"
assert len(optimizer.second_moment) == 3, f"Should have 3 second moment buffers, got {len(optimizer.second_moment)}"
print("✅ Moment buffers created correctly")
except Exception as e:
print(f"❌ Moment buffers failed: {e}")
raise
# Test step counting and bias correction
try:
assert optimizer.step_count == 1, f"Step count should be 1, got {optimizer.step_count}"
# Take another step
w1.grad = Variable(0.1)
w2.grad = Variable(0.2)
b.grad = Variable(0.05)
optimizer.step()
assert optimizer.step_count == 2, f"Step count should be 2, got {optimizer.step_count}"
print("✅ Step counting and bias correction work correctly")
except Exception as e:
print(f"❌ Step counting and bias correction failed: {e}")
raise
# Test adaptive learning rates
try:
# Adam should have different effective learning rates for different parameters
# This is tested implicitly by the parameter updates above
print("✅ Adaptive learning rates work correctly")
except Exception as e:
print(f"❌ Adaptive learning rates failed: {e}")
raise
print("🎯 Adam optimizer behavior:")
print(" Maintains first and second moment estimates")
print(" Applies bias correction for early training")
print(" Uses adaptive learning rates per parameter")
print(" Combines benefits of momentum and RMSprop")
print("📈 Progress: Adam Optimizer ✓")
# Run the test
test_adam_optimizer()In [ ]:
#| export
class StepLR:
"""
Step Learning Rate Scheduler
Decays learning rate by gamma every step_size epochs:
learning_rate = initial_lr * (gamma ^ (epoch // step_size))
"""
def __init__(self, optimizer: Union[SGD, Adam], step_size: int, gamma: float = 0.1):
"""
Initialize step learning rate scheduler.
Args:
optimizer: Optimizer to schedule
step_size: Number of epochs between decreases
gamma: Multiplicative factor for learning rate decay
TODO: Implement learning rate scheduler initialization.
APPROACH:
1. Store optimizer reference
2. Store scheduling parameters
3. Save initial learning rate
4. Initialize step counter
EXAMPLE:
```python
optimizer = SGD([w1, w2], learning_rate=0.1)
scheduler = StepLR(optimizer, step_size=10, gamma=0.1)
# In training loop:
for epoch in range(100):
train_one_epoch()
scheduler.step() # Update learning rate
```
HINTS:
- Store optimizer reference
- Save initial learning rate from optimizer
- Initialize step counter to 0
- gamma is the decay factor (0.1 = 10x reduction)
"""
### BEGIN SOLUTION
self.optimizer = optimizer
self.step_size = step_size
self.gamma = gamma
self.initial_lr = optimizer.learning_rate
self.step_count = 0
### END SOLUTION
def step(self) -> None:
"""
Update learning rate based on current step.
TODO: Implement learning rate update.
APPROACH:
1. Increment step counter
2. Calculate new learning rate using step decay formula
3. Update optimizer's learning rate
MATHEMATICAL FORMULATION:
new_lr = initial_lr * (gamma ^ ((step_count - 1) // step_size))
IMPLEMENTATION HINTS:
- Use // for integer division
- Use ** for exponentiation
- Update optimizer.learning_rate directly
"""
### BEGIN SOLUTION
self.step_count += 1
# Calculate new learning rate
decay_factor = self.gamma ** ((self.step_count - 1) // self.step_size)
new_lr = self.initial_lr * decay_factor
# Update optimizer's learning rate
self.optimizer.learning_rate = new_lr
### END SOLUTION
def get_lr(self) -> float:
"""
Get current learning rate.
TODO: Return current learning rate.
IMPLEMENTATION HINTS:
- Return optimizer.learning_rate
"""
### BEGIN SOLUTION
return self.optimizer.learning_rate
### END SOLUTIONIn [ ]:
def test_step_scheduler():
"""Test StepLR scheduler implementation"""
print("🔬 Unit Test: Step Learning Rate Scheduler...")
# Create test parameters and optimizer
w = Variable(1.0, requires_grad=True)
optimizer = SGD([w], learning_rate=0.1)
# Test scheduler initialization
try:
scheduler = StepLR(optimizer, step_size=10, gamma=0.1)
# Test initial learning rate
assert scheduler.get_lr() == 0.1, f"Initial learning rate should be 0.1, got {scheduler.get_lr()}"
print("✅ Initial learning rate is correct")
except Exception as e:
print(f"❌ Initial learning rate failed: {e}")
raise
# Test step-based decay
try:
# Steps 1-10: no decay (decay happens after step 10)
for i in range(10):
scheduler.step()
assert scheduler.get_lr() == 0.1, f"Learning rate should still be 0.1 after 10 steps, got {scheduler.get_lr()}"
# Step 11: decay should occur
scheduler.step()
expected_lr = 0.1 * 0.1 # 0.01
assert abs(scheduler.get_lr() - expected_lr) < 1e-6, f"Learning rate should be {expected_lr} after 11 steps, got {scheduler.get_lr()}"
print("✅ Step-based decay works correctly")
except Exception as e:
print(f"❌ Step-based decay failed: {e}")
raise
# Test multiple decay levels
try:
# Steps 12-20: should stay at 0.01
for i in range(9):
scheduler.step()
assert abs(scheduler.get_lr() - 0.01) < 1e-6, f"Learning rate should be 0.01 after 20 steps, got {scheduler.get_lr()}"
# Step 21: another decay
scheduler.step()
expected_lr = 0.01 * 0.1 # 0.001
assert abs(scheduler.get_lr() - expected_lr) < 1e-6, f"Learning rate should be {expected_lr} after 21 steps, got {scheduler.get_lr()}"
print("✅ Multiple decay levels work correctly")
except Exception as e:
print(f"❌ Multiple decay levels failed: {e}")
raise
# Test with different optimizer
try:
w2 = Variable(2.0, requires_grad=True)
adam_optimizer = Adam([w2], learning_rate=0.001)
adam_scheduler = StepLR(adam_optimizer, step_size=5, gamma=0.5)
# Test initial learning rate
assert adam_scheduler.get_lr() == 0.001, f"Initial Adam learning rate should be 0.001, got {adam_scheduler.get_lr()}"
# Test decay after 5 steps
for i in range(5):
adam_scheduler.step()
# Learning rate should still be 0.001 after 5 steps
assert adam_scheduler.get_lr() == 0.001, f"Adam learning rate should still be 0.001 after 5 steps, got {adam_scheduler.get_lr()}"
# Step 6: decay should occur
adam_scheduler.step()
expected_lr = 0.001 * 0.5 # 0.0005
assert abs(adam_scheduler.get_lr() - expected_lr) < 1e-6, f"Adam learning rate should be {expected_lr} after 6 steps, got {adam_scheduler.get_lr()}"
print("✅ Works with different optimizers")
except Exception as e:
print(f"❌ Different optimizers failed: {e}")
raise
print("🎯 Step learning rate scheduler behavior:")
print(" Reduces learning rate at regular intervals")
print(" Multiplies current rate by gamma factor")
print(" Works with any optimizer (SGD, Adam, etc.)")
print("📈 Progress: Step Learning Rate Scheduler ✓")
# Run the test
test_step_scheduler()In [ ]:
def train_simple_model():
"""
Complete training example using optimizers.
TODO: Implement a complete training loop.
APPROACH:
1. Create a simple model (linear regression)
2. Generate training data
3. Set up optimizer and scheduler
4. Train for several epochs
5. Show convergence
LEARNING OBJECTIVE:
- See how optimizers enable real learning
- Compare SGD vs Adam performance
- Understand the complete training workflow
"""
### BEGIN SOLUTION
print("Training simple linear regression model...")
# Create simple model: y = w*x + b
w = Variable(0.1, requires_grad=True) # Initialize near zero
b = Variable(0.0, requires_grad=True)
# Training data: y = 2*x + 1
x_data = [1.0, 2.0, 3.0, 4.0, 5.0]
y_data = [3.0, 5.0, 7.0, 9.0, 11.0]
# Try SGD first
print("\n🔍 Training with SGD...")
optimizer_sgd = SGD([w, b], learning_rate=0.01, momentum=0.9)
for epoch in range(60):
total_loss = 0
for x_val, y_val in zip(x_data, y_data):
# Forward pass
x = Variable(x_val, requires_grad=False)
y_target = Variable(y_val, requires_grad=False)
# Prediction: y = w*x + b
try:
from tinytorch.core.autograd import add, multiply, subtract
except ImportError:
setup_import_paths()
from autograd_dev import add, multiply, subtract
prediction = add(multiply(w, x), b)
# Loss: (prediction - target)^2
error = subtract(prediction, y_target)
loss = multiply(error, error)
# Backward pass
optimizer_sgd.zero_grad()
loss.backward()
optimizer_sgd.step()
total_loss += loss.data.data.item()
if epoch % 10 == 0:
print(f"Epoch {epoch}: Loss = {total_loss:.4f}, w = {w.data.data.item():.3f}, b = {b.data.data.item():.3f}")
sgd_final_w = w.data.data.item()
sgd_final_b = b.data.data.item()
# Reset parameters and try Adam
print("\n🔍 Training with Adam...")
w.data = Tensor(0.1)
b.data = Tensor(0.0)
optimizer_adam = Adam([w, b], learning_rate=0.01)
for epoch in range(60):
total_loss = 0
for x_val, y_val in zip(x_data, y_data):
# Forward pass
x = Variable(x_val, requires_grad=False)
y_target = Variable(y_val, requires_grad=False)
# Prediction: y = w*x + b
prediction = add(multiply(w, x), b)
# Loss: (prediction - target)^2
error = subtract(prediction, y_target)
loss = multiply(error, error)
# Backward pass
optimizer_adam.zero_grad()
loss.backward()
optimizer_adam.step()
total_loss += loss.data.data.item()
if epoch % 10 == 0:
print(f"Epoch {epoch}: Loss = {total_loss:.4f}, w = {w.data.data.item():.3f}, b = {b.data.data.item():.3f}")
adam_final_w = w.data.data.item()
adam_final_b = b.data.data.item()
print(f"\n📊 Results:")
print(f"Target: w = 2.0, b = 1.0")
print(f"SGD: w = {sgd_final_w:.3f}, b = {sgd_final_b:.3f}")
print(f"Adam: w = {adam_final_w:.3f}, b = {adam_final_b:.3f}")
return sgd_final_w, sgd_final_b, adam_final_w, adam_final_b
### END SOLUTIONIn [ ]:
def test_training_integration():
"""Test complete training integration with optimizers"""
print("🔬 Unit Test: Complete Training Integration...")
# Test training with SGD and Adam
try:
sgd_w, sgd_b, adam_w, adam_b = train_simple_model()
# Test SGD convergence
assert abs(sgd_w - 2.0) < 0.1, f"SGD should converge close to w=2.0, got {sgd_w}"
assert abs(sgd_b - 1.0) < 0.1, f"SGD should converge close to b=1.0, got {sgd_b}"
print("✅ SGD convergence works")
# Test Adam convergence (may be different due to adaptive learning rates)
assert abs(adam_w - 2.0) < 1.0, f"Adam should converge reasonably close to w=2.0, got {adam_w}"
assert abs(adam_b - 1.0) < 1.0, f"Adam should converge reasonably close to b=1.0, got {adam_b}"
print("✅ Adam convergence works")
except Exception as e:
print(f"❌ Training integration failed: {e}")
raise
# Test optimizer comparison
try:
# Both optimizers should achieve reasonable results
sgd_error = (sgd_w - 2.0)**2 + (sgd_b - 1.0)**2
adam_error = (adam_w - 2.0)**2 + (adam_b - 1.0)**2
# Both should have low error (< 0.1)
assert sgd_error < 0.1, f"SGD error should be < 0.1, got {sgd_error}"
assert adam_error < 1.0, f"Adam error should be < 1.0, got {adam_error}"
print("✅ Optimizer comparison works")
except Exception as e:
print(f"❌ Optimizer comparison failed: {e}")
raise
# Test gradient flow
try:
# Create a simple test to verify gradients flow correctly
w = Variable(1.0, requires_grad=True)
b = Variable(0.0, requires_grad=True)
# Set up simple gradients
w.grad = Variable(0.1)
b.grad = Variable(0.05)
# Test SGD step
sgd_optimizer = SGD([w, b], learning_rate=0.1)
original_w = w.data.data.item()
original_b = b.data.data.item()
sgd_optimizer.step()
# Check updates
assert w.data.data.item() != original_w, "SGD should update w"
assert b.data.data.item() != original_b, "SGD should update b"
print("✅ Gradient flow works correctly")
except Exception as e:
print(f"❌ Gradient flow failed: {e}")
raise
print("🎯 Training integration behavior:")
print(" Optimizers successfully minimize loss functions")
print(" SGD and Adam both converge to target values")
print(" Gradient computation and updates work correctly")
print(" Ready for real neural network training")
print("📈 Progress: Complete Training Integration ✓")
# Run the test
test_training_integration()In [ ]:
# =============================================================================
# STANDARDIZED MODULE TESTING - DO NOT MODIFY
# This cell is locked to ensure consistent testing across all TinyTorch modules
# =============================================================================
if __name__ == "__main__":
from tito.tools.testing import run_module_tests_auto
# Automatically discover and run all tests in this module
success = run_module_tests_auto("Optimizers")