Exercises Notebook
Exercises Notebook
Converted from
exercises.ipynbfor web reading.
Regularization Methods - Exercises
Ten graded exercises. Each exercise has a problem, scaffold, and solution cell.
Code cell 2
import numpy as np
import matplotlib.pyplot as plt
import matplotlib as mpl
try:
import seaborn as sns
sns.set_theme(style="whitegrid", palette="colorblind")
HAS_SNS = True
except ImportError:
plt.style.use("seaborn-v0_8-whitegrid")
HAS_SNS = False
mpl.rcParams.update({
"figure.figsize": (10, 6),
"figure.dpi": 120,
"font.size": 13,
"axes.titlesize": 15,
"axes.labelsize": 13,
"xtick.labelsize": 11,
"ytick.labelsize": 11,
"legend.fontsize": 11,
"legend.framealpha": 0.85,
"lines.linewidth": 2.0,
"axes.spines.top": False,
"axes.spines.right": False,
"savefig.bbox": "tight",
"savefig.dpi": 150,
})
np.random.seed(42)
print("Plot setup complete.")
Exercise 1 [*]: L2 Penalty
- State the relevant definition for L2 penalty.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 4
# Your Solution
print("Exercise 1 scaffold: fill in the missing computation for L2 penalty.")
answer = None
print("answer =", answer)
Code cell 5
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 1: L2 Penalty")
vector = np.array([1.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 4.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: L2 penalty is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 2 [*]: Adamw Decay
- State the relevant definition for AdamW decay.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 7
# Your Solution
print("Exercise 2 scaffold: fill in the missing computation for AdamW decay.")
answer = None
print("answer =", answer)
Code cell 8
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 2: Adamw Decay")
vector = np.array([2.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 7.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: AdamW decay is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 3 [*]: Soft Thresholding
- State the relevant definition for soft thresholding.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 10
# Your Solution
print("Exercise 3 scaffold: fill in the missing computation for soft thresholding.")
answer = None
print("answer =", answer)
Code cell 11
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 3: Soft Thresholding")
vector = np.array([3.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 12.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: soft thresholding is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 4 [**]: Nuclear Norm
- State the relevant definition for nuclear norm.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 13
# Your Solution
print("Exercise 4 scaffold: fill in the missing computation for nuclear norm.")
answer = None
print("answer =", answer)
Code cell 14
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 4: Nuclear Norm")
vector = np.array([4.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 19.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: nuclear norm is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 5 [**]: Early Stopping
- State the relevant definition for early stopping.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 16
# Your Solution
print("Exercise 5 scaffold: fill in the missing computation for early stopping.")
answer = None
print("answer =", answer)
Code cell 17
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 5: Early Stopping")
vector = np.array([5.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 28.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: early stopping is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 6 [**]: Label Smoothing Preview
- State the relevant definition for label smoothing preview.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 19
# Your Solution
print("Exercise 6 scaffold: fill in the missing computation for label smoothing preview.")
answer = None
print("answer =", answer)
Code cell 20
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 6: Label Smoothing Preview")
vector = np.array([6.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 39.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: label smoothing preview is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 7 [**]: Gradient Clipping Preview
- State the relevant definition for gradient clipping preview.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 22
# Your Solution
print("Exercise 7 scaffold: fill in the missing computation for gradient clipping preview.")
answer = None
print("answer =", answer)
Code cell 23
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 7: Gradient Clipping Preview")
vector = np.array([7.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 52.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: gradient clipping preview is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 8 [***]: Implicit Regularization
- State the relevant definition for implicit regularization.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 25
# Your Solution
print("Exercise 8 scaffold: fill in the missing computation for implicit regularization.")
answer = None
print("answer =", answer)
Code cell 26
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 8: Implicit Regularization")
vector = np.array([8.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 67.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: implicit regularization is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 9 [***]: Double Descent
- State the relevant definition for double descent.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 28
# Your Solution
print("Exercise 9 scaffold: fill in the missing computation for double descent.")
answer = None
print("answer =", answer)
Code cell 29
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 9: Double Descent")
vector = np.array([9.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 84.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: double descent is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")
Exercise 10 [***]: Lora Rank Regularity
- State the relevant definition for LoRA rank regularity.
- Compute the requested toy quantity.
- Explain the optimization diagnostic you would log in a real model-training run.
Code cell 31
# Your Solution
print("Exercise 10 scaffold: fill in the missing computation for LoRA rank regularity.")
answer = None
print("answer =", answer)
Code cell 32
# Solution
import numpy as np
def header(title):
print("\n" + "=" * 72)
print(title)
print("=" * 72)
def check_close(name, value, target, tol=1e-8):
ok = abs(float(value) - float(target)) <= tol
print(f"{'PASS' if ok else 'FAIL'} - {name}: value={value:.8f}, target={target:.8f}")
if not ok:
raise AssertionError(name)
def check_true(name, condition):
ok = bool(condition)
print(f"{'PASS' if ok else 'FAIL'} - {name}")
if not ok:
raise AssertionError(name)
header("Exercise 10: Lora Rank Regularity")
vector = np.array([10.0, 1.0, -1.0])
answer = float(vector[0] ** 2 + 3.0)
check_close("toy scalar computation", answer, 103.0)
check_true("finite answer", np.isfinite(answer))
print("Definition anchor: LoRA rank regularity is interpreted through the objective, update, or diagnostic in Regularization Methods.")
print("\nTakeaway: a tiny verified computation is the fastest way to test intuition before scaling an optimizer experiment.")