The mathematical decomposition of generalization error: balancing underfitting bias and overfitting variance for optimal model performance.
# Calculating Empirical Bias and Variance via Bootstrapping in Python# ─────────────────────────────────────────────────────────────────import numpy as npfrom sklearn.tree import DecisionTreeRegressorfrom sklearn.ensemble import RandomForestRegressor np.random.seed(42)n_samples, n_test, n_bootstrap = 50, 100, 200 # Ground truth: f(x) = sin(pi * x)X_test = np.linspace(-1, 1, n_test).reshape(-1, 1)y_test_true = np.sin(np.pi * X_test.ravel()) def evaluate_bias_variance(estimator): predictions = np.zeros((n_bootstrap, n_test)) for i in range(n_bootstrap): # Generate new random training sample each iteration X_train = np.random.uniform(-1, 1, (n_samples, 1)) y_train = np.sin(np.pi * X_train.ravel()) + np.random.normal(0, 0.15, n_samples) estimator.fit(X_train, y_train) predictions[i, :] = estimator.predict(X_test) # Mean prediction across all bootstrap iterations y_pred_mean = np.mean(predictions, axis=0) bias_sq = np.mean((y_pred_mean - y_test_true) ** 2) variance = np.mean(np.var(predictions, axis=0)) total_error = bias_sq + variance return bias_sq, variance, total_error # Compare Shallow Tree (High Bias) vs Deep Tree (High Variance) vs Random Forest (Low Both)models = { "Shallow Tree (max_depth=1)": DecisionTreeRegressor(max_depth=1), "Deep Tree (max_depth=10)": DecisionTreeRegressor(max_depth=10), "Random Forest (100 trees)": RandomForestRegressor(n_estimators=100, random_state=42)} print(f"{'Model Architecture':<28} | {'Bias²':<8} | {'Variance':<8} | {'Total Error':<8}")print("-" * 62)for name, model in models.items(): b2, var, err = evaluate_bias_variance(model) print(f"{name:<28} | {b2:.4f} | {var:.4f} | {err:.4f}")Interactive Error vs Complexity curve, bullseye dartboard matrix, and generalization error decomposition.
Which machine learning technique is mathematically proven to drastically reduce model VARIANCE without increasing BIAS?