Mastering model capacity, polynomial complexity, training vs validation loss curves, early stopping, and regularization techniques.
# Demonstrating Underfitting, Right Fit, and Overfitting in Python# ───────────────────────────────────────────────────────────────import numpy as npfrom sklearn.pipeline import make_pipelinefrom sklearn.preprocessing import PolynomialFeaturesfrom sklearn.linear_model import LinearRegression, Ridgefrom sklearn.metrics import mean_squared_error # 1. Generate non-linear ground truth: y = cos(1.5 * pi * x) + noisenp.random.seed(42)n_samples = 30X_train = np.sort(np.random.rand(n_samples))y_train = np.cos(1.5 * np.pi * X_train) + np.random.randn(n_samples) * 0.1 X_test = np.linspace(0, 1, 100)y_test_true = np.cos(1.5 * np.pi * X_test) # 2. Fit 3 Models with different capacitiesdegrees = [1, 4, 15]models = {}labels = {1: "Underfitting (Deg 1)", 4: "Right Fit (Deg 4)", 15: "Overfitting (Deg 15)"} print("Model Evaluation Summary:")print("-" * 55)for deg in degrees: pipeline = make_pipeline(PolynomialFeatures(deg), LinearRegression()) pipeline.fit(X_train[:, np.newaxis], y_train) train_pred = pipeline.predict(X_train[:, np.newaxis]) test_pred = pipeline.predict(X_test[:, np.newaxis]) train_mse = mean_squared_error(y_train, train_pred) test_mse = mean_squared_error(y_test_true, test_pred) print(f"{labels[deg]:<24} | Train MSE: {train_mse:.4f} | Test MSE: {test_mse:.4f}") # 3. Fixing Overfitting via L2 Regularization (Ridge Regression)ridge_pipeline = make_pipeline(PolynomialFeatures(15), Ridge(alpha=1.0))ridge_pipeline.fit(X_train[:, np.newaxis], y_train)ridge_test_mse = mean_squared_error(y_test_true, ridge_pipeline.predict(X_test[:, np.newaxis]))print("-" * 55)print(f"Fixed with Ridge (Deg 15) | Train MSE: {mean_squared_error(y_train, ridge_pipeline.predict(X_train[:, np.newaxis])):.4f} | Test MSE: {ridge_test_mse:.4f}")Interactive 2×3 visual geometry matrix, learning loss curves, and early stopping checkpoint arena.
A neural network achieves 99.8% accuracy on the Training set, but only 64.2% accuracy on the Validation set. What is the diagnosis and the most effective remedy?