|
| 1 | +import numpy as np |
| 2 | +import pandas as pd |
| 3 | + |
| 4 | + |
| 5 | +class RidgeRegression: |
| 6 | + def __init__( |
| 7 | + self, alpha: float = 0.001, lambda_: float = 0.1, iterations: int = 1000 |
| 8 | + ) -> None: |
| 9 | + """ |
| 10 | + Ridge Regression Constructor |
| 11 | + :param alpha: Learning rate for gradient descent |
| 12 | + :param lambda_: Regularization parameter (L2 regularization) |
| 13 | + :param iterations: Number of iterations for gradient descent |
| 14 | + """ |
| 15 | + self.alpha = alpha |
| 16 | + self.lambda_ = lambda_ |
| 17 | + self.iterations = iterations |
| 18 | + self.theta: np.ndarray | None = None |
| 19 | + |
| 20 | + def feature_scaling( |
| 21 | + self, features: np.ndarray |
| 22 | + ) -> tuple[np.ndarray, np.ndarray, np.ndarray]: |
| 23 | + """ |
| 24 | + Normalize features to have mean 0 and standard deviation 1. |
| 25 | +
|
| 26 | + :param features: Input features, shape (m, n) |
| 27 | + :return: Tuple containing: |
| 28 | + - Scaled features |
| 29 | + - Mean of each feature |
| 30 | + - Standard deviation of each feature |
| 31 | +
|
| 32 | + Example: |
| 33 | + >>> rr = RidgeRegression() |
| 34 | + >>> features = np.array([[1, 2], [2, 3], [4, 6]]) |
| 35 | + >>> scaled_features, mean, std = rr.feature_scaling(features) |
| 36 | + >>> np.allclose(scaled_features.mean(axis=0), 0) |
| 37 | + True |
| 38 | + >>> np.allclose(scaled_features.std(axis=0), 1) |
| 39 | + True |
| 40 | + """ |
| 41 | + mean = np.mean(features, axis=0) |
| 42 | + std = np.std(features, axis=0) |
| 43 | + |
| 44 | + # Avoid division by zero for constant features (std = 0) |
| 45 | + std[std == 0] = 1 # Set std=1 for constant features to avoid NaN |
| 46 | + |
| 47 | + scaled_features = (features - mean) / std |
| 48 | + return scaled_features, mean, std |
| 49 | + |
| 50 | + def fit(self, features: np.ndarray, target: np.ndarray) -> None: |
| 51 | + """ |
| 52 | + Fit the Ridge Regression model to the training data. |
| 53 | +
|
| 54 | + :param features: Input features, shape (m, n) |
| 55 | + :param target: Target values, shape (m,) |
| 56 | +
|
| 57 | + Example: |
| 58 | + >>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10) |
| 59 | + >>> features = np.array([[1, 2], [2, 3], [4, 6]]) |
| 60 | + >>> target = np.array([1, 2, 3]) |
| 61 | + >>> rr.fit(features, target) |
| 62 | + >>> rr.theta is not None |
| 63 | + True |
| 64 | + """ |
| 65 | + # Normalize features |
| 66 | + features_scaled, _mean, _std = self.feature_scaling(features) |
| 67 | + m, n = features_scaled.shape |
| 68 | + self.theta = np.zeros(n) # Initialize weights to zeros |
| 69 | + |
| 70 | + for _ in range(self.iterations): |
| 71 | + predictions = features_scaled.dot(self.theta) |
| 72 | + error = predictions - target |
| 73 | + |
| 74 | + # Compute gradient with L2 regularization |
| 75 | + gradient = (features_scaled.T.dot(error) + self.lambda_ * self.theta) / m |
| 76 | + self.theta -= self.alpha * gradient # Update weights |
| 77 | + |
| 78 | + def predict(self, features: np.ndarray) -> np.ndarray: |
| 79 | + """ |
| 80 | + Predict values using the trained model. |
| 81 | +
|
| 82 | + :param features: Input features, shape (m, n) |
| 83 | + :return: Predicted values, shape (m,) |
| 84 | +
|
| 85 | + Example: |
| 86 | + >>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10) |
| 87 | + >>> features = np.array([[1, 2], [2, 3], [4, 6]]) |
| 88 | + >>> target = np.array([1, 2, 3]) |
| 89 | + >>> rr.fit(features, target) |
| 90 | + >>> predictions = rr.predict(features) |
| 91 | + >>> predictions.shape == target.shape |
| 92 | + True |
| 93 | + """ |
| 94 | + if self.theta is None: |
| 95 | + raise ValueError("Model is not trained yet. Call the `fit` method first.") |
| 96 | + |
| 97 | + # Scale features using training data |
| 98 | + features_scaled, _mean, _std = self.feature_scaling(features) |
| 99 | + return features_scaled.dot(self.theta) |
| 100 | + |
| 101 | + def compute_cost(self, features: np.ndarray, target: np.ndarray) -> float: |
| 102 | + """ |
| 103 | + Compute the cost function with regularization. |
| 104 | +
|
| 105 | + :param features: Input features, shape (m, n) |
| 106 | + :param target: Target values, shape (m,) |
| 107 | + :return: Computed cost |
| 108 | +
|
| 109 | + Example: |
| 110 | + >>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10) |
| 111 | + >>> features = np.array([[1, 2], [2, 3], [4, 6]]) |
| 112 | + >>> target = np.array([1, 2, 3]) |
| 113 | + >>> rr.fit(features, target) |
| 114 | + >>> cost = rr.compute_cost(features, target) |
| 115 | + >>> isinstance(cost, float) |
| 116 | + True |
| 117 | + """ |
| 118 | + if self.theta is None: |
| 119 | + raise ValueError("Model is not trained yet. Call the `fit` method first.") |
| 120 | + |
| 121 | + # Scale features using training data |
| 122 | + features_scaled, _mean, _std = self.feature_scaling(features) |
| 123 | + m = len(target) |
| 124 | + predictions = features_scaled.dot(self.theta) |
| 125 | + cost = (1 / (2 * m)) * np.sum((predictions - target) ** 2) + ( |
| 126 | + self.lambda_ / (2 * m) |
| 127 | + ) * np.sum(self.theta**2) |
| 128 | + return cost |
| 129 | + |
| 130 | + def mean_absolute_error(self, y_true: np.ndarray, y_pred: np.ndarray) -> float: |
| 131 | + """ |
| 132 | + Compute Mean Absolute Error (MAE) between true and predicted values. |
| 133 | +
|
| 134 | + :param y_true: Actual target values, shape (m,) |
| 135 | + :param y_pred: Predicted target values, shape (m,) |
| 136 | + :return: MAE |
| 137 | +
|
| 138 | + Example: |
| 139 | + >>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10) |
| 140 | + >>> y_true = np.array([1, 2, 3]) |
| 141 | + >>> y_pred = np.array([1.1, 2.1, 2.9]) |
| 142 | + >>> mae = rr.mean_absolute_error(y_true, y_pred) |
| 143 | + >>> isinstance(mae, float) |
| 144 | + True |
| 145 | + """ |
| 146 | + return np.mean(np.abs(y_true - y_pred)) |
| 147 | + |
| 148 | + |
| 149 | +# Example usage |
| 150 | +if __name__ == "__main__": |
| 151 | + # Load dataset |
| 152 | + data = pd.read_csv( |
| 153 | + "https://raw.githubusercontent.com/yashLadha/The_Math_of_Intelligence/master/Week1/ADRvsRating.csv" |
| 154 | + ) |
| 155 | + data_x = data[["Rating"]].to_numpy() # Feature: Rating |
| 156 | + data_y = data["ADR"].to_numpy() # Target: ADR |
| 157 | + data_y = (data_y - np.mean(data_y)) / np.std(data_y) |
| 158 | + |
| 159 | + # Add bias term (intercept) to the feature matrix |
| 160 | + data_x = np.c_[np.ones(data_x.shape[0]), data_x] # Add intercept term |
| 161 | + |
| 162 | + # Initialize and train the Ridge Regression model |
| 163 | + model = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=1000) |
| 164 | + model.fit(data_x, data_y) |
| 165 | + |
| 166 | + # Predictions |
| 167 | + predictions = model.predict(data_x) |
| 168 | + |
| 169 | + # Results |
| 170 | + print("Optimized Weights:", model.theta) |
| 171 | + print("Cost:", model.compute_cost(data_x, data_y)) |
| 172 | + print("Mean Absolute Error:", model.mean_absolute_error(data_y, predictions)) |
0 commit comments