Skip to content

Commit f363673

Browse files
Harmanayapre-commit-ci[bot]cclauss
authored
Add Ridge Regression to Machine Learning (#12246)
* Fix issue #12108: Added Ridge Regression to Machine Learning * Added type hints and minor case improvements * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Resolved ruff checks * Added doctests * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Resolved ruff checks * Resolved mypy checks * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Resolved ruff checks * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * updating DIRECTORY.md * Comments should not force line wrapping * updating DIRECTORY.md * Fix feature scaling variable assignment in predictions * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> Co-authored-by: Christian Clauss <cclauss@me.com> Co-authored-by: cclauss <cclauss@users.noreply.github.com>
1 parent 2acda52 commit f363673

2 files changed

Lines changed: 173 additions & 0 deletions

File tree

‎DIRECTORY.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -741,6 +741,7 @@
741741
* [Q Learning](machine_learning/q_learning.py)
742742
* [Random Forest Classifier](machine_learning/random_forest_classifier.py)
743743
* [Random Forest Regressor](machine_learning/random_forest_regressor.py)
744+
* [Ridge Regression](machine_learning/ridge_regression.py)
744745
* [Rmsprop](machine_learning/rmsprop.py)
745746
* [Scoring Functions](machine_learning/scoring_functions.py)
746747
* [Self Organizing Map](machine_learning/self_organizing_map.py)
Lines changed: 172 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,172 @@
1+
import numpy as np
2+
import pandas as pd
3+
4+
5+
class RidgeRegression:
6+
def __init__(
7+
self, alpha: float = 0.001, lambda_: float = 0.1, iterations: int = 1000
8+
) -> None:
9+
"""
10+
Ridge Regression Constructor
11+
:param alpha: Learning rate for gradient descent
12+
:param lambda_: Regularization parameter (L2 regularization)
13+
:param iterations: Number of iterations for gradient descent
14+
"""
15+
self.alpha = alpha
16+
self.lambda_ = lambda_
17+
self.iterations = iterations
18+
self.theta: np.ndarray | None = None
19+
20+
def feature_scaling(
21+
self, features: np.ndarray
22+
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
23+
"""
24+
Normalize features to have mean 0 and standard deviation 1.
25+
26+
:param features: Input features, shape (m, n)
27+
:return: Tuple containing:
28+
- Scaled features
29+
- Mean of each feature
30+
- Standard deviation of each feature
31+
32+
Example:
33+
>>> rr = RidgeRegression()
34+
>>> features = np.array([[1, 2], [2, 3], [4, 6]])
35+
>>> scaled_features, mean, std = rr.feature_scaling(features)
36+
>>> np.allclose(scaled_features.mean(axis=0), 0)
37+
True
38+
>>> np.allclose(scaled_features.std(axis=0), 1)
39+
True
40+
"""
41+
mean = np.mean(features, axis=0)
42+
std = np.std(features, axis=0)
43+
44+
# Avoid division by zero for constant features (std = 0)
45+
std[std == 0] = 1 # Set std=1 for constant features to avoid NaN
46+
47+
scaled_features = (features - mean) / std
48+
return scaled_features, mean, std
49+
50+
def fit(self, features: np.ndarray, target: np.ndarray) -> None:
51+
"""
52+
Fit the Ridge Regression model to the training data.
53+
54+
:param features: Input features, shape (m, n)
55+
:param target: Target values, shape (m,)
56+
57+
Example:
58+
>>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10)
59+
>>> features = np.array([[1, 2], [2, 3], [4, 6]])
60+
>>> target = np.array([1, 2, 3])
61+
>>> rr.fit(features, target)
62+
>>> rr.theta is not None
63+
True
64+
"""
65+
# Normalize features
66+
features_scaled, _mean, _std = self.feature_scaling(features)
67+
m, n = features_scaled.shape
68+
self.theta = np.zeros(n) # Initialize weights to zeros
69+
70+
for _ in range(self.iterations):
71+
predictions = features_scaled.dot(self.theta)
72+
error = predictions - target
73+
74+
# Compute gradient with L2 regularization
75+
gradient = (features_scaled.T.dot(error) + self.lambda_ * self.theta) / m
76+
self.theta -= self.alpha * gradient # Update weights
77+
78+
def predict(self, features: np.ndarray) -> np.ndarray:
79+
"""
80+
Predict values using the trained model.
81+
82+
:param features: Input features, shape (m, n)
83+
:return: Predicted values, shape (m,)
84+
85+
Example:
86+
>>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10)
87+
>>> features = np.array([[1, 2], [2, 3], [4, 6]])
88+
>>> target = np.array([1, 2, 3])
89+
>>> rr.fit(features, target)
90+
>>> predictions = rr.predict(features)
91+
>>> predictions.shape == target.shape
92+
True
93+
"""
94+
if self.theta is None:
95+
raise ValueError("Model is not trained yet. Call the `fit` method first.")
96+
97+
# Scale features using training data
98+
features_scaled, _mean, _std = self.feature_scaling(features)
99+
return features_scaled.dot(self.theta)
100+
101+
def compute_cost(self, features: np.ndarray, target: np.ndarray) -> float:
102+
"""
103+
Compute the cost function with regularization.
104+
105+
:param features: Input features, shape (m, n)
106+
:param target: Target values, shape (m,)
107+
:return: Computed cost
108+
109+
Example:
110+
>>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10)
111+
>>> features = np.array([[1, 2], [2, 3], [4, 6]])
112+
>>> target = np.array([1, 2, 3])
113+
>>> rr.fit(features, target)
114+
>>> cost = rr.compute_cost(features, target)
115+
>>> isinstance(cost, float)
116+
True
117+
"""
118+
if self.theta is None:
119+
raise ValueError("Model is not trained yet. Call the `fit` method first.")
120+
121+
# Scale features using training data
122+
features_scaled, _mean, _std = self.feature_scaling(features)
123+
m = len(target)
124+
predictions = features_scaled.dot(self.theta)
125+
cost = (1 / (2 * m)) * np.sum((predictions - target) ** 2) + (
126+
self.lambda_ / (2 * m)
127+
) * np.sum(self.theta**2)
128+
return cost
129+
130+
def mean_absolute_error(self, y_true: np.ndarray, y_pred: np.ndarray) -> float:
131+
"""
132+
Compute Mean Absolute Error (MAE) between true and predicted values.
133+
134+
:param y_true: Actual target values, shape (m,)
135+
:param y_pred: Predicted target values, shape (m,)
136+
:return: MAE
137+
138+
Example:
139+
>>> rr = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=10)
140+
>>> y_true = np.array([1, 2, 3])
141+
>>> y_pred = np.array([1.1, 2.1, 2.9])
142+
>>> mae = rr.mean_absolute_error(y_true, y_pred)
143+
>>> isinstance(mae, float)
144+
True
145+
"""
146+
return np.mean(np.abs(y_true - y_pred))
147+
148+
149+
# Example usage
150+
if __name__ == "__main__":
151+
# Load dataset
152+
data = pd.read_csv(
153+
"https://raw.githubusercontent.com/yashLadha/The_Math_of_Intelligence/master/Week1/ADRvsRating.csv"
154+
)
155+
data_x = data[["Rating"]].to_numpy() # Feature: Rating
156+
data_y = data["ADR"].to_numpy() # Target: ADR
157+
data_y = (data_y - np.mean(data_y)) / np.std(data_y)
158+
159+
# Add bias term (intercept) to the feature matrix
160+
data_x = np.c_[np.ones(data_x.shape[0]), data_x] # Add intercept term
161+
162+
# Initialize and train the Ridge Regression model
163+
model = RidgeRegression(alpha=0.01, lambda_=0.1, iterations=1000)
164+
model.fit(data_x, data_y)
165+
166+
# Predictions
167+
predictions = model.predict(data_x)
168+
169+
# Results
170+
print("Optimized Weights:", model.theta)
171+
print("Cost:", model.compute_cost(data_x, data_y))
172+
print("Mean Absolute Error:", model.mean_absolute_error(data_y, predictions))

0 commit comments

Comments
 (0)