mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-07 20:10:20 +00:00
604 KiB
604 KiB
In [1]:
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as snsIn [2]:
df = pd.read_csv('data/googleplaystore_new.csv')In [3]:
df = df.dropna()
df = df.drop_duplicates()In [4]:
df.head(10)Out [4]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up |
In [5]:
# Getting the column for size, checking if it's ending with an M or a k and converting the Mb to kb with 1024* and then again from kb to just b by another 1024*
def parse_size(size_str):
if isinstance(size_str, str):
if size_str.endswith('M'):
return float(size_str[:-1]) * 1024 * 1024
elif size_str.endswith('k'):
return float(size_str[:-1]) * 1024
return size_strIn [6]:
df['Size in bytes'] = df['Size'].apply(parse_size)In [7]:
df[['Size', 'Size in bytes']].head(10)Out [7]:
| Size | Size in bytes | |
|---|---|---|
| 0 | 11k | 11264.0 |
| 1 | 21M | 22020096.0 |
| 2 | 41k | 41984.0 |
| 3 | 292k | 299008.0 |
| 4 | 3.0M | 3145728.0 |
| 5 | 2.5M | 2621440.0 |
| 6 | 8.2M | 8598323.2 |
| 7 | 695k | 711680.0 |
| 8 | 2.3M | 2411724.8 |
| 9 | 1.4M | 1468006.4 |
In [8]:
print(11*1024) # Just checking the kb and mb conversion happened properly, by checking 2 of the values manually.11264
In [9]:
print((21*1024)*1024) # Ideally this is the same as doing the calculation without the ( ) but just incase!22020096
In [10]:
print(1.4*1024*1024) # This one seemed odd... Checked on an online converted to double-check but turns out the 0.4 bytes shows up there too.1468006.4
In [11]:
df.head(10)Out [11]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 |
In [12]:
df['Numeric Installs'] = df['Installs'].str.replace('+', '').str.replace(',', '').astype(int)In [13]:
df.head(10)Out [13]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | Numeric Installs | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 | 1000000 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 | 1000000 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 | 5000000 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 | 1000000 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 | 1000000 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 | 1000 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 | 100000 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 | 10000000 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 | 100 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 | 1000000 |
In [14]:
df.to_csv('data/googleplaystore_new_new.csv', index=False)In [15]:
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression, Ridge, Lasso
from sklearn.preprocessing import PolynomialFeatures
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_scoreIn [16]:
df_new = pd.read_csv('data/googleplaystore_new_new.csv')In [17]:
columns_to_keep = ['Category', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Rating']
df_min = df_new[columns_to_keep]In [18]:
df_min.head(20)Out [18]:
| Category | Reviews | Content Rating | Size in bytes | Numeric Installs | Rating | |
|---|---|---|---|---|---|---|
| 0 | LIBRARIES_AND_DEMO | 20145 | Everyone | 11264.0 | 1000000 | 4.1 |
| 1 | BUSINESS | 46353 | Everyone | 22020096.0 | 1000000 | 4.3 |
| 2 | LIBRARIES_AND_DEMO | 58055 | Everyone | 41984.0 | 5000000 | 3.9 |
| 3 | LIBRARIES_AND_DEMO | 7750 | Everyone | 299008.0 | 1000000 | 3.8 |
| 4 | EDUCATION | 1619 | Everyone | 3145728.0 | 1000000 | 4.4 |
| 5 | LIBRARIES_AND_DEMO | 26 | Everyone | 2621440.0 | 1000 | 5.0 |
| 6 | BEAUTY | 473 | Mature 17+ | 8598323.2 | 100000 | 4.5 |
| 7 | COMMUNICATION | 124346 | Everyone | 711680.0 | 10000000 | 4.2 |
| 8 | EVENTS | 16 | Everyone | 2411724.8 | 100 | 5.0 |
| 9 | FINANCE | 13868 | Everyone | 1468006.4 | 1000000 | 4.1 |
| 10 | COMMUNICATION | 32254 | Everyone | 5767168.0 | 1000000 | 4.4 |
| 11 | COMMUNICATION | 125232 | Everyone | 2831155.2 | 10000000 | 4.2 |
| 12 | EDUCATION | 430 | Everyone | 538624.0 | 10000 | 4.0 |
| 13 | EDUCATION | 275 | Everyone | 2411724.8 | 50000 | 4.0 |
| 14 | BOOKS_AND_REFERENCE | 1778 | Mature 17+ | 5138022.4 | 500000 | 3.9 |
| 15 | BUSINESS | 2287 | Everyone | 1572864.0 | 1000000 | 4.4 |
| 16 | LIBRARIES_AND_DEMO | 126862 | Everyone | 638976.0 | 10000000 | 3.5 |
| 17 | COMMUNICATION | 255 | Everyone | 1677721.6 | 10000 | 4.1 |
| 18 | LIFESTYLE | 360 | Everyone | 4823449.6 | 10000 | 4.1 |
| 19 | EDUCATION | 656 | Everyone | 569344.0 | 10000 | 4.3 |
In [19]:
df_encoded = pd.get_dummies(df_min, columns=['Category', 'Content Rating'])In [20]:
X = df_encoded.drop('Rating', axis=1)
y = df_encoded['Rating']In [21]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)In [22]:
def evaluate_model(model, X_train, X_test, y_train, y_test, name):
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
results = {
'Model': name,
'MAE': mean_absolute_error(y_test, y_pred),
'RMSE': np.sqrt(mean_squared_error(y_test, y_pred)),
'R2': r2_score(y_test, y_pred)
}
return results, y_predIn [23]:
results = []In [24]:
lin_model = LinearRegression()
lin_res, y_pred_lin = evaluate_model(lin_model, X_train, X_test, y_train, y_test, "Linear Regression")
results.append(lin_res)In [25]:
X_testOut [25]:
| Reviews | Size in bytes | Numeric Installs | Category_ART_AND_DESIGN | Category_AUTO_AND_VEHICLES | Category_BEAUTY | Category_BOOKS_AND_REFERENCE | Category_BUSINESS | Category_COMICS | Category_COMMUNICATION | ... | Category_GAME | Category_HEALTH_AND_FITNESS | Category_HOUSE_AND_HOME | Category_LIBRARIES_AND_DEMO | Category_LIFESTYLE | Content Rating_Adults only 18+ | Content Rating_Everyone | Content Rating_Everyone 10+ | Content Rating_Mature 17+ | Content Rating_Teen | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 303 | 70782 | 52428800.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 805 | 132014 | 26214400.0 | 10000000 | False | False | False | False | False | False | True | ... | False | False | False | False | False | False | True | False | False | False |
| 352 | 58 | 15728640.0 | 10000 | False | False | False | False | False | False | False | ... | False | False | False | True | False | False | True | False | False | False |
| 952 | 1658 | 10171187.2 | 100000 | False | False | False | False | False | False | False | ... | False | False | False | False | True | False | True | False | False | False |
| 514 | 10852 | 18874368.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... |
| 1096 | 120 | 10485760.0 | 500 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | False | False | True | False |
| 551 | 27396 | 61865984.0 | 1000000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | False | False | True | False |
| 660 | 11506 | 15728640.0 | 100000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
| 655 | 1015 | 11534336.0 | 100000 | True | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 473 | 7976 | 46137344.0 | 500000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
333 rows × 26 columns
In [26]:
residual_linear = y_test - y_pred_lin
plt.figure(figsize=(8, 5))
plt.scatter(y_pred_lin, residual_linear, alpha=0.5)
plt.axhline(y=0, color='r', linestyle='--')
plt.title("Part E - Linear Regression: Residual Plot")
plt.xlabel("Predicted Rating")
plt.ylabel("Residual (y - y_hat)")
plt.tight_layout()
plt.show()In [27]:
poly_converter = PolynomialFeatures(degree=2, include_bias=False)
X_train_p = poly_converter.fit_transform(X_train)
X_test_p = poly_converter.transform(X_test)In [28]:
poly_model = LinearRegression()
poly_res, y_pred_poly = evaluate_model(poly_model, X_train_p, X_test_p, y_train, y_test, "Polynomial Regression")
results.append(poly_res)In [29]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_poly, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Polynomial Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [30]:
ridge_model = Ridge(alpha=1.0, solver='lsqr')
ridge_res, y_pred_ridge = evaluate_model(ridge_model, X_train, X_test, y_train, y_test, "Ridge Regression")
results.append(ridge_res)In [31]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_ridge, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Ridge Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [32]:
lasso_model = Lasso(alpha=0.001, max_iter=10000)
lasso_res, y_pred_lasso = evaluate_model(lasso_model, X_train, X_test, y_train, y_test, "Lasso Regression")
results.append(lasso_res)In [33]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_lasso, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Lasso Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [34]:
results_df = pd.DataFrame(results)
results_dfOut [34]:
| Model | MAE | RMSE | R2 | |
|---|---|---|---|---|
| 0 | Linear Regression | 0.280613 | 0.386712 | 0.146514 |
| 1 | Polynomial Regression | 0.295032 | 0.415806 | 0.013258 |
| 2 | Ridge Regression | 0.302947 | 0.416581 | 0.009580 |
| 3 | Lasso Regression | 0.281669 | 0.387097 | 0.144814 |
In [35]:
results_melted = results_df.melt(
id_vars='Model',
value_vars=['MAE', 'RMSE', 'R2'],
var_name='Metric',
value_name='Score'
)
plt.figure(figsize=(9, 5))
sns.barplot(data=results_melted, x='Model', y='Score', hue='Metric')
plt.title("Part E: Regression Model Comparison (MAE, RMSE, R2)")
plt.xticks(rotation=15)
plt.tight_layout()
plt.show()In [36]:
features_to_test = ['Reviews', 'Size in bytes', 'Numeric Installs']
feature_results = []In [37]:
for feature in features_to_test:
X_single = df_encoded[[feature]]
X_train_s, X_test_s, y_train_s, y_test_s = train_test_split(
X_single, y, test_size=0.3, random_state=101
)In [38]:
simple_model = LinearRegression()
simple_model.fit(X_train_s, y_train_s)
y_pred_s = simple_model.predict(X_test_s)In [39]:
feature_results.append({
'Feature': feature,
'MAE': mean_absolute_error(y_test_s, y_pred_s),
'RMSE': np.sqrt(mean_squared_error(y_test_s, y_pred_s)),
'R2': r2_score(y_test_s, y_pred_s)
})In [40]:
feature_results_df = pd.DataFrame(feature_results).sort_values('RMSE')
feature_results_dfOut [40]:
| Feature | MAE | RMSE | R2 | |
|---|---|---|---|---|
| 0 | Numeric Installs | 0.307489 | 0.417903 | 0.003281 |
In [41]:
feature_melted = feature_results_df.melt(
id_vars='Feature',
value_vars=['MAE', 'RMSE', 'R2'],
var_name='Metric',
value_name='Score'
)In [42]:
plt.figure(figsize=(9, 5))
sns.barplot(data=feature_melted, x='Feature', y='Score', hue='Metric')
plt.title("Part F: Single-Feature Comparison (Linear Regression)")
plt.xticks(rotation=15)
plt.tight_layout()
plt.show()In [43]:
best_feature = feature_results_df.iloc[0]['Feature']In [77]:
X_best = df_encoded[[best_feature]]
X_train_b, X_test_b, y_train_b, y_test_b = train_test_split(
X_best, y, test_size=0.3, random_state=101
)In [66]:
best_model = LinearRegression()
best_model.fit(X_train_b, y_train_b)
y_pred_b = best_model.predict(X_test_b)In [76]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test_b, y_pred_b, alpha=0.5)
plt.plot([y_test_b.min(), y_test_b.max()], [y_test_b.min(), y_test_b.max()], 'r--')
plt.title(f"Part F - Best Feature ({best_feature}): Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [47]:
from sklearn.model_selection import cross_val_score
from sklearn.preprocessing import StandardScalerIn [48]:
columns_to_keep_g = ['Category', 'Reviews', 'Content Rating', 'Rating', 'Numeric Installs', 'Size in bytes']
df_g = df_new[columns_to_keep_g]In [49]:
df_g_encoded = pd.get_dummies(df_g, columns=['Category', 'Content Rating'])In [50]:
X_g = df_g_encoded.drop('Size in bytes', axis=1) if 'Size in bytes' in df_g_encoded.columns else df_g_encoded
y_g = df_encoded['Size in bytes'] # Use Size in bytes from df_encodedIn [51]:
X_train_g, X_test_g, y_train_g, y_test_g = train_test_split(X_g, y_g, test_size=0.3, random_state=101)In [52]:
def evaluate_model_cv(model, X_train, X_test, y_train, y_test, name):
"""Evaluate model with both train/test split AND cross-validation"""
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
# Cross-validation scores (5-fold)
cv_scores = cross_val_score(model, X_train, y_train, cv=5, scoring='r2')
results = {
'Model': name,
'MAE (test)': mean_absolute_error(y_test, y_pred),
'RMSE (test)': np.sqrt(mean_squared_error(y_test, y_pred)),
'R2 (test)': r2_score(y_test, y_pred),
'CV R2 (mean)': cv_scores.mean(),
'CV R2 (std)': cv_scores.std()
}
return results, y_pred, cv_scoresIn [53]:
results_g = []In [54]:
lin_model_g = LinearRegression()
lin_res_g, y_pred_lin_g, cv_lin_g = evaluate_model_cv(lin_model_g, X_train_g, X_test_g, y_train_g, y_test_g, "Linear Regression")
results_g.append(lin_res_g)In [55]:
residual_lin_g = y_test_g - y_pred_lin_gIn [56]:
plt.figure(figsize=(8, 5))
plt.scatter(y_pred_lin_g, residual_lin_g, alpha=0.5)
plt.axhline(y=0, color='r', linestyle='--')
plt.title("Part G - Linear Regression: Residual Plot")
plt.xlabel("Predicted Size in Bytes")
plt.ylabel("Residual (y - y_hat)")
plt.tight_layout()
plt.show()In [57]:
poly_converter_g = PolynomialFeatures(degree=2, include_bias=False)
X_train_p_g = poly_converter_g.fit_transform(X_train_g)
X_test_p_g = poly_converter_g.transform(X_test_g)In [58]:
poly_model_g = LinearRegression()
poly_res_g, y_pred_poly_g, cv_poly_g = evaluate_model_cv(poly_model_g, X_train_p_g, X_test_p_g, y_train_g, y_test_g, "Polynomial Regression")
results_g.append(poly_res_g)In [59]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test_g, y_pred_poly_g, alpha=0.5)
plt.plot([y_test_g.min(), y_test_g.max()], [y_test_g.min(), y_test_g.max()], 'r--')
plt.title("Part G - Polynomial Regression: Actual vs Predicted")
plt.xlabel("True Size in Bytes")
plt.ylabel("Predicted Size in Bytes")
plt.tight_layout()
plt.show()In [60]:
ridge_model_g = Ridge(alpha=1.0, solver='lsqr')
ridge_res_g, y_pred_ridge_g, cv_ridge_g = evaluate_model_cv(ridge_model_g, X_train_g, X_test_g, y_train_g, y_test_g, "Ridge Regression")
results_g.append(ridge_res_g)In [61]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test_g, y_pred_ridge_g, alpha=0.5)
plt.plot([y_test_g.min(), y_test_g.max()], [y_test_g.min(), y_test_g.max()], 'r--')
plt.title("Part G - Ridge Regression: Actual vs Predicted")
plt.xlabel("True Size in Bytes")
plt.ylabel("Predicted Size in Bytes")
plt.tight_layout()
plt.show()In [62]:
results_df_g = pd.DataFrame(results_g)
results_df_gOut [62]:
| Model | MAE (test) | RMSE (test) | R2 (test) | CV R2 (mean) | CV R2 (std) | |
|---|---|---|---|---|---|---|
| 0 | Linear Regression | 1.458116e+07 | 1.906474e+07 | 0.292880 | 0.245829 | 0.067041 |
| 1 | Polynomial Regression | 1.708302e+07 | 2.663426e+07 | -0.380108 | -0.452607 | 0.900419 |
| 2 | Ridge Regression | 1.629031e+07 | 2.162718e+07 | 0.090021 | 0.090846 | 0.056075 |
In [63]:
results_melted_g = results_df_g.melt(
id_vars='Model',
value_vars=['MAE (test)', 'RMSE (test)', 'R2 (test)', 'CV R2 (mean)'],
var_name='Metric',
value_name='Score'
)In [64]:
plt.figure(figsize=(10, 6))
sns.barplot(data=results_melted_g, x='Model', y='Score', hue='Metric')
plt.title("Part G: Regression Model Comparison for Size in Bytes (with Cross-Validation)")
plt.xticks(rotation=15)
plt.tight_layout()
plt.show()