mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-07 20:10:20 +00:00
360 KiB
360 KiB
In [784]:
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as snsIn [785]:
df = pd.read_csv('data/googleplaystore_new.csv')In [786]:
df = df.dropna()
df = df.drop_duplicates()In [787]:
df.head(10)Out [787]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up |
In [788]:
# Getting the column for size, checking if it's ending with an M or a k and converting the Mb to kb with 1024* and then again from kb to just b by another 1024*
def parse_size(size_str):
if isinstance(size_str, str):
if size_str.endswith('M'):
return float(size_str[:-1]) * 1024 * 1024
elif size_str.endswith('k'):
return float(size_str[:-1]) * 1024
return size_strIn [789]:
df['Size in bytes'] = df['Size'].apply(parse_size)In [790]:
df[['Size', 'Size in bytes']].head(10)Out [790]:
| Size | Size in bytes | |
|---|---|---|
| 0 | 11k | 11264.0 |
| 1 | 21M | 22020096.0 |
| 2 | 41k | 41984.0 |
| 3 | 292k | 299008.0 |
| 4 | 3.0M | 3145728.0 |
| 5 | 2.5M | 2621440.0 |
| 6 | 8.2M | 8598323.2 |
| 7 | 695k | 711680.0 |
| 8 | 2.3M | 2411724.8 |
| 9 | 1.4M | 1468006.4 |
In [791]:
print(11*1024) # Just checking the kb and mb conversion happened properly, by checking 2 of the values manually.11264
In [792]:
print((21*1024)*1024) # Ideally this is the same as doing the calculation without the ( ) but just incase!22020096
In [793]:
print(1.4*1024*1024) # This one seemed odd... Checked on an online converted to double-check but turns out the 0.4 bytes shows up there too.1468006.4
In [794]:
df.head(10)Out [794]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 |
In [795]:
df['Numeric Installs'] = df['Installs'].str.replace('+', '').str.replace(',', '').astype(int)In [796]:
df.head(10)Out [796]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | Numeric Installs | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 | 1000000 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 | 1000000 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 | 5000000 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 | 1000000 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 | 1000000 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 | 1000 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 | 100000 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 | 10000000 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 | 100 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 | 1000000 |
In [797]:
df.to_csv('data/googleplaystore_new_new.csv', index=False)In [798]:
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression, Ridge, Lasso
from sklearn.preprocessing import PolynomialFeatures
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_scoreIn [799]:
df_new = pd.read_csv('data/googleplaystore_new_new.csv')In [800]:
columns_to_keep = ['Category', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Rating']
df_min = df_new[columns_to_keep]In [801]:
df_min.head(20)Out [801]:
| Category | Reviews | Content Rating | Size in bytes | Numeric Installs | Rating | |
|---|---|---|---|---|---|---|
| 0 | LIBRARIES_AND_DEMO | 20145 | Everyone | 11264.0 | 1000000 | 4.1 |
| 1 | BUSINESS | 46353 | Everyone | 22020096.0 | 1000000 | 4.3 |
| 2 | LIBRARIES_AND_DEMO | 58055 | Everyone | 41984.0 | 5000000 | 3.9 |
| 3 | LIBRARIES_AND_DEMO | 7750 | Everyone | 299008.0 | 1000000 | 3.8 |
| 4 | EDUCATION | 1619 | Everyone | 3145728.0 | 1000000 | 4.4 |
| 5 | LIBRARIES_AND_DEMO | 26 | Everyone | 2621440.0 | 1000 | 5.0 |
| 6 | BEAUTY | 473 | Mature 17+ | 8598323.2 | 100000 | 4.5 |
| 7 | COMMUNICATION | 124346 | Everyone | 711680.0 | 10000000 | 4.2 |
| 8 | EVENTS | 16 | Everyone | 2411724.8 | 100 | 5.0 |
| 9 | FINANCE | 13868 | Everyone | 1468006.4 | 1000000 | 4.1 |
| 10 | COMMUNICATION | 32254 | Everyone | 5767168.0 | 1000000 | 4.4 |
| 11 | COMMUNICATION | 125232 | Everyone | 2831155.2 | 10000000 | 4.2 |
| 12 | EDUCATION | 430 | Everyone | 538624.0 | 10000 | 4.0 |
| 13 | EDUCATION | 275 | Everyone | 2411724.8 | 50000 | 4.0 |
| 14 | BOOKS_AND_REFERENCE | 1778 | Mature 17+ | 5138022.4 | 500000 | 3.9 |
| 15 | BUSINESS | 2287 | Everyone | 1572864.0 | 1000000 | 4.4 |
| 16 | LIBRARIES_AND_DEMO | 126862 | Everyone | 638976.0 | 10000000 | 3.5 |
| 17 | COMMUNICATION | 255 | Everyone | 1677721.6 | 10000 | 4.1 |
| 18 | LIFESTYLE | 360 | Everyone | 4823449.6 | 10000 | 4.1 |
| 19 | EDUCATION | 656 | Everyone | 569344.0 | 10000 | 4.3 |
In [802]:
df_encoded = pd.get_dummies(df_min, columns=['Category', 'Content Rating'])In [803]:
X = df_encoded.drop('Rating', axis=1)
y = df_encoded['Rating']In [804]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)In [805]:
def evaluate_model(model, X_train, X_test, y_train, y_test, name):
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
results = {
'Model': name,
'MAE': mean_absolute_error(y_test, y_pred),
'RMSE': np.sqrt(mean_squared_error(y_test, y_pred)),
'R2': r2_score(y_test, y_pred)
}
return results, y_predIn [806]:
results = []In [807]:
lin_model = LinearRegression()
lin_res, y_pred_lin = evaluate_model(lin_model, X_train, X_test, y_train, y_test, "Linear Regression")
results.append(lin_res)In [808]:
X_testOut [808]:
| Reviews | Size in bytes | Numeric Installs | Category_ART_AND_DESIGN | Category_AUTO_AND_VEHICLES | Category_BEAUTY | Category_BOOKS_AND_REFERENCE | Category_BUSINESS | Category_COMICS | Category_COMMUNICATION | ... | Category_GAME | Category_HEALTH_AND_FITNESS | Category_HOUSE_AND_HOME | Category_LIBRARIES_AND_DEMO | Category_LIFESTYLE | Content Rating_Adults only 18+ | Content Rating_Everyone | Content Rating_Everyone 10+ | Content Rating_Mature 17+ | Content Rating_Teen | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 303 | 70782 | 52428800.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 805 | 132014 | 26214400.0 | 10000000 | False | False | False | False | False | False | True | ... | False | False | False | False | False | False | True | False | False | False |
| 352 | 58 | 15728640.0 | 10000 | False | False | False | False | False | False | False | ... | False | False | False | True | False | False | True | False | False | False |
| 952 | 1658 | 10171187.2 | 100000 | False | False | False | False | False | False | False | ... | False | False | False | False | True | False | True | False | False | False |
| 514 | 10852 | 18874368.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... |
| 1096 | 120 | 10485760.0 | 500 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | False | False | True | False |
| 551 | 27396 | 61865984.0 | 1000000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | False | False | True | False |
| 660 | 11506 | 15728640.0 | 100000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
| 655 | 1015 | 11534336.0 | 100000 | True | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 473 | 7976 | 46137344.0 | 500000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
333 rows × 26 columns
In [809]:
residual_linear = y_test - y_pred_lin
plt.figure(figsize=(8, 5))
plt.scatter(y_pred_lin, residual_linear, alpha=0.5)
plt.axhline(y=0, color='r', linestyle='--')
plt.title("Part E - Linear Regression: Residual Plot")
plt.xlabel("Predicted Rating")
plt.ylabel("Residual (y - y_hat)")
plt.tight_layout()
plt.show()In [810]:
poly_converter = PolynomialFeatures(degree=2, include_bias=False)
X_train_p = poly_converter.fit_transform(X_train)
X_test_p = poly_converter.transform(X_test)In [811]:
poly_model = LinearRegression()
poly_res, y_pred_poly = evaluate_model(poly_model, X_train_p, X_test_p, y_train, y_test, "Polynomial Regression")
results.append(poly_res)In [812]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_poly, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Polynomial Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [813]:
ridge_model = Ridge(alpha=1.0, solver='lsqr')
ridge_res, y_pred_ridge = evaluate_model(ridge_model, X_train, X_test, y_train, y_test, "Ridge Regression")
results.append(ridge_res)In [814]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_ridge, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Ridge Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [815]:
lasso_model = Lasso(alpha=0.001, max_iter=10000)
lasso_res, y_pred_lasso = evaluate_model(lasso_model, X_train, X_test, y_train, y_test, "Lasso Regression")
results.append(lasso_res)In [816]:
plt.figure(figsize=(6, 6))
plt.scatter(y_test, y_pred_lasso, alpha=0.5)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')
plt.title("Part E - Lasso Regression: Actual vs Predicted")
plt.xlabel("True Rating")
plt.ylabel("Predicted Rating")
plt.tight_layout()
plt.show()In [817]:
results_df = pd.DataFrame(results)
results_dfOut [817]:
| Model | MAE | RMSE | R2 | |
|---|---|---|---|---|
| 0 | Linear Regression | 0.280613 | 0.386712 | 0.146514 |
| 1 | Polynomial Regression | 0.295032 | 0.415806 | 0.013258 |
| 2 | Ridge Regression | 0.302947 | 0.416581 | 0.009580 |
| 3 | Lasso Regression | 0.281669 | 0.387097 | 0.144814 |
In [823]:
results_melted = results_df.melt(
id_vars='Model',
value_vars=['MAE', 'RMSE', 'R2'],
var_name='Metric',
value_name='Score'
)
plt.figure(figsize=(9, 5))
sns.barplot(data=results_melted, x='Model', y='Score', hue='Metric')
plt.title("Part E: Regression Model Comparison (MAE, RMSE, R2)")
plt.xticks(rotation=15)
plt.tight_layout()
plt.show()In [818]:
features_to_test = ['Reviews', 'Size in bytes', 'Numeric Installs']In [819]:
for i, NumericInstalls in enumerate(features_to_test, 1):
X_single = df_encoded[[NumericInstalls]]
X_train_s, X_test_s, y_train_s, y_test_s = train_test_split(X_single, y, test_size=0.3, random_state=101)In [820]:
simple_model = LinearRegression()
simple_model.fit(X_train_s, y_train_s)
y_pred_s = simple_model.predict(X_test_s)In [821]:
rmse = np.sqrt(mean_squared_error(y_test_s, y_pred_s))
print(f"RMSE using ONLY '{NumericInstalls}': {rmse:.4f}")RMSE using ONLY 'Numeric Installs': 0.4179
In [822]:
plt.subplot(1, 3, i)
sns.scatterplot(x=X_test_s[NumericInstalls], y=y_test_s, label='True Values')
sns.scatterplot(x=X_test_s[NumericInstalls], y=y_pred_s, label='Predicted Values', alpha=0.5)
sns.regplot(x=X_test_s[NumericInstalls], y=y_test_s, scatter_kws={'alpha':0.3}, line_kws={'color':'orange'})
plt.title(f"True vs Predicted Ratings using '{NumericInstalls}'")
plt.xlabel(NumericInstalls)
plt.ylabel('Rating')
plt.legend()
plt.show()In [822]: