mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-07 20:10:20 +00:00
152 KiB
152 KiB
In [371]:
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as snsIn [372]:
df = pd.read_csv('data/googleplaystore_new.csv')In [373]:
df = df.dropna()
df = df.drop_duplicates()In [374]:
df.head(10)Out [374]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up |
In [375]:
# Getting the column for size, checking if it's ending with an M or a k and converting the Mb to kb with 1024* and then again from kb to just b by another 1024*
def parse_size(size_str):
if isinstance(size_str, str):
if size_str.endswith('M'):
return float(size_str[:-1]) * 1024 * 1024
elif size_str.endswith('k'):
return float(size_str[:-1]) * 1024
return size_strIn [376]:
df['Size in bytes'] = df['Size'].apply(parse_size)In [377]:
df[['Size', 'Size in bytes']].head(10)Out [377]:
| Size | Size in bytes | |
|---|---|---|
| 0 | 11k | 11264.0 |
| 1 | 21M | 22020096.0 |
| 2 | 41k | 41984.0 |
| 3 | 292k | 299008.0 |
| 4 | 3.0M | 3145728.0 |
| 5 | 2.5M | 2621440.0 |
| 6 | 8.2M | 8598323.2 |
| 7 | 695k | 711680.0 |
| 8 | 2.3M | 2411724.8 |
| 9 | 1.4M | 1468006.4 |
In [378]:
print(11*1024) # Just checking the kb and mb conversion happened properly, by checking 2 of the values manually.11264
In [379]:
print((21*1024)*1024) # Ideally this is the same as doing the calculation without the ( ) but just incase!22020096
In [380]:
print(1.4*1024*1024) # This one seemed odd... Checked on an online converted to double-check but turns out the 0.4 bytes shows up there too.1468006.4
In [381]:
df.head(10)Out [381]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 |
In [382]:
df['Numeric Installs'] = df['Installs'].str.replace('+', '').str.replace(',', '').astype(int)In [383]:
df.head(10)Out [383]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | Numeric Installs | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 | 1000000 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 | 1000000 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 | 5000000 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 | 1000000 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 | 1000000 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 | 1000 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 | 100000 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 | 10000000 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 | 100 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 | 1000000 |
In [384]:
df.to_csv('data/googleplaystore_new_new.csv', index=False)In [385]:
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression, Ridge
from sklearn.preprocessing import PolynomialFeatures
from sklearn.metrics import mean_absolute_error, mean_squared_errorIn [386]:
df_new = pd.read_csv('data/googleplaystore_new_new.csv')In [387]:
columns_to_keep = ['Category', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Rating']
df_min = df_new[columns_to_keep]In [388]:
df_min.head(20)Out [388]:
| Category | Reviews | Content Rating | Size in bytes | Numeric Installs | Rating | |
|---|---|---|---|---|---|---|
| 0 | LIBRARIES_AND_DEMO | 20145 | Everyone | 11264.0 | 1000000 | 4.1 |
| 1 | BUSINESS | 46353 | Everyone | 22020096.0 | 1000000 | 4.3 |
| 2 | LIBRARIES_AND_DEMO | 58055 | Everyone | 41984.0 | 5000000 | 3.9 |
| 3 | LIBRARIES_AND_DEMO | 7750 | Everyone | 299008.0 | 1000000 | 3.8 |
| 4 | EDUCATION | 1619 | Everyone | 3145728.0 | 1000000 | 4.4 |
| 5 | LIBRARIES_AND_DEMO | 26 | Everyone | 2621440.0 | 1000 | 5.0 |
| 6 | BEAUTY | 473 | Mature 17+ | 8598323.2 | 100000 | 4.5 |
| 7 | COMMUNICATION | 124346 | Everyone | 711680.0 | 10000000 | 4.2 |
| 8 | EVENTS | 16 | Everyone | 2411724.8 | 100 | 5.0 |
| 9 | FINANCE | 13868 | Everyone | 1468006.4 | 1000000 | 4.1 |
| 10 | COMMUNICATION | 32254 | Everyone | 5767168.0 | 1000000 | 4.4 |
| 11 | COMMUNICATION | 125232 | Everyone | 2831155.2 | 10000000 | 4.2 |
| 12 | EDUCATION | 430 | Everyone | 538624.0 | 10000 | 4.0 |
| 13 | EDUCATION | 275 | Everyone | 2411724.8 | 50000 | 4.0 |
| 14 | BOOKS_AND_REFERENCE | 1778 | Mature 17+ | 5138022.4 | 500000 | 3.9 |
| 15 | BUSINESS | 2287 | Everyone | 1572864.0 | 1000000 | 4.4 |
| 16 | LIBRARIES_AND_DEMO | 126862 | Everyone | 638976.0 | 10000000 | 3.5 |
| 17 | COMMUNICATION | 255 | Everyone | 1677721.6 | 10000 | 4.1 |
| 18 | LIFESTYLE | 360 | Everyone | 4823449.6 | 10000 | 4.1 |
| 19 | EDUCATION | 656 | Everyone | 569344.0 | 10000 | 4.3 |
In [389]:
df_encoded = pd.get_dummies(df_min, columns=['Category', 'Content Rating'])In [390]:
X = df_encoded.drop('Rating', axis=1)
y = df_encoded['Rating']In [391]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)In [392]:
linear_model = LinearRegression()
linear_model.fit(X_train, y_train)
y_hat_linear = linear_model.predict(X_test)In [401]:
X_testOut [401]:
| Reviews | Size in bytes | Numeric Installs | Category_ART_AND_DESIGN | Category_AUTO_AND_VEHICLES | Category_BEAUTY | Category_BOOKS_AND_REFERENCE | Category_BUSINESS | Category_COMICS | Category_COMMUNICATION | ... | Category_GAME | Category_HEALTH_AND_FITNESS | Category_HOUSE_AND_HOME | Category_LIBRARIES_AND_DEMO | Category_LIFESTYLE | Content Rating_Adults only 18+ | Content Rating_Everyone | Content Rating_Everyone 10+ | Content Rating_Mature 17+ | Content Rating_Teen | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 303 | 70782 | 52428800.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 805 | 132014 | 26214400.0 | 10000000 | False | False | False | False | False | False | True | ... | False | False | False | False | False | False | True | False | False | False |
| 352 | 58 | 15728640.0 | 10000 | False | False | False | False | False | False | False | ... | False | False | False | True | False | False | True | False | False | False |
| 952 | 1658 | 10171187.2 | 100000 | False | False | False | False | False | False | False | ... | False | False | False | False | True | False | True | False | False | False |
| 514 | 10852 | 18874368.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... |
| 1096 | 120 | 10485760.0 | 500 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | False | False | True | False |
| 551 | 27396 | 61865984.0 | 1000000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | False | False | True | False |
| 660 | 11506 | 15728640.0 | 100000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
| 655 | 1015 | 11534336.0 | 100000 | True | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 473 | 7976 | 46137344.0 | 500000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
333 rows × 26 columns
In [393]:
print(f"Mean absolute error = {mean_absolute_error(y_test, y_hat_linear)}")
print(f"Root mean squared error = {np.sqrt(mean_squared_error(y_test, y_hat_linear))}")Mean absolute error = 0.28061256731153783 Root mean squared error = 0.38671187333256235
In [394]:
residual_error = y_test - y_hat_linear
plt.figure(figsize=(8,5))
plt.scatter(y_test, residual_error, alpha=0.5)
plt.axhline(y=0, color='r', linestyle='--') # Adds a red line at 0 error for reference
plt.title("Residual Error Plot (Linear Regression)")
plt.xlabel("True Rating Values (y)")
plt.ylabel("Residual Error (y - y_hat)")
plt.show()In [395]:
poly_converter = PolynomialFeatures(degree=2)
X_poly = poly_converter.fit_transform(X)In [396]:
X_train_p, X_test_p, y_train_p, y_test_p = train_test_split(X_poly, y, test_size=0.3, random_state=101)In [397]:
poly_model = LinearRegression()
poly_model.fit(X_train_p, y_train_p)
y_hat_poly = poly_model.predict(X_test_p)In [398]:
print(f"Mean absolute error = {mean_absolute_error(y_test_p, y_hat_poly)}")
print(f"Root mean squared error = {np.sqrt(mean_squared_error(y_test_p, y_hat_poly))}")
Mean absolute error = 0.29501714163585174 Root mean squared error = 0.4157783198958284
In [404]:
features_to_test = ['Reviews', 'Size in bytes', 'Numeric Installs']In [407]:
for feature in features_to_test:
X_single = df_encoded[[feature]]In [408]:
X_train_s, X_test_s, y_train_s, y_test_s = train_test_split(X_single, y, test_size=0.3, random_state=101)In [409]:
simple_model = LinearRegression()
simple_model.fit(X_train_s, y_train_s)Out [409]:
LinearRegression()In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
Parameters
In [410]:
y_pred_s = simple_model.predict(X_test_s)
rmse = np.sqrt(mean_squared_error(y_test_s, y_pred_s))In [411]:
print(f"RMSE using ONLY '{feature}': {rmse}")RMSE using ONLY 'Numeric Installs': 0.4179032096419041
In [ ]: