mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-08 04:10:20 +00:00
116 KiB
116 KiB
In [371]:
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as snsIn [372]:
df = pd.read_csv('data/googleplaystore_new.csv')In [373]:
df = df.dropna()
df = df.drop_duplicates()In [374]:
df.head(10)Out [374]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up |
In [375]:
# Getting the column for size, checking if it's ending with an M or a k and converting the Mb to kb with 1024* and then again from kb to just b by another 1024*
def parse_size(size_str):
if isinstance(size_str, str):
if size_str.endswith('M'):
return float(size_str[:-1]) * 1024 * 1024
elif size_str.endswith('k'):
return float(size_str[:-1]) * 1024
return size_strIn [376]:
df['Size in bytes'] = df['Size'].apply(parse_size)In [377]:
df[['Size', 'Size in bytes']].head(10)Out [377]:
| Size | Size in bytes | |
|---|---|---|
| 0 | 11k | 11264.0 |
| 1 | 21M | 22020096.0 |
| 2 | 41k | 41984.0 |
| 3 | 292k | 299008.0 |
| 4 | 3.0M | 3145728.0 |
| 5 | 2.5M | 2621440.0 |
| 6 | 8.2M | 8598323.2 |
| 7 | 695k | 711680.0 |
| 8 | 2.3M | 2411724.8 |
| 9 | 1.4M | 1468006.4 |
In [378]:
print(11*1024) # Just checking the kb and mb conversion happened properly, by checking 2 of the values manually.11264
In [379]:
print((21*1024)*1024) # Ideally this is the same as doing the calculation without the ( ) but just incase!22020096
In [380]:
print(1.4*1024*1024) # This one seemed odd... Checked on an online converted to double-check but turns out the 0.4 bytes shows up there too.1468006.4
In [381]:
df.head(10)Out [381]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 |
In [382]:
df['Numeric Installs'] = df['Installs'].str.replace('+', '').str.replace(',', '').astype(int)In [383]:
df.head(10)Out [383]:
| App | Category | Rating | Reviews | Size | Installs | Type | Price | Content Rating | Genres | Android Ver | Size in bytes | Numeric Installs | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 0 | Market Update Helper | LIBRARIES_AND_DEMO | 4.1 | 20145 | 11k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 11264.0 | 1000000 |
| 1 | SuperLivePro | BUSINESS | 4.3 | 46353 | 21M | 1,000,000+ | Free | 0 | Everyone | Business | 1.5 and up | 22020096.0 | 1000000 |
| 2 | Wifi Connect Library | LIBRARIES_AND_DEMO | 3.9 | 58055 | 41k | 5,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.5 and up | 41984.0 | 5000000 |
| 3 | Apk Installer | LIBRARIES_AND_DEMO | 3.8 | 7750 | 292k | 1,000,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 299008.0 | 1000000 |
| 4 | English speaking texts | EDUCATION | 4.4 | 1619 | 3.0M | 1,000,000+ | Free | 0 | Everyone | Education | 1.6 and up | 3145728.0 | 1000000 |
| 5 | Eternal life | LIBRARIES_AND_DEMO | 5.0 | 26 | 2.5M | 1,000+ | Free | 0 | Everyone | Libraries & Demo | 1.6 and up | 2621440.0 | 1000 |
| 6 | Dresses Ideas & Fashions +3000 | BEAUTY | 4.5 | 473 | 8.2M | 100,000+ | Free | 0 | Mature 17+ | Beauty | 1.6 and up | 8598323.2 | 100000 |
| 7 | GO Notifier | COMMUNICATION | 4.2 | 124346 | 695k | 10,000,000+ | Free | 0 | Everyone | Communication | 2.0 and up | 711680.0 | 10000000 |
| 8 | Prosperity | EVENTS | 5.0 | 16 | 2.3M | 100+ | Free | 0 | Everyone | Events | 2.0 and up | 2411724.8 | 100 |
| 9 | NSE Mobile Trading | FINANCE | 4.1 | 13868 | 1.4M | 1,000,000+ | Free | 0 | Everyone | Finance | 2.1 and up | 1468006.4 | 1000000 |
In [384]:
df.to_csv('data/googleplaystore_new_new.csv', index=False)In [385]:
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression, Ridge
from sklearn.preprocessing import PolynomialFeatures
from sklearn.metrics import mean_absolute_error, mean_squared_errorIn [386]:
df_new = pd.read_csv('data/googleplaystore_new_new.csv')In [387]:
columns_to_keep = ['Category', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Rating']
df_min = df_new[columns_to_keep]In [388]:
df_min.head(20)Out [388]:
| Category | Reviews | Content Rating | Size in bytes | Numeric Installs | Rating | |
|---|---|---|---|---|---|---|
| 0 | LIBRARIES_AND_DEMO | 20145 | Everyone | 11264.0 | 1000000 | 4.1 |
| 1 | BUSINESS | 46353 | Everyone | 22020096.0 | 1000000 | 4.3 |
| 2 | LIBRARIES_AND_DEMO | 58055 | Everyone | 41984.0 | 5000000 | 3.9 |
| 3 | LIBRARIES_AND_DEMO | 7750 | Everyone | 299008.0 | 1000000 | 3.8 |
| 4 | EDUCATION | 1619 | Everyone | 3145728.0 | 1000000 | 4.4 |
| 5 | LIBRARIES_AND_DEMO | 26 | Everyone | 2621440.0 | 1000 | 5.0 |
| 6 | BEAUTY | 473 | Mature 17+ | 8598323.2 | 100000 | 4.5 |
| 7 | COMMUNICATION | 124346 | Everyone | 711680.0 | 10000000 | 4.2 |
| 8 | EVENTS | 16 | Everyone | 2411724.8 | 100 | 5.0 |
| 9 | FINANCE | 13868 | Everyone | 1468006.4 | 1000000 | 4.1 |
| 10 | COMMUNICATION | 32254 | Everyone | 5767168.0 | 1000000 | 4.4 |
| 11 | COMMUNICATION | 125232 | Everyone | 2831155.2 | 10000000 | 4.2 |
| 12 | EDUCATION | 430 | Everyone | 538624.0 | 10000 | 4.0 |
| 13 | EDUCATION | 275 | Everyone | 2411724.8 | 50000 | 4.0 |
| 14 | BOOKS_AND_REFERENCE | 1778 | Mature 17+ | 5138022.4 | 500000 | 3.9 |
| 15 | BUSINESS | 2287 | Everyone | 1572864.0 | 1000000 | 4.4 |
| 16 | LIBRARIES_AND_DEMO | 126862 | Everyone | 638976.0 | 10000000 | 3.5 |
| 17 | COMMUNICATION | 255 | Everyone | 1677721.6 | 10000 | 4.1 |
| 18 | LIFESTYLE | 360 | Everyone | 4823449.6 | 10000 | 4.1 |
| 19 | EDUCATION | 656 | Everyone | 569344.0 | 10000 | 4.3 |
In [389]:
df_encoded = pd.get_dummies(df_min, columns=['Category', 'Content Rating'])In [390]:
X = df_encoded.drop('Rating', axis=1)
y = df_encoded['Rating']In [391]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)In [392]:
linear_model = LinearRegression()
linear_model.fit(X_train, y_train)
y_hat_linear = linear_model.predict(X_test)In [401]:
X_testOut [401]:
| Reviews | Size in bytes | Numeric Installs | Category_ART_AND_DESIGN | Category_AUTO_AND_VEHICLES | Category_BEAUTY | Category_BOOKS_AND_REFERENCE | Category_BUSINESS | Category_COMICS | Category_COMMUNICATION | ... | Category_GAME | Category_HEALTH_AND_FITNESS | Category_HOUSE_AND_HOME | Category_LIBRARIES_AND_DEMO | Category_LIFESTYLE | Content Rating_Adults only 18+ | Content Rating_Everyone | Content Rating_Everyone 10+ | Content Rating_Mature 17+ | Content Rating_Teen | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| 303 | 70782 | 52428800.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 805 | 132014 | 26214400.0 | 10000000 | False | False | False | False | False | False | True | ... | False | False | False | False | False | False | True | False | False | False |
| 352 | 58 | 15728640.0 | 10000 | False | False | False | False | False | False | False | ... | False | False | False | True | False | False | True | False | False | False |
| 952 | 1658 | 10171187.2 | 100000 | False | False | False | False | False | False | False | ... | False | False | False | False | True | False | True | False | False | False |
| 514 | 10852 | 18874368.0 | 1000000 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... | ... |
| 1096 | 120 | 10485760.0 | 500 | False | False | False | False | False | False | False | ... | False | False | False | False | False | False | False | False | True | False |
| 551 | 27396 | 61865984.0 | 1000000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | False | False | True | False |
| 660 | 11506 | 15728640.0 | 100000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
| 655 | 1015 | 11534336.0 | 100000 | True | False | False | False | False | False | False | ... | False | False | False | False | False | False | True | False | False | False |
| 473 | 7976 | 46137344.0 | 500000 | False | False | False | False | False | False | False | ... | False | True | False | False | False | False | True | False | False | False |
333 rows × 26 columns
In [393]:
print(f"Mean absolute error = {mean_absolute_error(y_test, y_hat_linear)}")
print(f"Root mean squared error = {np.sqrt(mean_squared_error(y_test, y_hat_linear))}")Mean absolute error = 0.28061256731153783 Root mean squared error = 0.38671187333256235
In [394]:
residual_error = y_test - y_hat_linear
plt.figure(figsize=(8,5))
plt.scatter(y_test, residual_error, alpha=0.5)
plt.axhline(y=0, color='r', linestyle='--') # Adds a red line at 0 error for reference
plt.title("Residual Error Plot (Linear Regression)")
plt.xlabel("True Rating Values (y)")
plt.ylabel("Residual Error (y - y_hat)")
plt.show()In [395]:
poly_converter = PolynomialFeatures(degree=2)
X_poly = poly_converter.fit_transform(X)In [396]:
X_train_p, X_test_p, y_train_p, y_test_p = train_test_split(X_poly, y, test_size=0.3, random_state=101)In [397]:
poly_model = LinearRegression()
poly_model.fit(X_train_p, y_train_p)
y_hat_poly = poly_model.predict(X_test_p)In [398]:
print(f"Mean absolute error = {mean_absolute_error(y_test_p, y_hat_poly)}")
print(f"Root mean squared error = {np.sqrt(mean_squared_error(y_test_p, y_hat_poly))}")
Mean absolute error = 0.29501714163585174 Root mean squared error = 0.4157783198958284
In [ ]: