mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-07 20:10:20 +00:00
95 KiB
95 KiB
In [19]:
import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LogisticRegression
from sklearn.neighbors import KNeighborsClassifier
from sklearn.metrics import accuracy_score, f1_score, confusion_matrix, ConfusionMatrixDisplay
import matplotlib.pyplot as pltIn [20]:
df = pd.read_csv('data/googleplaystore_new_new.csv')In [21]:
columns_to_keep = ['Rating', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Category']
df_part_a = df[columns_to_keep]In [22]:
X_raw = df_part_a.drop('Category', axis=1)
y = df_part_a['Category']In [23]:
X = pd.get_dummies(X_raw, columns=['Content Rating'])In [24]:
X_temp, X_test, y_temp, y_test = train_test_split(X, y, test_size=0.2, random_state=101)In [25]:
X_train, X_val, y_train, y_val = train_test_split(X_temp, y_temp, test_size=0.25, random_state=101)In [26]:
scaler = StandardScaler()
scaled_X_train = scaler.fit_transform(X_train)
scaled_X_val = scaler.transform(X_val)
scaled_X_test = scaler.transform(X_test)In [27]:
log_model = LogisticRegression(max_iter=1000)
log_model.fit(scaled_X_train, y_train)Out [27]:
LogisticRegression(max_iter=1000)In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
Parameters
In [28]:
y_val_pred_log = log_model.predict(scaled_X_val)In [29]:
log_val_f1 = f1_score(y_val, y_val_pred_log, average='weighted')
print(f"Validation Accuracy: {accuracy_score(y_val, y_val_pred_log):.4f}")
print(f"Validation F1-Score: {log_val_f1:.4f}")Validation Accuracy: 0.3333 Validation F1-Score: 0.3069
In [30]:
best_k = 1
best_f1 = 0In [31]:
for k in range(1, 15, 2):
knn_temp = KNeighborsClassifier(n_neighbors=k)
knn_temp.fit(scaled_X_train, y_train)
y_val_pred_knn = knn_temp.predict(scaled_X_val)
current_f1 = f1_score(y_val, y_val_pred_knn, average='weighted')
print(f"KNN (K={k}) Validation F1-Score: {current_f1:.4f}")
if current_f1 > best_f1:
best_f1 = current_f1
best_k = kKNN (K=1) Validation F1-Score: 0.3201 KNN (K=3) Validation F1-Score: 0.3245 KNN (K=5) Validation F1-Score: 0.2939 KNN (K=7) Validation F1-Score: 0.2916 KNN (K=9) Validation F1-Score: 0.3120 KNN (K=11) Validation F1-Score: 0.3097 KNN (K=13) Validation F1-Score: 0.3181
In [32]:
knn_final = KNeighborsClassifier(n_neighbors=best_k)
knn_final.fit(scaled_X_train, y_train)Out [32]:
KNeighborsClassifier(n_neighbors=3)In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
Parameters
In [32]:
In [32]: