mirror of
https://github.com/mudabbir-ahmad/UNI-PROG3-CW2-MLWP.git
synced 2026-10-07 20:10:20 +00:00
327 KiB
327 KiB
In [148]:
import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LogisticRegression
from sklearn.neighbors import KNeighborsClassifier
from sklearn.metrics import accuracy_score, f1_score, classification_report, confusion_matrix, ConfusionMatrixDisplay
import matplotlib.pyplot as pltIn [149]:
df = pd.read_csv('data/googleplaystore_new_new.csv')In [150]:
columns_to_keep = ['Rating', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Category']
df_part_a = df[columns_to_keep]In [151]:
X_raw = df_part_a.drop('Category', axis=1)
y = df_part_a['Category']In [152]:
X = pd.get_dummies(X_raw, columns=['Content Rating'])In [153]:
X_temp, X_test, y_temp, y_test = train_test_split(X, y, test_size=0.2, random_state=101)In [154]:
X_train, X_val, y_train, y_val = train_test_split(X_temp, y_temp, test_size=0.25, random_state=101)In [155]:
scaler = StandardScaler()
scaled_X_train = scaler.fit_transform(X_train)
scaled_X_val = scaler.transform(X_val)
scaled_X_test = scaler.transform(X_test)In [156]:
def eval_metrics(y_true, y_pred, label="Model"):
acc = accuracy_score(y_true, y_pred)
f1_w = f1_score(y_true, y_pred, average='weighted')
print(f"{label} -> Accuracy: {acc:.4f} | Weighted F1: {f1_w:.4f}")
return acc, f1_wIn [156]:
In [157]:
log_model = LogisticRegression(max_iter=1000, random_state=101)
log_model.fit(scaled_X_train, y_train)Out [157]:
LogisticRegression(max_iter=1000, random_state=101)In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
Parameters
In [158]:
y_val_pred_log = log_model.predict(scaled_X_val)
log_val_acc, log_val_f1 = eval_metrics(y_val, y_val_pred_log, "LogReg (Validation)")LogReg (Validation) -> Accuracy: 0.3333 | Weighted F1: 0.3069
In [159]:
y_test_pred_log = log_model.predict(scaled_X_test)
log_test_acc, log_test_f1 = eval_metrics(y_test, y_test_pred_log, "LogReg (Test)")LogReg (Test) -> Accuracy: 0.3108 | Weighted F1: 0.2646
In [160]:
log_probs = log_model.predict_proba(scaled_X_test)
log_pred_idx = np.argmax(log_probs, axis=1)
log_pred_conf = log_probs[np.arange(len(log_probs)), log_pred_idx]
log_pred_df = pd.DataFrame({
"True_Label": y_test.reset_index(drop=True),
"Predicted_Label": pd.Series(y_test_pred_log).reset_index(drop=True),
"Predicted_Confidence": log_pred_conf
})
log_pred_df["Correct"] = log_pred_df["True_Label"] == log_pred_df["Predicted_Label"]
display(log_pred_df.head(15))| True_Label | Predicted_Label | Predicted_Confidence | Correct | |
|---|---|---|---|---|
| 0 | FINANCE | FINANCE | 0.201206 | True |
| 1 | COMMUNICATION | FINANCE | 0.127063 | False |
| 2 | LIBRARIES_AND_DEMO | EDUCATION | 0.134790 | False |
| 3 | LIFESTYLE | EDUCATION | 0.146064 | False |
| 4 | EDUCATION | EDUCATION | 0.136238 | True |
| 5 | ART_AND_DESIGN | LIFESTYLE | 0.143387 | False |
| 6 | BEAUTY | EDUCATION | 0.125436 | False |
| 7 | DATING | DATING | 0.865158 | True |
| 8 | HOUSE_AND_HOME | LIFESTYLE | 0.120546 | False |
| 9 | FINANCE | HEALTH_AND_FITNESS | 0.274936 | False |
| 10 | HEALTH_AND_FITNESS | GAME | 0.407868 | False |
| 11 | EDUCATION | EDUCATION | 0.143047 | True |
| 12 | HEALTH_AND_FITNESS | HEALTH_AND_FITNESS | 0.198159 | True |
| 13 | FINANCE | EDUCATION | 0.136472 | False |
| 14 | FINANCE | FINANCE | 0.180731 | True |
In [161]:
best_k = 1
best_f1 = 0.0In [162]:
test_error_rates = []
k_candidates = list(range(1, 30)) # 1..29
for k in k_candidates:
knn_model = KNeighborsClassifier(n_neighbors=k)
knn_model.fit(scaled_X_train, y_train)
y_pred_test_k = knn_model.predict(scaled_X_test)
# Elbow metric based on test error (as requested)
test_error = 1 - accuracy_score(y_test, y_pred_test_k)
test_error_rates.append(test_error)
# Keep track of best k by validation weighted F1 (good model selection practice)
y_pred_val_k = knn_model.predict(scaled_X_val)
current_f1 = f1_score(y_val, y_pred_val_k, average='weighted')
if current_f1 > best_f1:
best_f1 = current_f1
best_k = kIn [163]:
knn_probs = knn_final.predict_proba(scaled_X_test)
knn_pred_idx = np.argmax(knn_probs, axis=1)
knn_pred_conf = knn_probs[np.arange(len(knn_probs)), knn_pred_idx]
knn_pred_df = pd.DataFrame({
"True_Label": y_test.reset_index(drop=True),
"Predicted_Label": pd.Series(y_test_pred_knn).reset_index(drop=True),
"Predicted_Confidence": knn_pred_conf
})
knn_pred_df["Correct"] = knn_pred_df["True_Label"] == knn_pred_df["Predicted_Label"]
display(knn_pred_df.head(15))| True_Label | Predicted_Label | Predicted_Confidence | Correct | |
|---|---|---|---|---|
| 0 | FINANCE | BEAUTY | 0.333333 | False |
| 1 | COMMUNICATION | ART_AND_DESIGN | 0.333333 | False |
| 2 | LIBRARIES_AND_DEMO | COMMUNICATION | 0.333333 | False |
| 3 | LIFESTYLE | BUSINESS | 0.333333 | False |
| 4 | EDUCATION | GAME | 0.666667 | False |
| 5 | ART_AND_DESIGN | AUTO_AND_VEHICLES | 0.333333 | False |
| 6 | BEAUTY | BEAUTY | 0.333333 | True |
| 7 | DATING | DATING | 1.000000 | True |
| 8 | HOUSE_AND_HOME | AUTO_AND_VEHICLES | 0.333333 | False |
| 9 | FINANCE | HEALTH_AND_FITNESS | 0.666667 | False |
| 10 | HEALTH_AND_FITNESS | FINANCE | 0.333333 | False |
| 11 | EDUCATION | HEALTH_AND_FITNESS | 0.666667 | False |
| 12 | HEALTH_AND_FITNESS | EDUCATION | 0.666667 | False |
| 13 | FINANCE | ART_AND_DESIGN | 0.333333 | False |
| 14 | FINANCE | FINANCE | 1.000000 | True |
In [164]:
best_acc_idx = int(np.argmax(knn_val_acc_scores))
best_f1_idx = int(np.argmax(knn_val_f1_scores))
plt.figure(figsize=(9, 5))
plt.plot(k_values, knn_val_acc_scores, marker='o', linewidth=2, label='Validation Accuracy')
plt.plot(k_values, knn_val_f1_scores, marker='s', linewidth=2, label='Validation Weighted F1')
# Highlight best metric points
plt.scatter(
k_values[best_acc_idx],
knn_val_acc_scores[best_acc_idx],
s=120,
color='green',
zorder=5,
label=f'Best Accuracy (k={k_values[best_acc_idx]})'
)
plt.scatter(
k_values[best_f1_idx],
knn_val_f1_scores[best_f1_idx],
s=120,
color='red',
zorder=5,
label=f'Best Weighted F1 (k={k_values[best_f1_idx]})'
)
plt.axvline(best_k, color='red', linestyle='--', alpha=0.5, label=f'Selected k = {best_k}')
plt.title('KNN Validation Performance vs K (Dynamic)')
plt.xlabel('Number of Neighbors (k)')
plt.ylabel('Score')
plt.xticks(k_values)
plt.grid(alpha=0.3)
plt.legend()
plt.show()In [165]:
plt.figure(figsize=(10, 6), dpi=120)
plt.plot(k_candidates, test_error_rates, label='test error')
plt.xlabel('k value')
plt.title('Elbow method for KNN')
plt.legend()
plt.grid(alpha=0.3)
plt.show()In [166]:
knn_final = KNeighborsClassifier(n_neighbors=best_k)
knn_final.fit(scaled_X_train, y_train)Out [166]:
KNeighborsClassifier(n_neighbors=3)In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
Parameters
In [167]:
y_val_pred_knn = knn_final.predict(scaled_X_val)
knn_val_acc, knn_val_f1 = eval_metrics(y_val, y_val_pred_knn, f"KNN k={best_k} (Validation)")KNN k=3 (Validation) -> Accuracy: 0.3288 | Weighted F1: 0.3245
In [168]:
y_test_pred_knn = knn_final.predict(scaled_X_test)
knn_test_acc, knn_test_f1 = eval_metrics(y_test, y_test_pred_knn, f"KNN k={best_k} (Test)")KNN k=3 (Test) -> Accuracy: 0.2523 | Weighted F1: 0.2288
In [169]:
results_df = pd.DataFrame([
{"Model": "Logistic Regression", "Split": "Validation", "Accuracy": log_val_acc, "Weighted F1": log_val_f1},
{"Model": "Logistic Regression", "Split": "Test", "Accuracy": log_test_acc, "Weighted F1": log_test_f1},
{"Model": f"KNN (k={best_k})", "Split": "Validation", "Accuracy": knn_val_acc, "Weighted F1": knn_val_f1},
{"Model": f"KNN (k={best_k})", "Split": "Test", "Accuracy": knn_test_acc, "Weighted F1": knn_test_f1},
])
results_dfOut [169]:
| Model | Split | Accuracy | Weighted F1 | |
|---|---|---|---|---|
| 0 | Logistic Regression | Validation | 0.333333 | 0.306923 |
| 1 | Logistic Regression | Test | 0.310811 | 0.264609 |
| 2 | KNN (k=3) | Validation | 0.328829 | 0.324541 |
| 3 | KNN (k=3) | Test | 0.252252 | 0.228784 |
In [170]:
top_n = 10
top_classes = y_test.value_counts().head(top_n).index.tolist()
mask = y_test.isin(top_classes)
y_test_top = y_test[mask]
y_pred_top = pd.Series(y_test_pred_knn, index=y_test.index)[mask]
cm_top = confusion_matrix(y_test_top, y_pred_top, labels=top_classes)
disp_top = ConfusionMatrixDisplay(confusion_matrix=cm_top, display_labels=top_classes)
fig, ax = plt.subplots(figsize=(8, 6))
disp_top.plot(
ax=ax,
cmap='viridis',
xticks_rotation=45,
values_format='d',
colorbar=False
)
ax.set_title(f'KNN Confusion Matrix (Top {top_n} Classes, k={best_k})')
plt.tight_layout()
plt.show()In [170]:
In [170]: