From 790aeb7346ed0d3544e6589d0adcd25141d8d393 Mon Sep 17 00:00:00 2001 From: mudabbir-ahmad Date: Sat, 25 Apr 2026 21:52:02 +0100 Subject: [PATCH] Part A Q2 --- Q2.ipynb | 2018 +++++++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 2006 insertions(+), 12 deletions(-) diff --git a/Q2.ipynb b/Q2.ipynb index ad45166..8697e43 100644 --- a/Q2.ipynb +++ b/Q2.ipynb @@ -12,6 +12,41 @@ ], "id": "bb3519b1aa083259" }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.510748800Z", + "start_time": "2026-04-25T20:51:48.489901400Z" + } + }, + "cell_type": "code", + "source": [ + "import pandas as pd\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.preprocessing import StandardScaler\n", + "from sklearn.linear_model import LogisticRegression\n", + "from sklearn.neighbors import KNeighborsClassifier\n", + "from sklearn.metrics import accuracy_score, f1_score, confusion_matrix, ConfusionMatrixDisplay\n", + "import matplotlib.pyplot as plt" + ], + "id": "76e70b2ed9af0b56", + "outputs": [], + "execution_count": 5 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.526865400Z", + "start_time": "2026-04-25T20:51:48.511753700Z" + } + }, + "cell_type": "code", + "source": "df = pd.read_csv('data/googleplaystore_new_new.csv')", + "id": "f44380615d3aba25", + "outputs": [], + "execution_count": 6 + }, { "metadata": {}, "cell_type": "markdown", @@ -26,12 +61,1961 @@ "id": "1baa7daa49445720" }, { - "metadata": {}, + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.534574700Z", + "start_time": "2026-04-25T20:51:48.527865900Z" + } + }, "cell_type": "code", + "source": [ + "columns_to_keep = ['Rating', 'Reviews', 'Content Rating', 'Size in bytes', 'Numeric Installs', 'Category']\n", + "df_part_a = df[columns_to_keep]" + ], + "id": "f6fd7137bf91f31e", "outputs": [], - "execution_count": null, - "source": "", - "id": "f6fd7137bf91f31e" + "execution_count": 7 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.542093600Z", + "start_time": "2026-04-25T20:51:48.535570300Z" + } + }, + "cell_type": "code", + "source": [ + "X_raw = df_part_a.drop('Category', axis=1)\n", + "y = df_part_a['Category']" + ], + "id": "b5daf475a5d5ca15", + "outputs": [], + "execution_count": 8 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.568373500Z", + "start_time": "2026-04-25T20:51:48.542093600Z" + } + }, + "cell_type": "code", + "source": "X = pd.get_dummies(X_raw, columns=['Content Rating'])", + "id": "99a441c665dcc19b", + "outputs": [], + "execution_count": 9 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.586290600Z", + "start_time": "2026-04-25T20:51:48.569378200Z" + } + }, + "cell_type": "code", + "source": "X_temp, X_test, y_temp, y_test = train_test_split(X, y, test_size=0.2, random_state=101)", + "id": "c6eb622c0f7ec63c", + "outputs": [], + "execution_count": 10 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.593412500Z", + "start_time": "2026-04-25T20:51:48.588327300Z" + } + }, + "cell_type": "code", + "source": "X_train, X_val, y_train, y_val = train_test_split(X_temp, y_temp, test_size=0.25, random_state=101)", + "id": "89e8f117f057e0e2", + "outputs": [], + "execution_count": 11 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.605804800Z", + "start_time": "2026-04-25T20:51:48.593412500Z" + } + }, + "cell_type": "code", + "source": [ + "scaler = StandardScaler()\n", + "scaled_X_train = scaler.fit_transform(X_train)\n", + "scaled_X_val = scaler.transform(X_val)\n", + "scaled_X_test = scaler.transform(X_test)" + ], + "id": "d529991171fafb3e", + "outputs": [], + "execution_count": 12 + }, + { + "metadata": {}, + "cell_type": "markdown", + "source": "## Logistic Regression", + "id": "66dc17cdf1a00a06" + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.668048800Z", + "start_time": "2026-04-25T20:51:48.607804Z" + } + }, + "cell_type": "code", + "source": [ + "log_model = LogisticRegression(max_iter=1000)\n", + "log_model.fit(scaled_X_train, y_train)" + ], + "id": "84128fa4e7823565", + "outputs": [ + { + "data": { + "text/plain": [ + "LogisticRegression(max_iter=1000)" + ], + "text/html": [ + "
LogisticRegression(max_iter=1000)
In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
" + ] + }, + "execution_count": 13, + "metadata": {}, + "output_type": "execute_result" + } + ], + "execution_count": 13 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.694210900Z", + "start_time": "2026-04-25T20:51:48.680578900Z" + } + }, + "cell_type": "code", + "source": "y_val_pred_log = log_model.predict(scaled_X_val)", + "id": "965e16ce48443c97", + "outputs": [], + "execution_count": 14 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.714470600Z", + "start_time": "2026-04-25T20:51:48.695211200Z" + } + }, + "cell_type": "code", + "source": [ + "log_val_f1 = f1_score(y_val, y_val_pred_log, average='weighted')\n", + "print(f\"Validation Accuracy: {accuracy_score(y_val, y_val_pred_log):.4f}\")\n", + "print(f\"Validation F1-Score: {log_val_f1:.4f}\")" + ], + "id": "9be803af2f34cd6", + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Validation Accuracy: 0.3333\n", + "Validation F1-Score: 0.3069\n" + ] + } + ], + "execution_count": 15 + }, + { + "metadata": {}, + "cell_type": "markdown", + "source": "## KNN", + "id": "2b7b85284cddecb2" + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.736063Z", + "start_time": "2026-04-25T20:51:48.715973100Z" + } + }, + "cell_type": "code", + "source": [ + "best_k = 1\n", + "best_f1 = 0" + ], + "id": "8c4835d7a413a6e6", + "outputs": [], + "execution_count": 16 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.791235200Z", + "start_time": "2026-04-25T20:51:48.736063Z" + } + }, + "cell_type": "code", + "source": [ + "for k in range(1, 15, 2):\n", + " knn_temp = KNeighborsClassifier(n_neighbors=k)\n", + " knn_temp.fit(scaled_X_train, y_train)\n", + " y_val_pred_knn = knn_temp.predict(scaled_X_val)\n", + "\n", + " current_f1 = f1_score(y_val, y_val_pred_knn, average='weighted')\n", + " print(f\"KNN (K={k}) Validation F1-Score: {current_f1:.4f}\")\n", + "\n", + " if current_f1 > best_f1:\n", + " best_f1 = current_f1\n", + " best_k = k" + ], + "id": "a8d69c56d1a21825", + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "KNN (K=1) Validation F1-Score: 0.3201\n", + "KNN (K=3) Validation F1-Score: 0.3245\n", + "KNN (K=5) Validation F1-Score: 0.2939\n", + "KNN (K=7) Validation F1-Score: 0.2916\n", + "KNN (K=9) Validation F1-Score: 0.3120\n", + "KNN (K=11) Validation F1-Score: 0.3097\n", + "KNN (K=13) Validation F1-Score: 0.3181\n" + ] + } + ], + "execution_count": 17 + }, + { + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.823877400Z", + "start_time": "2026-04-25T20:51:48.792239400Z" + } + }, + "cell_type": "code", + "source": [ + "knn_final = KNeighborsClassifier(n_neighbors=best_k)\n", + "knn_final.fit(scaled_X_train, y_train)" + ], + "id": "96026b80f506bf93", + "outputs": [ + { + "data": { + "text/plain": [ + "KNeighborsClassifier(n_neighbors=3)" + ], + "text/html": [ + "
KNeighborsClassifier(n_neighbors=3)
In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook.
On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org.
" + ] + }, + "execution_count": 18, + "metadata": {}, + "output_type": "execute_result" + } + ], + "execution_count": 18 }, { "metadata": {}, @@ -47,12 +2031,17 @@ "id": "f15aa6cca58826fc" }, { - "metadata": {}, + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.841953400Z", + "start_time": "2026-04-25T20:51:48.825877800Z" + } + }, "cell_type": "code", - "outputs": [], - "execution_count": null, "source": "", - "id": "65e55b70ad44b2dc" + "id": "65e55b70ad44b2dc", + "outputs": [], + "execution_count": 18 }, { "metadata": {}, @@ -68,12 +2057,17 @@ "id": "86c040f5e043a39c" }, { - "metadata": {}, + "metadata": { + "ExecuteTime": { + "end_time": "2026-04-25T20:51:48.861351100Z", + "start_time": "2026-04-25T20:51:48.842953500Z" + } + }, "cell_type": "code", - "outputs": [], - "execution_count": null, "source": "", - "id": "f3535461938f94f5" + "id": "f3535461938f94f5", + "outputs": [], + "execution_count": 18 } ], "metadata": {