diff --git a/Q1.ipynb b/Q1.ipynb
index 6d4510a..ea73877 100644
--- a/Q1.ipynb
+++ b/Q1.ipynb
@@ -2058,13 +2058,883 @@
],
"id": "eaefa3aed2214793"
},
+ {
+ "metadata": {},
+ "cell_type": "markdown",
+ "source": "## Part F",
+ "id": "3c89e2f8ae31de3b"
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:14:41.668654900Z",
+ "start_time": "2026-04-25T15:14:41.650113100Z"
+ }
+ },
+ "cell_type": "code",
+ "source": "features_to_test = ['Reviews', 'Size in bytes', 'Numeric Installs']",
+ "id": "ab09192a122e6e8f",
+ "outputs": [],
+ "execution_count": 404
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:15:26.186132400Z",
+ "start_time": "2026-04-25T15:15:26.176807100Z"
+ }
+ },
+ "cell_type": "code",
+ "source": [
+ "for feature in features_to_test:\n",
+ " X_single = df_encoded[[feature]]"
+ ],
+ "id": "3f65e25b63cc8e93",
+ "outputs": [],
+ "execution_count": 407
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:15:37.813266600Z",
+ "start_time": "2026-04-25T15:15:37.797080500Z"
+ }
+ },
+ "cell_type": "code",
+ "source": "X_train_s, X_test_s, y_train_s, y_test_s = train_test_split(X_single, y, test_size=0.3, random_state=101)",
+ "id": "213bcc196f94054d",
+ "outputs": [],
+ "execution_count": 408
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:15:59.633576900Z",
+ "start_time": "2026-04-25T15:15:59.612227200Z"
+ }
+ },
+ "cell_type": "code",
+ "source": [
+ "simple_model = LinearRegression()\n",
+ "simple_model.fit(X_train_s, y_train_s)"
+ ],
+ "id": "75e6c5289568ef2f",
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "LinearRegression()"
+ ],
+ "text/html": [
+ "
LinearRegression() In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook. On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org. \n",
+ "
\n",
+ "
\n",
+ " Parameters \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " fit_intercept\n",
+ " fit_intercept: bool, default=True Whether to calculate the intercept for this model. If set to False, no intercept will be used in calculations (i.e. data is expected to be centered). \n",
+ " \n",
+ " \n",
+ " True \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " copy_X\n",
+ " copy_X: bool, default=True If True, X will be copied; else, it may be overwritten. \n",
+ " \n",
+ " \n",
+ " True \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " tol\n",
+ " tol: float, default=1e-6 The precision of the solution (`coef_`) is determined by `tol` which specifies a different convergence criterion for the `lsqr` solver. `tol` is set as `atol` and `btol` of :func:`scipy.sparse.linalg.lsqr` when fitting on sparse training data. This parameter has no effect when fitting on dense data. .. versionadded:: 1.7 \n",
+ " \n",
+ " \n",
+ " 1e-06 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " n_jobs\n",
+ " n_jobs: int, default=None The number of jobs to use for the computation. This will only provide speedup in case of sufficiently large problems, that is if firstly `n_targets > 1` and secondly `X` is sparse or if `positive` is set to `True`. ``None`` means 1 unless in a :obj:`joblib.parallel_backend` context. ``-1`` means using all processors. See :term:`Glossary ` for more details. \n",
+ " \n",
+ " \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " positive\n",
+ " positive: bool, default=False When set to ``True``, forces the coefficients to be positive. This option is only supported for dense arrays. For a comparison between a linear regression model with positive constraints on the regression coefficients and a linear regression without such constraints, see :ref:`sphx_glr_auto_examples_linear_model_plot_nnls.py`. .. versionadded:: 0.24 \n",
+ " \n",
+ " \n",
+ " False \n",
+ " \n",
+ " \n",
+ " \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ]
+ },
+ "execution_count": 409,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "execution_count": 409
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:16:10.648492700Z",
+ "start_time": "2026-04-25T15:16:10.630533500Z"
+ }
+ },
+ "cell_type": "code",
+ "source": [
+ "y_pred_s = simple_model.predict(X_test_s)\n",
+ "rmse = np.sqrt(mean_squared_error(y_test_s, y_pred_s))"
+ ],
+ "id": "983db24d45054c97",
+ "outputs": [],
+ "execution_count": 410
+ },
+ {
+ "metadata": {
+ "ExecuteTime": {
+ "end_time": "2026-04-25T15:16:25.585898300Z",
+ "start_time": "2026-04-25T15:16:25.549110800Z"
+ }
+ },
+ "cell_type": "code",
+ "source": "print(f\"RMSE using ONLY '{feature}': {rmse}\")",
+ "id": "4cb7c407238e67c1",
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "RMSE using ONLY 'Numeric Installs': 0.4179032096419041\n"
+ ]
+ }
+ ],
+ "execution_count": 411
+ },
{
"metadata": {},
"cell_type": "code",
"outputs": [],
"execution_count": null,
"source": "",
- "id": "8e3d435421cb39d"
+ "id": "79c525159dd03465"
}
],
"metadata": {