{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport tensorflow as tf\n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer, Normalizer, MinMaxScaler\nfrom sklearn.compose import ColumnTransformer\nfrom scipy.stats import skew\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.ensemble import RandomForestRegressor","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:03:52.560402Z","iopub.execute_input":"2022-07-21T06:03:52.560770Z","iopub.status.idle":"2022-07-21T06:04:14.903498Z","shell.execute_reply.started":"2022-07-21T06:03:52.560692Z","shell.execute_reply":"2022-07-21T06:04:14.902481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/laptop-price-prediction/train.csv', index_col='Id')\ntest_df = pd.read_csv('../input/laptop-price-prediction/test.csv', index_col='Id')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:21.409616Z","iopub.execute_input":"2022-07-21T06:04:21.410314Z","iopub.status.idle":"2022-07-21T06:04:21.515230Z","shell.execute_reply.started":"2022-07-21T06:04:21.410287Z","shell.execute_reply":"2022-07-21T06:04:21.511351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:24.946149Z","iopub.execute_input":"2022-07-21T06:04:24.946578Z","iopub.status.idle":"2022-07-21T06:04:25.047691Z","shell.execute_reply.started":"2022-07-21T06:04:24.946555Z","shell.execute_reply":"2022-07-21T06:04:25.046561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:25.355615Z","iopub.execute_input":"2022-07-21T06:04:25.357310Z","iopub.status.idle":"2022-07-21T06:04:25.394630Z","shell.execute_reply.started":"2022-07-21T06:04:25.357260Z","shell.execute_reply":"2022-07-21T06:04:25.393479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:28.309836Z","iopub.execute_input":"2022-07-21T06:04:28.310180Z","iopub.status.idle":"2022-07-21T06:04:28.323185Z","shell.execute_reply.started":"2022-07-21T06:04:28.310155Z","shell.execute_reply":"2022-07-21T06:04:28.319192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = ['Pulgadas', 'RAM', 'Peso']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:29.060678Z","iopub.execute_input":"2022-07-21T06:04:29.061288Z","iopub.status.idle":"2022-07-21T06:04:29.080267Z","shell.execute_reply.started":"2022-07-21T06:04:29.061258Z","shell.execute_reply":"2022-07-21T06:04:29.070360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:29.744697Z","iopub.execute_input":"2022-07-21T06:04:29.745100Z","iopub.status.idle":"2022-07-21T06:04:29.772401Z","shell.execute_reply.started":"2022-07-21T06:04:29.745071Z","shell.execute_reply":"2022-07-21T06:04:29.771422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:29.939531Z","iopub.execute_input":"2022-07-21T06:04:29.940497Z","iopub.status.idle":"2022-07-21T06:04:29.966261Z","shell.execute_reply.started":"2022-07-21T06:04:29.940472Z","shell.execute_reply":"2022-07-21T06:04:29.964693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.RAM.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:30.152884Z","iopub.execute_input":"2022-07-21T06:04:30.153457Z","iopub.status.idle":"2022-07-21T06:04:30.160430Z","shell.execute_reply.started":"2022-07-21T06:04:30.153432Z","shell.execute_reply":"2022-07-21T06:04:30.159417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['RAM'] = train_df['RAM'].str.replace('GB', '').astype('int')\ntest_df['RAM'] = test_df['RAM'].str.replace('GB', '').astype('int')\ntrain_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:30.348905Z","iopub.execute_input":"2022-07-21T06:04:30.349285Z","iopub.status.idle":"2022-07-21T06:04:30.375068Z","shell.execute_reply.started":"2022-07-21T06:04:30.349262Z","shell.execute_reply":"2022-07-21T06:04:30.373347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.Peso.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:30.569159Z","iopub.execute_input":"2022-07-21T06:04:30.569526Z","iopub.status.idle":"2022-07-21T06:04:30.581543Z","shell.execute_reply.started":"2022-07-21T06:04:30.569498Z","shell.execute_reply":"2022-07-21T06:04:30.579869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Peso'] = train_df['Peso'].str.replace('kg', '').astype('float')\ntest_df['Peso'] = test_df['Peso'].str.replace('kg', '').astype('float')\ntrain_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:30.728473Z","iopub.execute_input":"2022-07-21T06:04:30.728819Z","iopub.status.idle":"2022-07-21T06:04:30.772350Z","shell.execute_reply.started":"2022-07-21T06:04:30.728790Z","shell.execute_reply":"2022-07-21T06:04:30.771029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(train_df[num_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:30.917290Z","iopub.execute_input":"2022-07-21T06:04:30.917677Z","iopub.status.idle":"2022-07-21T06:04:33.032849Z","shell.execute_reply.started":"2022-07-21T06:04:30.917647Z","shell.execute_reply":"2022-07-21T06:04:33.027765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline Linear Regression","metadata":{}},{"cell_type":"code","source":"X_whole_01 = train_df[num_cols]\ny_whole_01 = train_df['Precio']\n\nX_train_01, X_valid_01, y_train_01, y_valid_01 = train_test_split(X_whole_01,\n                                                                 y_whole_01,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\nlr_01 = LinearRegression()\n\nlr_01.fit(X_train_01, y_train_01)\n\ny_fit_01 = lr_01.predict(X_train_01)\ny_pred_01 = lr_01.predict(X_valid_01)\n\nrmse_fit_01 = mean_squared_error(y_train_01, y_fit_01, squared=False)\nrmse_pred_01 = mean_squared_error(y_valid_01, y_pred_01, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_01:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_01:.2f}\")\n\nsns.regplot(x=y_valid_01, y=y_pred_01);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_01,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_01,  color='green', marker='x', linestyle='None', label='y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_01-y_pred_01,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_02,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:04:33.035359Z","iopub.execute_input":"2022-07-21T06:04:33.035943Z","iopub.status.idle":"2022-07-21T06:04:34.767821Z","shell.execute_reply.started":"2022-07-21T06:04:33.035905Z","shell.execute_reply":"2022-07-21T06:04:34.763825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_01","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:03.109811Z","iopub.execute_input":"2022-07-21T06:05:03.110222Z","iopub.status.idle":"2022-07-21T06:05:03.119456Z","shell.execute_reply.started":"2022-07-21T06:05:03.110194Z","shell.execute_reply":"2022-07-21T06:05:03.118379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:04.167256Z","iopub.execute_input":"2022-07-21T06:05:04.167555Z","iopub.status.idle":"2022-07-21T06:05:04.203583Z","shell.execute_reply.started":"2022-07-21T06:05:04.167532Z","shell.execute_reply":"2022-07-21T06:05:04.202035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features_mask = train_df.dtypes == 'object'","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:05.029113Z","iopub.execute_input":"2022-07-21T06:05:05.031012Z","iopub.status.idle":"2022-07-21T06:05:05.037814Z","shell.execute_reply.started":"2022-07-21T06:05:05.030917Z","shell.execute_reply":"2022-07-21T06:05:05.036334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns[cat_features_mask].tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:05.660714Z","iopub.execute_input":"2022-07-21T06:05:05.662060Z","iopub.status.idle":"2022-07-21T06:05:05.668723Z","shell.execute_reply.started":"2022-07-21T06:05:05.662008Z","shell.execute_reply":"2022-07-21T06:05:05.668004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = train_df.select_dtypes('object').columns","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:06.233183Z","iopub.execute_input":"2022-07-21T06:05:06.233656Z","iopub.status.idle":"2022-07-21T06:05:06.238804Z","shell.execute_reply.started":"2022-07-21T06:05:06.233621Z","shell.execute_reply":"2022-07-21T06:05:06.238184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols_low_card = ['Fabricante', 'Tipo',  'OS']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:06.493890Z","iopub.execute_input":"2022-07-21T06:05:06.494414Z","iopub.status.idle":"2022-07-21T06:05:06.498343Z","shell.execute_reply.started":"2022-07-21T06:05:06.494388Z","shell.execute_reply":"2022-07-21T06:05:06.497227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols_hi_card = list(set(cat_cols) - set(cat_cols_low_card))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:06.704221Z","iopub.execute_input":"2022-07-21T06:05:06.704780Z","iopub.status.idle":"2022-07-21T06:05:06.710607Z","shell.execute_reply.started":"2022-07-21T06:05:06.704753Z","shell.execute_reply":"2022-07-21T06:05:06.709016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols_hi_card","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:06.899063Z","iopub.execute_input":"2022-07-21T06:05:06.899455Z","iopub.status.idle":"2022-07-21T06:05:06.906712Z","shell.execute_reply.started":"2022-07-21T06:05:06.899430Z","shell.execute_reply":"2022-07-21T06:05:06.904922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor col in cat_cols_low_card:\n    sns.boxplot(x=train_df[col], y=train_df['Precio'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:07.172405Z","iopub.execute_input":"2022-07-21T06:05:07.172739Z","iopub.status.idle":"2022-07-21T06:05:07.621579Z","shell.execute_reply.started":"2022-07-21T06:05:07.172716Z","shell.execute_reply":"2022-07-21T06:05:07.619744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def subplot_hist(df, cols, no_cols_plot=3, figsize=(15,6)):\n    \n    plt.figure(figsize=figsize)\n\n    for i in range(len(cols)):\n\n        plt.subplot(np.ceil(len(cols)/no_cols_plot).astype(int), no_cols_plot, i+1)\n        plt.hist(df[cols[i]])\n        plt.title(cols[i])  ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:07.919499Z","iopub.execute_input":"2022-07-21T06:05:07.919812Z","iopub.status.idle":"2022-07-21T06:05:07.927032Z","shell.execute_reply.started":"2022-07-21T06:05:07.919788Z","shell.execute_reply":"2022-07-21T06:05:07.925977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def subplot_box_num(df, cols, no_cols_plot=3, figsize=(15,6)):\n    \n    plt.figure(figsize=figsize)\n\n    for i in range(len(cols)):\n\n        plt.subplot(np.ceil(len(cols)/no_cols_plot).astype(int), no_cols_plot, i+1)\n        sns.boxplot(x=df[cols[i]])\n        plt.title(cols[i]) ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:08.821118Z","iopub.execute_input":"2022-07-21T06:05:08.822527Z","iopub.status.idle":"2022-07-21T06:05:08.827360Z","shell.execute_reply.started":"2022-07-21T06:05:08.822499Z","shell.execute_reply":"2022-07-21T06:05:08.826479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplot_hist(train_df, num_cols, figsize=(15,4))\nsubplot_box_num(train_df, num_cols, figsize=(15,4))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:09.013027Z","iopub.execute_input":"2022-07-21T06:05:09.013586Z","iopub.status.idle":"2022-07-21T06:05:09.780985Z","shell.execute_reply.started":"2022-07-21T06:05:09.013560Z","shell.execute_reply":"2022-07-21T06:05:09.780045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(20,5))\n# sns.boxplot(x=train_df['OS'], y=train_df['Precio']);\n\n\n\ndef subplot_box_cat(df, cols, no_cols_plot=3, figsize=(15,6)):\n    \n    plt.figure(figsize=figsize)\n\n    for i in range(len(cols)):\n\n        plt.subplot(np.ceil(len(cols)/no_cols_plot).astype(int), no_cols_plot, i+1)\n        sns.boxplot(x=df[cols[i]], y=df['Precio'])\n        plt.title(cols[i]) ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:09.782437Z","iopub.execute_input":"2022-07-21T06:05:09.783246Z","iopub.status.idle":"2022-07-21T06:05:09.790042Z","shell.execute_reply.started":"2022-07-21T06:05:09.783218Z","shell.execute_reply":"2022-07-21T06:05:09.788655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplot_box_cat(train_df, cat_cols_low_card, no_cols_plot=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:09.791892Z","iopub.execute_input":"2022-07-21T06:05:09.792311Z","iopub.status.idle":"2022-07-21T06:05:10.426382Z","shell.execute_reply.started":"2022-07-21T06:05:09.792279Z","shell.execute_reply":"2022-07-21T06:05:10.425536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adding only low cardinality columns - model_02","metadata":{}},{"cell_type":"code","source":"X_whole_02 = train_df[num_cols + cat_cols_low_card]\ny_whole_02 = train_df['Precio']\n\nX_train_02, X_valid_02, y_train_02, y_valid_02 = train_test_split(X_whole_02,\n                                                                 y_whole_02,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n\ncategorical_transformer_02 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_02 = Pipeline([\n    ('std_scaler', StandardScaler())\n])\n\npreprocessor_02 = ColumnTransformer([\n    ('num', numerical_transformer_02, num_cols),\n    ('cat', categorical_transformer_02, cat_cols_low_card)\n])\n\n\n# Linear Regression\n\nlr_02 = Pipeline([\n    ('preprocessor', preprocessor_02),\n    ('lr', LinearRegression())\n])\n\nlr_02.fit(X_train_02, y_train_02)\n\ny_fit_02 = lr_02.predict(X_train_02)\ny_pred_02 = lr_02.predict(X_valid_02)\n\nrmse_fit_02 = mean_squared_error(y_train_02, y_fit_02, squared=False)\nrmse_pred_02 = mean_squared_error(y_valid_02, y_pred_02, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_02:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_02:.2f}\")\n\nsns.regplot(x=y_valid_02, y=y_pred_02);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_02,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_02,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_02-y_pred_02,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_02,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:11.520516Z","iopub.execute_input":"2022-07-21T06:05:11.521925Z","iopub.status.idle":"2022-07-21T06:05:12.117545Z","shell.execute_reply.started":"2022-07-21T06:05:11.521860Z","shell.execute_reply":"2022-07-21T06:05:12.116891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Scaler","metadata":{}},{"cell_type":"code","source":"std_scal = StandardScaler()\nrob_scal = RobustScaler()\nnor_scal = Normalizer()\nminmax_scal = MinMaxScaler()\npow_scal = PowerTransformer()\n\ntest_std_scal = pd.DataFrame(std_scal.fit_transform(train_df[num_cols]), columns=num_cols, index=train_df.index)\ntest_rob_scal = pd.DataFrame(rob_scal.fit_transform(train_df[num_cols]), columns=num_cols, index=train_df.index)\ntest_nor_scal = pd.DataFrame(nor_scal.fit_transform(train_df[num_cols]), columns=num_cols, index=train_df.index)\ntest_minmax_scal = pd.DataFrame(minmax_scal.fit_transform(train_df[num_cols]), columns=num_cols, index=train_df.index)\ntest_pow_scal = pd.DataFrame(pow_scal.fit_transform(train_df[num_cols]), columns=num_cols, index=train_df.index)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:12.118813Z","iopub.execute_input":"2022-07-21T06:05:12.119745Z","iopub.status.idle":"2022-07-21T06:05:12.149249Z","shell.execute_reply.started":"2022-07-21T06:05:12.119721Z","shell.execute_reply":"2022-07-21T06:05:12.148552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Standard Scaler')\ndisplay(test_std_scal.skew())\nprint('-'*30)\nprint('Robust Scaler')\ndisplay(test_rob_scal.skew())\nprint('-'*30)\nprint('Normalizer')\ndisplay(test_nor_scal.skew())\nprint('-'*30)\nprint('Min Max Scaler')\ndisplay(test_minmax_scal.skew())\nprint('-'*30)\nprint('Power Transformer')\ndisplay(test_pow_scal.skew())","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:12.150198Z","iopub.execute_input":"2022-07-21T06:05:12.151470Z","iopub.status.idle":"2022-07-21T06:05:12.182136Z","shell.execute_reply.started":"2022-07-21T06:05:12.151427Z","shell.execute_reply":"2022-07-21T06:05:12.181205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplot_hist(test_std_scal, num_cols, 3, (15,4))\nsubplot_hist(test_pow_scal, num_cols, 3, (15,4))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:12.219515Z","iopub.execute_input":"2022-07-21T06:05:12.220634Z","iopub.status.idle":"2022-07-21T06:05:12.878061Z","shell.execute_reply.started":"2022-07-21T06:05:12.220596Z","shell.execute_reply":"2022-07-21T06:05:12.875008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_03\nSame as model_02, only change POwerTransformer instead of StandardScaler","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data\nX_whole_03 = train_df[num_cols + cat_cols_low_card]\ny_whole_03 = train_df['Precio']\n\nX_train_03, X_valid_03, y_train_03, y_valid_03 = train_test_split(X_whole_03,\n                                                                 y_whole_03,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_03 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_03 = Pipeline([\n    ('scaler', PowerTransformer())\n])\n\npreprocessor_03 = ColumnTransformer([\n    ('num', numerical_transformer_03, num_cols),\n    ('cat', categorical_transformer_03, cat_cols_low_card)\n])\n\n# Linear Regression\n\nlr_03 = Pipeline([\n    ('preprocessor', preprocessor_03),\n    ('lr', LinearRegression())\n])\n\nlr_03.fit(X_train_03, y_train_03)\n\n\ny_fit_03 = lr_03.predict(X_train_03)\ny_pred_03 = lr_03.predict(X_valid_03)\n\nrmse_fit_03 = mean_squared_error(y_train_03, y_fit_03, squared=False)\nrmse_pred_03 = mean_squared_error(y_valid_03, y_pred_03, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_03:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_03:.2f}\")\n\nsns.regplot(x=y_valid_03, y=y_pred_03);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_03,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_03,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_03-y_pred_03,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_03,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:13.249653Z","iopub.execute_input":"2022-07-21T06:05:13.249986Z","iopub.status.idle":"2022-07-21T06:05:13.868629Z","shell.execute_reply.started":"2022-07-21T06:05:13.249946Z","shell.execute_reply":"2022-07-21T06:05:13.867061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# High Cardinality Columns","metadata":{}},{"cell_type":"code","source":"cat_cols_hi_card","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:13.871389Z","iopub.execute_input":"2022-07-21T06:05:13.871807Z","iopub.status.idle":"2022-07-21T06:05:13.884948Z","shell.execute_reply.started":"2022-07-21T06:05:13.871772Z","shell.execute_reply":"2022-07-21T06:05:13.883783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ScreenResolution","metadata":{}},{"cell_type":"code","source":"print(train_df.ScreenResolution.nunique())\ntrain_df.ScreenResolution.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:05:14.420387Z","iopub.execute_input":"2022-07-21T06:05:14.420750Z","iopub.status.idle":"2022-07-21T06:05:14.432089Z","shell.execute_reply.started":"2022-07-21T06:05:14.420722Z","shell.execute_reply":"2022-07-21T06:05:14.431171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(train_df.ScreenResolution)):\n#     print(train_df.ScreenResolution[i], train_df['ScreenResolution'].loc[train_df['ScreenResolution'] == train_df.ScreenResolution[i]].count() < 10)\n    if train_df['ScreenResolution'].loc[train_df['ScreenResolution'] == train_df.ScreenResolution[i]].count() < 10:\n        train_df.ScreenResolution[i] = train_df.ScreenResolution[i].replace(train_df.ScreenResolution[i], 'Other')\n        \ntrain_df.ScreenResolution.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:28.692312Z","iopub.execute_input":"2022-07-21T06:06:28.692665Z","iopub.status.idle":"2022-07-21T06:06:29.037438Z","shell.execute_reply.started":"2022-07-21T06:06:28.692638Z","shell.execute_reply":"2022-07-21T06:06:29.036469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_cols_hi_card].describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:31.523657Z","iopub.execute_input":"2022-07-21T06:06:31.524412Z","iopub.status.idle":"2022-07-21T06:06:31.544787Z","shell.execute_reply.started":"2022-07-21T06:06:31.524379Z","shell.execute_reply":"2022-07-21T06:06:31.543766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CPU","metadata":{}},{"cell_type":"code","source":"print(train_df.CPU.nunique())\ntrain_df.CPU.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:31.827369Z","iopub.execute_input":"2022-07-21T06:06:31.827813Z","iopub.status.idle":"2022-07-21T06:06:31.841197Z","shell.execute_reply.started":"2022-07-21T06:06:31.827776Z","shell.execute_reply":"2022-07-21T06:06:31.840186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(train_df.CPU)):\n#     print(i, train_df.CPU[i], train_df['CPU'].loc[train_df['CPU'] == train_df.CPU[i]].count() < 10)\n\n    if train_df['CPU'].loc[train_df['CPU'] == train_df.CPU[i]].count() < 10:\n        train_df.CPU[i] = train_df.CPU[i].replace(train_df.CPU[i], 'Other')\n        \ntrain_df.CPU.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:31.966637Z","iopub.execute_input":"2022-07-21T06:06:31.967290Z","iopub.status.idle":"2022-07-21T06:06:32.422626Z","shell.execute_reply.started":"2022-07-21T06:06:31.967255Z","shell.execute_reply":"2022-07-21T06:06:32.420295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.CPU[3].replace(train_df.CPU[3], 'Other')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:32.424227Z","iopub.execute_input":"2022-07-21T06:06:32.424736Z","iopub.status.idle":"2022-07-21T06:06:32.431717Z","shell.execute_reply.started":"2022-07-21T06:06:32.424709Z","shell.execute_reply":"2022-07-21T06:06:32.430024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_cols_hi_card].describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:32.433559Z","iopub.execute_input":"2022-07-21T06:06:32.433928Z","iopub.status.idle":"2022-07-21T06:06:32.456590Z","shell.execute_reply.started":"2022-07-21T06:06:32.433895Z","shell.execute_reply":"2022-07-21T06:06:32.455653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Disco","metadata":{}},{"cell_type":"code","source":"print(train_df.Disco.nunique())\ntrain_df.Disco.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:33.415020Z","iopub.execute_input":"2022-07-21T06:06:33.416122Z","iopub.status.idle":"2022-07-21T06:06:33.426097Z","shell.execute_reply.started":"2022-07-21T06:06:33.416088Z","shell.execute_reply":"2022-07-21T06:06:33.425296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(train_df.Disco)):\n#     print(train_df.ScreenResolution[i], train_df['ScreenResolution'].loc[train_df['ScreenResolution'] == train_df.ScreenResolution[i]].count() < 10)\n    if train_df['Disco'].loc[train_df['Disco'] == train_df.Disco[i]].count() < 10:\n        train_df.Disco[i] = train_df.Disco[i].replace(train_df.Disco[i], 'Other')\n        \ntrain_df.Disco.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:33.583852Z","iopub.execute_input":"2022-07-21T06:06:33.585095Z","iopub.status.idle":"2022-07-21T06:06:34.020610Z","shell.execute_reply.started":"2022-07-21T06:06:33.585049Z","shell.execute_reply":"2022-07-21T06:06:34.018318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_cols_hi_card].describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:34.022065Z","iopub.execute_input":"2022-07-21T06:06:34.022442Z","iopub.status.idle":"2022-07-21T06:06:34.050481Z","shell.execute_reply.started":"2022-07-21T06:06:34.022415Z","shell.execute_reply":"2022-07-21T06:06:34.049071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GPU","metadata":{}},{"cell_type":"code","source":"print(train_df.GPU.nunique())\ntrain_df.GPU.value_counts().head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:34.560352Z","iopub.execute_input":"2022-07-21T06:06:34.560761Z","iopub.status.idle":"2022-07-21T06:06:34.574016Z","shell.execute_reply.started":"2022-07-21T06:06:34.560730Z","shell.execute_reply":"2022-07-21T06:06:34.572792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(train_df.GPU)):\n#     print(train_df.ScreenResolution[i], train_df['ScreenResolution'].loc[train_df['ScreenResolution'] == train_df.ScreenResolution[i]].count() < 10)\n    if train_df['GPU'].loc[train_df['GPU'] == train_df.GPU[i]].count() < 10:\n        train_df.GPU[i] = train_df.GPU[i].replace(train_df.GPU[i], 'Other')\n        \ntrain_df.GPU.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:34.755735Z","iopub.execute_input":"2022-07-21T06:06:34.756145Z","iopub.status.idle":"2022-07-21T06:06:35.323940Z","shell.execute_reply.started":"2022-07-21T06:06:34.756103Z","shell.execute_reply":"2022-07-21T06:06:35.322511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_cols_hi_card].describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:35.325921Z","iopub.execute_input":"2022-07-21T06:06:35.326307Z","iopub.status.idle":"2022-07-21T06:06:35.350324Z","shell.execute_reply.started":"2022-07-21T06:06:35.326269Z","shell.execute_reply":"2022-07-21T06:06:35.349219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplot_box_cat(train_df, cat_cols_hi_card, no_cols_plot=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:35.952762Z","iopub.execute_input":"2022-07-21T06:06:35.953102Z","iopub.status.idle":"2022-07-21T06:06:37.037438Z","shell.execute_reply.started":"2022-07-21T06:06:35.953079Z","shell.execute_reply":"2022-07-21T06:06:37.036285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_04\nPipeline with StandardScaler and all the columns","metadata":{}},{"cell_type":"code","source":"cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:37.040143Z","iopub.execute_input":"2022-07-21T06:06:37.041297Z","iopub.status.idle":"2022-07-21T06:06:37.048187Z","shell.execute_reply.started":"2022-07-21T06:06:37.041260Z","shell.execute_reply":"2022-07-21T06:06:37.047108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_cols].describe(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:37.049428Z","iopub.execute_input":"2022-07-21T06:06:37.049712Z","iopub.status.idle":"2022-07-21T06:06:37.088948Z","shell.execute_reply.started":"2022-07-21T06:06:37.049684Z","shell.execute_reply":"2022-07-21T06:06:37.087728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:37.090437Z","iopub.execute_input":"2022-07-21T06:06:37.090739Z","iopub.status.idle":"2022-07-21T06:06:37.104235Z","shell.execute_reply.started":"2022-07-21T06:06:37.090709Z","shell.execute_reply":"2022-07-21T06:06:37.103125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:37.358156Z","iopub.execute_input":"2022-07-21T06:06:37.358530Z","iopub.status.idle":"2022-07-21T06:06:37.366015Z","shell.execute_reply.started":"2022-07-21T06:06:37.358501Z","shell.execute_reply":"2022-07-21T06:06:37.365043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[num_cols + cat_cols_low_card + cat_cols_hi_card]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:37.950556Z","iopub.execute_input":"2022-07-21T06:06:37.950878Z","iopub.status.idle":"2022-07-21T06:06:37.958169Z","shell.execute_reply.started":"2022-07-21T06:06:37.950853Z","shell.execute_reply":"2022-07-21T06:06:37.956596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data\nX_whole_04 = train_df[num_cols + cat_cols_low_card + cat_cols_hi_card]\ny_whole_04 = train_df['Precio']\n\nX_train_04, X_valid_04, y_train_04, y_valid_04 = train_test_split(X_whole_04,\n                                                                 y_whole_04,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_04 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_04 = Pipeline([\n    ('scaler', StandardScaler())\n])\n\npreprocessor_04 = ColumnTransformer([\n    ('num', numerical_transformer_04, num_cols),\n    ('cat', categorical_transformer_04, cat_cols_low_card + cat_cols_hi_card)\n])\n\n# Linear Regression\n\nlr_04 = Pipeline([\n    ('preprocessor', preprocessor_04),\n    ('lr', LinearRegression())\n])\n\nlr_04.fit(X_train_04, y_train_04)\n\n\ny_fit_04 = lr_04.predict(X_train_04)\ny_pred_04 = lr_04.predict(X_valid_04)\n\nrmse_fit_04 = mean_squared_error(y_train_04, y_fit_04, squared=False)\nrmse_pred_04 = mean_squared_error(y_valid_04, y_pred_04, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_04:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_04:.2f}\")\n\nsns.regplot(x=y_valid_04, y=y_pred_04);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_04,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_04,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_04-y_pred_04,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_04,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:38.139868Z","iopub.execute_input":"2022-07-21T06:06:38.140247Z","iopub.status.idle":"2022-07-21T06:06:38.682114Z","shell.execute_reply.started":"2022-07-21T06:06:38.140219Z","shell.execute_reply":"2022-07-21T06:06:38.680542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_transformer_04.fit(X_train_04, y_train_04)\npd.DataFrame.sparse.from_spmatrix(categorical_transformer_04.transform(X_train_04))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:38.684579Z","iopub.execute_input":"2022-07-21T06:06:38.684980Z","iopub.status.idle":"2022-07-21T06:06:38.771299Z","shell.execute_reply.started":"2022-07-21T06:06:38.684919Z","shell.execute_reply":"2022-07-21T06:06:38.770212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_transformer_03.fit(X_train_03, y_train_03)\npd.DataFrame.sparse.from_spmatrix(categorical_transformer_03.transform(X_train_03))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:38.772805Z","iopub.execute_input":"2022-07-21T06:06:38.773119Z","iopub.status.idle":"2022-07-21T06:06:38.854591Z","shell.execute_reply.started":"2022-07-21T06:06:38.773090Z","shell.execute_reply":"2022-07-21T06:06:38.853511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(X_train_03.describe(include='object'))\ndisplay(X_train_04.describe(include='object'))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:38.856514Z","iopub.execute_input":"2022-07-21T06:06:38.856811Z","iopub.status.idle":"2022-07-21T06:06:38.890708Z","shell.execute_reply.started":"2022-07-21T06:06:38.856782Z","shell.execute_reply":"2022-07-21T06:06:38.889431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_05\nmodel_04 with PowerTransformer","metadata":{}},{"cell_type":"code","source":"# Data\nX_whole_05 = train_df[num_cols + cat_cols_low_card + cat_cols_hi_card]\ny_whole_05 = train_df['Precio']\n\nX_train_05, X_valid_05, y_train_05, y_valid_05 = train_test_split(X_whole_05,\n                                                                 y_whole_05,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_05 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_05 = Pipeline([\n    ('scaler', PowerTransformer())\n])\n\npreprocessor_05 = ColumnTransformer([\n    ('num', numerical_transformer_05, num_cols),\n    ('cat', categorical_transformer_05, cat_cols_low_card + cat_cols_hi_card)\n])\n\n# Linear Regression\n\nlr_05 = Pipeline([\n    ('preprocessor', preprocessor_05),\n    ('lr', LinearRegression())\n])\n\nlr_05.fit(X_train_05, y_train_05)\n\n\ny_fit_05 = lr_05.predict(X_train_05)\ny_pred_05 = lr_05.predict(X_valid_05)\n\nrmse_fit_05 = mean_squared_error(y_train_05, y_fit_05, squared=False)\nrmse_pred_05 = mean_squared_error(y_valid_05, y_pred_05, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_05:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_05:.2f}\")\n\nsns.regplot(x=y_valid_05, y=y_pred_05);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_05,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_05,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_05-y_pred_05,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_05,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:39.373596Z","iopub.execute_input":"2022-07-21T06:06:39.374427Z","iopub.status.idle":"2022-07-21T06:06:39.918681Z","shell.execute_reply.started":"2022-07-21T06:06:39.374394Z","shell.execute_reply":"2022-07-21T06:06:39.917799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_06\nmodel_04 with K Neighbour Regressor","metadata":{}},{"cell_type":"code","source":"# Data\nX_whole_06 = train_df[num_cols + cat_cols_low_card + cat_cols_hi_card]\ny_whole_06 = train_df['Precio']\n\nX_train_06, X_valid_06, y_train_06, y_valid_06 = train_test_split(X_whole_06,\n                                                                 y_whole_06,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_06 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_06 = Pipeline([\n    ('scaler', StandardScaler())\n])\n\npreprocessor_06 = ColumnTransformer([\n    ('num', numerical_transformer_06, num_cols),\n    ('cat', categorical_transformer_06, cat_cols_low_card + cat_cols_hi_card)\n])\n\n# Linear Regression\n\nkn_06 = Pipeline([\n    ('preprocessor', preprocessor_06),\n    ('kn', KNeighborsRegressor(n_neighbors=5))\n])\n\nkn_06.fit(X_train_06, y_train_06)\n\n\ny_fit_06 = kn_06.predict(X_train_06)\ny_pred_06 = kn_06.predict(X_valid_06)\n\nrmse_fit_06 = mean_squared_error(y_train_06, y_fit_06, squared=False)\nrmse_pred_06 = mean_squared_error(y_valid_06, y_pred_06, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_06:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_06:.2f}\")\n\nsns.regplot(x=y_valid_06, y=y_pred_06);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_06,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_06,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_06-y_pred_06,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_06,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:39.950005Z","iopub.execute_input":"2022-07-21T06:06:39.950501Z","iopub.status.idle":"2022-07-21T06:06:40.565181Z","shell.execute_reply.started":"2022-07-21T06:06:39.950464Z","shell.execute_reply":"2022-07-21T06:06:40.564308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_07\nmodel_04 wit Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"# Data\nX_whole_07 = train_df[num_cols + cat_cols_low_card + cat_cols_hi_card]\ny_whole_07 = train_df['Precio']\n\nX_train_07, X_valid_07, y_train_07, y_valid_07 = train_test_split(X_whole_07,\n                                                                 y_whole_07,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_07 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_07 = Pipeline([\n    ('scaler', StandardScaler())\n])\n\npreprocessor_07 = ColumnTransformer([\n    ('num', numerical_transformer_07, num_cols),\n    ('cat', categorical_transformer_07, cat_cols_low_card + cat_cols_hi_card)\n])\n\n# Linear Regression\n\nrf_07 = Pipeline([\n    ('preprocessor', preprocessor_07),\n    ('rf', RandomForestRegressor(n_estimators=100))\n])\n\nrf_07.fit(X_train_07, y_train_07)\n\n\ny_fit_07 = rf_07.predict(X_train_07)\ny_pred_07 = rf_07.predict(X_valid_07)\n\nrmse_fit_07 = mean_squared_error(y_train_07, y_fit_07, squared=False)\nrmse_pred_07 = mean_squared_error(y_valid_07, y_pred_07, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_07:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_07:.2f}\")\n\nsns.regplot(x=y_valid_07, y=y_pred_07);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_07,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_07,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_07-y_pred_07,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_07,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:41.722517Z","iopub.execute_input":"2022-07-21T06:06:41.723119Z","iopub.status.idle":"2022-07-21T06:06:43.161383Z","shell.execute_reply.started":"2022-07-21T06:06:41.723086Z","shell.execute_reply":"2022-07-21T06:06:43.160476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_08\nRandom Forest Regressor, Standard Scaler no high card columns","metadata":{}},{"cell_type":"code","source":"# Data\nX_whole_08 = train_df[num_cols + cat_cols_low_card]\ny_whole_08 = train_df['Precio']\n\nX_train_08, X_valid_08, y_train_08, y_valid_08 = train_test_split(X_whole_08,\n                                                                 y_whole_08,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_08 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_08 = Pipeline([\n    ('scaler', StandardScaler())\n])\n\npreprocessor_08 = ColumnTransformer([\n    ('num', numerical_transformer_08, num_cols),\n    ('cat', categorical_transformer_08, cat_cols_low_card)\n])\n\n# Linear Regression\n\nrf_08 = Pipeline([\n    ('preprocessor', preprocessor_08),\n    ('rf', RandomForestRegressor(n_estimators=100))\n])\n\nrf_08.fit(X_train_08, y_train_08)\n\n\ny_fit_08 = rf_08.predict(X_train_08)\ny_pred_08 = rf_08.predict(X_valid_08)\n\nrmse_fit_08 = mean_squared_error(y_train_08, y_fit_08, squared=False)\nrmse_pred_08 = mean_squared_error(y_valid_08, y_pred_08, squared=False)\n\nprint(f\"RMSE Training:    {rmse_fit_08:.2f}\")\nprint(f\"RMSE Validation:  {rmse_pred_08:.2f}\")\n\nsns.regplot(x=y_valid_08, y=y_pred_08);\nplt.xlabel('y_valid')\nplt.ylabel('y_pred')\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_08,  color='red', marker='o', linestyle='None', label='y_train')\nax = plt.plot(range(190), y_pred_08,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend()\n\nplt.figure(figsize=(20,6))\nax = plt.plot(range(190), y_valid_08-y_pred_08,  color='blue', marker='o', label='y_valid - y_pred')\n# ax = plt.plot(range(190), y_pred_08,  color='green', marker='x', linestyle='None', label='y_pred')\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:06:43.163386Z","iopub.execute_input":"2022-07-21T06:06:43.164253Z","iopub.status.idle":"2022-07-21T06:06:44.300680Z","shell.execute_reply.started":"2022-07-21T06:06:43.164216Z","shell.execute_reply":"2022-07-21T06:06:44.299180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model_09\nNeural Network","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Data\nX_whole_09 = train_df[num_cols + cat_cols_low_card + cat_cols_hi_card]\ny_whole_09 = train_df['Precio']\n\nX_train_09, X_valid_09, y_train_09, y_valid_09 = train_test_split(X_whole_09,\n                                                                 y_whole_09,\n                                                                 test_size=0.2,\n                                                                 random_state=0)\n\n# Preprocessing Pipeline\ncategorical_transformer_09 = Pipeline([\n    ('oh_encoder', OneHotEncoder(handle_unknown='ignore', drop='first'))\n])\n\nnumerical_transformer_09 = Pipeline([\n    ('scaler', StandardScaler())\n])\n\npreprocessor_09 = ColumnTransformer([\n    ('num', numerical_transformer_09, num_cols),\n    ('cat', categorical_transformer_09, cat_cols_low_card + cat_cols_hi_card)\n])\n\npreprocessor_09.fit(X_valid_09)\nX_train_09 = preprocessor_09.transform(X_train_09).toarray()\nX_valid_09 = preprocessor_09.transform(X_valid_09).toarray()\nX_test_09 = preprocessor_09.transform(test_df).toarray()\n\n\nDROPOUT_RATE = 0.2\n\nprint('Model Setup')\n\nmodel = tf.keras.Sequential([\n    tf.keras.Input(shape=(X_train_09.shape[1], )),\n#     tf.keras.layers.Dropout(DROPOUT_RATE),\n    tf.keras.layers.Dense(32, 'relu'),\n    tf.keras.layers.Dropout(DROPOUT_RATE),\n    tf.keras.layers.Normalization(),\n    tf.keras.layers.Dense(1024, 'relu'),\n    tf.keras.layers.Dropout(DROPOUT_RATE),\n    tf.keras.layers.Normalization(),\n    tf.keras.layers.Dense(1024, 'relu'),\n    tf.keras.layers.Dropout(DROPOUT_RATE),\n    tf.keras.layers.Normalization(),\n    tf.keras.layers.Dense(64, 'relu'),\n    tf.keras.layers.Dropout(DROPOUT_RATE),\n#     tf.keras.layers.Normalization(),\n#     tf.keras.layers.Dense(128, 'relu'),\n#     tf.keras.layers.Dropout(DROPOUT_RATE),\n#     tf.keras.layers.Normalization(),\n#     tf.keras.layers.Dense(64, 'relu'),\n#     tf.keras.layers.Dropout(DROPOUT_RATE),\n#     tf.keras.layers.Normalization(),\n#     tf.keras.layers.Dense(32, 'relu'),\n#     tf.keras.layers.Dropout(DROPOUT_RATE),\n    tf.keras.layers.Dense(1, 'linear')\n])\n\nmodel.summary()\n\nprint('Model Compile')\n\nmodel.compile(loss='mse',\n             optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n             metrics=[tf.keras.metrics.RootMeanSquaredError()])\n\nprint('Model Fit')\n\nhistory = model.fit(x=X_train_09,\n                   y=y_train_09,\n                   validation_data=(X_valid_09, y_valid_09),\n                   epochs=1000,\n                   verbose=0,\n                   batch_size=100)\n\nprint(f\"Last RMSE:            {history.history['root_mean_squared_error'][-1]:.2f}\")\nprint(f\"Last Validation RMSE: {history.history['val_root_mean_squared_error'][-1]:.2f}\")  \n\nrmse_fit_09 = history.history['root_mean_squared_error'][-1]\nrmse_pred_09 = history.history['val_root_mean_squared_error'][-1]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:14:54.527416Z","iopub.execute_input":"2022-07-21T06:14:54.527800Z","iopub.status.idle":"2022-07-21T06:21:36.069148Z","shell.execute_reply.started":"2022-07-21T06:14:54.527771Z","shell.execute_reply":"2022-07-21T06:21:36.067693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot Utility\ndef plot_graphs(history, string):\n  plt.plot(history.history[string])\n  plt.plot(history.history['val_'+string])\n  plt.xlabel(\"Epochs\")\n  plt.ylabel(string)\n  plt.legend([string, 'val_'+string])\n  plt.show()\n\n    \nprint(f\"Last RMSE:            {history.history['root_mean_squared_error'][-1]:.2f}\")\nprint(f\"Last Validation RMSE: {history.history['val_root_mean_squared_error'][-1]:.2f}\")  \nprint(f\"Accuracy on Validation:   {model.evaluate(X_valid_09, y_valid_09)[1]:.2f}\")\n    \n# Plot the accuracy and loss history\nplot_graphs(history, 'root_mean_squared_error')\nplot_graphs(history, 'loss')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.071555Z","iopub.execute_input":"2022-07-21T06:21:36.071914Z","iopub.status.idle":"2022-07-21T06:21:36.545571Z","shell.execute_reply.started":"2022-07-21T06:21:36.071885Z","shell.execute_reply":"2022-07-21T06:21:36.544219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_09.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.547132Z","iopub.execute_input":"2022-07-21T06:21:36.547414Z","iopub.status.idle":"2022-07-21T06:21:36.555029Z","shell.execute_reply.started":"2022-07-21T06:21:36.547387Z","shell.execute_reply":"2022-07-21T06:21:36.553502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison","metadata":{}},{"cell_type":"code","source":"rmse_fit_list = []\nrmse_pred_list = []\nfor i in range(1,10):\n    rmse_fit_list.append('rmse_fit_0' + str(i))\n    rmse_pred_list.append('rmse_pred_0' + str(i))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.560904Z","iopub.execute_input":"2022-07-21T06:21:36.561299Z","iopub.status.idle":"2022-07-21T06:21:36.569232Z","shell.execute_reply.started":"2022-07-21T06:21:36.561268Z","shell.execute_reply":"2022-07-21T06:21:36.567808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(rmse_pred_list)):\n    print('model_0'+str(i+1) + ' - ' + f\"RMSE Training: {globals()[rmse_fit_list[i]]:.2f} - RMSE Validation: {globals()[rmse_pred_list[i]]:.2f}\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.570236Z","iopub.execute_input":"2022-07-21T06:21:36.570543Z","iopub.status.idle":"2022-07-21T06:21:36.583718Z","shell.execute_reply.started":"2022-07-21T06:21:36.570514Z","shell.execute_reply":"2022-07-21T06:21:36.583074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Predictions on Test Set","metadata":{}},{"cell_type":"code","source":"# model_07\ny_pred_test = rf_07.predict(test_df)\n\nsubmission_df = pd.DataFrame({\n    'Id': test_df.index,\n    'Precio': y_pred_test\n})\n\nsubmission_df.to_csv('nb_submission_07.csv', index=False)\nprint(y_pred_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.585196Z","iopub.execute_input":"2022-07-21T06:21:36.585510Z","iopub.status.idle":"2022-07-21T06:21:36.627785Z","shell.execute_reply.started":"2022-07-21T06:21:36.585483Z","shell.execute_reply":"2022-07-21T06:21:36.626321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_06\ny_pred_test = kn_06.predict(test_df)\n\nsubmission_df = pd.DataFrame({\n    'Id': test_df.index,\n    'Precio': y_pred_test\n})\n\nsubmission_df.to_csv('nb_submission_02.csv', index=False)\nprint(y_pred_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.629721Z","iopub.execute_input":"2022-07-21T06:21:36.630173Z","iopub.status.idle":"2022-07-21T06:21:36.664901Z","shell.execute_reply.started":"2022-07-21T06:21:36.630142Z","shell.execute_reply":"2022-07-21T06:21:36.664142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_09\ny_pred_test = model.predict(X_test_09)\n\nsubmission_df = pd.DataFrame({\n    'Id': test_df.index,\n    'Precio': y_pred_test.flatten()\n})\n\nsubmission_df.to_csv('nb_submission_09.csv', index=False)\nprint(y_pred_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T06:21:36.666151Z","iopub.execute_input":"2022-07-21T06:21:36.666885Z","iopub.status.idle":"2022-07-21T06:21:36.948258Z","shell.execute_reply.started":"2022-07-21T06:21:36.666854Z","shell.execute_reply":"2022-07-21T06:21:36.947187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}