{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T15:46:33.608841Z","iopub.execute_input":"2022-08-11T15:46:33.609286Z","iopub.status.idle":"2022-08-11T15:46:33.626468Z","shell.execute_reply.started":"2022-08-11T15:46:33.609243Z","shell.execute_reply":"2022-08-11T15:46:33.625468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import fastai\nfrom fastai import *\nfrom fastai.tabular.all import *\nfastai.__version__","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.628528Z","iopub.execute_input":"2022-08-11T15:46:33.629778Z","iopub.status.idle":"2022-08-11T15:46:33.642341Z","shell.execute_reply.started":"2022-08-11T15:46:33.629719Z","shell.execute_reply":"2022-08-11T15:46:33.640979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/home-data-for-ml-course/train.csv')\ndf_test = pd.read_csv('../input/home-data-for-ml-course/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.644111Z","iopub.execute_input":"2022-08-11T15:46:33.645744Z","iopub.status.idle":"2022-08-11T15:46:33.708338Z","shell.execute_reply.started":"2022-08-11T15:46:33.645689Z","shell.execute_reply":"2022-08-11T15:46:33.706965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.711871Z","iopub.execute_input":"2022-08-11T15:46:33.714035Z","iopub.status.idle":"2022-08-11T15:46:33.731873Z","shell.execute_reply.started":"2022-08-11T15:46:33.713970Z","shell.execute_reply":"2022-08-11T15:46:33.730369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dep = 'SalePrice'\nprocs = [Categorify, FillMissing, Normalize] \nsplits = RandomSplitter(valid_pct = 0.2)(range_of(df_train))\ncont_names, cat_names = cont_cat_split(df_train, 1, dep)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.733599Z","iopub.execute_input":"2022-08-11T15:46:33.734174Z","iopub.status.idle":"2022-08-11T15:46:33.753084Z","shell.execute_reply.started":"2022-08-11T15:46:33.734140Z","shell.execute_reply":"2022-08-11T15:46:33.751899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_names","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.754386Z","iopub.execute_input":"2022-08-11T15:46:33.755499Z","iopub.status.idle":"2022-08-11T15:46:33.765593Z","shell.execute_reply.started":"2022-08-11T15:46:33.755465Z","shell.execute_reply":"2022-08-11T15:46:33.762196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cont_names","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.767083Z","iopub.execute_input":"2022-08-11T15:46:33.768286Z","iopub.status.idle":"2022-08-11T15:46:33.779000Z","shell.execute_reply.started":"2022-08-11T15:46:33.768239Z","shell.execute_reply":"2022-08-11T15:46:33.777581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def xs_y(df):\n    xs = df[cat_names+cont_names].copy()\n    return xs,df[dep] if dep in df else None\ntst_xs,_ = xs_y(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.781109Z","iopub.execute_input":"2022-08-11T15:46:33.781841Z","iopub.status.idle":"2022-08-11T15:46:33.798338Z","shell.execute_reply.started":"2022-08-11T15:46:33.781797Z","shell.execute_reply":"2022-08-11T15:46:33.796998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to = TabularPandas(df_train,procs,cat_names,cont_names,y_names=dep,splits=splits)\nto.show(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:33.800335Z","iopub.execute_input":"2022-08-11T15:46:33.801038Z","iopub.status.idle":"2022-08-11T15:46:34.040210Z","shell.execute_reply.started":"2022-08-11T15:46:33.800989Z","shell.execute_reply":"2022-08-11T15:46:34.039366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = to.dataloaders()\ndls.valid.show_batch()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:34.043965Z","iopub.execute_input":"2022-08-11T15:46:34.044774Z","iopub.status.idle":"2022-08-11T15:46:34.163914Z","shell.execute_reply.started":"2022-08-11T15:46:34.044737Z","shell.execute_reply":"2022-08-11T15:46:34.162716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, y_train = to.train.xs, to.train.y\nx_test, y_test = to.valid.xs, to.valid.y","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:34.165374Z","iopub.execute_input":"2022-08-11T15:46:34.166350Z","iopub.status.idle":"2022-08-11T15:46:34.178969Z","shell.execute_reply.started":"2022-08-11T15:46:34.166304Z","shell.execute_reply":"2022-08-11T15:46:34.177718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf_classifier = RandomForestRegressor(n_estimators=100, min_samples_leaf = 5, n_jobs=-1)\nrf_classifier.fit(x_train, y_train)\n\nfrom sklearn.metrics import mean_absolute_error\ny_pred = rf_classifier.predict(x_test)\nmean_absolute_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:34.180555Z","iopub.execute_input":"2022-08-11T15:46:34.181460Z","iopub.status.idle":"2022-08-11T15:46:35.050984Z","shell.execute_reply.started":"2022-08-11T15:46:34.181424Z","shell.execute_reply":"2022-08-11T15:46:35.049543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_test = TabularPandas(df_test, procs, cat_names, cont_names)\n# to_test['Id'] = df_test[\"Id\"]\ns = set(x_test)\ndifference = [x for x in to_test.xs if x not in s]\noutcome = rf_classifier.predict(to_test.xs.drop(columns=difference))\noutput= pd.DataFrame({'Id':df_test[\"Id\"], 'SalePrice': outcome.astype(float)})\noutput.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:35.054608Z","iopub.execute_input":"2022-08-11T15:46:35.055847Z","iopub.status.idle":"2022-08-11T15:46:35.348408Z","shell.execute_reply.started":"2022-08-11T15:46:35.055792Z","shell.execute_reply":"2022-08-11T15:46:35.346938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importance = pd.DataFrame(dict(cols=x_train.columns, imp=rf_classifier.feature_importances_));","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:35.350668Z","iopub.execute_input":"2022-08-11T15:46:35.351544Z","iopub.status.idle":"2022-08-11T15:46:35.460677Z","shell.execute_reply.started":"2022-08-11T15:46:35.351473Z","shell.execute_reply":"2022-08-11T15:46:35.459821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importance.sort_values(by=\"imp\", ascending=False).head(30)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:35.462414Z","iopub.execute_input":"2022-08-11T15:46:35.462862Z","iopub.status.idle":"2022-08-11T15:46:35.478816Z","shell.execute_reply.started":"2022-08-11T15:46:35.462819Z","shell.execute_reply":"2022-08-11T15:46:35.477584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#keep variables with importance > 0.001\nto_keep = list(importance[importance.imp >= 0.001].cols) + ['SalePrice']\nprint('Number of features to keep: ', len(to_keep))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:35.480304Z","iopub.execute_input":"2022-08-11T15:46:35.481296Z","iopub.status.idle":"2022-08-11T15:46:35.493938Z","shell.execute_reply.started":"2022-08-11T15:46:35.481247Z","shell.execute_reply":"2022-08-11T15:46:35.492622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to_keep.remove('LotFrontage_na')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:35.496397Z","iopub.execute_input":"2022-08-11T15:46:35.497294Z","iopub.status.idle":"2022-08-11T15:46:35.531503Z","shell.execute_reply.started":"2022-08-11T15:46:35.497249Z","shell.execute_reply":"2022-08-11T15:46:35.529970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_2 = df_train[to_keep]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:50.011908Z","iopub.execute_input":"2022-08-11T15:46:50.012920Z","iopub.status.idle":"2022-08-11T15:46:50.023396Z","shell.execute_reply.started":"2022-08-11T15:46:50.012844Z","shell.execute_reply":"2022-08-11T15:46:50.021576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dep = 'SalePrice'\nprocs = [Categorify, FillMissing, Normalize] \nsplits = RandomSplitter(valid_pct = 0.2)(range_of(df_train_2))\ncont_names, cat_names = cont_cat_split(df_train_2, 1, dep)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:57.982393Z","iopub.execute_input":"2022-08-11T15:46:57.982935Z","iopub.status.idle":"2022-08-11T15:46:57.997801Z","shell.execute_reply.started":"2022-08-11T15:46:57.982888Z","shell.execute_reply":"2022-08-11T15:46:57.996514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to = TabularPandas(df_train_2,procs,cat_names,cont_names,y_names=dep,splits=splits)\nto.show(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:58.837984Z","iopub.execute_input":"2022-08-11T15:46:58.838395Z","iopub.status.idle":"2022-08-11T15:46:58.966761Z","shell.execute_reply.started":"2022-08-11T15:46:58.838360Z","shell.execute_reply":"2022-08-11T15:46:58.965668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = to.dataloaders()\ndls.valid.show_batch()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:46:59.440410Z","iopub.execute_input":"2022-08-11T15:46:59.440851Z","iopub.status.idle":"2022-08-11T15:46:59.502005Z","shell.execute_reply.started":"2022-08-11T15:46:59.440816Z","shell.execute_reply":"2022-08-11T15:46:59.501025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, y_train = to.train.xs, to.train.y\nx_test, y_test = to.valid.xs, to.valid.y","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:47:00.282218Z","iopub.execute_input":"2022-08-11T15:47:00.283962Z","iopub.status.idle":"2022-08-11T15:47:00.294476Z","shell.execute_reply.started":"2022-08-11T15:47:00.283916Z","shell.execute_reply":"2022-08-11T15:47:00.293475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf_classifier = RandomForestRegressor(n_estimators=100, min_samples_leaf = 5, n_jobs=-1)\nrf_classifier.fit(x_train, y_train)\n\nfrom sklearn.metrics import mean_absolute_error\ny_pred = rf_classifier.predict(x_test)\nmean_absolute_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:47:07.294167Z","iopub.execute_input":"2022-08-11T15:47:07.294572Z","iopub.status.idle":"2022-08-11T15:47:08.059774Z","shell.execute_reply.started":"2022-08-11T15:47:07.294539Z","shell.execute_reply":"2022-08-11T15:47:08.058710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_test = TabularPandas(df_test, procs, cat_names, cont_names)\n# to_test['Id'] = df_test[\"Id\"]\ns = set(x_test)\ndifference = [x for x in to_test.xs if x not in s]\noutcome = rf_classifier.predict(to_test.xs.drop(columns=difference))\noutput= pd.DataFrame({'Id':df_test[\"Id\"], 'SalePrice': outcome.astype(float)})\noutput.to_csv('./submission_features_removed_RF_regressor.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:47:04.369110Z","iopub.execute_input":"2022-08-11T15:47:04.369979Z","iopub.status.idle":"2022-08-11T15:47:04.623920Z","shell.execute_reply.started":"2022-08-11T15:47:04.369928Z","shell.execute_reply":"2022-08-11T15:47:04.622504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}