{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T09:23:05.404697Z","iopub.execute_input":"2022-08-10T09:23:05.405118Z","iopub.status.idle":"2022-08-10T09:23:05.413181Z","shell.execute_reply.started":"2022-08-10T09:23:05.405080Z","shell.execute_reply":"2022-08-10T09:23:05.411920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV\nfrom xgboost import XGBRegressor\nimport eli5\nfrom eli5.sklearn import PermutationImportance\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.425747Z","iopub.execute_input":"2022-08-10T09:23:05.426383Z","iopub.status.idle":"2022-08-10T09:23:05.432349Z","shell.execute_reply.started":"2022-08-10T09:23:05.426346Z","shell.execute_reply":"2022-08-10T09:23:05.431458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#add data\ntrain_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv').drop('Id',axis=1)\ntest_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.454186Z","iopub.execute_input":"2022-08-10T09:23:05.454922Z","iopub.status.idle":"2022-08-10T09:23:05.507370Z","shell.execute_reply.started":"2022-08-10T09:23:05.454871Z","shell.execute_reply":"2022-08-10T09:23:05.506117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for later use\ntest_ids = test_data.pop('Id')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.512993Z","iopub.execute_input":"2022-08-10T09:23:05.513691Z","iopub.status.idle":"2022-08-10T09:23:05.541879Z","shell.execute_reply.started":"2022-08-10T09:23:05.513609Z","shell.execute_reply":"2022-08-10T09:23:05.540719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.543956Z","iopub.execute_input":"2022-08-10T09:23:05.544624Z","iopub.status.idle":"2022-08-10T09:23:05.554025Z","shell.execute_reply.started":"2022-08-10T09:23:05.544576Z","shell.execute_reply":"2022-08-10T09:23:05.552740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.584685Z","iopub.execute_input":"2022-08-10T09:23:05.585310Z","iopub.status.idle":"2022-08-10T09:23:05.696956Z","shell.execute_reply.started":"2022-08-10T09:23:05.585275Z","shell.execute_reply":"2022-08-10T09:23:05.695753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.699080Z","iopub.execute_input":"2022-08-10T09:23:05.701356Z","iopub.status.idle":"2022-08-10T09:23:05.825895Z","shell.execute_reply.started":"2022-08-10T09:23:05.701315Z","shell.execute_reply":"2022-08-10T09:23:05.824583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.827825Z","iopub.execute_input":"2022-08-10T09:23:05.828757Z","iopub.status.idle":"2022-08-10T09:23:05.854280Z","shell.execute_reply.started":"2022-08-10T09:23:05.828707Z","shell.execute_reply":"2022-08-10T09:23:05.853321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.856790Z","iopub.execute_input":"2022-08-10T09:23:05.857602Z","iopub.status.idle":"2022-08-10T09:23:05.882776Z","shell.execute_reply.started":"2022-08-10T09:23:05.857565Z","shell.execute_reply":"2022-08-10T09:23:05.880443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"na_not_missing_cols = []\nwith open('../input/house-prices-advanced-regression-techniques/data_description.txt') as file:\n    curr_label = None\n    for line in file:\n        if ':' in line:\n            curr_label = line.split(':')[0]\n        elif 'NA' in line:\n            na_not_missing_cols.append(curr_label)\ntrain_data[na_not_missing_cols].info()\nfor col in na_not_missing_cols:\n    train_data[col].fillna('NA',inplace=True)\n    test_data[col].fillna('NA',inplace=True)\nprint('Above cols in train/test contain nulls after processing: ' + str((train_data[na_not_missing_cols].isnull().any().any())) or (test_data[na_not_missing_cols].isnull().any().any()))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.884252Z","iopub.execute_input":"2022-08-10T09:23:05.884892Z","iopub.status.idle":"2022-08-10T09:23:05.916494Z","shell.execute_reply.started":"2022-08-10T09:23:05.884852Z","shell.execute_reply":"2022-08-10T09:23:05.915254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mark nulls\nna_cols = [col for col in train_data.columns if train_data[col].isnull().any()]\n\nnums = list(train_data.select_dtypes(exclude='object').columns)\ncats = list(train_data.select_dtypes(exclude=np.number).columns)\nnums.remove('SalePrice') # target\nnums.remove('MSSubClass') # categorical\n\nimputer = SimpleImputer() # default mean\ntrain_data[nums] = pd.DataFrame(imputer.fit_transform(train_data[nums]),columns=nums)\ntest_data[nums] = pd.DataFrame(imputer.transform(test_data[nums]),columns=nums) # reuse mean from training\nimputer = SimpleImputer(strategy='most_frequent')\ntrain_data[cats] = pd.DataFrame(imputer.fit_transform(train_data[cats]),columns=cats)\ntest_data[cats] = pd.DataFrame(imputer.transform(test_data[cats]),columns=cats) # reuse mode from training\nfor col in na_cols: # mark nulls (binary-encoded categorical features)\n    train_data['na_'+col] = train_data[col].isnull().astype('int64')\n    test_data['na_'+col] = test_data[col].isnull().astype('int64')\nprint('Any nulls in train/test after processing: ' + str((train_data.isnull().any().any())) or (test_data.isnull().any().any()))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:05.918555Z","iopub.execute_input":"2022-08-10T09:23:05.919059Z","iopub.status.idle":"2022-08-10T09:23:06.022179Z","shell.execute_reply.started":"2022-08-10T09:23:05.919016Z","shell.execute_reply":"2022-08-10T09:23:06.020891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search = ['Street',\n          'LotShape',\n          'Utilities',\n          'LandSlope',\n          'BsmtFinType1',\n          'BsmtFinType2',\n          'BsmtExposure',\n          'CentralAir',\n          'Functional'\n          'PavedDrive'] # looked at data_description.txt to determine which were obviously ordinal\nordinal_cols = []\nall_categories = []\nwith open('../input/house-prices-advanced-regression-techniques/data_description.txt') as file:\n    searching = False\n    curr_label = None\n    curr_categories = []\n    for line in file:\n        line = line.strip('\\n').strip()\n        if line == '':\n            if curr_categories: # empty string and non-empty list indicates end of categories\n                ordinal_cols.append(curr_label)\n                all_categories.append(curr_categories)\n                searching = False\n                curr_label = None\n                curr_categories = []\n        elif ':' in line: # checking for ordinal categorical variables\n            curr_label = line.split(':')[0]\n            if 'Overall' in curr_label:\n                pass\n            elif 'Q' in curr_label:\n                searching = True\n            elif curr_label[-4:] == 'Cond':\n                searching = True\n            else:\n                for string in search:\n                    if string in curr_label:\n                        search.remove(string)\n                        searching = True\n            if not searching:\n                curr_label = None\n        elif searching: # not an empty string but searching means currently on a category\n            split = line.split()\n            if len(split)>1:\n                curr_categories.append(split[0])\n                \nfor i,label in enumerate(ordinal_cols):\n    print(label + '-'*(15-len(label)) + str(all_categories[i]))\n    cats.remove(label)\n    \nencoder = OrdinalEncoder(categories=all_categories) # thankfully no errors caused by unknown values (typos)\n                \ntrain_data[ordinal_cols] = encoder.fit_transform(train_data[ordinal_cols])\ntest_data[ordinal_cols] = encoder.transform(test_data[ordinal_cols])\n\ntrain_data[ordinal_cols].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:06.024460Z","iopub.execute_input":"2022-08-10T09:23:06.024876Z","iopub.status.idle":"2022-08-10T09:23:06.114941Z","shell.execute_reply.started":"2022-08-10T09:23:06.024841Z","shell.execute_reply":"2022-08-10T09:23:06.113876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data[cats].nunique()) # none of them need target encoding\n# graphing to see if there is anything worth dropping\ndef bar_plot(col):\n    grouped = train_data.SalePrice.groupby(train_data[col]).mean().reset_index()\n    sns.barplot(x=grouped[col],y=grouped.SalePrice,order=grouped.sort_values('SalePrice')[col])\n    plt.title(col+' vs. SalePrice')\n    plt.xticks(rotation=60)\n\nplt.figure(figsize=(20,40))\nplt.subplots_adjust(wspace=0.5,hspace=1)\ni=1\nfor col in cats:\n    plt.subplot(9,3,i)\n    bar_plot(col)\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:23:06.117171Z","iopub.execute_input":"2022-08-10T09:23:06.117540Z","iopub.status.idle":"2022-08-10T09:23:10.009451Z","shell.execute_reply.started":"2022-08-10T09:23:06.117506Z","shell.execute_reply":"2022-08-10T09:23:10.008216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}