{"cells":[{"metadata":{"trusted":true,"_uuid":"0637b9817480f61811d4c80b85d769f155ca663a"},"cell_type":"markdown","source":"Hello All ,\nThis kernel is realted to the iowa dataset where we use cardinality to hot encode the string colums and use imputation to fit to the NAN values .Please have a look at the code below : \n\n**Your submission scored 17266.76666, which is an improvement of your previous score of 18282.02181. Great job!**\nWe hope to improve the model further as the course progresses :-).\n\n"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3b87a95f0848537909ec68927b41da8e067eaa3e"},"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\n#load data\niowa_train_path='../input/train.csv'\niowa_test_path='../input/test.csv'\niowa_train_data=pd.read_csv(iowa_train_path)\niowa_test_data=pd.read_csv(iowa_test_path)\n#set the values \ntrain_x = iowa_train_data.drop(['SalePrice'], axis=1)\ntrain_y=iowa_train_data.SalePrice\ntest_x=iowa_test_data\n#now we wont be dropping the colums with the text \n#instead use the one hot encoding \n#for train set \ntrain_low_cardinality_cols = [cname for cname in train_x.columns if \n                                train_x[cname].nunique() < 10 and\n                                train_x[cname].dtype == \"object\"]\ntrain_numeric_cols = [cname for cname in train_x.columns if \n                                train_x[cname].dtype in ['int64', 'float64']]\ntrain_total_cols=train_low_cardinality_cols+train_numeric_cols\nprint(len(train_total_cols))\n#thus we have dropped 3 columns with low cardinality \ntrain_x_cardinal = train_x[train_total_cols]\ntest_x_cardinal = test_x[train_total_cols]\n#test set \ntrain_x_one_hot_encoded=pd.get_dummies(train_x_cardinal)\ntest_x_one_hot_encoded=pd.get_dummies(test_x_cardinal)\n#now aligning the two \ntrain_x_final, test_x_final = train_x_one_hot_encoded.align(test_x_one_hot_encoded,\n                                                                    join='left', \n                                                              axis=1)\n#import the imputer \nfrom sklearn.impute import SimpleImputer\nmy_imputer = SimpleImputer()\n\nimputed_train_x_plus = train_x_final.copy()\nimputed_test_x_plus=test_x_final.copy()\ncols_with_missing = {col for col in train_x_final.columns \n                                 if train_x_final[col].isnull().any()}\nfor col in cols_with_missing:\n    imputed_train_x_plus[col + '_was_missing'] = imputed_train_x_plus[col].isnull()\n    imputed_test_x_plus[col + '_was_missing'] = imputed_test_x_plus[col].isnull()\n\n\n#start with the imputer operation\nimputed_train_x_plus = my_imputer.fit_transform(imputed_train_x_plus)\nimputed_test_x_plus=my_imputer.transform(imputed_test_x_plus)\n#now generate the model random forest \nmodel = RandomForestRegressor(random_state=1)\nmodel.fit(imputed_train_x_plus, train_y)\npredictions = model.predict(imputed_test_x_plus)\n\n#getting the data ready for submission \noutput = pd.DataFrame({'Id': test_x.Id,\n                       'SalePrice': predictions})\noutput.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d3d11093ad74f8e9d4867105d39a6d31ef83aef"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7555d3951383f209e7e528f2dcce89a8611329b1"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}