{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport math\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport sys\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T09:40:53.419021Z","iopub.execute_input":"2022-07-24T09:40:53.419712Z","iopub.status.idle":"2022-07-24T09:40:53.434142Z","shell.execute_reply.started":"2022-07-24T09:40:53.419609Z","shell.execute_reply":"2022-07-24T09:40:53.432797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")\ndf_test = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.479583Z","iopub.execute_input":"2022-07-24T09:40:53.479993Z","iopub.status.idle":"2022-07-24T09:40:53.527126Z","shell.execute_reply.started":"2022-07-24T09:40:53.479961Z","shell.execute_reply":"2022-07-24T09:40:53.525834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def isNAN(x):\n    return isinstance(x, float) and math.isnan(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.539716Z","iopub.execute_input":"2022-07-24T09:40:53.540111Z","iopub.status.idle":"2022-07-24T09:40:53.546429Z","shell.execute_reply.started":"2022-07-24T09:40:53.540067Z","shell.execute_reply":"2022-07-24T09:40:53.545076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def CompareNAN(column1, column2):\n    column1 = [not isNAN(x) for x in column1]\n    column2 = [not isNAN(x) for x in column2]\n    return column1 == column2","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.549088Z","iopub.execute_input":"2022-07-24T09:40:53.550079Z","iopub.status.idle":"2022-07-24T09:40:53.561604Z","shell.execute_reply.started":"2022-07-24T09:40:53.550025Z","shell.execute_reply":"2022-07-24T09:40:53.560307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_irrelevent_columns(train, test):\n    removed_columns = []\n    for column in train.columns:\n        if len([x for x in train[column] == train[column].mode()[0] if x]) / len(train[column]) > 0.99:\n            train.drop(column, axis = 1, inplace = True)\n            test.drop(column, axis = 1, inplace = True)\n            removed_columns.append(column)\n    \n    return removed_columns","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.598253Z","iopub.execute_input":"2022-07-24T09:40:53.598984Z","iopub.status.idle":"2022-07-24T09:40:53.608148Z","shell.execute_reply.started":"2022-07-24T09:40:53.598929Z","shell.execute_reply":"2022-07-24T09:40:53.606833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NANToString(train, test, column, add_indicator = False, indicator_name = \"\"):    \n    train.loc[:,column] = [x if not isNAN(x) else \"NAN\" for x in train[column]]\n    test.loc[:,column] = [x if not isNAN(x) else \"NAN\" for x in test[column]]\n    \n    if add_indicator:\n        indicator_name = indicator_name if indicator_name else column\n        train[\"is_\"+indicator_name+\"_exists\"] = [not x == \"NAN\" for x in train[column]]\n        test[\"is_\"+indicator_name+\"_exists\"] = [not x == \"NAN\" for x in test[column]]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.717382Z","iopub.execute_input":"2022-07-24T09:40:53.718238Z","iopub.status.idle":"2022-07-24T09:40:53.725537Z","shell.execute_reply.started":"2022-07-24T09:40:53.718193Z","shell.execute_reply":"2022-07-24T09:40:53.724550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NANToMode(train, test, column, add_indicator = False, indicator_name = \"\"):\n    \n    indicator_name = indicator_name if indicator_name else column\n    if add_indicator:\n        train[\"is_\"+indicator_name+\"_exists\"] = [not isNAN(x) for x in train[column]]\n        test[\"is_\"+indicator_name+\"_exists\"] = [not isNAN(x) for x in test[column]]\n    \n    mode_value = train[column].mode()[0]\n    \n    train.loc[:,column] = [x if not isNAN(x) else mode_value for x in train[column]]\n    test.loc[:,column] = [x if not isNAN(x) and x in train[column] else mode_value for x in test[column]]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.776766Z","iopub.execute_input":"2022-07-24T09:40:53.777742Z","iopub.status.idle":"2022-07-24T09:40:53.785389Z","shell.execute_reply.started":"2022-07-24T09:40:53.777679Z","shell.execute_reply":"2022-07-24T09:40:53.784174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NANToMedian(train, test, column, add_indicator = False):\n    \n    if add_indicator:\n        train[\"is_\"+column+\"_exists\"] = [not isNAN(x) for x in train[column]]\n        test[\"is_\"+column+\"_exists\"] = [not isNAN(x) for x in test[column]]\n    \n    median_value = train[column].median()\n    \n    train.loc[:,column] = [x if not isNAN(x) else median_value for x in train[column]]\n    test.loc[:,column] = [x if not isNAN(x) else median_value for x in test[column]]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.861350Z","iopub.execute_input":"2022-07-24T09:40:53.861764Z","iopub.status.idle":"2022-07-24T09:40:53.868189Z","shell.execute_reply.started":"2022-07-24T09:40:53.861728Z","shell.execute_reply":"2022-07-24T09:40:53.867170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NANToZero(train, test, column, add_indicator = False):\n    \n    if add_indicator:\n        train[\"is_\"+column+\"_exists\"] = [not isNAN(x) for x in train[column]]\n        test[\"is_\"+column+\"_exists\"] = [not isNAN(x) for x in test[column]]\n    \n    train.loc[:,column] = [x if not isNAN(x) else 0 for x in train[column]]\n    test.loc[:,column] = [x if not isNAN(x) else 0 for x in test[column]]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.919170Z","iopub.execute_input":"2022-07-24T09:40:53.919574Z","iopub.status.idle":"2022-07-24T09:40:53.926464Z","shell.execute_reply.started":"2022-07-24T09:40:53.919540Z","shell.execute_reply":"2022-07-24T09:40:53.925161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ProcessMiscFeature(train, test):  \n    \n    train.loc[:,\"MiscFeature\"] = [x if not isinstance(x, float) else \"NAN\" for x in train.MiscFeature]\n    test.loc[:,\"MiscFeature\"] = [x if not isinstance(x, float) and x in set(train.MiscFeature) else \"NAN\" for x in test.MiscFeature]\n    \n    misc_train = pd.get_dummies(train.MiscFeature, prefix='MiscFeature') \n    \n    if \"MiscFeature_NAN\" in misc_train.columns:\n        misc_train.drop(\"MiscFeature_NAN\", inplace = True, axis = 1)\n    misc_train.drop(\"MiscFeature_Othr\", inplace = True, axis = 1)\n    \n    misc_test = pd.get_dummies(test.MiscFeature, prefix='MiscFeature')  \n     \n    if \"MiscFeature_NAN\" in misc_test.columns:\n        misc_test.drop(\"MiscFeature_NAN\", inplace = True, axis = 1)\n    misc_test.drop(\"MiscFeature_Othr\", inplace = True, axis = 1)\n    \n    train = pd.concat([train, misc_train], axis = 1)\n    test = pd.concat([test,misc_test], axis = 1)\n    \n    return train, test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:53.985236Z","iopub.execute_input":"2022-07-24T09:40:53.985742Z","iopub.status.idle":"2022-07-24T09:40:53.994561Z","shell.execute_reply.started":"2022-07-24T09:40:53.985699Z","shell.execute_reply":"2022-07-24T09:40:53.993553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def FinalPreprocess(train, test):\n    for column in train.columns:\n        if column in test.columns:\n            NANToMode(train, test, column)\n        else:\n            mode_value = train[column].mode()[0]\n            test[column] = [mode_value for x in range(test.shape[0])]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.051161Z","iopub.execute_input":"2022-07-24T09:40:54.051587Z","iopub.status.idle":"2022-07-24T09:40:54.059520Z","shell.execute_reply.started":"2022-07-24T09:40:54.051545Z","shell.execute_reply":"2022-07-24T09:40:54.057850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndf_train.drop(\"SalePrice\", axis = True, inplace = True)\n\nNANToString(df_train, df_test, \"Alley\", add_indicator = True)\nNANToString(df_train, df_test, \"BsmtCond\", add_indicator = True, indicator_name = \"Bsmt\")\nNANToString(df_train, df_test, \"BsmtExposure\")\nNANToString(df_train, df_test, \"BsmtFinType1\")\nNANToString(df_train, df_test, \"BsmtFinType2\")\nNANToString(df_train, df_test, \"BsmtQual\")\nNANToString(df_train, df_test, \"Fence\", add_indicator = True)\nNANToString(df_train, df_test, \"FireplaceQu\", add_indicator = True)\nNANToString(df_train, df_test, \"GarageCond\", add_indicator = True, indicator_name = \"Garage\")\nNANToString(df_train, df_test, \"GarageFinish\")\nNANToString(df_train, df_test, \"GarageQual\")\nNANToString(df_train, df_test, \"GarageType\")\nNANToString(df_train, df_test, \"GarageYrBlt\")\nNANToZero(df_train, df_test, \"MasVnrArea\")\nNANToString(df_train, df_test, \"MasVnrType\", add_indicator = True, indicator_name = \"Vnr\")\nNANToString(df_train, df_test, \"PoolQC\")\ndf_train, df_test = ProcessMiscFeature(df_train, df_test)\n\nNANToMedian(df_train, df_test, \"LotFrontage\")\nFinalPreprocess(df_train, df_test)\nremoved_columns = remove_irrelevent_columns(df_train, df_test)\nprint(removed_columns)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.149811Z","iopub.execute_input":"2022-07-24T09:40:54.150205Z","iopub.status.idle":"2022-07-24T09:40:54.160846Z","shell.execute_reply.started":"2022-07-24T09:40:54.150172Z","shell.execute_reply":"2022-07-24T09:40:54.159436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_df(train, test):\n    \n    labels = train.SalePrice.copy()\n    \n    train.drop(\"Id\", axis = True, inplace = True)\n    train.drop(\"SalePrice\", axis = True, inplace = True)\n    test.drop(\"Id\", axis = True, inplace = True)\n    \n    NANToString(train, test, \"Alley\", add_indicator = True)\n    NANToString(train, test, \"BsmtCond\", add_indicator = True, indicator_name = \"Bsmt\")\n    NANToString(train, test, \"BsmtExposure\")\n    NANToString(train, test, \"BsmtFinType1\")\n    NANToString(train, test, \"BsmtFinType2\")\n    NANToString(train, test, \"BsmtQual\")\n    NANToString(train, test, \"Fence\", add_indicator = True)\n    NANToString(train, test, \"FireplaceQu\", add_indicator = True)\n    NANToString(train, test, \"GarageCond\", add_indicator = True, indicator_name = \"Garage\")\n    NANToString(train, test, \"GarageFinish\")\n    NANToString(train, test, \"GarageQual\")\n    NANToString(train, test, \"GarageType\")\n    NANToString(train, test, \"GarageYrBlt\")\n    NANToZero(train, test, \"MasVnrArea\")\n    NANToString(train, test, \"MasVnrType\", add_indicator = True, indicator_name = \"Vnr\")\n    NANToString(train, test, \"PoolQC\")\n    \n    train, test = ProcessMiscFeature(train, test)\n    \n    NANToMedian(train, test, \"LotFrontage\")\n    \n    FinalPreprocess(train, test)\n    remove_irrelevent_columns(train, test)\n    \n    return train, test, labels","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.251034Z","iopub.execute_input":"2022-07-24T09:40:54.251441Z","iopub.status.idle":"2022-07-24T09:40:54.268255Z","shell.execute_reply.started":"2022-07-24T09:40:54.251410Z","shell.execute_reply":"2022-07-24T09:40:54.266869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data, Test_Data, Labels = process_df(df_train, df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.344047Z","iopub.execute_input":"2022-07-24T09:40:54.344485Z","iopub.status.idle":"2022-07-24T09:40:54.702987Z","shell.execute_reply.started":"2022-07-24T09:40:54.344450Z","shell.execute_reply":"2022-07-24T09:40:54.701809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pprint\n\nIsNullCount = {}\ndf = Train_Data.isnull()\nfor column in Train_Data.columns:\n    IsNullCount[column] = sum(df[column])*100/len(df[column])\n\npp = pprint.PrettyPrinter()\npp.pprint(IsNullCount)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.705166Z","iopub.execute_input":"2022-07-24T09:40:54.705520Z","iopub.status.idle":"2022-07-24T09:40:54.732783Z","shell.execute_reply.started":"2022-07-24T09:40:54.705489Z","shell.execute_reply":"2022-07-24T09:40:54.731428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pprint\n\nIsNullCount = {}\ndf = Test_Data.isnull()\nfor column in Test_Data.columns:\n    IsNullCount[column] = sum(df[column])*100/len(df[column])\n\npp = pprint.PrettyPrinter()\npp.pprint(IsNullCount)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T09:40:54.734512Z","iopub.execute_input":"2022-07-24T09:40:54.735035Z","iopub.status.idle":"2022-07-24T09:40:54.765774Z","shell.execute_reply.started":"2022-07-24T09:40:54.734976Z","shell.execute_reply":"2022-07-24T09:40:54.764657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train_Data, Test_Data, Labels","metadata":{}},{"cell_type":"code","source":"Train_Data.to_pickle(\"Train_Data.pkl\")\nTest_Data.to_pickle(\"Test_Data.pkl\")\nLabels.to_pickle(\"Labels.pkl\")","metadata":{},"execution_count":null,"outputs":[]}]}