{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>**Random Forest with advanced data augmentation**</center>\n\n![](https://mljar.com/images/machine-learning/random_forest_logo.png)\n\n## **Index:**\n\n- [Importing necessary libraries](#import)\n- [Importing the data](#data)\n- [Raw Data Visualization](#vis)\n    - [Scatterplot of raw data](#scatt)\n    - [Null values heat-map](#heat)\n    - [Comparison of Null values between training and testing data](#barnull)\n- [Data Pre-processing](#preprocess)\n- [Splitting data into x (Values) and y (labels)](#split)\n- [Creating and fitting the model](#model)\n- [Submission](#submit)","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"\n## **Importing necessary libraries** <a id=\"import\"></a>","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport sklearn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T12:28:06.178033Z","iopub.execute_input":"2022-07-12T12:28:06.178453Z","iopub.status.idle":"2022-07-12T12:28:06.184056Z","shell.execute_reply.started":"2022-07-12T12:28:06.178418Z","shell.execute_reply":"2022-07-12T12:28:06.182871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Importing the data** <a id=\"data\"></a>","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:06.732071Z","iopub.execute_input":"2022-07-12T12:28:06.732650Z","iopub.status.idle":"2022-07-12T12:28:06.740393Z","shell.execute_reply.started":"2022-07-12T12:28:06.732604Z","shell.execute_reply":"2022-07-12T12:28:06.739228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataTrain = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ndataTrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:07.301053Z","iopub.execute_input":"2022-07-12T12:28:07.301823Z","iopub.status.idle":"2022-07-12T12:28:07.341247Z","shell.execute_reply.started":"2022-07-12T12:28:07.301786Z","shell.execute_reply":"2022-07-12T12:28:07.340053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataTest = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')\ndataTest.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:07.672760Z","iopub.execute_input":"2022-07-12T12:28:07.673711Z","iopub.status.idle":"2022-07-12T12:28:07.713202Z","shell.execute_reply.started":"2022-07-12T12:28:07.673674Z","shell.execute_reply":"2022-07-12T12:28:07.712297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Raw Data Visualization** <a id=\"vis\"></a>","metadata":{}},{"cell_type":"markdown","source":"#### ***Scatterplot of raw data*** <a id=\"scatt\"></a>","metadata":{}},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11.7,8.27)})\nsns.scatterplot(x=dataTrain.Id, y=dataTrain.SalePrice, size=dataTrain.SalePrice, hue=dataTrain.OverallCond, style=dataTrain.YrSold, sizes=(60,300), palette=\"magma\")\nplt.title('Scatterplot of raw data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:08.037706Z","iopub.execute_input":"2022-07-12T12:28:08.038728Z","iopub.status.idle":"2022-07-12T12:28:08.624888Z","shell.execute_reply.started":"2022-07-12T12:28:08.038686Z","shell.execute_reply":"2022-07-12T12:28:08.624045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Null values heat-map*** <a id=\"heat\"></a>","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, sharex=True, figsize=(20,10))\nsns.heatmap(ax=axes[0], yticklabels=False, data=dataTrain.isnull(), cbar=False, cmap=\"viridis\")\nsns.heatmap(ax=axes[1], yticklabels=False, data=dataTest.isnull(), cbar=False, cmap=\"tab20c\")\naxes[0].set_title('Heatmap of missing values in training data')\naxes[1].set_title('Heatmap of missing values in testing data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:08.626297Z","iopub.execute_input":"2022-07-12T12:28:08.626628Z","iopub.status.idle":"2022-07-12T12:28:09.776797Z","shell.execute_reply.started":"2022-07-12T12:28:08.626598Z","shell.execute_reply":"2022-07-12T12:28:09.775724Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Comparison of Null values between training and testing data*** <a id=\"barnull\"></a>","metadata":{}},{"cell_type":"code","source":"def show_values(axs, orient=\"v\", space=.01):\n    def _single(ax):\n        if orient == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height() + (p.get_height()*0.01)\n                value = '{:.1f}'.format(p.get_height())\n                ax.text(_x, _y, value, ha=\"center\") \n        elif orient == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height() - (p.get_height()*0.5)\n                value = '{:.1f}'.format(p.get_width())\n                ax.text(_x, _y, value, ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _single(ax)\n    else:\n        _single(axs)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T12:28:09.778580Z","iopub.execute_input":"2022-07-12T12:28:09.778966Z","iopub.status.idle":"2022-07-12T12:28:09.787936Z","shell.execute_reply.started":"2022-07-12T12:28:09.778924Z","shell.execute_reply":"2022-07-12T12:28:09.786778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, sharex=True, figsize=(20,10))\nnanTrain = {}\nfor column in dataTrain.columns[1:]:\n    perc =  dataTrain[column].isna().sum()/len(dataTrain[column])\n    if perc >= 0.01:\n        nanTrain[str(column)] = perc\nnanTrain = {key: value*100 for key, value in sorted(nanTrain.items(), key=lambda item: item[1], reverse=True)}\na = sns.barplot(ax=axes[0], y=list(nanTrain.keys()), x=list(nanTrain.values()), palette=\"coolwarm\", ci=None)\nplt.xlabel(\"NaN Values (%)\")\nplt.ylabel(\"Labels\")\nplt.title('NaN values in training data:')\n#===================================================================================================\nnanTest = {}\nfor column in dataTest.columns[1:]:\n    perc =  dataTest[column].isna().sum()/len(dataTest[column])\n    if perc >= 0.01:\n        nanTest[str(column)] = perc\nnanTest = {key: value*100 for key, value in sorted(nanTest.items(), key=lambda item: item[1], reverse=True)}\nb = sns.barplot(ax=axes[1], y=list(nanTest.keys()), x=list(nanTest.values()), palette=\"flare\", ci=None)\n\naxes[0].set_title('Missing data in training set')\naxes[1].set_title('Missing data in training set')\naxes[0].set_xlabel('NaN Values (%)')\naxes[0].set_ylabel('Labels')\naxes[1].set_xlabel('NaN Values (%)')\naxes[1].set_ylabel('Labels')\n\nshow_values(a, \"h\", space=0.3)\nshow_values(b, \"h\", space=0.3)\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T12:28:09.789394Z","iopub.execute_input":"2022-07-12T12:28:09.790957Z","iopub.status.idle":"2022-07-12T12:28:10.448575Z","shell.execute_reply.started":"2022-07-12T12:28:09.790913Z","shell.execute_reply":"2022-07-12T12:28:10.447253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Data Pre-processing*** <a id=\"preprocess\"></a>","metadata":{}},{"cell_type":"markdown","source":"#### Checking Correlation between labels and the target label.","metadata":{}},{"cell_type":"code","source":"THRESHOLD = 0.5\n\ndata = dataTrain.corr()[\"SalePrice\"].sort_values(ascending=False)\nindices = data.index\nlabels = []\ncorr = []\nfor i in range(1, len(indices)):\n    if data[indices[i]]>THRESHOLD:\n        labels.append(indices[i])\n        corr.append(data[i])\nsns.barplot(x=corr, y=labels)\nplt.title('Lables with correlation coefficient > Threshold (0.5)')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T12:28:10.452420Z","iopub.execute_input":"2022-07-12T12:28:10.453116Z","iopub.status.idle":"2022-07-12T12:28:10.738775Z","shell.execute_reply.started":"2022-07-12T12:28:10.453067Z","shell.execute_reply":"2022-07-12T12:28:10.737597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dropping the columns that have insignificant correlation with out target variable (unnecessary columns).","metadata":{}},{"cell_type":"code","source":"unnecessary = []\nlab = dataTrain.SalePrice\nidCol = dataTest.Id\ndataTrain = dataTrain.drop(columns=[str(item) for item in dataTrain.columns[1:] if str(item) not in labels])\ndataTest = dataTest.drop(columns=[str(item) for item in dataTest.columns[1:] if str(item) not in labels])\ndataTrain = dataTrain.drop(columns=['Id'])\ndataTest = dataTest.drop(columns=['Id'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:10.740183Z","iopub.execute_input":"2022-07-12T12:28:10.740994Z","iopub.status.idle":"2022-07-12T12:28:10.751601Z","shell.execute_reply.started":"2022-07-12T12:28:10.740956Z","shell.execute_reply":"2022-07-12T12:28:10.750494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Filling the null values with backward fill. It will backward fill the NaN values that are present in the pandas dataframe.","metadata":{}},{"cell_type":"code","source":"dataTrain = dataTrain.fillna(method='bfill')\ndataTest = dataTest.fillna(method='bfill')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:10.753138Z","iopub.execute_input":"2022-07-12T12:28:10.753463Z","iopub.status.idle":"2022-07-12T12:28:10.767998Z","shell.execute_reply.started":"2022-07-12T12:28:10.753436Z","shell.execute_reply":"2022-07-12T12:28:10.766996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking if further NaN values persist.","metadata":{}},{"cell_type":"code","source":"sum(dataTrain.isnull().sum()), sum(dataTest.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:10.769380Z","iopub.execute_input":"2022-07-12T12:28:10.769890Z","iopub.status.idle":"2022-07-12T12:28:10.783648Z","shell.execute_reply.started":"2022-07-12T12:28:10.769849Z","shell.execute_reply":"2022-07-12T12:28:10.782451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Splitting data into x (Values) and y (labels)*** <a id=\"split\"></a>","metadata":{}},{"cell_type":"code","source":"yTrain = lab\nxTest = dataTest.to_numpy()\nxTrain = dataTrain.to_numpy()\nxTrain.shape, yTrain.shape, xTest.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:10.784982Z","iopub.execute_input":"2022-07-12T12:28:10.785307Z","iopub.status.idle":"2022-07-12T12:28:10.793056Z","shell.execute_reply.started":"2022-07-12T12:28:10.785281Z","shell.execute_reply":"2022-07-12T12:28:10.792249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Creating and fitting the model*** <a id=\"model\"></a>","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(max_depth=15)\nmodel.fit(xTrain,yTrain)\npreds = model.predict(xTrain)\nprint('R2 Score: ', sklearn.metrics.r2_score(yTrain,preds)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:28:10.794335Z","iopub.execute_input":"2022-07-12T12:28:10.794833Z","iopub.status.idle":"2022-07-12T12:28:13.687400Z","shell.execute_reply.started":"2022-07-12T12:28:10.794803Z","shell.execute_reply":"2022-07-12T12:28:13.686616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Submission*** <a id=\"submit\"></a>","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame(columns=[\"Id\",\"SalePrice\"])\noutput[\"Id\"] = idCol\noutput[\"SalePrice\"] = model.predict(xTest)\noutput[\"Id\"] = output[\"Id\"].astype(int)\noutput","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:36:56.728708Z","iopub.execute_input":"2022-07-12T12:36:56.729080Z","iopub.status.idle":"2022-07-12T12:36:57.114757Z","shell.execute_reply.started":"2022-07-12T12:36:56.729050Z","shell.execute_reply":"2022-07-12T12:36:57.113776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False)\nprint('Submission succesful!')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:37:01.128247Z","iopub.execute_input":"2022-07-12T12:37:01.128617Z","iopub.status.idle":"2022-07-12T12:37:01.138450Z","shell.execute_reply.started":"2022-07-12T12:37:01.128587Z","shell.execute_reply":"2022-07-12T12:37:01.137320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center><b>Thank you!</b></center>\n![](https://hippocampus-garden.com/static/b28b1df984ab3ae46f7396f81efba766/80893/ogp.jpg)","metadata":{}}]}