{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-28T17:43:53.755341Z","iopub.execute_input":"2022-06-28T17:43:53.755722Z","iopub.status.idle":"2022-06-28T17:43:53.764827Z","shell.execute_reply.started":"2022-06-28T17:43:53.755691Z","shell.execute_reply":"2022-06-28T17:43:53.763569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sharing the Code for Basic EDA & Prediction (using Python) - Do upvote if it helps and share any feedback for improvement :-)","metadata":{}},{"cell_type":"code","source":"# importing basic libraries & setting directory\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:43:59.911730Z","iopub.execute_input":"2022-06-28T17:43:59.912137Z","iopub.status.idle":"2022-06-28T17:43:59.919193Z","shell.execute_reply.started":"2022-06-28T17:43:59.912102Z","shell.execute_reply":"2022-06-28T17:43:59.918128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import make_pipeline\nfrom sklearn.compose import make_column_transformer\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import KBinsDiscretizer, OneHotEncoder\n\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:44:02.247067Z","iopub.execute_input":"2022-06-28T17:44:02.247499Z","iopub.status.idle":"2022-06-28T17:44:02.253229Z","shell.execute_reply.started":"2022-06-28T17:44:02.247461Z","shell.execute_reply":"2022-06-28T17:44:02.252301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/delhihousepriceprediction/DHP_Test.csv')\ntrain = pd.read_csv('/kaggle/input/delhihousepriceprediction/DHP_Train_Revised.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:57:51.290735Z","iopub.execute_input":"2022-06-28T17:57:51.291168Z","iopub.status.idle":"2022-06-28T17:57:51.313514Z","shell.execute_reply.started":"2022-06-28T17:57:51.291130Z","shell.execute_reply":"2022-06-28T17:57:51.312016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:01.973288Z","iopub.execute_input":"2022-06-28T17:45:01.973889Z","iopub.status.idle":"2022-06-28T17:45:02.005546Z","shell.execute_reply.started":"2022-06-28T17:45:01.973848Z","shell.execute_reply":"2022-06-28T17:45:02.004700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:11.418462Z","iopub.execute_input":"2022-06-28T17:45:11.419272Z","iopub.status.idle":"2022-06-28T17:45:11.438037Z","shell.execute_reply.started":"2022-06-28T17:45:11.419226Z","shell.execute_reply":"2022-06-28T17:45:11.436708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the first 7 few rows\ntrain.head(7)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:19.914515Z","iopub.execute_input":"2022-06-28T17:45:19.914939Z","iopub.status.idle":"2022-06-28T17:45:19.937942Z","shell.execute_reply.started":"2022-06-28T17:45:19.914904Z","shell.execute_reply":"2022-06-28T17:45:19.936542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# showing the last 3 rows\ntrain.tail(3)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:24.604350Z","iopub.execute_input":"2022-06-28T17:45:24.604753Z","iopub.status.idle":"2022-06-28T17:45:24.627012Z","shell.execute_reply.started":"2022-06-28T17:45:24.604719Z","shell.execute_reply":"2022-06-28T17:45:24.625807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking column-wise count of null values\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:29.911710Z","iopub.execute_input":"2022-06-28T17:45:29.912106Z","iopub.status.idle":"2022-06-28T17:45:29.925465Z","shell.execute_reply.started":"2022-06-28T17:45:29.912071Z","shell.execute_reply":"2022-06-28T17:45:29.924107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test the same for TEST dataset also if any null\n\ntest.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:35.121985Z","iopub.execute_input":"2022-06-28T17:45:35.122465Z","iopub.status.idle":"2022-06-28T17:45:35.134899Z","shell.execute_reply.started":"2022-06-28T17:45:35.122419Z","shell.execute_reply":"2022-06-28T17:45:35.133940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the variable name and datatype of each column\ntrain.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:43.389891Z","iopub.execute_input":"2022-06-28T17:45:43.390418Z","iopub.status.idle":"2022-06-28T17:45:43.402461Z","shell.execute_reply.started":"2022-06-28T17:45:43.390367Z","shell.execute_reply":"2022-06-28T17:45:43.400928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# little extra details compared to dtypes()\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:48.375768Z","iopub.execute_input":"2022-06-28T17:45:48.376167Z","iopub.status.idle":"2022-06-28T17:45:48.404271Z","shell.execute_reply.started":"2022-06-28T17:45:48.376133Z","shell.execute_reply":"2022-06-28T17:45:48.402864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting an overall summary of all columns of the dataset\ntrain.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:45:54.037583Z","iopub.execute_input":"2022-06-28T17:45:54.037988Z","iopub.status.idle":"2022-06-28T17:45:54.099567Z","shell.execute_reply.started":"2022-06-28T17:45:54.037952Z","shell.execute_reply":"2022-06-28T17:45:54.098334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gets details about the categorical variables also\ntrain.describe(include = 'all')","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:03.130467Z","iopub.execute_input":"2022-06-28T17:46:03.130927Z","iopub.status.idle":"2022-06-28T17:46:03.194314Z","shell.execute_reply.started":"2022-06-28T17:46:03.130888Z","shell.execute_reply":"2022-06-28T17:46:03.193014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Details for 1 column\n\ntrain.MEDV.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:10.173214Z","iopub.execute_input":"2022-06-28T17:46:10.173582Z","iopub.status.idle":"2022-06-28T17:46:10.186352Z","shell.execute_reply.started":"2022-06-28T17:46:10.173551Z","shell.execute_reply":"2022-06-28T17:46:10.184771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking head of a particular column of the dataset\ntrain['TAX'].head()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:15.587278Z","iopub.execute_input":"2022-06-28T17:46:15.587703Z","iopub.status.idle":"2022-06-28T17:46:15.595697Z","shell.execute_reply.started":"2022-06-28T17:46:15.587662Z","shell.execute_reply":"2022-06-28T17:46:15.594671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding unique values of a particular column of the dataset\ntrain['TAX'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:20.723374Z","iopub.execute_input":"2022-06-28T17:46:20.724202Z","iopub.status.idle":"2022-06-28T17:46:20.732836Z","shell.execute_reply.started":"2022-06-28T17:46:20.724145Z","shell.execute_reply":"2022-06-28T17:46:20.731770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding unique values of a particular column of the dataset and printing it as a list\ntrain['TAX'].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:26.027245Z","iopub.execute_input":"2022-06-28T17:46:26.027635Z","iopub.status.idle":"2022-06-28T17:46:26.039066Z","shell.execute_reply.started":"2022-06-28T17:46:26.027586Z","shell.execute_reply":"2022-06-28T17:46:26.037479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shows Dimension of the dataset - Rows and Columns\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:35.253577Z","iopub.execute_input":"2022-06-28T17:46:35.254125Z","iopub.status.idle":"2022-06-28T17:46:35.260891Z","shell.execute_reply.started":"2022-06-28T17:46:35.254073Z","shell.execute_reply":"2022-06-28T17:46:35.259709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just checking the number of rows of the dataset\ntrain.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:39.385849Z","iopub.execute_input":"2022-06-28T17:46:39.386274Z","iopub.status.idle":"2022-06-28T17:46:39.393236Z","shell.execute_reply.started":"2022-06-28T17:46:39.386238Z","shell.execute_reply":"2022-06-28T17:46:39.392061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just checking the number of columns of the dataset\ntrain.shape[1]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:43.999079Z","iopub.execute_input":"2022-06-28T17:46:43.999976Z","iopub.status.idle":"2022-06-28T17:46:44.005922Z","shell.execute_reply.started":"2022-06-28T17:46:43.999930Z","shell.execute_reply":"2022-06-28T17:46:44.005061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  count of Number of Non-Missing/Not NUll Values for each Variable\ntrain.count()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:48.643492Z","iopub.execute_input":"2022-06-28T17:46:48.643868Z","iopub.status.idle":"2022-06-28T17:46:48.653750Z","shell.execute_reply.started":"2022-06-28T17:46:48.643837Z","shell.execute_reply":"2022-06-28T17:46:48.652918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to find duplicates if any \ntrain.duplicated()\n# train.drop_duplicates()   # to remove the duplicates\n# train.drop_duplicates(subset='Col-Name-1')  # if want to remove duplicate from a particualr column\n## which will keep the 1st occurrence and delete the rest rows\n# train.drop_duplicates(subset='Col-Name-1' , inplace=False) # if we dont want to remove permanently from the file","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:46:55.165571Z","iopub.execute_input":"2022-06-28T17:46:55.165990Z","iopub.status.idle":"2022-06-28T17:46:55.180302Z","shell.execute_reply.started":"2022-06-28T17:46:55.165954Z","shell.execute_reply":"2022-06-28T17:46:55.179090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train.duplicated()].shape","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:47:00.347535Z","iopub.execute_input":"2022-06-28T17:47:00.348449Z","iopub.status.idle":"2022-06-28T17:47:00.359696Z","shell.execute_reply.started":"2022-06-28T17:47:00.348406Z","shell.execute_reply":"2022-06-28T17:47:00.358457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Detecting Outliers\n\n    it can be done by determining the determine the upper cut off and the lower cutoff using any of the 3 Methods            \n                                        Percentile Method\n                                        IQR Method\n                                        Standard Deviation Method    ","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:47:17.078978Z","iopub.execute_input":"2022-06-28T17:47:17.079380Z","iopub.status.idle":"2022-06-28T17:47:17.086734Z","shell.execute_reply.started":"2022-06-28T17:47:17.079342Z","shell.execute_reply":"2022-06-28T17:47:17.085463Z"}}},{"cell_type":"code","source":"# determine the upper cut off and the lower cutoff using IQR(Interquartile range) Method\n\np0=train.MEDV.min()\np100=train.MEDV.max()\nq1=train.MEDV.quantile(0.25)\nq2=train.MEDV.quantile(0.5)\nq3=train.MEDV.quantile(0.75)\niqr=q3-q1\n# Now since we have all the values we need to find the lower cutoff(lc) and the upper cutoff(uc) of the values.\nlc = q1 - 1.5*iqr\nuc = q3 + 1.5*iqr\n\n# If lc < p0 → There are NO Outliers on the lower side\n# If uc > p100 → There are NO Outliers on the higher side\n\nprint( \"p0 = \" , p0 ,\", p100 = \" , p100 ,\", lc = \" , lc ,\", uc = \" , uc)\n\n# Finally doing a box plot of the Column/Variable of the dataset\ntrain.MEDV.plot(kind='box')","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:47:32.974127Z","iopub.execute_input":"2022-06-28T17:47:32.974520Z","iopub.status.idle":"2022-06-28T17:47:33.206922Z","shell.execute_reply.started":"2022-06-28T17:47:32.974486Z","shell.execute_reply":"2022-06-28T17:47:33.205701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Handling / Treating Outliers --- Do not worry about the data loss as here we are not going to remove any value from the variable but rather clip them\n\nif outlier on lower side - clip on lowerside or vice versa\n","metadata":{}},{"cell_type":"code","source":"# Here outlier was on upper side so clipped on upper side and then plotting again to see of outlier present or not\n\ntrain.MEDV.clip(upper=uc,inplace=True)\ntrain.MEDV.plot(kind='box')","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:47:54.722010Z","iopub.execute_input":"2022-06-28T17:47:54.722378Z","iopub.status.idle":"2022-06-28T17:47:54.893661Z","shell.execute_reply.started":"2022-06-28T17:47:54.722347Z","shell.execute_reply":"2022-06-28T17:47:54.892428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Detecting Missing Values¶","metadata":{}},{"cell_type":"code","source":"# seeing all the columns and records to see if any null/missing values\ntrain.isna()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:48:10.485221Z","iopub.execute_input":"2022-06-28T17:48:10.485649Z","iopub.status.idle":"2022-06-28T17:48:10.516426Z","shell.execute_reply.started":"2022-06-28T17:48:10.485589Z","shell.execute_reply":"2022-06-28T17:48:10.515241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# finding a count of total missing values\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:48:17.835344Z","iopub.execute_input":"2022-06-28T17:48:17.835746Z","iopub.status.idle":"2022-06-28T17:48:17.846717Z","shell.execute_reply.started":"2022-06-28T17:48:17.835710Z","shell.execute_reply":"2022-06-28T17:48:17.845824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# finding a % of the missing value column-wise\ntrain.isna().sum()/train.shape[0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Treating Missing Values ➼➼➼➼ handling / treating the missing values\n\n                                        ➼➼➼➼        Drop the variable\n                                        ➼➼➼➼        Drop the observation(s)\n                                        ➼➼➼➼        Missing Value Imputation\n\nHere we dont have any missing values - so nothing to do Else we could have done -- Data Imputation is done on the Series. Here we replace the missing values with some value which could be static, mean, median, mode, or an output of a predictive model\n\nIf it is a categorical variable, let’s impute the values by mode. If it is a continuous variable, let’s impute the values by mean.\n\nif the % of any column/variable is too high to impute say > 65% we can drop that column/variable also\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"## imputing using Mode()  if it was on a Continuous Variables\n\n# train.MEDV.mode()[0]\n# train.MEDV.fillna(train.MEDV.mode()[0],inplace=True)\n\n## or using a mean\n\n# train.MEDV.fillna(train.MEDV.mean()[0],inplace=True)\n\n## or dropping the column/variable if the % is too high for missing values\n\n# train.dropna(axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:48:54.945883Z","iopub.execute_input":"2022-06-28T17:48:54.946255Z","iopub.status.idle":"2022-06-28T17:48:54.951939Z","shell.execute_reply.started":"2022-06-28T17:48:54.946222Z","shell.execute_reply":"2022-06-28T17:48:54.951028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.MEDV.hist()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:48:59.193705Z","iopub.execute_input":"2022-06-28T17:48:59.194789Z","iopub.status.idle":"2022-06-28T17:48:59.348275Z","shell.execute_reply.started":"2022-06-28T17:48:59.194746Z","shell.execute_reply":"2022-06-28T17:48:59.347185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(7)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:05.044096Z","iopub.execute_input":"2022-06-28T17:49:05.044793Z","iopub.status.idle":"2022-06-28T17:49:05.070741Z","shell.execute_reply.started":"2022-06-28T17:49:05.044738Z","shell.execute_reply":"2022-06-28T17:49:05.069394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# renaming of single column - but not saving in the file (inplace = TRUE) will save it in the file\ntrain.rename(columns = {'RIVER_FLG': \"RIVER_FLG1\"} )\t","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:10.693296Z","iopub.execute_input":"2022-06-28T17:49:10.693727Z","iopub.status.idle":"2022-06-28T17:49:10.728159Z","shell.execute_reply.started":"2022-06-28T17:49:10.693691Z","shell.execute_reply":"2022-06-28T17:49:10.726873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"◘◘◘ Analysis Using Charts ➼➼➼➼\n\n                            ♥♥    Univariate Analysis     ➼➼➼➼        Here we take 1 variable and plot charts on it\n\n                                    ♥    Histogram    (For Continuous Variables)\n                                    ♥    BoxPlot     (For Continuous Variables)                                    \n\n                            ♥♥    Bivariate Analysis          ➼➼➼➼        Here we take 2 variable and plot charts on it (can be both Categorical and Numerical)\n\n                                    ♥    Scatter Plot    (For Numerical & Numerical Variables)\n                                    ♥    Correlation Matrix with a Heatmap     (find correlation between all the numeric variables)\n                                    ♥    HeatMap (to visualize the correlation between the different numerical columns of the data)\n                                    ♥    Bar & Line Chart                     (for Numerical & Categorical)\n                                    ♥    Cross Tab & HeatMap on top             (for Categorical & Categorical)","metadata":{}},{"cell_type":"code","source":"# Plotting Histogram for MEDV column from the dataset\ntrain.MEDV.hist()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:28.076822Z","iopub.execute_input":"2022-06-28T17:49:28.077209Z","iopub.status.idle":"2022-06-28T17:49:28.284194Z","shell.execute_reply.started":"2022-06-28T17:49:28.077177Z","shell.execute_reply":"2022-06-28T17:49:28.283310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# another method to plot Histogram\ntrain.MEDV.plot(kind='hist' , grid = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:36.475747Z","iopub.execute_input":"2022-06-28T17:49:36.476682Z","iopub.status.idle":"2022-06-28T17:49:36.706950Z","shell.execute_reply.started":"2022-06-28T17:49:36.476604Z","shell.execute_reply":"2022-06-28T17:49:36.705601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Another way to Plot HISTOGRAM using \"matplotlib\"\nplt.hist(train.MEDV)\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:41.328819Z","iopub.execute_input":"2022-06-28T17:49:41.329198Z","iopub.status.idle":"2022-06-28T17:49:41.541944Z","shell.execute_reply.started":"2022-06-28T17:49:41.329168Z","shell.execute_reply":"2022-06-28T17:49:41.540695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BoxPlot\ntrain.MEDV.plot(kind='box')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:46.785514Z","iopub.execute_input":"2022-06-28T17:49:46.786451Z","iopub.status.idle":"2022-06-28T17:49:46.950359Z","shell.execute_reply.started":"2022-06-28T17:49:46.786391Z","shell.execute_reply":"2022-06-28T17:49:46.949471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Another way to plot BoxPlot\nplt.boxplot(train.MEDV)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:52.337088Z","iopub.execute_input":"2022-06-28T17:49:52.337468Z","iopub.status.idle":"2022-06-28T17:49:52.461539Z","shell.execute_reply.started":"2022-06-28T17:49:52.337426Z","shell.execute_reply":"2022-06-28T17:49:52.460068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use of GroupBy for Categorical Variables to see Categorization & Distribution of data\ntrain.groupby('RAD').MEDV.count().plot(kind='pie')\t\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:49:58.354111Z","iopub.execute_input":"2022-06-28T17:49:58.354514Z","iopub.status.idle":"2022-06-28T17:49:58.518595Z","shell.execute_reply.started":"2022-06-28T17:49:58.354478Z","shell.execute_reply":"2022-06-28T17:49:58.517294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of RAD\nsns.countplot(train.RAD)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:03.252287Z","iopub.execute_input":"2022-06-28T17:50:03.252713Z","iopub.status.idle":"2022-06-28T17:50:03.455483Z","shell.execute_reply.started":"2022-06-28T17:50:03.252672Z","shell.execute_reply":"2022-06-28T17:50:03.454339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('ZN').RAD.count().plot(kind='pie')\t\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:08.891563Z","iopub.execute_input":"2022-06-28T17:50:08.891983Z","iopub.status.idle":"2022-06-28T17:50:09.156439Z","shell.execute_reply.started":"2022-06-28T17:50:08.891947Z","shell.execute_reply":"2022-06-28T17:50:09.154953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scatter Plot \n# plotting Numerical against Numerical\n\ntrain.plot(x='RIVER_FLG',y='TAX',kind = 'scatter')","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:14.483444Z","iopub.execute_input":"2022-06-28T17:50:14.484261Z","iopub.status.idle":"2022-06-28T17:50:14.686340Z","shell.execute_reply.started":"2022-06-28T17:50:14.484216Z","shell.execute_reply":"2022-06-28T17:50:14.685262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Correlation Matrix ➼➼➼➼ Finding a correlation between all the numeric variables.","metadata":{}},{"cell_type":"code","source":"train.select_dtypes(['float64' , 'int64']).corr()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:29.257437Z","iopub.execute_input":"2022-06-28T17:50:29.258229Z","iopub.status.idle":"2022-06-28T17:50:29.290493Z","shell.execute_reply.started":"2022-06-28T17:50:29.258185Z","shell.execute_reply":"2022-06-28T17:50:29.288879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Heatmap ➼➼➼➼ Creating a heatmap using Seaborn on the top of the correlation matrix obtained above to visualize the correlation, between the different numerical columns of the data","metadata":{}},{"cell_type":"code","source":"sns.heatmap(train.select_dtypes(['float64' , 'int64']).corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:42.582120Z","iopub.execute_input":"2022-06-28T17:50:42.582545Z","iopub.status.idle":"2022-06-28T17:50:43.154885Z","shell.execute_reply.started":"2022-06-28T17:50:42.582508Z","shell.execute_reply":"2022-06-28T17:50:43.153429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"◘ ◘ ◘ use of GroupBy for Numerical & Categorical Variables to see Composition of data ➼➼➼➼\n◘ ◘ ◘ use of Bar & Line Chart & Area Chart & Pie Chart ➼➼➼➼¶\n","metadata":{}},{"cell_type":"code","source":"# Bar Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\ntrain.groupby('CRIM').TAX.sum().plot(kind='bar')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:50:57.455975Z","iopub.execute_input":"2022-06-28T17:50:57.456359Z","iopub.status.idle":"2022-06-28T17:51:01.569386Z","shell.execute_reply.started":"2022-06-28T17:50:57.456325Z","shell.execute_reply":"2022-06-28T17:51:01.567777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Another BAR plot method show Comparison between CRIM (Categorical) and Tax (Numerical)\n\nsummary=train.groupby('CRIM').TAX.sum()\nplt.bar(x=summary.index , height=summary.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:08.462948Z","iopub.execute_input":"2022-06-28T17:51:08.463384Z","iopub.status.idle":"2022-06-28T17:51:09.401472Z","shell.execute_reply.started":"2022-06-28T17:51:08.463347Z","shell.execute_reply":"2022-06-28T17:51:09.400630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SNS method for BAR plot to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\nsns.barplot(x=summary.index , y=summary.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:13.315963Z","iopub.execute_input":"2022-06-28T17:51:13.316510Z","iopub.status.idle":"2022-06-28T17:51:18.355457Z","shell.execute_reply.started":"2022-06-28T17:51:13.316459Z","shell.execute_reply":"2022-06-28T17:51:18.354573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Line Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\ntrain.groupby('CRIM').TAX.sum().plot(kind='line')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:18.357688Z","iopub.execute_input":"2022-06-28T17:51:18.358077Z","iopub.status.idle":"2022-06-28T17:51:18.495316Z","shell.execute_reply.started":"2022-06-28T17:51:18.358043Z","shell.execute_reply":"2022-06-28T17:51:18.493352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pie Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\ntrain.groupby('CRIM').TAX.sum().plot(kind='pie')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:24.063960Z","iopub.execute_input":"2022-06-28T17:51:24.064333Z","iopub.status.idle":"2022-06-28T17:51:27.071961Z","shell.execute_reply.started":"2022-06-28T17:51:24.064300Z","shell.execute_reply":"2022-06-28T17:51:27.070746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Area Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\ntrain.groupby('CRIM').TAX.sum().plot(kind='area')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:31.618177Z","iopub.execute_input":"2022-06-28T17:51:31.618607Z","iopub.status.idle":"2022-06-28T17:51:31.749868Z","shell.execute_reply.started":"2022-06-28T17:51:31.618568Z","shell.execute_reply":"2022-06-28T17:51:31.749000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Horizontal Bar Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\ntrain.groupby('CRIM').TAX.sum().plot(kind='barh')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:38.096761Z","iopub.execute_input":"2022-06-28T17:51:38.097242Z","iopub.status.idle":"2022-06-28T17:51:42.163385Z","shell.execute_reply.started":"2022-06-28T17:51:38.097197Z","shell.execute_reply":"2022-06-28T17:51:42.162250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SNS Box Plot Chart to show Comparison between CRIM (Categorical) and Tax (Numerical)\n\nsns.boxplot(x='CRIM',y='TAX',data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:44.455227Z","iopub.execute_input":"2022-06-28T17:51:44.455601Z","iopub.status.idle":"2022-06-28T17:51:52.125789Z","shell.execute_reply.started":"2022-06-28T17:51:44.455570Z","shell.execute_reply":"2022-06-28T17:51:52.124679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CrossTab to show Comparison between CRIM (Categorical) and MEDV (Numerical)\n\npd.crosstab(train.CRIM,train.MEDV)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:52.127504Z","iopub.execute_input":"2022-06-28T17:51:52.128038Z","iopub.status.idle":"2022-06-28T17:51:52.233014Z","shell.execute_reply.started":"2022-06-28T17:51:52.128004Z","shell.execute_reply":"2022-06-28T17:51:52.231960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# HeatMap on top of the Cross Tab\n\nsns.heatmap(pd.crosstab(train.CRIM,train.MEDV))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:51:55.965917Z","iopub.execute_input":"2022-06-28T17:51:55.966315Z","iopub.status.idle":"2022-06-28T17:51:56.773717Z","shell.execute_reply.started":"2022-06-28T17:51:55.966279Z","shell.execute_reply":"2022-06-28T17:51:56.772426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:01.524479Z","iopub.execute_input":"2022-06-28T17:52:01.524913Z","iopub.status.idle":"2022-06-28T17:52:01.535331Z","shell.execute_reply.started":"2022-06-28T17:52:01.524876Z","shell.execute_reply":"2022-06-28T17:52:01.534181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:06.651371Z","iopub.execute_input":"2022-06-28T17:52:06.651759Z","iopub.status.idle":"2022-06-28T17:52:06.666378Z","shell.execute_reply.started":"2022-06-28T17:52:06.651725Z","shell.execute_reply":"2022-06-28T17:52:06.665525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Engineering¶","metadata":{}},{"cell_type":"code","source":"# splitting the target and independent variables of the training set\n# droppng the MEDV column from the training dataset as it is missing in Test and trying to preduct it in the train dataset\n\nY = train['MEDV']\nX = train.drop(['MEDV'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:20.072120Z","iopub.execute_input":"2022-06-28T17:52:20.072521Z","iopub.status.idle":"2022-06-28T17:52:20.079815Z","shell.execute_reply.started":"2022-06-28T17:52:20.072482Z","shell.execute_reply":"2022-06-28T17:52:20.078629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the head of Y after the split\n\nY.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:32.810592Z","iopub.execute_input":"2022-06-28T17:52:32.811001Z","iopub.status.idle":"2022-06-28T17:52:32.819883Z","shell.execute_reply.started":"2022-06-28T17:52:32.810964Z","shell.execute_reply":"2022-06-28T17:52:32.818703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the head of X after the split\n\nX.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:37.728998Z","iopub.execute_input":"2022-06-28T17:52:37.729382Z","iopub.status.idle":"2022-06-28T17:52:37.750853Z","shell.execute_reply.started":"2022-06-28T17:52:37.729351Z","shell.execute_reply":"2022-06-28T17:52:37.749230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the dimension of X after the split\n\nX.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:47.513844Z","iopub.execute_input":"2022-06-28T17:52:47.514535Z","iopub.status.idle":"2022-06-28T17:52:47.521987Z","shell.execute_reply.started":"2022-06-28T17:52:47.514479Z","shell.execute_reply":"2022-06-28T17:52:47.520804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the dimension of Y after the split\n\nY.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:52.196515Z","iopub.execute_input":"2022-06-28T17:52:52.197055Z","iopub.status.idle":"2022-06-28T17:52:52.205692Z","shell.execute_reply.started":"2022-06-28T17:52:52.197002Z","shell.execute_reply":"2022-06-28T17:52:52.204256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Y dont have any columns after the split - so converting it to a DataFrame to have atleast 1 column\n\nY = pd.DataFrame(Y)\nY.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:52:58.057315Z","iopub.execute_input":"2022-06-28T17:52:58.058051Z","iopub.status.idle":"2022-06-28T17:52:58.066445Z","shell.execute_reply.started":"2022-06-28T17:52:58.058013Z","shell.execute_reply":"2022-06-28T17:52:58.064934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting to list for numerical and categorical variables\n\nnumerical_features = [c for c, dtype in zip(X.columns, X.dtypes)\n                     if dtype.kind in ['i','f'] and c !='ID']\ncategorical_features = [c for c, dtype in zip(X.columns, X.dtypes)\n                     if dtype.kind not in ['i','f']]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:53:03.386414Z","iopub.execute_input":"2022-06-28T17:53:03.386840Z","iopub.status.idle":"2022-06-28T17:53:03.393585Z","shell.execute_reply.started":"2022-06-28T17:53:03.386802Z","shell.execute_reply":"2022-06-28T17:53:03.392686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking list of numerical variables\n\nnumerical_features","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:53:07.649439Z","iopub.execute_input":"2022-06-28T17:53:07.650152Z","iopub.status.idle":"2022-06-28T17:53:07.657761Z","shell.execute_reply.started":"2022-06-28T17:53:07.650108Z","shell.execute_reply":"2022-06-28T17:53:07.656785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking list of categorical variables\n\ncategorical_features","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:53:12.911652Z","iopub.execute_input":"2022-06-28T17:53:12.912073Z","iopub.status.idle":"2022-06-28T17:53:12.919841Z","shell.execute_reply.started":"2022-06-28T17:53:12.912037Z","shell.execute_reply":"2022-06-28T17:53:12.918548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# splitting to test and train dataset again for the TRAINING dataset\n\nfrom sklearn.model_selection import train_test_split\n\n# create train test split\nX_train, X_test, Y_train, Y_test = train_test_split( X,  Y, test_size=0.3, random_state=4)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:53:22.517254Z","iopub.execute_input":"2022-06-28T17:53:22.517658Z","iopub.status.idle":"2022-06-28T17:53:22.525985Z","shell.execute_reply.started":"2022-06-28T17:53:22.517593Z","shell.execute_reply":"2022-06-28T17:53:22.524804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since this is for prediction of Continuous Value (and not state of variable) - so Regression Decision Tree to be used\n\nimport sklearn.tree as tree\nreg = tree.DecisionTreeRegressor(max_depth = 3)\nreg.fit(X_train, Y_train)\nreg.score(X_test, Y_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:53:30.240766Z","iopub.execute_input":"2022-06-28T17:53:30.241147Z","iopub.status.idle":"2022-06-28T17:53:30.286836Z","shell.execute_reply.started":"2022-06-28T17:53:30.241114Z","shell.execute_reply":"2022-06-28T17:53:30.285883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the Regression Score of the model - which is basically MSE of the Test dataset of the Model\n\nreg.score(X_test, Y_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:54:50.721055Z","iopub.execute_input":"2022-06-28T17:54:50.721434Z","iopub.status.idle":"2022-06-28T17:54:50.733728Z","shell.execute_reply.started":"2022-06-28T17:54:50.721402Z","shell.execute_reply":"2022-06-28T17:54:50.732579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# giving a name to the predictors and printing top 5 predictors by feature importance \n\nreg.feature_importances_\npd.Series(reg.feature_importances_, index = X.columns) . sort_values(ascending = False) . head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:54:56.131065Z","iopub.execute_input":"2022-06-28T17:54:56.131858Z","iopub.status.idle":"2022-06-28T17:54:56.141958Z","shell.execute_reply.started":"2022-06-28T17:54:56.131807Z","shell.execute_reply":"2022-06-28T17:54:56.140898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = reg.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:55:01.143793Z","iopub.execute_input":"2022-06-28T17:55:01.144204Z","iopub.status.idle":"2022-06-28T17:55:01.151296Z","shell.execute_reply.started":"2022-06-28T17:55:01.144170Z","shell.execute_reply":"2022-06-28T17:55:01.149980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:55:39.137246Z","iopub.execute_input":"2022-06-28T17:55:39.137613Z","iopub.status.idle":"2022-06-28T17:55:39.146327Z","shell.execute_reply.started":"2022-06-28T17:55:39.137581Z","shell.execute_reply":"2022-06-28T17:55:39.145217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading the TEST dataset\n\nprint(test.head(5))\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:07.207119Z","iopub.execute_input":"2022-06-28T17:58:07.207518Z","iopub.status.idle":"2022-06-28T17:58:07.219599Z","shell.execute_reply.started":"2022-06-28T17:58:07.207481Z","shell.execute_reply":"2022-06-28T17:58:07.218708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction of Test dataset based on the Model used on the Training Dataset\n\ny_pred_test = reg.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:10.266574Z","iopub.execute_input":"2022-06-28T17:58:10.267324Z","iopub.status.idle":"2022-06-28T17:58:10.274499Z","shell.execute_reply.started":"2022-06-28T17:58:10.267282Z","shell.execute_reply":"2022-06-28T17:58:10.273686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the values what is predicted on the test dataset based on the model that was applied on training set \n\ny_pred_test","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:17.707482Z","iopub.execute_input":"2022-06-28T17:58:17.708472Z","iopub.status.idle":"2022-06-28T17:58:17.717849Z","shell.execute_reply.started":"2022-06-28T17:58:17.708430Z","shell.execute_reply":"2022-06-28T17:58:17.716299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting the test predictions to a dataframe for storage\n\ny_pred_test_dataframe = pd.DataFrame(y_pred_test, columns=['MEDV'])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:23.084160Z","iopub.execute_input":"2022-06-28T17:58:23.084558Z","iopub.status.idle":"2022-06-28T17:58:23.091434Z","shell.execute_reply.started":"2022-06-28T17:58:23.084523Z","shell.execute_reply":"2022-06-28T17:58:23.089981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the 1st few rows\n\ny_pred_test_dataframe.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:26.902142Z","iopub.execute_input":"2022-06-28T17:58:26.902549Z","iopub.status.idle":"2022-06-28T17:58:26.913074Z","shell.execute_reply.started":"2022-06-28T17:58:26.902515Z","shell.execute_reply":"2022-06-28T17:58:26.912075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# populating the test dataset with target value\n\ntest_with_target_populated =  pd.concat([test, y_pred_test_dataframe], axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:30.036265Z","iopub.execute_input":"2022-06-28T17:58:30.037183Z","iopub.status.idle":"2022-06-28T17:58:30.044823Z","shell.execute_reply.started":"2022-06-28T17:58:30.037123Z","shell.execute_reply":"2022-06-28T17:58:30.043844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_output =  test_with_target_populated[[\"ID\", \"MEDV\"]]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:32.612551Z","iopub.execute_input":"2022-06-28T17:58:32.613031Z","iopub.status.idle":"2022-06-28T17:58:32.621158Z","shell.execute_reply.started":"2022-06-28T17:58:32.612991Z","shell.execute_reply":"2022-06-28T17:58:32.619862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_output.to_csv(\"sai_submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T17:58:34.187058Z","iopub.execute_input":"2022-06-28T17:58:34.190085Z","iopub.status.idle":"2022-06-28T17:58:34.200239Z","shell.execute_reply.started":"2022-06-28T17:58:34.190019Z","shell.execute_reply":"2022-06-28T17:58:34.198654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}