{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n\"\"\"\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\"\"\"\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T09:07:58.239933Z","iopub.execute_input":"2022-07-31T09:07:58.240512Z","iopub.status.idle":"2022-07-31T09:07:58.255012Z","shell.execute_reply.started":"2022-07-31T09:07:58.240438Z","shell.execute_reply":"2022-07-31T09:07:58.254375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **Importing Data Sets**","metadata":{}},{"cell_type":"code","source":"#Importing Dataset\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n#For finding mode\nfrom scipy import stats\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import StandardScaler\n\n\n\n\ndf_train_original = pd.read_csv(r'../input/house-prices-advanced-regression-techniques/train.csv')\ndf_test_original = pd.read_csv(r'../input/house-prices-advanced-regression-techniques/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:58.259006Z","iopub.execute_input":"2022-07-31T09:07:58.259446Z","iopub.status.idle":"2022-07-31T09:07:59.113818Z","shell.execute_reply.started":"2022-07-31T09:07:58.259413Z","shell.execute_reply":"2022-07-31T09:07:59.112704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Copying the datasets\n\ndf_train = df_train_original.copy()\ndf_test = df_test_original.copy()\nprint('Training data set shape: {}'.format(df_train.shape))\nprint('Testing data set shape: {}'.format(df_test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.115211Z","iopub.execute_input":"2022-07-31T09:07:59.115507Z","iopub.status.idle":"2022-07-31T09:07:59.123171Z","shell.execute_reply.started":"2022-07-31T09:07:59.115475Z","shell.execute_reply":"2022-07-31T09:07:59.121914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.125420Z","iopub.execute_input":"2022-07-31T09:07:59.125703Z","iopub.status.idle":"2022-07-31T09:07:59.164355Z","shell.execute_reply.started":"2022-07-31T09:07:59.125672Z","shell.execute_reply":"2022-07-31T09:07:59.163270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.165675Z","iopub.execute_input":"2022-07-31T09:07:59.165941Z","iopub.status.idle":"2022-07-31T09:07:59.192856Z","shell.execute_reply.started":"2022-07-31T09:07:59.165906Z","shell.execute_reply":"2022-07-31T09:07:59.191681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Null Values for Data Sets**","metadata":{}},{"cell_type":"code","source":"#Function for Checking Null Values of Dataset and showing graph \ndef Null_Analysis(df):    \n    columns_with_nullValues =  df_train.columns[df_train.isnull().any()] \n    columns_with_nullValues_count=  df_train[columns_with_nullValues].isnull().sum() \n    columns_with_nullValues_count_percentage= df_train[columns_with_nullValues].isnull().sum() * 100 / df_train.shape[0] \n\n\n\n    NullValues_Result = pd.concat([columns_with_nullValues_count,columns_with_nullValues_count_percentage], axis=1,join='inner') \n    NullValues_Result.columns = ['Count', 'Percentage'] \n    NullValues_Result['Percentage'] = round(NullValues_Result['Percentage'],2) \n    NullValues_Result\n    return NullValues_Result\n    \ndef Null_Analysis_Graph(df):\n    NullValues_Result= Null_Analysis(df)\n    NullValues_Result['Percentage'].hist(bins=10)\n    plt.xlabel('Missing Values Percentages')\n    plt.ylabel('Frequency')\n    plt.title('Histogram of Missing Values Percentages')\n    plt.show()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.194387Z","iopub.execute_input":"2022-07-31T09:07:59.194731Z","iopub.status.idle":"2022-07-31T09:07:59.206584Z","shell.execute_reply.started":"2022-07-31T09:07:59.194687Z","shell.execute_reply":"2022-07-31T09:07:59.205726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Training Data Set\nNull_Analysis(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.208630Z","iopub.execute_input":"2022-07-31T09:07:59.209611Z","iopub.status.idle":"2022-07-31T09:07:59.255354Z","shell.execute_reply.started":"2022-07-31T09:07:59.209559Z","shell.execute_reply":"2022-07-31T09:07:59.254374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Null_Analysis_Graph(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.256929Z","iopub.execute_input":"2022-07-31T09:07:59.257334Z","iopub.status.idle":"2022-07-31T09:07:59.517402Z","shell.execute_reply.started":"2022-07-31T09:07:59.257291Z","shell.execute_reply":"2022-07-31T09:07:59.516484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test Data Set\nNull_Analysis(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.518627Z","iopub.execute_input":"2022-07-31T09:07:59.518866Z","iopub.status.idle":"2022-07-31T09:07:59.552118Z","shell.execute_reply.started":"2022-07-31T09:07:59.518837Z","shell.execute_reply":"2022-07-31T09:07:59.551159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Null_Analysis_Graph(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.553586Z","iopub.execute_input":"2022-07-31T09:07:59.553834Z","iopub.status.idle":"2022-07-31T09:07:59.761179Z","shell.execute_reply.started":"2022-07-31T09:07:59.553804Z","shell.execute_reply":"2022-07-31T09:07:59.760024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Frequency of columns according to different dtypes","metadata":{}},{"cell_type":"code","source":"# store columns with specific data type\ninteger_columns = df_train.select_dtypes(include=['int64']).columns\nfloat_columns = df_train.select_dtypes(include=['float64']).columns\nobject_columns = df_train.select_dtypes(include=['object']).columns\n\nprint(\"No of Integer type columns: {}\".format(len(integer_columns)))\nprint(\"No of Float type columns: {}\".format(len(float_columns)))\nprint(\"No of String type columns: {}\".format(len(object_columns)))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.762685Z","iopub.execute_input":"2022-07-31T09:07:59.763035Z","iopub.status.idle":"2022-07-31T09:07:59.780138Z","shell.execute_reply.started":"2022-07-31T09:07:59.762986Z","shell.execute_reply":"2022-07-31T09:07:59.778770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data=df_train, x=\"SalePrice\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:07:59.781874Z","iopub.execute_input":"2022-07-31T09:07:59.782280Z","iopub.status.idle":"2022-07-31T09:08:00.178152Z","shell.execute_reply.started":"2022-07-31T09:07:59.782182Z","shell.execute_reply":"2022-07-31T09:08:00.177282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['SalePrice']].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:00.181525Z","iopub.execute_input":"2022-07-31T09:08:00.181787Z","iopub.status.idle":"2022-07-31T09:08:00.201310Z","shell.execute_reply.started":"2022-07-31T09:08:00.181755Z","shell.execute_reply":"2022-07-31T09:08:00.200447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for float_col in float_columns:\n    sns.relplot(data=df_train, x=float_col, y='SalePrice')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:00.202694Z","iopub.execute_input":"2022-07-31T09:08:00.203463Z","iopub.status.idle":"2022-07-31T09:08:01.146573Z","shell.execute_reply.started":"2022-07-31T09:08:00.203399Z","shell.execute_reply":"2022-07-31T09:08:01.145482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for integer_col in integer_columns:\n    sns.relplot(data=df_train, x=integer_col, y='SalePrice')","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-31T09:08:01.148040Z","iopub.execute_input":"2022-07-31T09:08:01.148395Z","iopub.status.idle":"2022-07-31T09:08:14.034063Z","shell.execute_reply.started":"2022-07-31T09:08:01.148359Z","shell.execute_reply":"2022-07-31T09:08:14.033043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Concatinating two dataframes**","metadata":{}},{"cell_type":"code","source":"#First we should remove 'SalePrice' from Training set before concatinating\ndf_train_SalePrice=df_train['SalePrice']\ndf_train.drop(['SalePrice'], axis=1, inplace=True)\n\nprint('Training data set shape: {}'.format(df_train.shape))\nprint('Testing data set shape: {}'.format(df_test.shape))\n\ndf_train_len = df_train.shape[0]\ndf_test_len = df_test.shape[0]\n\n#Concatinating Training and Test dataset\nframes = [df_train,df_test]\ndf = pd.concat(frames)\nprint('After Concatination')\nprint('Data set shape: {}'.format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.036006Z","iopub.execute_input":"2022-07-31T09:08:14.036378Z","iopub.status.idle":"2022-07-31T09:08:14.065999Z","shell.execute_reply.started":"2022-07-31T09:08:14.036332Z","shell.execute_reply":"2022-07-31T09:08:14.065030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Checking for null values**","metadata":{}},{"cell_type":"code","source":"columns_with_nullValues =  df.columns[df.isnull().any()]\ncolumns_with_nullValues_count=  df[columns_with_nullValues].isnull().sum()\ncolumns_with_nullValues_count_percentage= df[columns_with_nullValues].isnull().sum() * 100 / df.shape[0]\n\nNullValues_Result = pd.concat([columns_with_nullValues_count,columns_with_nullValues_count_percentage], axis=1,join='inner')\nNullValues_Result.columns = ['Count', 'Percentage']\nNullValues_Result['Percentage'] = round(NullValues_Result['Percentage'],2)\nNullValues_Result","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.067274Z","iopub.execute_input":"2022-07-31T09:08:14.067511Z","iopub.status.idle":"2022-07-31T09:08:14.131545Z","shell.execute_reply.started":"2022-07-31T09:08:14.067480Z","shell.execute_reply":"2022-07-31T09:08:14.130509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nNullValues_Result['Percentage'].hist(bins=10)\nplt.xlabel('Missing Values Percentages')\nplt.ylabel('Frequency')\nplt.title('Histogram of Missing Values Percentages')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.132731Z","iopub.execute_input":"2022-07-31T09:08:14.132963Z","iopub.status.idle":"2022-07-31T09:08:14.348310Z","shell.execute_reply.started":"2022-07-31T09:08:14.132934Z","shell.execute_reply":"2022-07-31T09:08:14.347353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dropping columns with large number of missing values","metadata":{}},{"cell_type":"code","source":"#From the histogram , dropping columns with more than 40percent missing values makes sense\ncolumns_to_drop = NullValues_Result[NullValues_Result['Percentage']>=40].index\n\n\nprint('Columns with more than 40% missing values:)')\nfor col in columns_to_drop:\n    print(col)\nprint('\\nTotal columns found: {}'.format(len(columns_to_drop)))\n\n#Dropping columns from the main dataset 'df'\ndf.drop(columns_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.349421Z","iopub.execute_input":"2022-07-31T09:08:14.349645Z","iopub.status.idle":"2022-07-31T09:08:14.362112Z","shell.execute_reply.started":"2022-07-31T09:08:14.349606Z","shell.execute_reply":"2022-07-31T09:08:14.361112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Removing the nullValues >40% from 'columns_with_nullValues'\ncolumns_with_nullValues_lessthan_40 = [col for col in columns_with_nullValues if col not in columns_to_drop]\n\n'''\nlen(columns_with_nullValues)=34\nlen(columns_to_drop)=5\n=>Result should be 34-5 = 29\n'''\n\nlen(columns_with_nullValues_lessthan_40)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.363793Z","iopub.execute_input":"2022-07-31T09:08:14.364790Z","iopub.status.idle":"2022-07-31T09:08:14.372517Z","shell.execute_reply.started":"2022-07-31T09:08:14.364725Z","shell.execute_reply":"2022-07-31T09:08:14.371628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can see what kind datatypes those columns with missing values has\ndf[columns_with_nullValues_lessthan_40].dtypes.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.374051Z","iopub.execute_input":"2022-07-31T09:08:14.375019Z","iopub.status.idle":"2022-07-31T09:08:14.387281Z","shell.execute_reply.started":"2022-07-31T09:08:14.374972Z","shell.execute_reply":"2022-07-31T09:08:14.386185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[columns_with_nullValues_lessthan_40].info()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.389121Z","iopub.execute_input":"2022-07-31T09:08:14.389727Z","iopub.status.idle":"2022-07-31T09:08:14.419897Z","shell.execute_reply.started":"2022-07-31T09:08:14.389677Z","shell.execute_reply":"2022-07-31T09:08:14.418820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#As those columns has either type as 'Object' or 'float64'. \n#For those missing values (Operation to perform) \n#    1.mode for 'Object'  2.mean for 'float64'\nfor col in columns_with_nullValues_lessthan_40:\n    res = df[[col]].dtypes == 'object'\n    #If data type of the particular column is 'object'\n    if res.bool():\n        #Find its mode and replace the missing values with the mode\n        col_mode_temp = stats.mode(df[col])\n        df[col] = df[col].fillna(col_mode_temp[0][0])\n        continue\n    #If data type of the particular column is 'float64' i.e not 'object'\n    \n    col_mean_temp = np.mean(df[col])\n    df[col] = df[col].fillna(col_mean_temp)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.421773Z","iopub.execute_input":"2022-07-31T09:08:14.422415Z","iopub.status.idle":"2022-07-31T09:08:14.513344Z","shell.execute_reply.started":"2022-07-31T09:08:14.422364Z","shell.execute_reply":"2022-07-31T09:08:14.512327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.514753Z","iopub.execute_input":"2022-07-31T09:08:14.515283Z","iopub.status.idle":"2022-07-31T09:08:14.522431Z","shell.execute_reply.started":"2022-07-31T09:08:14.515215Z","shell.execute_reply":"2022-07-31T09:08:14.521542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"count=0\nfor col in df.columns:\n    res = df[[col]].dtypes.isin(['float64','int64'])\n    if res.bool():\n        count+=1\nprint(count)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.523975Z","iopub.execute_input":"2022-07-31T09:08:14.524848Z","iopub.status.idle":"2022-07-31T09:08:14.602372Z","shell.execute_reply.started":"2022-07-31T09:08:14.524801Z","shell.execute_reply":"2022-07-31T09:08:14.601403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Selecting the categorical features and creating dummy variables","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.603554Z","iopub.execute_input":"2022-07-31T09:08:14.603790Z","iopub.status.idle":"2022-07-31T09:08:14.636956Z","shell.execute_reply.started":"2022-07-31T09:08:14.603763Z","shell.execute_reply":"2022-07-31T09:08:14.636047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extracing the categorical features\ncategorical_columns = df.select_dtypes(include=['object']).columns\n\n#Creating dummy variables for the extracted categorical features\ndummies = pd.get_dummies(df[categorical_columns], drop_first=True)\n\n#Dropping the categorical feature columns from the dataset , and then concatinating the dummies\ndf.drop(categorical_columns, axis=1, inplace=True)\ndf_final = pd.concat([df, dummies], axis = 'columns')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.639394Z","iopub.execute_input":"2022-07-31T09:08:14.640047Z","iopub.status.idle":"2022-07-31T09:08:14.700102Z","shell.execute_reply.started":"2022-07-31T09:08:14.640000Z","shell.execute_reply":"2022-07-31T09:08:14.699090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting the dataset to training and test back again","metadata":{}},{"cell_type":"code","source":"df_train = df_final.iloc[:df_train_len,1:]\ndf_test = df_final.iloc[df_test_len+1:,1:]\nprint('df_train shape: {}'.format(df_train.shape))\nprint('df_test shape: {}'.format(df_test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.702004Z","iopub.execute_input":"2022-07-31T09:08:14.702616Z","iopub.status.idle":"2022-07-31T09:08:14.718089Z","shell.execute_reply.started":"2022-07-31T09:08:14.702571Z","shell.execute_reply":"2022-07-31T09:08:14.717210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Null Values ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Correlation btw features\n\ndf['SalePrice']=df_train_original['SalePrice']\ncorrmat = df.corr()\n  \nf, ax = plt.subplots(figsize =(9, 8))\nsns.heatmap(corrmat, ax = ax, cmap =\"YlGnBu\", linewidths = 0.1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:14.719878Z","iopub.execute_input":"2022-07-31T09:08:14.720254Z","iopub.status.idle":"2022-07-31T09:08:15.973169Z","shell.execute_reply.started":"2022-07-31T09:08:14.720196Z","shell.execute_reply":"2022-07-31T09:08:15.972295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['SalePrice'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:15.974844Z","iopub.execute_input":"2022-07-31T09:08:15.975361Z","iopub.status.idle":"2022-07-31T09:08:16.236757Z","shell.execute_reply.started":"2022-07-31T09:08:15.975313Z","shell.execute_reply":"2022-07-31T09:08:16.236124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:16.237858Z","iopub.execute_input":"2022-07-31T09:08:16.238172Z","iopub.status.idle":"2022-07-31T09:08:16.301120Z","shell.execute_reply.started":"2022-07-31T09:08:16.238144Z","shell.execute_reply":"2022-07-31T09:08:16.300246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncorrmat = df.corr()\n  \nf, ax = plt.subplots(figsize =(9, 8))\nsns.heatmap(corrmat, ax = ax, cmap =\"YlGnBu\", linewidths = 0.1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:16.302454Z","iopub.execute_input":"2022-07-31T09:08:16.302697Z","iopub.status.idle":"2022-07-31T09:08:17.549537Z","shell.execute_reply.started":"2022-07-31T09:08:16.302667Z","shell.execute_reply":"2022-07-31T09:08:17.548877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['SalePrice'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:17.550809Z","iopub.execute_input":"2022-07-31T09:08:17.551150Z","iopub.status.idle":"2022-07-31T09:08:17.796692Z","shell.execute_reply.started":"2022-07-31T09:08:17.551121Z","shell.execute_reply":"2022-07-31T09:08:17.795656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:17.797896Z","iopub.execute_input":"2022-07-31T09:08:17.798134Z","iopub.status.idle":"2022-07-31T09:08:17.864537Z","shell.execute_reply.started":"2022-07-31T09:08:17.798103Z","shell.execute_reply":"2022-07-31T09:08:17.863492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Scaling the features\nStandardized_Scaler = StandardScaler().fit(df_train)\ndf_train_scaled = Standardized_Scaler.transform(df_train)\ndf_test_scaled = Standardized_Scaler.transform(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:17.865639Z","iopub.execute_input":"2022-07-31T09:08:17.865854Z","iopub.status.idle":"2022-07-31T09:08:17.907785Z","shell.execute_reply.started":"2022-07-31T09:08:17.865827Z","shell.execute_reply":"2022-07-31T09:08:17.906748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Min Maxing the features\ndf_train_minmaxed = (df_train - df_train.min()) / (df_train.max() - df_train.min())\ndf_test_minmaxed = (df_test - df_test.min()) / (df_test.max() - df_test.min())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:17.909127Z","iopub.execute_input":"2022-07-31T09:08:17.909364Z","iopub.status.idle":"2022-07-31T09:08:17.944556Z","shell.execute_reply.started":"2022-07-31T09:08:17.909337Z","shell.execute_reply":"2022-07-31T09:08:17.943679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linear Regression model","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#Getting the training dataset target\ndf_train_target=df_train_original['SalePrice']\n\n\nlinearRegression_model = LinearRegression()\nlinearRegression_model.fit(df_train_scaled,df_train_target)\n\nprint('Linear Regression score:)')\nprint('Training Set score: %.3f' % linearRegression_model.score(df_train_scaled, df_train_target))\npred = linearRegression_model.predict(df_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:17.945937Z","iopub.execute_input":"2022-07-31T09:08:17.946183Z","iopub.status.idle":"2022-07-31T09:08:18.005312Z","shell.execute_reply.started":"2022-07-31T09:08:17.946146Z","shell.execute_reply":"2022-07-31T09:08:18.004341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor \n\nregressor = RandomForestRegressor(n_estimators= 10, random_state=0)\nregressor.fit(df_train_scaled, df_train_target)\n\nprint('Random Forest score:)')\nprint('Training Set score: %.3f' % regressor.score(df_train_scaled, df_train_target))\n\npred= regressor.predict(df_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:18.006783Z","iopub.execute_input":"2022-07-31T09:08:18.007353Z","iopub.status.idle":"2022-07-31T09:08:18.451434Z","shell.execute_reply.started":"2022-07-31T09:08:18.007307Z","shell.execute_reply":"2022-07-31T09:08:18.450251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\nregressor = GradientBoostingRegressor( random_state=0)\nregressor.fit(df_train_scaled, df_train_target)\n\nprint('Random Forest score:)')\nprint('Training Set score: %.3f' % regressor.score(df_train_scaled, df_train_target))\n\npred= regressor.predict(df_test_scaled)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:18.453199Z","iopub.execute_input":"2022-07-31T09:08:18.453508Z","iopub.status.idle":"2022-07-31T09:08:19.577520Z","shell.execute_reply.started":"2022-07-31T09:08:18.453471Z","shell.execute_reply":"2022-07-31T09:08:19.576166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_id = df_test_original[['Id']]\nsubmission_result= pd.DataFrame(pred)\nsubmission_df = pd.concat([submission_id, submission_result], axis=1, join='inner')\nsubmission_df.columns = ['Id', 'SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:19.581902Z","iopub.execute_input":"2022-07-31T09:08:19.582172Z","iopub.status.idle":"2022-07-31T09:08:19.591206Z","shell.execute_reply.started":"2022-07-31T09:08:19.582143Z","shell.execute_reply":"2022-07-31T09:08:19.590048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing to csv.file","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:19.592419Z","iopub.execute_input":"2022-07-31T09:08:19.592673Z","iopub.status.idle":"2022-07-31T09:08:19.611568Z","shell.execute_reply.started":"2022-07-31T09:08:19.592643Z","shell.execute_reply":"2022-07-31T09:08:19.610426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:08:19.613830Z","iopub.execute_input":"2022-07-31T09:08:19.614183Z","iopub.status.idle":"2022-07-31T09:08:19.621960Z","shell.execute_reply.started":"2022-07-31T09:08:19.614115Z","shell.execute_reply":"2022-07-31T09:08:19.620921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}