{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T18:26:55.815981Z","iopub.execute_input":"2022-08-13T18:26:55.816495Z","iopub.status.idle":"2022-08-13T18:26:55.828800Z","shell.execute_reply.started":"2022-08-13T18:26:55.816461Z","shell.execute_reply":"2022-08-13T18:26:55.827939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n\ntrain.tail()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:55.994074Z","iopub.execute_input":"2022-08-13T18:26:55.994760Z","iopub.status.idle":"2022-08-13T18:26:56.071539Z","shell.execute_reply.started":"2022-08-13T18:26:55.994719Z","shell.execute_reply":"2022-08-13T18:26:56.070939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.163403Z","iopub.execute_input":"2022-08-13T18:26:56.163778Z","iopub.status.idle":"2022-08-13T18:26:56.193812Z","shell.execute_reply.started":"2022-08-13T18:26:56.163741Z","shell.execute_reply":"2022-08-13T18:26:56.192853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.315513Z","iopub.execute_input":"2022-08-13T18:26:56.316025Z","iopub.status.idle":"2022-08-13T18:26:56.339096Z","shell.execute_reply.started":"2022-08-13T18:26:56.315989Z","shell.execute_reply":"2022-08-13T18:26:56.338112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.487282Z","iopub.execute_input":"2022-08-13T18:26:56.487837Z","iopub.status.idle":"2022-08-13T18:26:56.587001Z","shell.execute_reply.started":"2022-08-13T18:26:56.487801Z","shell.execute_reply":"2022-08-13T18:26:56.585885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum() #checking the missing value of training data","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.621954Z","iopub.execute_input":"2022-08-13T18:26:56.622280Z","iopub.status.idle":"2022-08-13T18:26:56.636300Z","shell.execute_reply.started":"2022-08-13T18:26:56.622248Z","shell.execute_reply":"2022-08-13T18:26:56.635091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['LotFrontage']= train['LotFrontage'].fillna(train['LotFrontage'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.743554Z","iopub.execute_input":"2022-08-13T18:26:56.744203Z","iopub.status.idle":"2022-08-13T18:26:56.751496Z","shell.execute_reply.started":"2022-08-13T18:26:56.744140Z","shell.execute_reply":"2022-08-13T18:26:56.750568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:56.899516Z","iopub.execute_input":"2022-08-13T18:26:56.900872Z","iopub.status.idle":"2022-08-13T18:26:56.957729Z","shell.execute_reply.started":"2022-08-13T18:26:56.900823Z","shell.execute_reply":"2022-08-13T18:26:56.956584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.distplot(train['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.023877Z","iopub.execute_input":"2022-08-13T18:26:57.024198Z","iopub.status.idle":"2022-08-13T18:26:57.372486Z","shell.execute_reply.started":"2022-08-13T18:26:57.024168Z","shell.execute_reply":"2022-08-13T18:26:57.371463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum() #checking the missing value of testing data","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.374842Z","iopub.execute_input":"2022-08-13T18:26:57.375199Z","iopub.status.idle":"2022-08-13T18:26:57.391056Z","shell.execute_reply.started":"2022-08-13T18:26:57.375154Z","shell.execute_reply":"2022-08-13T18:26:57.390179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['LotFrontage'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.392657Z","iopub.execute_input":"2022-08-13T18:26:57.393802Z","iopub.status.idle":"2022-08-13T18:26:57.403285Z","shell.execute_reply.started":"2022-08-13T18:26:57.393758Z","shell.execute_reply":"2022-08-13T18:26:57.402531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['LotFrontage'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.405338Z","iopub.execute_input":"2022-08-13T18:26:57.406441Z","iopub.status.idle":"2022-08-13T18:26:57.414863Z","shell.execute_reply.started":"2022-08-13T18:26:57.406399Z","shell.execute_reply":"2022-08-13T18:26:57.413904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['LotFrontage']= test['LotFrontage'].fillna(test['LotFrontage'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.512375Z","iopub.execute_input":"2022-08-13T18:26:57.512669Z","iopub.status.idle":"2022-08-13T18:26:57.519226Z","shell.execute_reply.started":"2022-08-13T18:26:57.512636Z","shell.execute_reply":"2022-08-13T18:26:57.518092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.653759Z","iopub.execute_input":"2022-08-13T18:26:57.654054Z","iopub.status.idle":"2022-08-13T18:26:57.757440Z","shell.execute_reply.started":"2022-08-13T18:26:57.654023Z","shell.execute_reply":"2022-08-13T18:26:57.756125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dealing with categorical data (foor data train)\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder \nlabel_encoder= LabelEncoder()\n\ntrain['Fence']= label_encoder.fit_transform(train['Fence'])\ntrain['SaleType']=label_encoder.fit_transform(train['SaleType'])\ntrain['SaleCondition']=label_encoder.fit_transform(train['SaleCondition'])\ntrain['GarageType']=label_encoder.fit_transform(train['GarageType'])\ntrain['Street']=label_encoder.fit_transform(train['Street'])\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.766312Z","iopub.execute_input":"2022-08-13T18:26:57.766706Z","iopub.status.idle":"2022-08-13T18:26:57.782297Z","shell.execute_reply.started":"2022-08-13T18:26:57.766654Z","shell.execute_reply":"2022-08-13T18:26:57.781344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dealing with categorical data (foor testing data)\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder \nlabel_encoder= LabelEncoder()\n\ntest['Fence']= label_encoder.fit_transform(test['Fence'])\ntest['SaleType']=label_encoder.fit_transform(test['SaleType'])\ntest['SaleCondition']=label_encoder.fit_transform(test['SaleCondition'])\ntest['GarageType']=label_encoder.fit_transform(test['GarageType'])\ntest['Street']=label_encoder.fit_transform(test['Street'])\n\ntest['Fence']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.879765Z","iopub.execute_input":"2022-08-13T18:26:57.880534Z","iopub.status.idle":"2022-08-13T18:26:57.897287Z","shell.execute_reply.started":"2022-08-13T18:26:57.880496Z","shell.execute_reply":"2022-08-13T18:26:57.895963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns \n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:57.992118Z","iopub.execute_input":"2022-08-13T18:26:57.992468Z","iopub.status.idle":"2022-08-13T18:26:57.997492Z","shell.execute_reply.started":"2022-08-13T18:26:57.992432Z","shell.execute_reply":"2022-08-13T18:26:57.996534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We select some numerical and categorical parameter to be put in multilinear equation\n#the parameters that we choose were table : 1,3,4,5,49,50, 51, 56, 58, 61, 62, 71, 73, so on\n\ndata_train = train.iloc[:, [1,3,4,5,17,18,19,20,26,34,36,37,38,43,44,45,46, 47, 48,49,50,51,52,54,56,59,61,62,67,68,69,71,75,76,77,78,79]].values\n\ndata_test = test.iloc[:, [1,3,4,5,17,18,19,20,26,36,37,38,34,43,44,45,46,47, 48, 49,50,51,52,54,56,59,61,62,67,68,69,71,75,76,77,78,79]].values","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.114242Z","iopub.execute_input":"2022-08-13T18:26:58.115459Z","iopub.status.idle":"2022-08-13T18:26:58.126373Z","shell.execute_reply.started":"2022-08-13T18:26:58.115408Z","shell.execute_reply":"2022-08-13T18:26:58.125538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**First thing first, I want to evaluate model that we use to predict the house price, since the data type is estimation with several independent variables, so I will use Multiple Linear Regresssion, I will evaluate the data trajn first**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.248532Z","iopub.execute_input":"2022-08-13T18:26:58.248975Z","iopub.status.idle":"2022-08-13T18:26:58.253347Z","shell.execute_reply.started":"2022-08-13T18:26:58.248945Z","shell.execute_reply":"2022-08-13T18:26:58.252564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train= train['SalePrice']\ny_train.sum #y train is the real house price","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.359616Z","iopub.execute_input":"2022-08-13T18:26:58.360753Z","iopub.status.idle":"2022-08-13T18:26:58.368050Z","shell.execute_reply.started":"2022-08-13T18:26:58.360690Z","shell.execute_reply":"2022-08-13T18:26:58.367119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.467502Z","iopub.execute_input":"2022-08-13T18:26:58.467796Z","iopub.status.idle":"2022-08-13T18:26:58.475566Z","shell.execute_reply.started":"2022-08-13T18:26:58.467766Z","shell.execute_reply":"2022-08-13T18:26:58.474665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dealing with missing value for training data\nfrom sklearn.impute import SimpleImputer\nimputer = SimpleImputer(missing_values=np.nan, strategy='mean')\nimputer.fit(data_train)\ndata_train = imputer.transform(data_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.573794Z","iopub.execute_input":"2022-08-13T18:26:58.574324Z","iopub.status.idle":"2022-08-13T18:26:58.583334Z","shell.execute_reply.started":"2022-08-13T18:26:58.574264Z","shell.execute_reply":"2022-08-13T18:26:58.582725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, Xtest, y_training, y_test = train_test_split(data_train, y_train, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.752766Z","iopub.execute_input":"2022-08-13T18:26:58.753221Z","iopub.status.idle":"2022-08-13T18:26:58.759924Z","shell.execute_reply.started":"2022-08-13T18:26:58.753188Z","shell.execute_reply":"2022-08-13T18:26:58.758939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import linear_model\nregr=linear_model.LinearRegression()\nregr.fit(X_train,y_training)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:58.991500Z","iopub.execute_input":"2022-08-13T18:26:58.992560Z","iopub.status.idle":"2022-08-13T18:26:59.010865Z","shell.execute_reply.started":"2022-08-13T18:26:58.992509Z","shell.execute_reply":"2022-08-13T18:26:59.008510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction= regr.predict(Xtest)\ny_prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.021220Z","iopub.execute_input":"2022-08-13T18:26:59.022294Z","iopub.status.idle":"2022-08-13T18:26:59.065797Z","shell.execute_reply.started":"2022-08-13T18:26:59.022227Z","shell.execute_reply":"2022-08-13T18:26:59.064401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mse=np.square(np.subtract(y_test,y_prediction)).mean()\nrmse=np.sqrt(mse)/1000\nrmse","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.132823Z","iopub.execute_input":"2022-08-13T18:26:59.133089Z","iopub.status.idle":"2022-08-13T18:26:59.151997Z","shell.execute_reply.started":"2022-08-13T18:26:59.133061Z","shell.execute_reply":"2022-08-13T18:26:59.150997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(y_test, y_prediction)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.232754Z","iopub.execute_input":"2022-08-13T18:26:59.233074Z","iopub.status.idle":"2022-08-13T18:26:59.246904Z","shell.execute_reply.started":"2022-08-13T18:26:59.233044Z","shell.execute_reply":"2022-08-13T18:26:59.246154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.337850Z","iopub.execute_input":"2022-08-13T18:26:59.338881Z","iopub.status.idle":"2022-08-13T18:26:59.346009Z","shell.execute_reply.started":"2022-08-13T18:26:59.338836Z","shell.execute_reply":"2022-08-13T18:26:59.345251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dealing with missing value for testing data\nfrom sklearn.impute import SimpleImputer\nimputer = SimpleImputer(missing_values=np.nan, strategy='mean')\nimputer.fit(data_test)\ndata_test = imputer.transform(data_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.428440Z","iopub.execute_input":"2022-08-13T18:26:59.428783Z","iopub.status.idle":"2022-08-13T18:26:59.437749Z","shell.execute_reply.started":"2022-08-13T18:26:59.428750Z","shell.execute_reply":"2022-08-13T18:26:59.436735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **we're going to use multiple linear regression to predict the house prices**","metadata":{}},{"cell_type":"code","source":"#x_train, x_test, y_tr = train_tet_split(data_train, data_test,y_train)\nregr.fit(data_train,y_train)\ny_prediction_test=regr.predict(data_test)\ny_prediction_test","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:26:59.525766Z","iopub.execute_input":"2022-08-13T18:26:59.526871Z","iopub.status.idle":"2022-08-13T18:26:59.547409Z","shell.execute_reply.started":"2022-08-13T18:26:59.526827Z","shell.execute_reply":"2022-08-13T18:26:59.546453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"my_submission = pd.DataFrame({'Id': test[\"Id\"], 'SalePrice' : y_prediction_test.ravel()})\n\nmy_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:27:27.475086Z","iopub.execute_input":"2022-08-13T18:27:27.475876Z","iopub.status.idle":"2022-08-13T18:27:27.489303Z","shell.execute_reply.started":"2022-08-13T18:27:27.475827Z","shell.execute_reply":"2022-08-13T18:27:27.488152Z"},"trusted":true},"execution_count":null,"outputs":[]}]}