{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-04T13:59:36.156278Z","iopub.execute_input":"2022-07-04T13:59:36.156649Z","iopub.status.idle":"2022-07-04T13:59:36.16391Z","shell.execute_reply.started":"2022-07-04T13:59:36.156615Z","shell.execute_reply":"2022-07-04T13:59:36.162592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# House Prices Prediction","metadata":{}},{"cell_type":"markdown","source":"### Importing","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n#import bamboolib as bam\nimport seaborn as sns\nimport plotly.express as px\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:59:36.165176Z","iopub.execute_input":"2022-07-04T13:59:36.165762Z","iopub.status.idle":"2022-07-04T13:59:36.181612Z","shell.execute_reply.started":"2022-07-04T13:59:36.165717Z","shell.execute_reply":"2022-07-04T13:59:36.180373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:00:55.850064Z","iopub.execute_input":"2022-07-04T14:00:55.85053Z","iopub.status.idle":"2022-07-04T14:00:55.937984Z","shell.execute_reply.started":"2022-07-04T14:00:55.850495Z","shell.execute_reply":"2022-07-04T14:00:55.936708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Explore the data","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:01:57.553248Z","iopub.execute_input":"2022-07-04T14:01:57.553621Z","iopub.status.idle":"2022-07-04T14:01:57.582645Z","shell.execute_reply.started":"2022-07-04T14:01:57.553587Z","shell.execute_reply":"2022-07-04T14:01:57.58119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:02:05.617156Z","iopub.execute_input":"2022-07-04T14:02:05.617867Z","iopub.status.idle":"2022-07-04T14:02:05.656585Z","shell.execute_reply.started":"2022-07-04T14:02:05.617829Z","shell.execute_reply":"2022-07-04T14:02:05.655368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:02:11.619234Z","iopub.execute_input":"2022-07-04T14:02:11.619788Z","iopub.status.idle":"2022-07-04T14:02:11.762206Z","shell.execute_reply.started":"2022-07-04T14:02:11.619739Z","shell.execute_reply":"2022-07-04T14:02:11.760999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:02:16.578233Z","iopub.execute_input":"2022-07-04T14:02:16.579276Z","iopub.status.idle":"2022-07-04T14:02:16.596318Z","shell.execute_reply.started":"2022-07-04T14:02:16.57924Z","shell.execute_reply":"2022-07-04T14:02:16.595442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Pandas profiling - optional","metadata":{}},{"cell_type":"code","source":"# import pandas_profiling \n\n# profile = droppedDf.profile_report(title='Pandas Profiling Report')\n# profile.to_file(output_file=\"Data_Profiling_v3.html\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:02:34.994438Z","iopub.execute_input":"2022-07-04T14:02:34.994835Z","iopub.status.idle":"2022-07-04T14:02:35.000527Z","shell.execute_reply.started":"2022-07-04T14:02:34.994803Z","shell.execute_reply":"2022-07-04T14:02:34.998924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### EDA for categorical and numerical features","metadata":{}},{"cell_type":"markdown","source":"Show the distrubition of target value that column \"SalePrice\"","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(df, x='SalePrice')\nfig","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:02:54.381001Z","iopub.execute_input":"2022-07-04T14:02:54.382321Z","iopub.status.idle":"2022-07-04T14:02:55.485451Z","shell.execute_reply.started":"2022-07-04T14:02:54.382279Z","shell.execute_reply":"2022-07-04T14:02:55.484435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are many outliers. The distrubition is like a normal distrubition. The feature is target value. Because of we will not make changing.","metadata":{}},{"cell_type":"markdown","source":"We will use features that have correlation with \"SalePrice\".","metadata":{}},{"cell_type":"code","source":"corr = df.corr()\ng = sns.heatmap(corr,  vmax=.3, center=0,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5},fmt='.2f', cmap='coolwarm')\nsns.despine()\ng.figure.set_size_inches(14,10)\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:03:29.180283Z","iopub.execute_input":"2022-07-04T14:03:29.180677Z","iopub.status.idle":"2022-07-04T14:03:29.750001Z","shell.execute_reply.started":"2022-07-04T14:03:29.180642Z","shell.execute_reply":"2022-07-04T14:03:29.748533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This heatmap denotes the correlation of features with 'SalePrice'. From 0 to positive 0.3, positive correlation increases; from 0 to nearly -0.5, there is a negative correlation which indicates negative relationship. It also helps to identify outliers. Features which don't have high relationship with SalePrice will be dropped.  \n\nFeatures below will be dropped:\n\"Id\", \"MSSubClass\", \"MSZoning\", \"Street\", \"LandContour\", \"Utilities\", \"LandSlope\", \"Condition1\", \"Condition2\", \"BldgType\", \"OverallCond\", \"RoofStyle\",\"RoofMatl\", \"Exterior1st\", \"Exterior2nd\",\"MasVnrType\", \"ExterCond\", \"Foundation\", \"BsmtCond\", \"BsmtExposure\", \"BsmtFinType1\",\"BsmtFinType2\", \"BsmtFinSF2\", \"BsmtUnfSF\", \"Heating\", \"Electrical\", \"LowQualFinSF\", \"BsmtFullBath\", \"BsmtHalfBath\", \"HalfBath\", \"SaleCondition\", \"SaleType\", \"YrSold\", \"MoSold\", \"MiscVal\", \"MiscFeature\", \"Fence\", \"PoolQC\", \"PoolArea\", \"ScreenPorch\", \"3SsnPorch\", \"EnclosedPorch\", \"OpenPorchSF\", \"WoodDeckSF\", \"PavedDrive\", \"GarageCond\", \"GarageQual\", \"GarageType\", \"FireplaceQu\", \"Functional\", \"KitchenAbvGr\", \"BedroomAbvGr\"","metadata":{}},{"cell_type":"code","source":"dropColumns = [\"Id\", \"MSSubClass\", \"MSZoning\", \"Street\", \"LandContour\", \"Utilities\", \"LandSlope\", \"Condition1\", \"Condition2\", \"BldgType\", \"OverallCond\", \"RoofStyle\", \n               \"RoofMatl\", \"Exterior1st\", \"Exterior2nd\",\"MasVnrType\", \"ExterCond\", \"Foundation\", \"BsmtCond\", \"BsmtExposure\", \"BsmtFinType1\",\n              \"BsmtFinType2\", \"BsmtFinSF2\", \"BsmtUnfSF\", \"Heating\", \"Electrical\", \"LowQualFinSF\", \"BsmtFullBath\", \"BsmtHalfBath\", \"HalfBath\"] + [\"SaleCondition\", \"SaleType\", \"YrSold\", \"MoSold\", \"MiscVal\", \"MiscFeature\", \"Fence\", \"PoolQC\", \"PoolArea\", \"ScreenPorch\", \"3SsnPorch\", \"EnclosedPorch\", \"OpenPorchSF\", \"WoodDeckSF\", \"PavedDrive\", \"GarageCond\", \"GarageQual\", \"GarageType\", \"FireplaceQu\", \"Functional\", \"KitchenAbvGr\", \"BedroomAbvGr\"]\n\ndroppedDf = df.drop(columns=dropColumns, axis=1)\ndroppedDf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:03:44.69691Z","iopub.execute_input":"2022-07-04T14:03:44.697358Z","iopub.status.idle":"2022-07-04T14:03:44.73302Z","shell.execute_reply.started":"2022-07-04T14:03:44.697327Z","shell.execute_reply":"2022-07-04T14:03:44.731476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_list=droppedDf.columns.tolist()\n\ndef splitCategoricalAndNumericalData(col_list: list):\n    cat_cols=[]\n    num_cols=[]\n    for i in col_list:\n        if df[i].dtype == \"object\":\n            cat_cols.append(i)\n        elif df[i].dtype == \"int64\" or df[i].dtype == \"float64\":\n            num_cols.append(i)\n    return cat_cols, num_cols\n\ncat_cols, num_cols = splitCategoricalAndNumericalData(col_list)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:03:51.725517Z","iopub.execute_input":"2022-07-04T14:03:51.725939Z","iopub.status.idle":"2022-07-04T14:03:51.734647Z","shell.execute_reply.started":"2022-07-04T14:03:51.725905Z","shell.execute_reply":"2022-07-04T14:03:51.732935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rowNumberOfPlot = len(num_cols)//2\n\ndef multiplePlot(row, col, df, columns):\n    fig, axs = plt.subplots(row, col,figsize=(25,row*10))\n    for i in range(col):\n        for j in range(row):\n            featureName = columns[i*row + j]\n            df_i = df[featureName]\n            axs[j, i].hist(df_i,bins=100,color=\"orange\")\n            axs[j, i].set_title(\"Frequency - {}\".format(featureName))\n            axs[j, i].set(xlabel=featureName, ylabel=\"Frequency\")\n\nmultiplePlot(rowNumberOfPlot,2,df,num_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:03:56.068432Z","iopub.execute_input":"2022-07-04T14:03:56.068849Z","iopub.status.idle":"2022-07-04T14:04:02.473642Z","shell.execute_reply.started":"2022-07-04T14:03:56.068815Z","shell.execute_reply":"2022-07-04T14:04:02.472646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rowNumberOfPlot = len(cat_cols)//2\n\n# multiplePlot(rowNumberOfPlot,2,df,cat_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:04:24.867313Z","iopub.execute_input":"2022-07-04T14:04:24.867698Z","iopub.status.idle":"2022-07-04T14:04:24.873694Z","shell.execute_reply.started":"2022-07-04T14:04:24.867668Z","shell.execute_reply":"2022-07-04T14:04:24.872104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"droppedDf.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:04:26.732291Z","iopub.execute_input":"2022-07-04T14:04:26.732736Z","iopub.status.idle":"2022-07-04T14:04:26.746814Z","shell.execute_reply.started":"2022-07-04T14:04:26.732699Z","shell.execute_reply":"2022-07-04T14:04:26.745516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Filling the missing values by examining the graphics.","metadata":{}},{"cell_type":"markdown","source":"#### Impute missing values","metadata":{}},{"cell_type":"markdown","source":"\"Alley\" feature contains 1369 missing values. These values will be assigned to other class named \"NO\".","metadata":{}},{"cell_type":"code","source":"droppedDf[\"Alley\"].fillna(\"NO\", inplace=True)\ndroppedDf[\"Alley\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:04:55.744428Z","iopub.execute_input":"2022-07-04T14:04:55.744805Z","iopub.status.idle":"2022-07-04T14:04:55.754443Z","shell.execute_reply.started":"2022-07-04T14:04:55.744773Z","shell.execute_reply":"2022-07-04T14:04:55.753531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"LotFrontage\" feature contains 259 missing values. These values will be filled by the mean of the feature according to graph.","metadata":{}},{"cell_type":"code","source":"droppedDf[\"LotFrontage\"].fillna(df.LotFrontage.mean(), inplace=True)\ndroppedDf[\"LotFrontage\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:05:05.918245Z","iopub.execute_input":"2022-07-04T14:05:05.918646Z","iopub.status.idle":"2022-07-04T14:05:05.929417Z","shell.execute_reply.started":"2022-07-04T14:05:05.918615Z","shell.execute_reply":"2022-07-04T14:05:05.928436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"GarageFinish\" feature contains 81 missing values. These values will be assigned to other class named \"NO\".","metadata":{}},{"cell_type":"code","source":"droppedDf[\"GarageFinish\"].fillna(\"NO\", inplace=True)\ndroppedDf[\"GarageFinish\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:05:19.158017Z","iopub.execute_input":"2022-07-04T14:05:19.158452Z","iopub.status.idle":"2022-07-04T14:05:19.167855Z","shell.execute_reply.started":"2022-07-04T14:05:19.158409Z","shell.execute_reply":"2022-07-04T14:05:19.166975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"GarageYrBlt\" feature contains 81 missing values. These values will be filled by the mean of the feature according to graph.\n","metadata":{}},{"cell_type":"code","source":"droppedDf[\"GarageYrBlt\"].fillna(df.GarageYrBlt.mean(), inplace=True)\ndroppedDf[\"GarageYrBlt\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:05:39.168438Z","iopub.execute_input":"2022-07-04T14:05:39.168852Z","iopub.status.idle":"2022-07-04T14:05:39.177878Z","shell.execute_reply.started":"2022-07-04T14:05:39.168818Z","shell.execute_reply":"2022-07-04T14:05:39.17683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"BsmtQual\" feature contains 37 missing values. These values will be assigned to other class named \"NO\".","metadata":{}},{"cell_type":"code","source":"droppedDf[\"BsmtQual\"].fillna(\"NO\", inplace=True)\ndroppedDf[\"BsmtQual\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:07:11.400662Z","iopub.execute_input":"2022-07-04T14:07:11.401051Z","iopub.status.idle":"2022-07-04T14:07:11.41116Z","shell.execute_reply.started":"2022-07-04T14:07:11.401021Z","shell.execute_reply":"2022-07-04T14:07:11.409825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"MasVnrArea\" feature contains 8 missing values. Since the number of missing values is low, these values will be filled with the most repeated value, zero.","metadata":{}},{"cell_type":"code","source":"droppedDf[\"MasVnrArea\"].fillna(0, inplace=True)\ndroppedDf[\"MasVnrArea\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:07:25.816157Z","iopub.execute_input":"2022-07-04T14:07:25.816542Z","iopub.status.idle":"2022-07-04T14:07:25.82523Z","shell.execute_reply.started":"2022-07-04T14:07:25.81651Z","shell.execute_reply":"2022-07-04T14:07:25.824375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When \"MasVnrAreaCatg\" is examined, it is deemed appropriate to divide it into 3 different categories.  ","metadata":{}},{"cell_type":"code","source":"droppedDf['MasVnrAreaCatg'] = np.where(droppedDf.MasVnrArea>1000,'BIG',\n                                      np.where(droppedDf.MasVnrArea>500,'MEDIUM',\n                                              np.where(droppedDf.MasVnrArea>0,'SMALL','NO')))\ndroppedDf['MasVnrAreaCatg'].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:07:38.40737Z","iopub.execute_input":"2022-07-04T14:07:38.407782Z","iopub.status.idle":"2022-07-04T14:07:38.419926Z","shell.execute_reply.started":"2022-07-04T14:07:38.40775Z","shell.execute_reply":"2022-07-04T14:07:38.418621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking the filled values.","metadata":{}},{"cell_type":"code","source":"droppedDf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:07:50.078848Z","iopub.execute_input":"2022-07-04T14:07:50.079226Z","iopub.status.idle":"2022-07-04T14:07:50.09718Z","shell.execute_reply.started":"2022-07-04T14:07:50.079195Z","shell.execute_reply":"2022-07-04T14:07:50.095749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"markdown","source":"### Prepare the input data","metadata":{}},{"cell_type":"code","source":"inputDf = droppedDf.drop(['SalePrice'],axis=1)\ninputDf = inputDf.iloc[[0]].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:09.719263Z","iopub.execute_input":"2022-07-04T14:08:09.719717Z","iopub.status.idle":"2022-07-04T14:08:09.728379Z","shell.execute_reply.started":"2022-07-04T14:08:09.719678Z","shell.execute_reply":"2022-07-04T14:08:09.727119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in inputDf:\n    if inputDf[i].dtype == \"object\":\n        inputDf[i] = droppedDf[i].mode()[0]\n    elif inputDf[i].dtype == \"int64\" or inputDf[i].dtype == \"float64\":\n        inputDf[i] = droppedDf[i].mean()\ninputDf\n\nobj_feat = list(inputDf.loc[:, inputDf.dtypes == 'object'].columns.values)\nfor feature in obj_feat:\n    inputDf[feature] = inputDf[feature].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:13.17749Z","iopub.execute_input":"2022-07-04T14:08:13.177885Z","iopub.status.idle":"2022-07-04T14:08:13.213313Z","shell.execute_reply.started":"2022-07-04T14:08:13.177854Z","shell.execute_reply":"2022-07-04T14:08:13.211985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing the libraries","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:22.160971Z","iopub.execute_input":"2022-07-04T14:08:22.161352Z","iopub.status.idle":"2022-07-04T14:08:22.582921Z","shell.execute_reply.started":"2022-07-04T14:08:22.16132Z","shell.execute_reply":"2022-07-04T14:08:22.581814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = droppedDf.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:29.623245Z","iopub.execute_input":"2022-07-04T14:08:29.623637Z","iopub.status.idle":"2022-07-04T14:08:29.62935Z","shell.execute_reply.started":"2022-07-04T14:08:29.623607Z","shell.execute_reply":"2022-07-04T14:08:29.628154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Converting the \"object\" type data to \"category\" for LightGBM model.","metadata":{}},{"cell_type":"code","source":"obj_feat = list(df.loc[:, df.dtypes == 'object'].columns.values)\nfor feature in obj_feat:\n    df[feature] = df[feature].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:39.494936Z","iopub.execute_input":"2022-07-04T14:08:39.495337Z","iopub.status.idle":"2022-07-04T14:08:39.516569Z","shell.execute_reply.started":"2022-07-04T14:08:39.495303Z","shell.execute_reply":"2022-07-04T14:08:39.51561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To define the input and output feature\nx = df.drop(['SalePrice'],axis=1)\ny = df.SalePrice\n\n# train and test split\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size=0.30,random_state=1)\nx.iloc[0].index","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:08:55.233705Z","iopub.execute_input":"2022-07-04T14:08:55.234107Z","iopub.status.idle":"2022-07-04T14:08:55.255348Z","shell.execute_reply.started":"2022-07-04T14:08:55.234068Z","shell.execute_reply":"2022-07-04T14:08:55.254154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = lgb.LGBMRegressor(max_depth=5, \n                          n_estimators = 100, \n                          learning_rate = 0.2,\n                          min_child_samples = 30)\nmodel.fit(x_train, y_train)\n\npred_y_train = model.predict(x_train)\npred_y_test = model.predict(x_test)\n\nr2_train = metrics.r2_score(y_train, pred_y_train)\nr2_test = metrics.r2_score(y_test, pred_y_test)\n\nmsle_train =metrics.mean_squared_log_error(y_train, pred_y_train)\nmsle_test =metrics.mean_squared_log_error(y_test, pred_y_test)\n\nprint(f\"Train r2 = {r2_train:.2f} \\nTest r2 = {r2_test:.2f}\")\nprint(f\"Train msle = {msle_train:.2f} \\nTest msle = {msle_test:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:09:02.105076Z","iopub.execute_input":"2022-07-04T14:09:02.105465Z","iopub.status.idle":"2022-07-04T14:09:02.294403Z","shell.execute_reply.started":"2022-07-04T14:09:02.10543Z","shell.execute_reply":"2022-07-04T14:09:02.29345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model development using GridSearchCV","metadata":{}},{"cell_type":"markdown","source":"Above, the hyperparameters of the model were chosen intuitively. At this stage, optimum hyperparameter coefficients were selected by applying Grid Search Cross Validation.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparams = [{\"max_depth\":[3, 5], \n            \"n_estimators\" : [50, 100], \n            \"learning_rate\" : [0.1, 0.2],\n            \"min_child_samples\" : [20, 10]}]\n\ngs_knn = GridSearchCV(model,\n                      param_grid=params,\n                      cv=5)\n\ngs_knn.fit(x_train, y_train)\ngs_knn.score(x_train, y_train)\n\npred_y_train = model.predict(x_train)\npred_y_test = model.predict(x_test)\n\nr2_train = metrics.r2_score(y_train, pred_y_train)\nr2_test = metrics.r2_score(y_test, pred_y_test)\n\nmsle_train =metrics.mean_squared_log_error(y_train, pred_y_train)\nmsle_test =metrics.mean_squared_log_error(y_test, pred_y_test)\n\nprint(f\"Train r2 = {r2_train:.2f} \\nTest r2 = {r2_test:.2f}\")\nprint(f\"Train msle = {msle_train:.2f} \\nTest msle = {msle_test:.2f}\")\n\ngs_knn.best_params_\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:09:22.327685Z","iopub.execute_input":"2022-07-04T14:09:22.328204Z","iopub.status.idle":"2022-07-04T14:09:27.653459Z","shell.execute_reply.started":"2022-07-04T14:09:22.32816Z","shell.execute_reply":"2022-07-04T14:09:27.652415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A similar result was observed between the heuristic and GridSearchCV when the results were compared.","metadata":{}},{"cell_type":"markdown","source":"#### Feature importance","metadata":{}},{"cell_type":"markdown","source":"The first 20 values that affect the model are observed. These variables will be used in the deployment phase.","metadata":{}},{"cell_type":"code","source":"#show top 20 feature importance\nimport seaborn as sns\nfeature_imp = pd.Series(model.feature_importances_,index = x_train.columns).sort_values(ascending=False)[:20]\nplt.figure(figsize=(9,5))\nsns.barplot(x=feature_imp,y=feature_imp.index)\nplt.xlabel(\"Feature Importance Scores\")\nplt.ylabel(\"Features\")\nplt.title(\"Feature Importance\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:09:45.588739Z","iopub.execute_input":"2022-07-04T14:09:45.589115Z","iopub.status.idle":"2022-07-04T14:09:45.922698Z","shell.execute_reply.started":"2022-07-04T14:09:45.589077Z","shell.execute_reply":"2022-07-04T14:09:45.921465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model save and load","metadata":{}},{"cell_type":"markdown","source":"The trained model can be saved and uploaded via Pickle library.","metadata":{}},{"cell_type":"code","source":"# save the model to disk\nimport pickle\n\nfilename = 'trained_model.model'\npickle.dump(model, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:10:09.410011Z","iopub.execute_input":"2022-07-04T14:10:09.41044Z","iopub.status.idle":"2022-07-04T14:10:09.425182Z","shell.execute_reply.started":"2022-07-04T14:10:09.410407Z","shell.execute_reply":"2022-07-04T14:10:09.424193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the model from disk\nimport pickle\n\nfilename = 'trained_model.model'\n\nloaded_model = pickle.load(open(filename, 'rb'))\nresult = loaded_model.score(x_test, y_test)\nprint(result)\n\n# predict\nprint(loaded_model.predict(inputDf))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T14:10:18.792134Z","iopub.execute_input":"2022-07-04T14:10:18.792552Z","iopub.status.idle":"2022-07-04T14:10:18.913524Z","shell.execute_reply.started":"2022-07-04T14:10:18.792516Z","shell.execute_reply":"2022-07-04T14:10:18.912507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For demo:\n\n[![Open in Streamlit](https://static.streamlit.io/badges/streamlit_badge_black_white.svg)](https://share.streamlit.io/uzunb/house-prices-prediction-lgbm/main/1_%F0%9F%92%BB_Enter_Page.py)\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}