{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Importing libs**","metadata":{}},{"cell_type":"code","source":"#============ Standart\nimport os\nimport warnings\nimport numpy as np \nimport pandas as pd\nfrom math import sqrt\nfrom scipy.special import boxcox1p\nfrom scipy.stats import norm, skew\n\n\n#============= Viz tools\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n\n#============= Models and tests\nfrom sklearn.svm import SVC\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import KFold, GridSearchCV, cross_val_score\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, accuracy_score, log_loss\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis, QuadraticDiscriminantAnalysis\n\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T20:03:36.601755Z","iopub.execute_input":"2022-08-08T20:03:36.602148Z","iopub.status.idle":"2022-08-08T20:03:36.610902Z","shell.execute_reply.started":"2022-08-08T20:03:36.602118Z","shell.execute_reply":"2022-08-08T20:03:36.609521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Importing and shaping the data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')\ny_val = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:36.617572Z","iopub.execute_input":"2022-08-08T20:03:36.618023Z","iopub.status.idle":"2022-08-08T20:03:36.682554Z","shell.execute_reply.started":"2022-08-08T20:03:36.617983Z","shell.execute_reply":"2022-08-08T20:03:36.681628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ID_test = test['Id'] # Saving ID to test accuracy later\ny_train = train['SalePrice'] # Saving target for training later\n\ntrain.drop('Id', inplace=True, axis=1)\ntest.drop('Id', inplace=True, axis=1)\n\n\nfull_data = pd.concat([train, test])\nfull_data.drop('SalePrice', inplace=True, axis=1)\n\nfull_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:36.684672Z","iopub.execute_input":"2022-08-08T20:03:36.685651Z","iopub.status.idle":"2022-08-08T20:03:36.717630Z","shell.execute_reply.started":"2022-08-08T20:03:36.685603Z","shell.execute_reply":"2022-08-08T20:03:36.716297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Defining which features i'll use**\n\n* **MSZoning**: Identifies the general zoning classification of the sale.\n* **LotArea**: Lot size in square feet\n* **Street**: Type of road access to property\n* **LandContour**: Flatness of the property\n* **LotConfig**: Lot configuration\n* **OverallQual**: Rates the overall material and finish of the house\n* **OverallCond**: Rates the overall condition of the house\n* **YearBuilt**: Original construction date\n* **YearRemodAdd**: Remodel date (same as construction date if no remodeling or additions)\n* **FullBath**: Full bathrooms above grade\n* **1stFlrSF**: First Floor square feet\n* **2ndFlrSF**: Second floor square feet\n* **GrLivArea**: Above grade (ground) living area square feet\n* **GarageArea**: Size of garage in square feet\n* **TotRmsAbvGrd**: Total rooms above grade (does not include bathrooms)","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize= (12 , 12))\nsns.heatmap(train.corr(),cmap=\"Blues\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:36.719486Z","iopub.execute_input":"2022-08-08T20:03:36.720384Z","iopub.status.idle":"2022-08-08T20:03:37.685667Z","shell.execute_reply.started":"2022-08-08T20:03:36.720333Z","shell.execute_reply":"2022-08-08T20:03:37.684444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining features\nfeatures = ['MSZoning',\n            'LotArea',\n            'Street',\n            'LandContour',\n            'LotConfig',\n            'OverallQual',\n            'OverallCond',\n            'YearBuilt',\n            'YearRemodAdd',\n            'FullBath',\n            '1stFlrSF',\n            '2ndFlrSF',\n            'GrLivArea',\n            'GarageArea',\n            'TotRmsAbvGrd']\n\nfull_data = full_data[features]\n\n\n# Separating features\nnum_types = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\nnum_features = []\ncat_features = []\n\nfor col in full_data.columns:\n    if full_data[col].dtype in num_types:\n        num_features.append(col)\n    elif full_data[col].dtype == 'object':\n        cat_features.append(col)\n        \n\nfull_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:37.688557Z","iopub.execute_input":"2022-08-08T20:03:37.689597Z","iopub.status.idle":"2022-08-08T20:03:37.708215Z","shell.execute_reply.started":"2022-08-08T20:03:37.689545Z","shell.execute_reply":"2022-08-08T20:03:37.707399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling NA values\nfull_data['MSZoning'] = full_data['MSZoning'].fillna(full_data['MSZoning'].mode()[0])\nfull_data['GarageArea'] = full_data['GarageArea'].fillna(full_data['GarageArea'].median())\n\n\n# Looking for skewness on num feats\n\nskew = full_data[num_features].apply(lambda x: skew(x.dropna())).sort_values(ascending=False)\ndf_skew = pd.DataFrame({'Skew': skew})\n\nprint(df_skew.head())\n\nlambda_skew = 0.15\nfor feat in skew.index:\n    full_data[feat] = boxcox1p(full_data[feat], lambda_skew)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:37.709503Z","iopub.execute_input":"2022-08-08T20:03:37.710002Z","iopub.status.idle":"2022-08-08T20:03:37.732927Z","shell.execute_reply.started":"2022-08-08T20:03:37.709970Z","shell.execute_reply":"2022-08-08T20:03:37.731749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data = pd.get_dummies(full_data).reset_index(drop=True)\n\n\ntrain = full_data[:len(train)]\ntest = full_data[len(train):]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:37.734378Z","iopub.execute_input":"2022-08-08T20:03:37.734739Z","iopub.status.idle":"2022-08-08T20:03:37.752350Z","shell.execute_reply.started":"2022-08-08T20:03:37.734706Z","shell.execute_reply":"2022-08-08T20:03:37.751195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decision_tree_model = DecisionTreeRegressor(random_state=13)\nclassifiers = [\n    DecisionTreeRegressor(random_state=13),\n    GridSearchCV(decision_tree_model , {\n    \"max_depth\" : [6,7,8,9,10,11,12],\n    \"min_samples_split\": [6,7,8,9,10],\n    \"min_samples_leaf\" : [5,7,8,9,10]\n    },verbose = 1),\n    RandomForestRegressor(random_state=13),\n    GradientBoostingRegressor(loss='huber',random_state=13),\n    XGBRegressor(objective='reg:squarederror',random_state=13),\n    LGBMRegressor(objective='regression', verbose=1,random_state=13)\n]\n\nkf = KFold(n_splits=12, random_state=13, shuffle=True)\n\ndef cross_val_rmse(model, X=train , Y=y_train):\n    rmse = np.sqrt(-cross_val_score(model,X, Y, scoring=\"neg_mean_squared_error\", cv=kf))\n    return (rmse)\n\n\nacc_dict = {}\n\ngraph_cols = [\"Model\", \"RMSE\"]\ngraph = pd.DataFrame(columns=graph_cols)\n\nX_train, X_test = full_data[:len(train)], full_data[len(train):]\ny_train, y_test = y_train, y_val.iloc[:,1]\n\n\n\nfor clf in classifiers:\n    name = clf.__class__.__name__\n    acc = cross_val_rmse(clf).mean()\n    if name in acc_dict:\n        acc_dict[name] += acc\n    else:\n        acc_dict[name] = acc\n\nfor clf in acc_dict:\n    acc_dict[clf] = acc_dict[clf] / 10.0\n    new_data = pd.DataFrame([[clf, acc_dict[clf]]], columns=graph_cols)\n    graph = graph.append(new_data)\n\nplt.xlabel('RMSE')\nplt.title('Model results')\n\nsns.set_color_codes(\"muted\")\nsns.barplot(x='RMSE', y='Model', data=graph, color=\"r\")\n\n\nmodel = GradientBoostingRegressor(loss='huber',random_state=15)\nmodel.fit(train, y_train)\n\npredict = model.predict(test)\ncc = sqrt(mean_squared_error(predict, y_val.iloc[:,1]))\n\nprint(cc)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:03:37.754727Z","iopub.execute_input":"2022-08-08T20:03:37.755706Z","iopub.status.idle":"2022-08-08T20:05:51.164877Z","shell.execute_reply.started":"2022-08-08T20:03:37.755654Z","shell.execute_reply":"2022-08-08T20:05:51.163653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'Id': ID_test,\n                       'SalePrice': predict})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:08:51.943999Z","iopub.execute_input":"2022-08-08T20:08:51.944627Z","iopub.status.idle":"2022-08-08T20:08:51.956234Z","shell.execute_reply.started":"2022-08-08T20:08:51.944586Z","shell.execute_reply":"2022-08-08T20:08:51.955323Z"},"trusted":true},"execution_count":null,"outputs":[]}]}