{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T16:45:58.833368Z","iopub.execute_input":"2022-07-07T16:45:58.833944Z","iopub.status.idle":"2022-07-07T16:45:58.879515Z","shell.execute_reply.started":"2022-07-07T16:45:58.833831Z","shell.execute_reply":"2022-07-07T16:45:58.878256Z"}}},{"cell_type":"code","source":"train=pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest=pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:46:00.650986Z","iopub.execute_input":"2022-07-07T16:46:00.652403Z","iopub.status.idle":"2022-07-07T16:46:00.732324Z","shell.execute_reply.started":"2022-07-07T16:46:00.65236Z","shell.execute_reply":"2022-07-07T16:46:00.730848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:46:02.360718Z","iopub.execute_input":"2022-07-07T16:46:02.361526Z","iopub.status.idle":"2022-07-07T16:46:02.412203Z","shell.execute_reply.started":"2022-07-07T16:46:02.361482Z","shell.execute_reply":"2022-07-07T16:46:02.410962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target='SalePrice'","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:46:03.595542Z","iopub.execute_input":"2022-07-07T16:46:03.595893Z","iopub.status.idle":"2022-07-07T16:46:03.601613Z","shell.execute_reply.started":"2022-07-07T16:46:03.595865Z","shell.execute_reply":"2022-07-07T16:46:03.600593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install featurewiz --upgrade","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:46:07.689786Z","iopub.execute_input":"2022-07-07T16:46:07.690515Z","iopub.status.idle":"2022-07-07T16:46:47.759288Z","shell.execute_reply.started":"2022-07-07T16:46:07.690483Z","shell.execute_reply":"2022-07-07T16:46:47.757805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import featurewiz as FW\ntrainm, testm = FW.featurewiz(dataname=train, target=target, corr_limit=0.70, verbose=2, sep=',', \n\n\t\theader=0, test_data=test, feature_engg=['groupby','target'], category_encoders='CatBoostEncoder',\n\n\t\tdask_xgboost_flag=False, nrows=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:46:47.763266Z","iopub.execute_input":"2022-07-07T16:46:47.763766Z","iopub.status.idle":"2022-07-07T16:48:14.345618Z","shell.execute_reply.started":"2022-07-07T16:46:47.763699Z","shell.execute_reply":"2022-07-07T16:48:14.343962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrainm=pd.read_csv('../input/private/Trainm.csv')\ntestm=pd.read_csv('../input/private/Testm.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:58:37.598999Z","iopub.execute_input":"2022-07-09T11:58:37.599633Z","iopub.status.idle":"2022-07-09T11:58:37.723893Z","shell.execute_reply.started":"2022-07-09T11:58:37.599594Z","shell.execute_reply":"2022-07-09T11:58:37.722931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainm.to_csv('Trainm.csv')\ntestm.to_csv('Testm.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:58:38.478719Z","iopub.execute_input":"2022-07-09T11:58:38.479672Z","iopub.status.idle":"2022-07-09T11:58:38.909057Z","shell.execute_reply.started":"2022-07-09T11:58:38.479615Z","shell.execute_reply":"2022-07-09T11:58:38.907883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainm.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:58:43.862747Z","iopub.execute_input":"2022-07-09T11:58:43.863518Z","iopub.status.idle":"2022-07-09T11:58:43.891825Z","shell.execute_reply.started":"2022-07-09T11:58:43.863478Z","shell.execute_reply":"2022-07-09T11:58:43.890703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testm.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:58:47.238704Z","iopub.execute_input":"2022-07-09T11:58:47.239063Z","iopub.status.idle":"2022-07-09T11:58:47.265324Z","shell.execute_reply.started":"2022-07-09T11:58:47.239033Z","shell.execute_reply":"2022-07-09T11:58:47.26428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"modeltype = 'Regression'\ntarget='SalePrice'","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:31.027125Z","iopub.execute_input":"2022-07-09T11:59:31.027954Z","iopub.status.idle":"2022-07-09T11:59:31.035818Z","shell.execute_reply.started":"2022-07-09T11:59:31.027883Z","shell.execute_reply":"2022-07-09T11:59:31.034961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:33.575422Z","iopub.execute_input":"2022-07-09T11:59:33.576099Z","iopub.status.idle":"2022-07-09T11:59:33.603679Z","shell.execute_reply.started":"2022-07-09T11:59:33.576063Z","shell.execute_reply":"2022-07-09T11:59:33.602695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test.shape)\nprint(testm.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:43.194148Z","iopub.execute_input":"2022-07-09T11:59:43.194582Z","iopub.status.idle":"2022-07-09T11:59:43.204777Z","shell.execute_reply.started":"2022-07-09T11:59:43.194543Z","shell.execute_reply":"2022-07-09T11:59:43.203539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if isinstance(target, str):\n    cols = [x for x in list(trainm) if x not in [target]]\nelse:\n    cols = [x for x in list(trainm) if x not in target]\nX = trainm[cols]\ny = trainm[target]\ntrain.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:45.867823Z","iopub.execute_input":"2022-07-09T11:59:45.868707Z","iopub.status.idle":"2022-07-09T11:59:45.881533Z","shell.execute_reply.started":"2022-07-09T11:59:45.868658Z","shell.execute_reply":"2022-07-09T11:59:45.880575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###  find the y values that have low samples ####\nif modeltype != 'Regression':\n    low_counts = y.value_counts()[(y.value_counts()<=1).values].index\n    print(len(low_counts))\n    ## You need to remove those rows that have just one sample ##\n    X = X[~(y.isin(low_counts))]\n    y = y[~(y.isin(low_counts))]\nX.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:48.734944Z","iopub.execute_input":"2022-07-09T11:59:48.736423Z","iopub.status.idle":"2022-07-09T11:59:48.744606Z","shell.execute_reply.started":"2022-07-09T11:59:48.736381Z","shell.execute_reply":"2022-07-09T11:59:48.743667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nif modeltype == 'Regression':\n    X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.1, random_state=999)\nelse:\n    X_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.1,stratify=y, random_state=999)\nprint(X_train.shape, X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:59:51.937982Z","iopub.execute_input":"2022-07-09T11:59:51.938834Z","iopub.status.idle":"2022-07-09T11:59:51.951643Z","shell.execute_reply.started":"2022-07-09T11:59:51.938795Z","shell.execute_reply":"2022-07-09T11:59:51.950577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:51:32.812581Z","iopub.execute_input":"2022-07-09T11:51:32.814964Z","iopub.status.idle":"2022-07-09T11:51:32.85764Z","shell.execute_reply.started":"2022-07-09T11:51:32.814923Z","shell.execute_reply":"2022-07-09T11:51:32.856727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nfrom sklearn.model_selection import RandomizedSearchCV\nimport lightgbm as lgb\nlgb=lgb.LGBMRegressor()\nparameters = {'num_leaves':[20,40,60,80,100], 'min_child_samples':[5,10,15],'max_depth':[-1,5,10,20],\n             'learning_rate':[0.05,0.1,0.2],'reg_alpha':[0,0.01,0.03],'metric':['rmse']}\nclf=RandomizedSearchCV(lgb,parameters,scoring='neg_mean_squared_error',n_iter=100)\nclf.fit(X=X_train, y=y_train)\nprint(clf.best_params_)\npredicted=clf.predict(X_test)\n# Calculate MAE\nrmse_pred = mean_absolute_error(y_test,predicted) \nprint(\"Root Mean Absolute Error:\" , np.sqrt(rmse_pred))\nend = time.time()\nprint('Execution time is:')\nprint(end - start)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:52:12.936593Z","iopub.execute_input":"2022-07-09T12:52:12.936963Z","iopub.status.idle":"2022-07-09T12:55:14.919221Z","shell.execute_reply.started":"2022-07-09T12:52:12.936928Z","shell.execute_reply":"2022-07-09T12:55:14.91835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_regressor_lgbm=clf.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:58:09.629397Z","iopub.execute_input":"2022-07-09T12:58:09.630367Z","iopub.status.idle":"2022-07-09T12:58:09.635082Z","shell.execute_reply.started":"2022-07-09T12:58:09.630319Z","shell.execute_reply":"2022-07-09T12:58:09.634093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get predictions\ny_pred_test_lgbm = best_regressor_lgbm.predict(testm)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:59:04.401421Z","iopub.execute_input":"2022-07-09T12:59:04.401769Z","iopub.status.idle":"2022-07-09T12:59:04.42432Z","shell.execute_reply.started":"2022-07-09T12:59:04.401737Z","shell.execute_reply":"2022-07-09T12:59:04.423579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"light_clf.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:09:25.474974Z","iopub.execute_input":"2022-07-09T12:09:25.475586Z","iopub.status.idle":"2022-07-09T12:09:25.482384Z","shell.execute_reply.started":"2022-07-09T12:09:25.475548Z","shell.execute_reply":"2022-07-09T12:09:25.481472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import mean_squared_error as MSE\ny_pred_light = light_clf.predict(X_test)\nrmse = np.sqrt(MSE(y_test, y_pred_light))\nprint(\"RMSE : % f\" %(rmse))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:14:37.313298Z","iopub.execute_input":"2022-07-09T12:14:37.31365Z","iopub.status.idle":"2022-07-09T12:14:37.328933Z","shell.execute_reply.started":"2022-07-09T12:14:37.31362Z","shell.execute_reply":"2022-07-09T12:14:37.327893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RandomizedSearchCV\nimport time\n\n# A parameter grid for XGBoost\nparams = {\n    'n_estimators':[500],\n    'min_child_weight':[4,5], \n    'gamma':[i/10.0 for i in range(3,6)],  \n    'subsample':[i/10.0 for i in range(6,11)],\n    'colsample_bytree':[i/10.0 for i in range(6,11)], \n    'max_depth': [2,3,4,6,7],\n    'objective': ['reg:squarederror', 'reg:tweedie'],\n    'booster': ['gbtree', 'gblinear'],\n    'eval_metric': ['rmse'],\n    'eta': [i/10.0 for i in range(3,6)],\n}\n\nreg = XGBRegressor()\n\n# run randomized search\nn_iter_search = 100\nrandom_search = RandomizedSearchCV(reg, param_distributions=params,\n                                   n_iter=n_iter_search, cv=5, scoring='neg_mean_squared_error')\n\nstart = time.time()\nrandom_search.fit(X_train, y_train)\nprint(\"RandomizedSearchCV took %.2f seconds for %d candidates\"\n      \" parameter settings.\" % ((time.time() - start), n_iter_search))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:17:39.771295Z","iopub.execute_input":"2022-07-09T12:17:39.771645Z","iopub.status.idle":"2022-07-09T12:31:17.198268Z","shell.execute_reply.started":"2022-07-09T12:17:39.771616Z","shell.execute_reply":"2022-07-09T12:31:17.197378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_regressor = random_search.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:31:17.202015Z","iopub.execute_input":"2022-07-09T12:31:17.203795Z","iopub.status.idle":"2022-07-09T12:31:17.208985Z","shell.execute_reply.started":"2022-07-09T12:31:17.203758Z","shell.execute_reply":"2022-07-09T12:31:17.207928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n# Get predictions\ny_pred = best_regressor.predict(X_test)\n# Calculate MAE\nrmse_pred = mean_absolute_error(y_test, y_pred) \n\nprint(\"Root Mean Absolute Error:\" , np.sqrt(rmse_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:33:20.295807Z","iopub.execute_input":"2022-07-09T12:33:20.296789Z","iopub.status.idle":"2022-07-09T12:33:20.310895Z","shell.execute_reply.started":"2022-07-09T12:33:20.29675Z","shell.execute_reply":"2022-07-09T12:33:20.310023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = xgb_r.predict(X_test)\nrmse = np.sqrt(MSE(y_test, pred))\nprint(\"RMSE : % f\" %(rmse))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:15:33.658353Z","iopub.execute_input":"2022-07-09T12:15:33.658695Z","iopub.status.idle":"2022-07-09T12:15:33.673858Z","shell.execute_reply.started":"2022-07-09T12:15:33.658667Z","shell.execute_reply":"2022-07-09T12:15:33.672965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get predictions\ny_pred_test = best_regressor.predict(testm)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:36:28.220991Z","iopub.execute_input":"2022-07-09T12:36:28.221343Z","iopub.status.idle":"2022-07-09T12:36:28.239749Z","shell.execute_reply.started":"2022-07-09T12:36:28.221315Z","shell.execute_reply":"2022-07-09T12:36:28.238955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv('../input/house-prices-advanced-regression-techniques/sample_submission.csv')\nsubmission['SalePrice']=y_pred_test","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:36:30.722726Z","iopub.execute_input":"2022-07-09T12:36:30.723379Z","iopub.status.idle":"2022-07-09T12:36:30.732784Z","shell.execute_reply.started":"2022-07-09T12:36:30.72334Z","shell.execute_reply":"2022-07-09T12:36:30.731652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission_xg.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:37:04.699283Z","iopub.execute_input":"2022-07-09T12:37:04.699887Z","iopub.status.idle":"2022-07-09T12:37:04.711001Z","shell.execute_reply.started":"2022-07-09T12:37:04.699849Z","shell.execute_reply":"2022-07-09T12:37:04.709944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_light = light_clf.predict(testm)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:37:50.45293Z","iopub.execute_input":"2022-07-09T12:37:50.453969Z","iopub.status.idle":"2022-07-09T12:37:50.47371Z","shell.execute_reply.started":"2022-07-09T12:37:50.453897Z","shell.execute_reply":"2022-07-09T12:37:50.472465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_light.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:37:59.309404Z","iopub.execute_input":"2022-07-09T12:37:59.309745Z","iopub.status.idle":"2022-07-09T12:37:59.316151Z","shell.execute_reply.started":"2022-07-09T12:37:59.309716Z","shell.execute_reply":"2022-07-09T12:37:59.315169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv('../input/house-prices-advanced-regression-techniques/sample_submission.csv')\nsubmission['SalePrice']=y_pred_test_lgbm","metadata":{"execution":{"iopub.status.busy":"2022-07-09T12:59:46.665714Z","iopub.execute_input":"2022-07-09T12:59:46.666381Z","iopub.status.idle":"2022-07-09T12:59:46.677296Z","shell.execute_reply.started":"2022-07-09T12:59:46.666343Z","shell.execute_reply":"2022-07-09T12:59:46.676401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission_lgbm_updated.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:00:02.831496Z","iopub.execute_input":"2022-07-09T13:00:02.831843Z","iopub.status.idle":"2022-07-09T13:00:02.844819Z","shell.execute_reply.started":"2022-07-09T13:00:02.831813Z","shell.execute_reply":"2022-07-09T13:00:02.843723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_xg=pd.read_csv('./submission_xg.csv')\nsub_lgbm=pd.read_csv('./submission_lgbm_updated.csv')\nsub_final=pd.read_csv('../input/house-prices-advanced-regression-techniques/sample_submission.csv')\nsub_final['SalePrice']=sub_xg['SalePrice']*0.4+sub_lgbm['SalePrice']*0.5\nsub_final.to_csv('sub_final.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:36:15.226002Z","iopub.execute_input":"2022-07-09T13:36:15.226351Z","iopub.status.idle":"2022-07-09T13:36:15.24764Z","shell.execute_reply.started":"2022-07-09T13:36:15.22632Z","shell.execute_reply":"2022-07-09T13:36:15.246544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}