{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing the Dependencies\nimport numpy as np\nimport pandas as pd\npd.options.display.max_rows = 10\npd.options.display.max_columns = 300\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nimport matplotlib\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nfrom sklearn.impute import SimpleImputer\n!pip install feature_engine\nfrom feature_engine.encoding import OneHotEncoder\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom plotly.subplots import make_subplots\nfrom plotly import graph_objects as go\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T09:51:58.992253Z","iopub.execute_input":"2022-07-31T09:51:58.992605Z","iopub.status.idle":"2022-07-31T09:52:08.616830Z","shell.execute_reply.started":"2022-07-31T09:51:58.992575Z","shell.execute_reply":"2022-07-31T09:52:08.615636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data = pd.read_csv( '../input/widsdatathon2022/train.csv' )\nData_id_df = Data[['id']]\nData_target_df = Data[['site_eui']]\nTest_Data = pd.read_csv( '../input/widsdatathon2022/test.csv' )\nData","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:08.619660Z","iopub.execute_input":"2022-07-31T09:52:08.620441Z","iopub.status.idle":"2022-07-31T09:52:09.216287Z","shell.execute_reply.started":"2022-07-31T09:52:08.620394Z","shell.execute_reply":"2022-07-31T09:52:09.215229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.218010Z","iopub.execute_input":"2022-07-31T09:52:09.218681Z","iopub.status.idle":"2022-07-31T09:52:09.229974Z","shell.execute_reply.started":"2022-07-31T09:52:09.218643Z","shell.execute_reply":"2022-07-31T09:52:09.228912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.233484Z","iopub.execute_input":"2022-07-31T09:52:09.234194Z","iopub.status.idle":"2022-07-31T09:52:09.281325Z","shell.execute_reply.started":"2022-07-31T09:52:09.234156Z","shell.execute_reply":"2022-07-31T09:52:09.280315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ordinal_feature_names = [ 'Year_Factor' ]\nnominal_feature_names = [ 'State_Factor', 'building_class', 'facility_type'  ] \ndate_time_feature_names = []\nnumerical_feature_name = [ 'floor_area', 'year_built' ,'energy_star_rating', 'ELEVATION', 'january_min_temp', 'january_avg_temp', 'january_max_temp',\n                           'february_min_temp', 'february_avg_temp', 'february_max_temp', 'march_min_temp', 'march_avg_temp', 'march_max_temp',\n                            'april_min_temp', 'april_avg_temp', 'april_max_temp', 'may_min_temp', 'may_avg_temp', 'may_max_temp', 'june_min_temp',\n                            'june_avg_temp', 'june_max_temp', 'july_min_temp', 'july_avg_temp', 'july_max_temp', 'august_min_temp', 'august_avg_temp',\n                               'august_max_temp', 'september_min_temp', 'september_avg_temp', 'september_max_temp', 'october_min_temp', 'october_avg_temp',\n                               'october_max_temp', 'november_min_temp', 'november_avg_temp', 'november_max_temp', 'december_min_temp', 'december_avg_temp',\n                               'december_max_temp', 'cooling_degree_days', 'heating_degree_days','precipitation_inches', 'snowfall_inches', 'snowdepth_inches',\n                               'avg_temp', 'days_below_30F', 'days_below_20F', 'days_below_10F', 'days_below_0F', 'days_above_80F', 'days_above_90F',\n                               'days_above_100F', 'days_above_110F', 'direction_max_wind_speed', 'direction_peak_wind_speed', 'max_wind_speed', 'days_with_fog']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.282720Z","iopub.execute_input":"2022-07-31T09:52:09.283597Z","iopub.status.idle":"2022-07-31T09:52:09.292564Z","shell.execute_reply.started":"2022-07-31T09:52:09.283561Z","shell.execute_reply":"2022-07-31T09:52:09.291729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"df_temp = Data[['facility_type','floor_area','site_eui']].groupby( by='facility_type' ).mean().reset_index()\n\nfig = px.histogram(df_temp, x=\"facility_type\", y=\"site_eui\").update_xaxes( categoryorder='total ascending' )\nfig.update_layout( height=800 )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.294745Z","iopub.execute_input":"2022-07-31T09:52:09.295622Z","iopub.status.idle":"2022-07-31T09:52:09.370603Z","shell.execute_reply.started":"2022-07-31T09:52:09.295594Z","shell.execute_reply":"2022-07-31T09:52:09.369717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(df_temp, x=\"facility_type\", y=\"floor_area\").update_xaxes( categoryorder='total ascending' )\nfig.update_layout( height=800 )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.371834Z","iopub.execute_input":"2022-07-31T09:52:09.373059Z","iopub.status.idle":"2022-07-31T09:52:09.427453Z","shell.execute_reply.started":"2022-07-31T09:52:09.373013Z","shell.execute_reply":"2022-07-31T09:52:09.426477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.sunburst(Data, path=['State_Factor', 'building_class', 'facility_type'], values='site_eui')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:09.429064Z","iopub.execute_input":"2022-07-31T09:52:09.429396Z","iopub.status.idle":"2022-07-31T09:52:10.755363Z","shell.execute_reply.started":"2022-07-31T09:52:09.429362Z","shell.execute_reply":"2022-07-31T09:52:10.754479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.sunburst(Data, path=['State_Factor', 'building_class', 'facility_type'], values='avg_temp')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:10.756621Z","iopub.execute_input":"2022-07-31T09:52:10.758094Z","iopub.status.idle":"2022-07-31T09:52:12.088156Z","shell.execute_reply.started":"2022-07-31T09:52:10.758049Z","shell.execute_reply":"2022-07-31T09:52:12.084926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(Data, x=\"site_eui\", lines=True, marginal=\"histogram\")\nfig.update_layout( height=1000 )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:12.092697Z","iopub.execute_input":"2022-07-31T09:52:12.093615Z","iopub.status.idle":"2022-07-31T09:52:12.242258Z","shell.execute_reply.started":"2022-07-31T09:52:12.093531Z","shell.execute_reply":"2022-07-31T09:52:12.241393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(Data, x=\"site_eui\", color='building_class', facet_row='State_Factor', lines=True, marginal=\"histogram\")\nfig.update_layout( height=1500 )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:12.243545Z","iopub.execute_input":"2022-07-31T09:52:12.244497Z","iopub.status.idle":"2022-07-31T09:52:12.572075Z","shell.execute_reply.started":"2022-07-31T09:52:12.244460Z","shell.execute_reply":"2022-07-31T09:52:12.570907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_histograms( df ):\n  i=0\n  fig = make_subplots( rows=1, cols=df.shape[1],subplot_titles=df.columns.values )\n  for feature in df.columns.values:\n    fig.add_trace( go.Histogram( x=df[feature], name=feature), row=1,col=i+1 )\n    i = i+1\n\n  no_of_features = len(df.columns.values)\n  fig.update_layout( bargap=0.2, width= no_of_features*800, height=700 )\n  fig.show()\n  return","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:12.573868Z","iopub.execute_input":"2022-07-31T09:52:12.574234Z","iopub.status.idle":"2022-07-31T09:52:12.582502Z","shell.execute_reply.started":"2022-07-31T09:52:12.574199Z","shell.execute_reply":"2022-07-31T09:52:12.581381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_missing( df ):  \n  for column in df.columns.values:\n    print( \"Feature:- \", column )\n    print(\"No. of Unique Values:-\", len( list(df[column].unique())) )\n    print(\"Unique Values:-\", list( df[column].unique() ) )\n    print('Percentage of Missing Values:- ', df[column].isnull().sum()/df[column].shape[0]*100 )\n    print(\"--------------------------------------------------------------\")\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:12.584359Z","iopub.execute_input":"2022-07-31T09:52:12.584733Z","iopub.status.idle":"2022-07-31T09:52:12.593347Z","shell.execute_reply.started":"2022-07-31T09:52:12.584677Z","shell.execute_reply":"2022-07-31T09:52:12.592413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_histograms( Data[nominal_feature_names] )","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:12.595119Z","iopub.execute_input":"2022-07-31T09:52:12.596276Z","iopub.status.idle":"2022-07-31T09:52:13.540614Z","shell.execute_reply.started":"2022-07-31T09:52:12.596212Z","shell.execute_reply":"2022-07-31T09:52:13.539718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_unique_count_skew( df ):  \n  for column in df.columns.values:\n    print( \"Feature:- \", column )\n    print(\"No. of Unique Values:-\", len( list(df[column].unique())) )\n    print(\"Skewness:-\", df[column].skew(skipna = True) )\n    print('Percentage of Missing Values:- ', df[column].isnull().sum()/df[column].shape[0]*100 )\n    print(\"--------------------------------------------------------------\")\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:13.542288Z","iopub.execute_input":"2022-07-31T09:52:13.542897Z","iopub.status.idle":"2022-07-31T09:52:13.550074Z","shell.execute_reply.started":"2022-07-31T09:52:13.542859Z","shell.execute_reply":"2022-07-31T09:52:13.549170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# year_built = 0 | don't makes any sense | replacing them with mode value\nData['year_built'].replace( to_replace=[0],  value=Data['year_built'].mode(), inplace=True )","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:13.551599Z","iopub.execute_input":"2022-07-31T09:52:13.552232Z","iopub.status.idle":"2022-07-31T09:52:13.563715Z","shell.execute_reply.started":"2022-07-31T09:52:13.552198Z","shell.execute_reply":"2022-07-31T09:52:13.562564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_histograms( Data[numerical_feature_name] )","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:52:13.565930Z","iopub.execute_input":"2022-07-31T09:52:13.566659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data.columns.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_scatter_trend( df, columns ):\n  i=0\n  fig = make_subplots( rows=1, cols=len(columns),subplot_titles=columns )\n  for feature in columns:\n    fig.add_trace( go.Scatter( x=df[feature] , y=df['site_eui'], mode='markers', ), row=1,col=i+1 )\n    i = i+1\n\n  no_of_features = len(columns)\n  fig.update_layout( bargap=0.2, width= no_of_features*800, height=700 )\n  fig.show()\n  return\n\nplot_scatter_trend( Data, columns=['Year_Factor', 'year_built', 'energy_star_rating', 'avg_temp', 'direction_max_wind_speed', 'days_with_fog' ] )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_avg_temp=[ 'january_avg_temp', 'february_avg_temp', 'march_avg_temp', 'april_avg_temp', 'may_avg_temp', 'june_avg_temp', 'july_avg_temp', 'august_avg_temp', \n          'september_avg_temp', 'october_avg_temp', 'november_avg_temp', 'december_avg_temp' ]\n# fig = make_subplots( rows=1, cols=Data.shape[1],subplot_titles=columns)\nfig = go.Figure()\nfor feature in columns_avg_temp:\n    fig.add_trace( go.Histogram( x=Data[feature], name=feature )  )\n\nfig.update_layout(barmode='overlay')\nfig.update_traces(opacity=0.75)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_avg_temp=[ 'january_avg_temp', 'february_avg_temp', 'march_avg_temp', 'april_avg_temp', 'may_avg_temp', 'june_avg_temp', 'july_avg_temp', 'august_avg_temp', \n          'september_avg_temp', 'october_avg_temp', 'november_avg_temp', 'december_avg_temp' ]\n# fig = make_subplots( rows=1, cols=Data.shape[1],subplot_titles=columns)\nfig = go.Figure()\nfor feature in columns_avg_temp:\n    fig.add_trace( go.Box( x=Data[feature], name=feature)  )\n\nfig.update_layout( height=1500 )\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_temp = Data[['year_built','site_eui','energy_star_rating','building_class']]\ndf_temp.dropna(axis=0,inplace=True)\nfig = px.scatter(df_temp, x=\"year_built\", y=\"energy_star_rating\",  size=\"site_eui\", color=\"building_class\", size_max=30)\nfig.update_layout( height=1000 )\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Coorealation Matrix\npx.imshow( Data[numerical_feature_name].corr(),color_continuous_scale='RdBu_r', width=1200, height=1000 )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_columns = ['building_class','ELEVATION', 'cooling_degree_days', 'heating_degree_days' ,'precipitation_inches', 'snowfall_inches', 'snowdepth_inches', 'avg_temp', 'site_eui' ]\nsns.pairplot(Data[temp_columns] , hue='building_class', size=4.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with sns.axes_style('white'):\n    sns.jointplot(\"heating_degree_days\", \"avg_temp\", Data, kind='kde');\n    \n\nwith sns.axes_style('white'):\n    sns.jointplot(\"snowfall_inches\", \"snowdepth_inches\", Data, kind='kde');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Pre-Processing","metadata":{}},{"cell_type":"code","source":"# Features with large no of missing values\nlarge_missing_features = ['energy_star_rating', 'direction_max_wind_speed', 'direction_peak_wind_speed', 'max_wind_speed', 'days_with_fog']\n# Data.drop( columns=large_missing_features, inplace=True )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data.drop(columns=['id','site_eui'],inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ordinal and Nominal Features","metadata":{}},{"cell_type":"code","source":"print( unique_missing(Data[nominal_feature_names]) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data[nominal_feature_names]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from feature_engine.encoding import OneHotEncoder\nohe_encoder = OneHotEncoder(variables=['State_Factor','building_class'], drop_last=False)\nohe_encoder.fit(Data)\nData = ohe_encoder.transform( Data )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from feature_engine.encoding import OrdinalEncoder\nord_encoder = OrdinalEncoder(variables=['facility_type'],encoding_method='arbitrary')\nord_encoder.fit(Data)\nData = ord_encoder.transform( Data )\n\nData","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( unique_missing(Data[ordinal_feature_names]) )\nData[ordinal_feature_names]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Numerical Features","metadata":{}},{"cell_type":"code","source":"print( missing_unique_count_skew(Data[numerical_feature_name]) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data[numerical_feature_name].describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imputing Missing Values\n(Features having small count of missing values)","metadata":{}},{"cell_type":"code","source":"from feature_engine.imputation import MeanMedianImputer\n\nCI_year_built = MeanMedianImputer( variables='year_built' )\nCI_year_built.fit(Data)\nData = CI_year_built.transform(Data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Features with Large Number of Missing Values\n# high_missing_values_features = [ 'energy_star_rating', 'direction_max_wind_speed', 'direction_peak_wind_speed', 'max_wind_speed', 'days_with_fog' ]\n# Data.drop( columns=high_missing_values_features, inplace=True )\n# numerical_feature_name = list( set(numerical_feature_name)-set(high_missing_values_features) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Test Split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train,X_test,y_train,y_test = train_test_split( Data, Data_target_df, test_size=0.25, random_state=10 )\nX_train.shape,X_test.shape,y_train.shape,y_test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imputing Missing Values\n(Features having large count of missing values)","metadata":{}},{"cell_type":"code","source":"from sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom xgboost import XGBRegressor\n\nimputer_high_mis_val_var = IterativeImputer( estimator=XGBRegressor(), max_iter=5, skip_complete=True, verbose=2)\nimputer_high_mis_val_var.fit( X_train )\nX_train = pd.DataFrame(imputer_high_mis_val_var.transform(X_train), columns=X_train.columns)\nX_test = pd.DataFrame(imputer_high_mis_val_var.transform(X_test), columns=X_test.columns)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.to_csv('X_train_preprocessed.csv')\nX_test.to_csv('X_test_preprocessed.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assumptions of Common Machine Learning Models","metadata":{}},{"cell_type":"code","source":"data=X_train.copy()\ndata['site_eui']=y_train['site_eui'].values\ndata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assumption 1: There is a Linear Relationship between the Independent and Dependent Variables.","metadata":{}},{"cell_type":"code","source":"numerical_feature_name.append('site_eui')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig_sns = sns.pairplot(data, x_vars=list(set(numerical_feature_name)-set(['site_eui'])), y_vars='site_eui')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[numerical_feature_name].corr()[['site_eui']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Target Variable is not linearly correlated with any of numerical features","metadata":{}},{"cell_type":"markdown","source":"### Assumption 2: No Multicollinearity | VIF","metadata":{}},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\n\ndef VIF( df ):\n    vif = pd.DataFrame()\n    vif['Features'] = df.columns.values\n    vif[\"VIF Value\"] = [variance_inflation_factor(df.values, i) for i in range(len(df.columns))]\n    return vif\n    \nVIF_df = VIF( data[numerical_feature_name] )\nVIF_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(x=VIF_df[\"Features\"], y=np.log(VIF_df['VIF Value']))\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VIF = 1 → No correlation","metadata":{}},{"cell_type":"code","source":"VIF_df[ VIF_df['VIF Value']<=1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VIF = 1 to 5 → Moderate correlation","metadata":{}},{"cell_type":"code","source":"VIF_df[ (VIF_df['VIF Value']>1) & (VIF_df['VIF Value']<5) ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VIF >10 → High correlation","metadata":{}},{"cell_type":"code","source":"VIF_df[ VIF_df['VIF Value']>10 ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary Report DataFrame\nreport_df = VIF_df.copy()\nreport_df['Correlation with Target Variable'] = data[numerical_feature_name].corr()['site_eui'].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assumption 3: No Autocorrelation","metadata":{}},{"cell_type":"code","source":"temp = []\nfor columns in report_df['Features'].values:\n    temp.append(data[columns].autocorr())\n\nreport_df['Autocorrelation Lag_1'] = temp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(x=report_df[\"Features\"], y=report_df['Autocorrelation Lag_1'])\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Autocorrelation is not present","metadata":{}},{"cell_type":"code","source":"from statsmodels.stats.stattools import durbin_watson\ntemp = []\nfor columns in report_df['Features'].values:\n    temp.append( durbin_watson(data[columns].values) )\n\nreport_df['Durbin – Watson (DW) statistic'] = temp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DW = 2, implies no autocorrelation","metadata":{}},{"cell_type":"code","source":"report_df[ report_df['Durbin – Watson (DW) statistic']==2 ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0 < DW < 2 implies positive autocorrelation","metadata":{}},{"cell_type":"code","source":"report_df[ report_df['Durbin – Watson (DW) statistic']<2 ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" 2 < DW < 4 indicates negative autocorrelation","metadata":{}},{"cell_type":"code","source":"report_df[ (report_df['Durbin – Watson (DW) statistic']>2) & (report_df['Durbin – Watson (DW) statistic']<4)  ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assumption 4: Mean of Residuals","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nregr = LinearRegression()\nregr.fit(X_train,y_train)\ny_pred = regr.predict(X_train)\nresiduals = y_train.values-y_pred\nresiduals = list(residuals.reshape(1,len(residuals))[0])\nmean_residuals = np.mean( residuals )\nprint(\"Mean of Residuals\",(mean_residuals))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mean of Residuals is very small","metadata":{}},{"cell_type":"markdown","source":"### Assumption 5: Residuals should be Homoskedastic","metadata":{}},{"cell_type":"code","source":"y_pred = list(y_pred.reshape(1,len(y_pred))[0])\nfig = px.scatter( x=list(y_train['site_eui'].values), y=y_pred, trendline=\"ols\")\nfig.update_layout(xaxis_title='Fitted Values', yaxis_title='Residuals', title='Residuals vs Fitted Values Plot' )\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assumption 6: Normal Distribution","metadata":{}},{"cell_type":"code","source":"p = sns.distplot(residuals,kde=True)\np = plt.title('Normality of error terms/residuals')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the Q-Q plot to graphically check for the hypothesis\nfrom scipy import stats\nfor columns in numerical_feature_name:\n    print(columns)\n    res = stats.probplot(data[columns], plot=plt)\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Models and Evaluation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\ndef Evaluation(model,X_train,X_test,y_train,y_test,hypertuning=False):\n  if hypertuning==True:\n    print(\"Param for GS\", model.best_params_)\n    print(\"CV score for GS\", model.best_score_)\n\n  print( \"-----------------------------------------------------------------------------------------------------------\")\n  #print( model )\n  print( \" For Train Set :  \")\n  y_pred = model.predict(X_train)\n\n  rmse_train = mean_squared_error( y_train, y_pred, squared=False )\n  print(\"Train RMSE = \", rmse_train )\n  r2_score_train = r2_score( y_train, y_pred )\n  print( \"Train R2 Score: \", r2_score_train )\n    \n  print( \" For Test Set :  \")\n  y_pred = model.predict(X_test)\n  rmse_test = mean_squared_error( y_test, y_pred, squared=False )\n  print(\"Test RMSE = \", rmse_test )\n  r2_score_test = r2_score( y_test, y_pred )\n  print( \"Test R2 Score: \", r2_score_test )\n\n  print('------------------------------------------------------------------------------------------------------------')\n  print(\"\\n\")\n\n  return  rmse_train, rmse_test, r2_score_train, r2_score_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.linear_model import ElasticNet, Lasso, LinearRegression, LogisticRegression, Ridge, SGDRegressor\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, ExtraTreesRegressor, AdaBoostRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_models_with_default_paramters(X_train,X_test,y_train,y_test):\n  models_default = [ RandomForestRegressor(), XGBRegressor(base_score=0.5), \n                    GradientBoostingRegressor(), ExtraTreesRegressor(),\n                     ElasticNet(), Lasso(), Ridge(), LinearRegression(),\n                     SGDRegressor(),AdaBoostRegressor(),\n                    # KNeighborsRegressor(), # SVR(), \n                     ]\n\n  RMSE_train = []\n  RMSE_test = []\n  R2_Score_train = []\n  R2_Score_test = []\n  Model_Name = []\n\n  for model in models_default:\n    print(model)\n    Model_Name.append( model )\n\n    model.fit(X_train, y_train['site_eui'].ravel())\n    rmse_train, rmse_test, r2_score_train, r2_score_test = Evaluation(model,X_train,X_test,y_train,y_test,False)\n    \n    RMSE_train.append( rmse_train )\n    RMSE_test.append( rmse_test )\n    R2_Score_train.append( r2_score_train )\n    R2_Score_test.append( r2_score_test )\n    \n  results = pd.DataFrame()\n  results['Model_Name'] = Model_Name\n\n  train_test_RMSE_difference = np.subtract(RMSE_train,RMSE_test)  # To Check Overfitting/Underfitting\n  train_test_r2_score_difference = np.subtract(r2_score_train,r2_score_test)  # To Check Overfitting/Underfitting\n\n  results['RMSE on Test Set'] = RMSE_test\n  results['RMSE on Train Set'] = RMSE_train\n  results['Difference of RMSE on train and test set'] = train_test_RMSE_difference\n    \n  results['R2 Score on Test Set'] = r2_score_test\n  results['R2 Score on Train Set'] = r2_score_train\n  results['Difference of R2 Score on train and test set'] = train_test_r2_score_difference\n\n  results = results.sort_values(by=['RMSE on Test Set','Difference of RMSE on train and test set'],ascending = [True, False]) \n\n  return results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = apply_models_with_default_paramters(X_train,X_test,y_train,y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying Catboost Algorithm","metadata":{}},{"cell_type":"code","source":"X_train['facility_type']=X_train['facility_type'].astype(int)\nX_test['facility_type']=X_test['facility_type'].astype(int)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CBR = CatBoostRegressor( eval_metric='RMSE', random_seed=10 )\nCBR.fit(X_train, y_train, eval_set=(X_test,y_test),cat_features=['facility_type'], use_best_model=True, verbose=True)\n\ny_train_pred = CBR.predict(X_train)\ny_test_pred = CBR.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hypertuning Using Optuna","metadata":{}},{"cell_type":"code","source":"!pip install optuna\nimport optuna\nfrom sklearn.model_selection import cross_val_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define the Tuning function so that it can be reused","metadata":{}},{"cell_type":"code","source":"def tune(objective):\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(objective, n_trials=15)\n    \n    params = study.best_params\n    best_score = study.best_value\n    print(f\"Best score: {best_score}\\n\")\n    print(f\"Optimized parameters: {params}\\n\")\n    return params","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define Objective of each model","metadata":{}},{"cell_type":"code","source":"def randomforest_objective(trial):\n    _n_estimators = trial.suggest_int(\"n_estimators\", 50, 1000)\n    _max_depth = trial.suggest_int(\"max_depth\", 2, 20)\n    _min_samp_split = trial.suggest_int(\"min_samples_split\", 2, 10)\n    _min_samples_leaf = trial.suggest_int(\"min_samples_leaf\", 2, 10)\n    #_max_features = trial.suggest_int(\"max_features\", 5, 50)\n\n    rf = RandomForestRegressor( max_depth=_max_depth,\n                                min_samples_split=_min_samp_split,\n                                min_samples_leaf=_min_samples_leaf,\n                                #max_features=_max_features,\n                                n_estimators=_n_estimators\n                              )\n    scores = cross_val_score( rf, X_train, y_train, cv=3, scoring=\"neg_root_mean_squared_error\"  )\n    return scores.mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tuning Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"# best_params_RF = tune( randomforest_objective )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Output of Tuning ( best_params_RF ) | RandomForest is working very slow","metadata":{}},{"cell_type":"code","source":"# [I 2022-07-28 04:08:54,872] A new study created in memory with name: no-name-630d1ca7-714e-4459-80ab-05d41bcd694a\n# [I 2022-07-28 04:33:03,855] Trial 0 finished with value: -44.87386641835478 and parameters: {'n_estimators': 617, 'max_depth': 10, 'min_samples_split': 6, 'min_samples_leaf': 5}. Best is trial 0 with value: -44.87386641835478.\n# [I 2022-07-28 04:45:04,724] Trial 1 finished with value: -43.77271070037237 and parameters: {'n_estimators': 188, 'max_depth': 20, 'min_samples_split': 4, 'min_samples_leaf': 8}. Best is trial 1 with value: -43.77271070037237.\n# [I 2022-07-28 05:01:09,984] Trial 2 finished with value: -43.52224729224213 and parameters: {'n_estimators': 311, 'max_depth': 14, 'min_samples_split': 9, 'min_samples_leaf': 3}. Best is trial 2 with value: -43.52224729224213.\n# [I 2022-07-28 05:08:04,089] Trial 3 finished with value: -44.019138170470214 and parameters: {'n_estimators': 160, 'max_depth': 11, 'min_samples_split': 3, 'min_samples_leaf': 2}. Best is trial 2 with value: -43.52224729224213.\n# [I 2022-07-28 05:30:13,704] Trial 4 finished with value: -43.41654345351314 and parameters: {'n_estimators': 356, 'max_depth': 18, 'min_samples_split': 4, 'min_samples_leaf': 5}. Best is trial 4 with value: -43.41654345351314.\n# [I 2022-07-28 05:36:50,708] Trial 5 finished with value: -49.190115248890336 and parameters: {'n_estimators': 423, 'max_depth': 4, 'min_samples_split': 5, 'min_samples_leaf': 2}. Best is trial 4 with value: -43.41654345351314.\n# [I 2022-07-28 05:57:46,917] Trial 6 finished with value: -46.06399958456198 and parameters: {'n_estimators': 673, 'max_depth': 8, 'min_samples_split': 7, 'min_samples_leaf': 8}. Best is trial 4 with value: -43.41654345351314.\n# [I 2022-07-28 06:04:09,282] Trial 7 finished with value: -53.07774494321758 and parameters: {'n_estimators': 829, 'max_depth': 2, 'min_samples_split': 7, 'min_samples_leaf': 10}. Best is trial 4 with value: -43.41654345351314.\n# [I 2022-07-28 06:13:50,677] Trial 8 finished with value: -49.32954837953239 and parameters: {'n_estimators': 613, 'max_depth': 4, 'min_samples_split': 2, 'min_samples_leaf': 9}. Best is trial 4 with value: -43.41654345351314.\n# [I 2022-07-28 06:21:54,058] Trial 9 finished with value: -45.047269158318386 and parameters: {'n_estimators': 191, 'max_depth': 11, 'min_samples_split': 8, 'min_samples_leaf': 10}. Best is trial 4 with value: -43.41654345351314.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tuning XGBoost Regressor","metadata":{}},{"cell_type":"code","source":"def xgboost_objective(trial):\n    _n_estimators = trial.suggest_int(\"n_estimators\", 50, 2000)\n    _max_depth = trial.suggest_int(\"max_depth\", 2, 20)\n    _learning_rate = trial.suggest_float( 'learning_rate', 0.05, 0.30 )\n    _min_child_weight=trial.suggest_int(  'min_child_weight' , 1 , 7 )\n    _gamma=trial.suggest_float('gamma', 0.0, 0.4)\n    _colsample_bytree=trial.suggest_float('colsample_bytree', 0.3, 0.7)\n\n    xgb = XGBRegressor( n_estimators=_n_estimators,\n                        max_depth=_max_depth,\n                        learning_rate=_learning_rate,\n                           min_child_weight = _min_child_weight,\n                           gamma = _gamma, \n                           colsample_bytree = _colsample_bytree,\n                           #max_features=_max_features,\n                            random_state=10\n                              )\n    scores = cross_val_score( xgb, X_train, y_train, cv=3, scoring=\"neg_root_mean_squared_error\"  )\n    return scores.mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params_xgb = tune( xgboost_objective )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Output of Tuning ( best_params_xgb )","metadata":{}},{"cell_type":"code","source":"# [I 2022-07-28 09:24:19,487] A new study created in memory with name: no-name-9e4530fa-dd1e-4633-87c3-25a2d7f9739f\n# [I 2022-07-28 09:28:59,818] Trial 0 finished with value: -40.809074273464404 and parameters: {'n_estimators': 426, 'max_depth': 7}. Best is trial 0 with value: -40.809074273464404.\n# [I 2022-07-28 09:36:53,598] Trial 1 finished with value: -40.85250917243983 and parameters: {'n_estimators': 496, 'max_depth': 10}. Best is trial 0 with value: -40.809074273464404.\n# [I 2022-07-28 09:37:34,650] Trial 2 finished with value: -42.156828257564726 and parameters: {'n_estimators': 86, 'max_depth': 5}. Best is trial 0 with value: -40.809074273464404.\n# [I 2022-07-28 09:39:26,729] Trial 3 finished with value: -43.36985830999338 and parameters: {'n_estimators': 518, 'max_depth': 2}. Best is trial 0 with value: -40.809074273464404.\n# [I 2022-07-28 09:41:49,745] Trial 4 finished with value: -41.708108377177155 and parameters: {'n_estimators': 111, 'max_depth': 13}. Best is trial 0 with value: -40.809074273464404.\n# [I 2022-07-28 09:49:21,820] Trial 5 finished with value: -40.740714381400885 and parameters: {'n_estimators': 688, 'max_depth': 7}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 10:04:55,666] Trial 6 finished with value: -42.93454980158758 and parameters: {'n_estimators': 438, 'max_depth': 20}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 10:08:09,354] Trial 7 finished with value: -41.325662163117165 and parameters: {'n_estimators': 228, 'max_depth': 9}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 10:27:58,096] Trial 8 finished with value: -42.63241820454128 and parameters: {'n_estimators': 737, 'max_depth': 19}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 10:37:33,919] Trial 9 finished with value: -41.26169552691501 and parameters: {'n_estimators': 548, 'max_depth': 11}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:00:28,717] Trial 10 finished with value: -42.02260589200941 and parameters: {'n_estimators': 855, 'max_depth': 16}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:09:48,827] Trial 11 finished with value: -40.769306837097595 and parameters: {'n_estimators': 1000, 'max_depth': 6}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:16:05,606] Trial 12 finished with value: -41.6406131117315 and parameters: {'n_estimators': 985, 'max_depth': 4}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:24:12,741] Trial 13 finished with value: -40.760321407003666 and parameters: {'n_estimators': 734, 'max_depth': 7}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:40:22,009] Trial 14 finished with value: -41.931185768484895 and parameters: {'n_estimators': 708, 'max_depth': 14}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:48:49,718] Trial 15 finished with value: -40.7629538317194 and parameters: {'n_estimators': 676, 'max_depth': 8}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:52:53,200] Trial 16 finished with value: -42.01819362842996 and parameters: {'n_estimators': 810, 'max_depth': 3}. Best is trial 5 with value: -40.740714381400885.\n# [I 2022-07-28 11:59:56,806] Trial 17 finished with value: -40.73506414335266 and parameters: {'n_estimators': 641, 'max_depth': 7}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:11:04,708] Trial 18 finished with value: -41.35326178678547 and parameters: {'n_estimators': 579, 'max_depth': 12}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:13:16,202] Trial 19 finished with value: -41.49518020168499 and parameters: {'n_estimators': 277, 'max_depth': 5}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:17:57,178] Trial 20 finished with value: -41.33021616259831 and parameters: {'n_estimators': 330, 'max_depth': 9}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:24:48,953] Trial 21 finished with value: -40.739700446410566 and parameters: {'n_estimators': 627, 'max_depth': 7}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:31:35,501] Trial 22 finished with value: -40.73956115230045 and parameters: {'n_estimators': 623, 'max_depth': 7}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:41:22,782] Trial 23 finished with value: -40.85377426386809 and parameters: {'n_estimators': 620, 'max_depth': 10}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:48:02,579] Trial 24 finished with value: -41.01618836251545 and parameters: {'n_estimators': 852, 'max_depth': 5}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:50:13,354] Trial 25 finished with value: -43.29436840100234 and parameters: {'n_estimators': 611, 'max_depth': 2}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 12:56:13,846] Trial 26 finished with value: -41.27254158191043 and parameters: {'n_estimators': 425, 'max_depth': 9}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 13:03:34,674] Trial 27 finished with value: -40.82018628833438 and parameters: {'n_estimators': 785, 'max_depth': 6}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 13:19:20,046] Trial 28 finished with value: -42.23757362231371 and parameters: {'n_estimators': 637, 'max_depth': 15}. Best is trial 17 with value: -40.73506414335266.\n# [I 2022-07-28 13:24:07,093] Trial 29 finished with value: -40.87969161935255 and parameters: {'n_estimators': 385, 'max_depth': 8}. Best is trial 17 with value: -40.73506414335266.\n\n# Best score: -40.73506414335266\n# Optimized parameters: {'n_estimators': 641, 'max_depth': 7}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    param = {\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.5),\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"bootstrap_type\": trial.suggest_categorical(\"bootstrap_type\", [\"Bayesian\", \"Bernoulli\", \"MVS\"]),\n        \"used_ram_limit\": \"12gb\", 'verbose':0, 'eval_metric':'RMSE','task_type':\"GPU\"\n    }\n\n    if param[\"bootstrap_type\"] == \"Bayesian\":\n        param[\"bagging_temperature\"] = trial.suggest_float(\"bagging_temperature\", 0, 1)\n    elif param[\"bootstrap_type\"] == \"Bernoulli\":\n        param[\"subsample\"] = trial.suggest_float(\"subsample\", 0.1, 1)\n\n    CBR = CatBoostRegressor(**param)\n\n    CBR.fit(X_train, y_train, eval_set=[(X_test, y_test)], use_best_model=True ,verbose=0, early_stopping_rounds=100)\n\n    y_pred = CBR.predict(X_test)\n    rmse = mean_squared_error( y_test, y_pred, squared=False )\n    return rmse","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=50)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# [I 2022-07-29 17:54:18,902] A new study created in memory with name: no-name-eedd9c7b-0394-46a7-8c6c-ca013e1505b3\n# [I 2022-07-29 17:54:26,369] Trial 0 finished with value: 49.58111223963894 and parameters: {'learning_rate': 0.013831432790481997, 'depth': 5, 'bootstrap_type': 'MVS'}. Best is trial 0 with value: 49.58111223963894.\n# [I 2022-07-29 17:54:40,483] Trial 1 finished with value: 45.387067886496176 and parameters: {'learning_rate': 0.09396338825063771, 'depth': 12, 'bootstrap_type': 'MVS'}. Best is trial 1 with value: 45.387067886496176.\n# [I 2022-07-29 17:54:46,164] Trial 2 finished with value: 44.323981093529554 and parameters: {'learning_rate': 0.35530763583293856, 'depth': 5, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.4366344370472929}. Best is trial 2 with value: 44.323981093529554.\n# [I 2022-07-29 17:55:05,393] Trial 3 finished with value: 42.842758033288554 and parameters: {'learning_rate': 0.1614373054564734, 'depth': 12, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.12356192506074615}. Best is trial 3 with value: 42.842758033288554.\n# [I 2022-07-29 17:55:12,048] Trial 4 finished with value: 44.08377949388904 and parameters: {'learning_rate': 0.24866669766462365, 'depth': 5, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.4011256912505806}. Best is trial 3 with value: 42.842758033288554.\n# [I 2022-07-29 17:55:15,941] Trial 5 finished with value: 45.75292125398273 and parameters: {'learning_rate': 0.4093416858734526, 'depth': 3, 'bootstrap_type': 'Bernoulli', 'subsample': 0.8354709538497008}. Best is trial 3 with value: 42.842758033288554.\n# [I 2022-07-29 17:55:17,494] Trial 6 finished with value: 51.49891773481429 and parameters: {'learning_rate': 0.17887487466952953, 'depth': 2, 'bootstrap_type': 'MVS'}. Best is trial 3 with value: 42.842758033288554.\n# [I 2022-07-29 17:55:32,210] Trial 7 finished with value: 42.50810242391222 and parameters: {'learning_rate': 0.1142511705148167, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.18122255085035577}. Best is trial 7 with value: 42.50810242391222.\n# [I 2022-07-29 17:55:35,036] Trial 8 finished with value: 49.55056947704093 and parameters: {'learning_rate': 0.4967446079779853, 'depth': 4, 'bootstrap_type': 'MVS'}. Best is trial 7 with value: 42.50810242391222.\n# [I 2022-07-29 17:55:42,309] Trial 9 finished with value: 44.174333467514806 and parameters: {'learning_rate': 0.45533316206533153, 'depth': 12, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.2226022382589382}. Best is trial 7 with value: 42.50810242391222.\n# [I 2022-07-29 17:55:53,427] Trial 10 finished with value: 47.142929419451406 and parameters: {'learning_rate': 0.012322417223639484, 'depth': 9, 'bootstrap_type': 'Bernoulli', 'subsample': 0.14655125759165472}. Best is trial 7 with value: 42.50810242391222.\n# [I 2022-07-29 17:56:05,683] Trial 11 finished with value: 42.63412417230119 and parameters: {'learning_rate': 0.15680966160750354, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.012776358880761562}. Best is trial 7 with value: 42.50810242391222.\n# [I 2022-07-29 17:56:17,104] Trial 12 finished with value: 42.442512364142075 and parameters: {'learning_rate': 0.11569520664624709, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.023833721988887193}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:56:23,156] Trial 13 finished with value: 44.44697018304328 and parameters: {'learning_rate': 0.2775779878201836, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.908852413650316}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:56:31,722] Trial 14 finished with value: 43.42111249383329 and parameters: {'learning_rate': 0.08432874759464284, 'depth': 7, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.23137517667956}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:56:46,487] Trial 15 finished with value: 43.35183106661987 and parameters: {'learning_rate': 0.09770977417023348, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.7372593924346121}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:56:52,664] Trial 16 finished with value: 44.78154099306481 and parameters: {'learning_rate': 0.23152102771599536, 'depth': 7, 'bootstrap_type': 'Bernoulli', 'subsample': 0.29039119390987295}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:02,697] Trial 17 finished with value: 43.27786712494999 and parameters: {'learning_rate': 0.3174010815394219, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.09801066131862789}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:13,288] Trial 18 finished with value: 43.27473619999658 and parameters: {'learning_rate': 0.06939876598012232, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.33673822881530596}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:24,419] Trial 19 finished with value: 42.71833179725281 and parameters: {'learning_rate': 0.20367230357266675, 'depth': 11, 'bootstrap_type': 'Bernoulli', 'subsample': 0.9955137336498117}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:34,243] Trial 20 finished with value: 42.97261501405963 and parameters: {'learning_rate': 0.12512545429916008, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.0001327935941669925}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:48,621] Trial 21 finished with value: 42.686004121353506 and parameters: {'learning_rate': 0.14818623884624205, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.006093994939965597}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:57:59,942] Trial 22 finished with value: 43.37860535299753 and parameters: {'learning_rate': 0.05880340943465262, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.6152488889114407}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:58:09,468] Trial 23 finished with value: 43.02743250615373 and parameters: {'learning_rate': 0.20219066679845987, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.17525637358757234}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:58:27,559] Trial 24 finished with value: 42.610391059600886 and parameters: {'learning_rate': 0.1376294848555814, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.008658179398171421}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:58:42,446] Trial 25 finished with value: 43.039748212887396 and parameters: {'learning_rate': 0.12602036988648554, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.3086800654548808}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:01,509] Trial 26 finished with value: 42.9776350553335 and parameters: {'learning_rate': 0.04721050217570187, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.1066166027150034}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:13,832] Trial 27 finished with value: 43.084688268892904 and parameters: {'learning_rate': 0.12005019435731752, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.26761165069328785}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:15,706] Trial 28 finished with value: 49.209120352896335 and parameters: {'learning_rate': 0.2811651089383239, 'depth': 6, 'bootstrap_type': 'MVS'}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:35,230] Trial 29 finished with value: 43.835333624047706 and parameters: {'learning_rate': 0.03785543200084196, 'depth': 11, 'bootstrap_type': 'Bernoulli', 'subsample': 0.5790108845934975}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:37,855] Trial 30 finished with value: 48.185482926679235 and parameters: {'learning_rate': 0.19243661993157965, 'depth': 6, 'bootstrap_type': 'MVS'}. Best is trial 12 with value: 42.442512364142075.\n# [I 2022-07-29 17:59:49,341] Trial 31 finished with value: 42.33186908472313 and parameters: {'learning_rate': 0.1507273421789995, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.008160705966694475}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:00,190] Trial 32 finished with value: 43.07897902244302 and parameters: {'learning_rate': 0.09977923805995376, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.09137460940028787}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:14,483] Trial 33 finished with value: 42.72927617239883 and parameters: {'learning_rate': 0.13674816457107422, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.06978380658152464}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:26,140] Trial 34 finished with value: 43.366781502460896 and parameters: {'learning_rate': 0.23145430548284207, 'depth': 12, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.1782444404891412}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:35,555] Trial 35 finished with value: 43.026138616010115 and parameters: {'learning_rate': 0.16072695030537476, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.02645295267369738}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:53,723] Trial 36 finished with value: 42.553673736296375 and parameters: {'learning_rate': 0.08265238356880457, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.1708804241287512}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:00:55,501] Trial 37 finished with value: 53.935377959598554 and parameters: {'learning_rate': 0.03339316309021684, 'depth': 1, 'bootstrap_type': 'MVS'}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:01:05,480] Trial 38 finished with value: 43.526131025242336 and parameters: {'learning_rate': 0.08730788633888985, 'depth': 7, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.1600739159749563}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:01:23,102] Trial 39 finished with value: 43.28675411603855 and parameters: {'learning_rate': 0.10962266744151272, 'depth': 12, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.5173257307982788}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:01:30,759] Trial 40 finished with value: 44.11266717721328 and parameters: {'learning_rate': 0.18038548224114304, 'depth': 10, 'bootstrap_type': 'Bernoulli', 'subsample': 0.5196715660465031}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:01:49,763] Trial 41 finished with value: 42.68822355027801 and parameters: {'learning_rate': 0.07085627857247963, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.07360421144644141}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:02:11,412] Trial 42 finished with value: 43.044814044640994 and parameters: {'learning_rate': 0.14237828452987944, 'depth': 12, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.18214027840166974}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:02:25,255] Trial 43 finished with value: 42.86797928478824 and parameters: {'learning_rate': 0.175670258694326, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.07756649549130837}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:02:36,966] Trial 44 finished with value: 42.85365516059054 and parameters: {'learning_rate': 0.22038151958616942, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.00031765586453797595}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:02:50,626] Trial 45 finished with value: 44.58532091295561 and parameters: {'learning_rate': 0.01755680291439171, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.38473601965442633}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:03:17,306] Trial 46 finished with value: 45.417081002718966 and parameters: {'learning_rate': 0.07843087323018191, 'depth': 12, 'bootstrap_type': 'MVS'}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:03:23,465] Trial 47 finished with value: 45.05693053089929 and parameters: {'learning_rate': 0.10830158975568026, 'depth': 4, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.1517471376363396}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:03:28,133] Trial 48 finished with value: 43.024441462973634 and parameters: {'learning_rate': 0.35821652003535775, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.2638306437575908}. Best is trial 31 with value: 42.33186908472313.\n# [I 2022-07-29 18:03:42,742] Trial 49 finished with value: 42.67190029978086 and parameters: {'learning_rate': 0.16430001372567382, 'depth': 11, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.13264564137913962}. Best is trial 31 with value: 42.33186908472313.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# [I 2022-07-29 13:14:08,553] A new study created in memory with name: no-name-1cddbb24-3faa-490f-9594-703eba1af986\n# [I 2022-07-29 13:14:18,276] Trial 0 finished with value: 44.54046255525308 and parameters: {'colsample_bylevel': 0.48866670955378627, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.7230783374990317}. Best is trial 0 with value: 44.54046255525308.\n# [I 2022-07-29 13:14:30,489] Trial 1 finished with value: 43.38562368917585 and parameters: {'colsample_bylevel': 0.16178647806025026, 'depth': 6, 'bootstrap_type': 'MVS'}. Best is trial 1 with value: 43.38562368917585.\n# [I 2022-07-29 13:14:35,424] Trial 2 finished with value: 48.588823981552544 and parameters: {'colsample_bylevel': 0.38319194841973375, 'depth': 1, 'bootstrap_type': 'MVS'}. Best is trial 1 with value: 43.38562368917585.\n# [I 2022-07-29 13:14:40,267] Trial 3 finished with value: 49.51530511652361 and parameters: {'colsample_bylevel': 0.1649822633063968, 'depth': 1, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.5768248302999776}. Best is trial 1 with value: 43.38562368917585.\n# [I 2022-07-29 13:14:55,641] Trial 4 finished with value: 42.783969905561854 and parameters: {'colsample_bylevel': 0.2028160358079163, 'depth': 7, 'bootstrap_type': 'MVS'}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:15:19,040] Trial 5 finished with value: 43.37038040538437 and parameters: {'colsample_bylevel': 0.09896468759827876, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.8047832160643467}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:15:28,653] Trial 6 finished with value: 48.128728054844736 and parameters: {'colsample_bylevel': 0.01189623193954003, 'depth': 7, 'bootstrap_type': 'Bernoulli', 'subsample': 0.13451195174647532}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:15:35,608] Trial 7 finished with value: 48.417830509937026 and parameters: {'colsample_bylevel': 0.014483457784828244, 'depth': 3, 'bootstrap_type': 'MVS'}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:15:56,305] Trial 8 finished with value: 43.8463787999906 and parameters: {'colsample_bylevel': 0.4794025323037158, 'depth': 10, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.3638012564398838}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:16:03,782] Trial 9 finished with value: 45.4610075992958 and parameters: {'colsample_bylevel': 0.21659574765865547, 'depth': 3, 'bootstrap_type': 'MVS'}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:16:54,303] Trial 10 finished with value: 43.28608085287827 and parameters: {'colsample_bylevel': 0.3242678860192164, 'depth': 12, 'bootstrap_type': 'Bernoulli', 'subsample': 0.9913535001083628}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:17:35,141] Trial 11 finished with value: 44.08911052094434 and parameters: {'colsample_bylevel': 0.3263870204897605, 'depth': 12, 'bootstrap_type': 'Bernoulli', 'subsample': 0.9404647554462564}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:18:33,627] Trial 12 finished with value: 43.43161097478708 and parameters: {'colsample_bylevel': 0.30582792100106426, 'depth': 12, 'bootstrap_type': 'Bernoulli', 'subsample': 0.9732850960057506}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:18:41,820] Trial 13 finished with value: 44.7474941942326 and parameters: {'colsample_bylevel': 0.27794701086145723, 'depth': 5, 'bootstrap_type': 'Bernoulli', 'subsample': 0.5646348745777019}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:19:26,413] Trial 14 finished with value: 43.452976776357325 and parameters: {'colsample_bylevel': 0.383396913167113, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:19:35,092] Trial 15 finished with value: 44.614903369861416 and parameters: {'colsample_bylevel': 0.2178502919811528, 'depth': 5, 'bootstrap_type': 'Bernoulli', 'subsample': 0.628184227938466}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:19:51,423] Trial 16 finished with value: 45.44966568571024 and parameters: {'colsample_bylevel': 0.3832272131389196, 'depth': 11, 'bootstrap_type': 'Bernoulli', 'subsample': 0.23439388610319206}. Best is trial 4 with value: 42.783969905561854.\n# [I 2022-07-29 13:20:12,201] Trial 17 finished with value: 42.752647463730646 and parameters: {'colsample_bylevel': 0.11207871382700671, 'depth': 8, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:20:33,338] Trial 18 finished with value: 42.94204212145742 and parameters: {'colsample_bylevel': 0.0994784128909381, 'depth': 8, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:20:48,523] Trial 19 finished with value: 43.40662118631061 and parameters: {'colsample_bylevel': 0.06857057177993339, 'depth': 7, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:20:58,416] Trial 20 finished with value: 44.080088322950004 and parameters: {'colsample_bylevel': 0.15022298860955174, 'depth': 5, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:21:19,756] Trial 21 finished with value: 42.879842163076624 and parameters: {'colsample_bylevel': 0.0954509096322517, 'depth': 8, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:21:41,174] Trial 22 finished with value: 42.763106509832134 and parameters: {'colsample_bylevel': 0.204411109506259, 'depth': 8, 'bootstrap_type': 'MVS'}. Best is trial 17 with value: 42.752647463730646.\n# [I 2022-07-29 13:22:12,648] Trial 23 finished with value: 42.442471379827474 and parameters: {'colsample_bylevel': 0.21923834852823085, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:22:43,769] Trial 24 finished with value: 42.62918086203817 and parameters: {'colsample_bylevel': 0.23782330226610962, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:23:23,337] Trial 25 finished with value: 42.70989274848187 and parameters: {'colsample_bylevel': 0.26581791705331265, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:23:52,722] Trial 26 finished with value: 43.4432506099702 and parameters: {'colsample_bylevel': 0.2650387998126775, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:24:20,693] Trial 27 finished with value: 43.011157669755015 and parameters: {'colsample_bylevel': 0.24341336640972364, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:25:00,267] Trial 28 finished with value: 43.45282249726105 and parameters: {'colsample_bylevel': 0.2887012574136051, 'depth': 11, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:25:14,505] Trial 29 finished with value: 43.69502179582406 and parameters: {'colsample_bylevel': 0.3504797018966508, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.037302242915029316}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:26:22,534] Trial 30 finished with value: 42.981398766117145 and parameters: {'colsample_bylevel': 0.24585448441814617, 'depth': 11, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:26:54,446] Trial 31 finished with value: 42.606725887433704 and parameters: {'colsample_bylevel': 0.14764513167713503, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:27:10,320] Trial 32 finished with value: 43.046125510772946 and parameters: {'colsample_bylevel': 0.18089596375364253, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:28:17,028] Trial 33 finished with value: 42.7661141701208 and parameters: {'colsample_bylevel': 0.14400167655469767, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:28:29,015] Trial 34 finished with value: 43.394954897029386 and parameters: {'colsample_bylevel': 0.23890086863073495, 'depth': 6, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:29:00,243] Trial 35 finished with value: 42.640063241510866 and parameters: {'colsample_bylevel': 0.1793652709438572, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:29:22,841] Trial 36 finished with value: 43.989993085128106 and parameters: {'colsample_bylevel': 0.1825752973650276, 'depth': 9, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.9635754002488535}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:29:53,614] Trial 37 finished with value: 42.63664978901446 and parameters: {'colsample_bylevel': 0.12590883944666564, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:30:09,338] Trial 38 finished with value: 43.42681008463609 and parameters: {'colsample_bylevel': 0.12740671638977102, 'depth': 7, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:30:21,599] Trial 39 finished with value: 44.68201583144805 and parameters: {'colsample_bylevel': 0.04773853135578858, 'depth': 6, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.018145404683223287}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:32:03,135] Trial 40 finished with value: 42.853057214735955 and parameters: {'colsample_bylevel': 0.07040757452326488, 'depth': 11, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:32:34,266] Trial 41 finished with value: 42.479392814664045 and parameters: {'colsample_bylevel': 0.17697546634625286, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:32:54,880] Trial 42 finished with value: 42.70894401616073 and parameters: {'colsample_bylevel': 0.13752459556033966, 'depth': 8, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:34:01,278] Trial 43 finished with value: 42.730499183895496 and parameters: {'colsample_bylevel': 0.1698875695818864, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 23 with value: 42.442471379827474.\n# [I 2022-07-29 13:34:32,877] Trial 44 finished with value: 42.42985200344688 and parameters: {'colsample_bylevel': 0.21969749037839517, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 44 with value: 42.42985200344688.\n# [I 2022-07-29 13:34:45,468] Trial 45 finished with value: 43.41408827780399 and parameters: {'colsample_bylevel': 0.21984899964411234, 'depth': 7, 'bootstrap_type': 'MVS'}. Best is trial 44 with value: 42.42985200344688.\n# [I 2022-07-29 13:35:03,261] Trial 46 finished with value: 42.92778083533445 and parameters: {'colsample_bylevel': 0.20158559796521977, 'depth': 9, 'bootstrap_type': 'MVS'}. Best is trial 44 with value: 42.42985200344688.\n# [I 2022-07-29 13:35:16,247] Trial 47 finished with value: 43.450493724256205 and parameters: {'colsample_bylevel': 0.46955316386350604, 'depth': 8, 'bootstrap_type': 'Bayesian', 'bagging_temperature': 0.28566967033909696}. Best is trial 44 with value: 42.42985200344688.\n# [I 2022-07-29 13:36:32,699] Trial 48 finished with value: 42.842102136123536 and parameters: {'colsample_bylevel': 0.1617542546580976, 'depth': 10, 'bootstrap_type': 'MVS'}. Best is trial 44 with value: 42.42985200344688.\n# [I 2022-07-29 13:36:38,079] Trial 49 finished with value: 48.89469406555535 and parameters: {'colsample_bylevel': 0.23111337147710878, 'depth': 1, 'bootstrap_type': 'MVS'}. Best is trial 44 with value: 42.42985200344688.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = study.best_params\nbest_score = study.best_value\nprint(f\"Best score: {best_score}\\n\")\nprint(f\"Optimized parameters: {params}\\n\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best score: 42.42985200344688\n# Optimized parameters: {'colsample_bylevel': 0.21969749037839517, 'depth': 9, 'bootstrap_type': 'MVS'}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_optimization_history(study)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study, target=lambda t: t.duration.total_seconds(), target_name=\"site_eui\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\nX = X_train.to_numpy()\ny = y_train.to_numpy()\nkf = KFold(n_splits=20)\n\nfor train_index, test_index in kf.split(X):\n  #  print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    X_train_kf, X_test_kf = X[train_index], X[test_index]\n    y_train_kf, y_test_kf = y[train_index], y[test_index]\n    \n    CBR = CatBoostRegressor( learning_rate= 0.1507273421789995, depth=9, bootstrap_type='Bayesian', bagging_temperature= 0.008160705966694475,\n                            od_type='Iter', metric_period = 200, od_wait=100, task_type=\"GPU\",\n                            used_ram_limit=\"10gb\", eval_metric='RMSE',random_seed=10)\n    CBR.fit( X_train_kf, y_train_kf, eval_set=(X_test_kf,y_test_kf), use_best_model=True )\n    \n    y_pred = CBR.predict(X_test)\n    rmse = mean_squared_error( y_test, y_pred, squared=False )\n    \n    print(\"RMSE on Test Set\", rmse )\n    print('--------------------------')\n    print('\\n')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nlist_objects = { 'final_model_CBR': CBR, 'ohe_encoder':ohe_encoder, 'ord_encoder':ord_encoder,'CI_year_built':CI_year_built}\nwith open('model_CatBR.pkl', 'wb') as files:\n  pickle.dump(list_objects, files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Model","metadata":{}},{"cell_type":"code","source":"with open('model_CatBR.pkl', 'rb') as file:\n   dict = pickle.load(file)\n    \nfinal_model_CBR = dict['final_model_CBR']\n\n# final_model_CBR.fit(X_train, y_train, eval_set=(X_test,y_test),cat_features=['facility_type'], use_best_model=True, verbose=False)\n\ny_train_pred = final_model_CBR.predict(X_train)\ny_test_pred = final_model_CBR.predict(X_test)\n\nrmse_train, rmse_test, r2_score_train, r2_score_test = Evaluation(final_model_CBR,X_train,X_test,y_train,y_test,False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(y_test_pred, kde=True) # Blue\nsns.distplot(y_test, kde=True) # Red\nsns.set(rc={'figure.figsize':(35,15)})\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Saving necessary Data and trained objects","metadata":{}},{"cell_type":"code","source":"X_train.to_csv('X_train_preprocessed.csv', encoding = 'utf-8-sig', index=False) \nX_test.to_csv('X_test_preprocessed.csv', encoding = 'utf-8-sig', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving the model","metadata":{}},{"cell_type":"code","source":"# import pickle\n# list_objects = { 'final_model_CBR': CBR, ;'ohe_encoder':ohe_encoder, 'ord_encoder':ord_encoder,'CI_year_built':CI_year_built}\n# with open('model_CatBR.pkl', 'wb') as files:\n#   pickle.dump(list_objects, files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explainable AI","metadata":{}},{"cell_type":"code","source":"!pip install shap\nimport shap\nshap.initjs()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data = X_train.sample(50)\nsample_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('model_CatBR.pkl', 'rb') as file:\n   dict = pickle.load(file)\n    \nfinal_model_CBR = dict['final_model_CBR']\nfinal_model_CBR","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap_values = shap.TreeExplainer(final_model_CBR).shap_values(sample_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values, sample_data, plot_type=\"bar\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values, sample_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.dependence_plot('facility_type', shap_values, sample_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.dependence_plot('energy_star_rating', shap_values, sample_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# row = 111\n# shap.plots._waterfall.waterfall_legacy(shap.TreeExplainer(final_model_CBR).expected_value[0], \n#                                        shap_values[row],\n#                                        feature_names=sample_set.columns.tolist()\n#                                       )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(final_model_CBR.predict(sample_data)[49])\n# shap.initjs()\n# shap.force_plot(shap.TreeExplainer(final_model_CBR).expected_value[0], shap_values[1][10], sample_data.iloc[10])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap.plots._waterfall.waterfall_legacy(shap.TreeExplainer(final_model_CBR).expected_value[0], \n#                                        shap_values[row],\n#                                        feature_names=sample_set.columns.tolist()\n#                                       )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap.decision_plot(shap.TreeExplainer(final_model_CBR).expected_value[0], \n#                    shap_values[start:shap.decision_plot(shap.TreeExplainer(final_model_CBR).expected_value[0], \n#                    shap_values[start:limit], \n#                    feature_names=sample_set.columns.tolist())], \n#                    feature_names=sample_set.columns.tolist())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"feature_importance_df = pd.DataFrame()\nfeature_importance_df['feature_name'] = X_train.columns.values\nfeature_importance_df['feature_importance_value'] = final_model_CBR.feature_importances_\nfeature_importance_df.sort_values(by='feature_importance_value',inplace=True)\nfig = px.bar( x=feature_importance_df['feature_name'], y=feature_importance_df['feature_importance_value'] )\nfig.update_layout( height=1000 )\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importance_df = feature_importance_df[feature_importance_df['feature_importance_value'] >= 0.8]\n# X_train_top = X_train[feature_importance_df['feature_name']]\n# X_test_top = X_test[feature_importance_df['feature_name']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data = pd.read_csv( '../input/widsdatathon2022/train.csv' )\nData['days_with_fog'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_columns=['heating_degree_days', 'snowdepth_inches', 'snowfall_inches', 'days_with_fog', 'State_Factor', 'february_avg_temp',\n               'building_class', 'ELEVATION', 'year_built', 'floor_area', 'energy_star_rating', 'facility_type']\n\nData = pd.read_csv( '../input/widsdatathon2022/train.csv' )\nprint(Data)\ndata_top = Data[ top_columns ]\n\nord_encoder_top = OrdinalEncoder(variables=['facility_type'],encoding_method='arbitrary')\nord_encoder_top.fit(data_top)\ndata_top = ord_encoder_top.transform( data_top )\n\nohe_encoder_top = OneHotEncoder(variables=['State_Factor','building_class'], drop_last=False)\nohe_encoder_top.fit(data_top)\ndata_top = ohe_encoder_top.transform( data_top )\n\n\nX_train_top,X_test_top,y_train,y_test = train_test_split( data_top, Data_target_df, test_size=0.25, random_state=10 )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\nX = X_train_top.to_numpy()\ny = y_train.to_numpy()\nkf = KFold(n_splits=20)\n\nfor train_index, test_index in kf.split(X):\n  #  print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    X_train_kf, X_test_kf = X[train_index], X[test_index]\n    y_train_kf, y_test_kf = y[train_index], y[test_index]\n    \n    CBR = CatBoostRegressor( learning_rate= 0.1507273421789995, depth=9, bootstrap_type='Bayesian', bagging_temperature= 0.008160705966694475,\n                            od_type='Iter', metric_period = 200, od_wait=100, task_type=\"GPU\",\n                            used_ram_limit=\"10gb\", eval_metric='RMSE',random_seed=10)\n    CBR.fit( X_train_kf, y_train_kf, eval_set=(X_test_kf,y_test_kf), use_best_model=True )\n    \n    y_pred = CBR.predict(X_test_top)\n    rmse = mean_squared_error( y_test, y_pred, squared=False )\n    \n    print(\"RMSE on Test Set\", rmse )\n    print('--------------------------')\n    print('\\n')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nlist_objects = { 'final_model_CBR_top': CBR ,'ohe_encoder_top':ohe_encoder_top, 'ord_encoder_top':ord_encoder_top}\nwith open('model_CatBR_top.pkl', 'wb') as files:\n  pickle.dump(list_objects, files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File Submission","metadata":{}},{"cell_type":"code","source":"Test_df = pd.read_csv(\"../input/widsdatathon2022/test.csv\")\nTest_df_id = Test_df[['id']]\nTest_df.drop(columns=['id'],inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique_missing( Test_df )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data['year_built'].replace( to_replace=[0],  value=Data['year_built'].mode(), inplace=True )\nTest_df = ohe_encoder.transform( Test_df )\nTest_df = ord_encoder.transform( Test_df )\nTest_df = CI_year_built.transform(Test_df)\nTest_df = pd.DataFrame(imputer_high_mis_val_var.transform(Test_df), columns=Test_df.columns)\n# list_objects = { 'final_model_CBR': CBR, 'ohe_encoder':ohe_encoder, 'ord_encoder':ord_encoder,'CI_year_built':CI_year_built}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df = pd.DataFrame()\npredictions_df['site_eui'] = final_model_CBR.predict(Test_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission_Data = pd.concat([Test_df_id,predictions_df], axis='columns')\nSubmission_Data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission_Data.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission_Data.to_csv('widsdatathon2022_Shirsh_Submission_File.csv', encoding = 'utf-8-sig', index=False) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"numerical_feature_name.remove('site_eui')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_num = X_train[numerical_feature_name] \nX_test_num = X_test[numerical_feature_name]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n# Scale data before applying PCA\nscaler=StandardScaler()\n \n# Use fit and transform method\nscaler.fit(X_train_num)\n# X_train_num=scaler.transform(X_train_num)\n# X_test_num=scaler.transform(X_test_num)\nX_train_num = pd.DataFrame( data=scaler.transform(X_train_num), columns=X_train_num.columns.values )\nX_test_num = pd.DataFrame( data=scaler.transform(X_test_num), columns=X_test_num.columns.values )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_num","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the n_components=3\nprincipal=PCA(n_components=20)\nprincipal.fit(X_train_num)\nX_train_num_pca=principal.transform(X_train_num)\nX_test_num_pca=principal.transform(X_test_num)\n\ncolumns = ['PCA_1','PCA_2','PCA_3','PCA_4','PCA_5','PCA_6','PCA_7','PCA_8','PCA_9','PCA_10',\n           'PCA_11','PCA_12','PCA_13','PCA_14','PCA_15','PCA_16','PCA_17','PCA_18','PCA_19','PCA_20']\n\nX_train_num_pca = pd.DataFrame( data=X_train_num_pca, columns=columns )\nX_test_num_pca = pd.DataFrame( data=X_test_num_pca, columns=columns )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how much variance is explained by each principal component\nprincipal.explained_variance_ratio_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(principal.explained_variance_ratio_).sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rest_columns = list(set(X_train.columns.values) - set( numerical_feature_name ))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_pca = pd.concat( [ X_train_num_pca, X_train[rest_columns] ], axis=1)\nX_test_pca = pd.concat( [ X_test_num_pca, X_test[rest_columns] ], axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = apply_models_with_default_paramters(X_train_pca,X_test_pca,y_train,y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CBR = CatBoostRegressor( eval_metric='RMSE', random_seed=10 )\nCBR.fit(X_train_pca, y_train, eval_set=(X_test_pca,y_test),cat_features=['facility_type'], use_best_model=True, verbose=True)\n\n# y_train_pred = CBR.predict(X_train)\n# y_test_pred = CBR.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}