{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9824960,"sourceType":"datasetVersion","datasetId":5964495}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### INTRODUCTION","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# IMPORT","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom catboost import CatBoostClassifier,CatBoostRegressor\nfrom lightgbm import *\nfrom tqdm import tqdm\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import OneHotEncoder,LabelEncoder\nfrom sklearn.impute import KNNImputer,SimpleImputer\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport logging\nimport gc\nimport optuna\nfrom sklearn.linear_model import *\nfrom sklearn.naive_bayes import *\nfrom sklearn.tree import *\nfrom sklearn.ensemble import *\nfrom sklearn.svm import *\nfrom sklearn.neighbors import *\nfrom datetime import timedelta\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import train_test_split,cross_val_score\nfrom sklearn.metrics import mean_absolute_percentage_error,mean_absolute_error,r2_score,f1_score\nimport dill as pickle\nimport warnings\nimport json\nfrom datetime import datetime\nimport optuna.visualization as vis\nfrom sklearn.metrics import cohen_kappa_score\nfrom imblearn.over_sampling import SMOTE\nfrom concurrent.futures import ThreadPoolExecutor\nlogging.basicConfig(filename='example.log', \n                    encoding='utf-8', \n                    level=logging.DEBUG,\n                    format=\"%(asctime)s:%(levelname)s:%(message)s\"\n                   )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:44.657343Z","iopub.execute_input":"2024-11-24T06:40:44.657747Z","iopub.status.idle":"2024-11-24T06:40:50.475445Z","shell.execute_reply.started":"2024-11-24T06:40:44.657685Z","shell.execute_reply":"2024-11-24T06:40:50.474100Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PARAMETERS","metadata":{}},{"cell_type":"code","source":"N_TRAILS = 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.477887Z","iopub.execute_input":"2024-11-24T06:40:50.478619Z","iopub.status.idle":"2024-11-24T06:40:50.483874Z","shell.execute_reply.started":"2024-11-24T06:40:50.478568Z","shell.execute_reply":"2024-11-24T06:40:50.482671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LOAD DATA AND EDA","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.485218Z","iopub.execute_input":"2024-11-24T06:40:50.485554Z","iopub.status.idle":"2024-11-24T06:40:50.572774Z","shell.execute_reply.started":"2024-11-24T06:40:50.485520Z","shell.execute_reply":"2024-11-24T06:40:50.571588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.574869Z","iopub.execute_input":"2024-11-24T06:40:50.575210Z","iopub.status.idle":"2024-11-24T06:40:50.627407Z","shell.execute_reply.started":"2024-11-24T06:40:50.575177Z","shell.execute_reply":"2024-11-24T06:40:50.626273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.628888Z","iopub.execute_input":"2024-11-24T06:40:50.629333Z","iopub.status.idle":"2024-11-24T06:40:50.663950Z","shell.execute_reply.started":"2024-11-24T06:40:50.629288Z","shell.execute_reply":"2024-11-24T06:40:50.662808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.665667Z","iopub.execute_input":"2024-11-24T06:40:50.666082Z","iopub.status.idle":"2024-11-24T06:40:50.681559Z","shell.execute_reply.started":"2024-11-24T06:40:50.666047Z","shell.execute_reply":"2024-11-24T06:40:50.680237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mapper_data = data_dict[~data_dict['Value Labels'].isna()]\nmapper = {}\nfor i,row in mapper_data.iterrows():\n    if pd.isna(row['Values']):\n        continue\n    vals = row['Values'].split(',')\n    labels = row['Value Labels'].split(',')\n    labels = [i.split('=')[-1] for i in labels]\n    mapper[row['Field']] = {int(val):lab for val,lab in zip(vals,labels)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.682954Z","iopub.execute_input":"2024-11-24T06:40:50.683289Z","iopub.status.idle":"2024-11-24T06:40:50.697397Z","shell.execute_reply.started":"2024-11-24T06:40:50.683258Z","shell.execute_reply":"2024-11-24T06:40:50.696046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for key,val in mapper.items():\n    train_df[key] = train_df[key].apply(lambda x: val[x] if not pd.isna(x) else None)\n\nfor key,val in mapper.items():\n    try:\n        test_df[key] = test_df[key].apply(lambda x: val[x] if not pd.isna(x) else None)\n    except KeyError:\n        pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.698941Z","iopub.execute_input":"2024-11-24T06:40:50.699416Z","iopub.status.idle":"2024-11-24T06:40:50.809165Z","shell.execute_reply.started":"2024-11-24T06:40:50.699367Z","shell.execute_reply":"2024-11-24T06:40:50.807951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"to_drop = list(set(train_df.columns).difference(list(test_df.columns)))\nto_drop.remove(\"sii\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.810430Z","iopub.execute_input":"2024-11-24T06:40:50.810786Z","iopub.status.idle":"2024-11-24T06:40:50.816972Z","shell.execute_reply.started":"2024-11-24T06:40:50.810751Z","shell.execute_reply":"2024-11-24T06:40:50.815186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEASON_COLS = [\n    \"Basic_Demos-Enroll_Season\", \"CGAS-Season\", \"Physical-Season\",\n    \"Fitness_Endurance-Season\", \"FGC-Season\", \"BIA-Season\",\n    \"PAQ_A-Season\", \"PAQ_C-Season\", \"SDS-Season\", \"PreInt_EduHx-Season\"\n]\ntrain_df = train_df[list(test_df.columns)+[\"sii\"]]\ntrain_df = train_df.drop(columns=SEASON_COLS)\ntest_df = test_df.drop(columns=SEASON_COLS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.822632Z","iopub.execute_input":"2024-11-24T06:40:50.823132Z","iopub.status.idle":"2024-11-24T06:40:50.837420Z","shell.execute_reply.started":"2024-11-24T06:40:50.823096Z","shell.execute_reply":"2024-11-24T06:40:50.836312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Dataset Info:\\n\")\ntrain_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.839178Z","iopub.execute_input":"2024-11-24T06:40:50.839622Z","iopub.status.idle":"2024-11-24T06:40:50.873267Z","shell.execute_reply.started":"2024-11-24T06:40:50.839572Z","shell.execute_reply":"2024-11-24T06:40:50.871974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nDataset Description:\\n\")\ntrain_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.874598Z","iopub.execute_input":"2024-11-24T06:40:50.874947Z","iopub.status.idle":"2024-11-24T06:40:50.986758Z","shell.execute_reply.started":"2024-11-24T06:40:50.874913Z","shell.execute_reply":"2024-11-24T06:40:50.985544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nMissing Values Summary:\\n\")\nmissing_values = train_df.isna().sum()\nmissing_values[missing_values > 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:50.988233Z","iopub.execute_input":"2024-11-24T06:40:50.988587Z","iopub.status.idle":"2024-11-24T06:40:51.004545Z","shell.execute_reply.started":"2024-11-24T06:40:50.988551Z","shell.execute_reply":"2024-11-24T06:40:51.003182Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train data","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:40:51.006546Z","iopub.execute_input":"2024-11-24T06:40:51.007037Z","iopub.status.idle":"2024-11-24T06:42:21.524373Z","shell.execute_reply.started":"2024-11-24T06:40:51.006977Z","shell.execute_reply":"2024-11-24T06:42:21.523065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.merge(train_df, train_ts, how=\"left\", on='id')\ntest = pd.merge(test_df, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\n# train = train.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.526274Z","iopub.execute_input":"2024-11-24T06:42:21.526619Z","iopub.status.idle":"2024-11-24T06:42:21.565636Z","shell.execute_reply.started":"2024-11-24T06:42:21.526586Z","shell.execute_reply":"2024-11-24T06:42:21.564605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ssi_enc = {0:\"_None\",1:\"Mild\",2:\"Moderate\",3:\"Severe\"}\nssi_dec = {val:key for key,val in ssi_enc.items()}\ntrain['sii'] = train['sii'].replace({\"None\":\"_None\"})\ntrain['sii'] = train['sii'].apply(lambda x : ssi_enc[x] if not pd.isna(x) else np.nan )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.567045Z","iopub.execute_input":"2024-11-24T06:42:21.567390Z","iopub.status.idle":"2024-11-24T06:42:21.577782Z","shell.execute_reply.started":"2024-11-24T06:42:21.567355Z","shell.execute_reply":"2024-11-24T06:42:21.576391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.to_csv(\"child_data.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.579475Z","iopub.execute_input":"2024-11-24T06:42:21.579895Z","iopub.status.idle":"2024-11-24T06:42:21.943088Z","shell.execute_reply.started":"2024-11-24T06:42:21.579855Z","shell.execute_reply":"2024-11-24T06:42:21.942068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# serieses = os.listdir('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\n# serieses_aggs_list =[]\n# for series in tqdm(serieses):\n#     files = os.listdir(os.path.join('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet',series))\n#     data = pd.DataFrame()\n#     id = series.replace(\"id=\",\"\").strip()\n#     for i in files:\n#         tmp = pd.read_parquet(os.path.join('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet',series,i))\n#         data = pd.concat([data,tmp])\n#     linspaces = np.linspace(0, data.shape[0], num=10)\n#     to_concats = []\n#     for j,indx in enumerate(linspaces):\n#         temp_data = data.iloc[int(indx):int(data.shape[0]//10),:].copy()\n#         temp_data['time_of_day_hour'] = pd.to_datetime(temp_data['time_of_day']).dt.hour.astype(str)\n#         temp_data = temp_data.drop(columns=['time_of_day','step'])\n#         if temp_data.value_counts('time_of_day_hour').empty:\n#             time_of_day_hour_mod = pd.Series({f\"time_of_day_hour_{j}\":np.nan})\n#         else:\n#             time_of_day_hour_mod = pd.Series({f\"time_of_day_hour_{j}\":temp_data.value_counts('time_of_day_hour').tolist()[0]})    \n#         temp_data = temp_data.drop(columns=['time_of_day_hour'])\n\n#         mean = temp_data.mean()\n#         mean.index = [i+f\"_mean_{j}\" for i in mean.index]\n\n#         min = temp_data.min()\n#         min.index = [i+f\"_min_{j}\" for i in min.index]\n\n#         max = temp_data.max()\n#         max.index = [i+f\"_max_{j}\" for i in max.index]\n\n#         quantile_25 = temp_data.quantile(0.25)\n#         quantile_25.index = [i+f\"_quantile_25_{j}\" for i in quantile_25.index]\n\n#         quantile_50 = temp_data.quantile(0.50)\n#         quantile_50.index = [i+f\"_quantile_25_{j}\" for i in quantile_50.index]\n\n#         quantile_75 = temp_data.quantile(0.75)\n#         quantile_75.index = [i+f\"_quantile_25_{j}\" for i in quantile_75.index]\n        \n#         count = temp_data.count()\n#         count.index = [i+f\"_count_{j}\" for i in count.index]\n        \n#         std = data.std()\n#         std.index = [i+f\"_std_{j}\" for i in std.index]\n        \n#         var = data.var()\n#         var.index = [i+f\"_var_{j}\" for i in var.index]\n#         to_concat = pd.concat([mean,std,var,count,min,max,quantile_25,quantile_50,quantile_75,time_of_day_hour_mod])   \n#         to_concats.append(to_concat)\n#     to_concat = pd.concat(to_concats)\n#     to_concat['id'] = id\n#     serieses_aggs_list.append(to_concat)\n# serieses_aggs = pd.concat(serieses_aggs_list,axis=1)\n# serieses_aggs = serieses_aggs.T.reset_index()\n# serieses_aggs = serieses_aggs.drop(columns='index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.944444Z","iopub.execute_input":"2024-11-24T06:42:21.944839Z","iopub.status.idle":"2024-11-24T06:42:21.951878Z","shell.execute_reply.started":"2024-11-24T06:42:21.944802Z","shell.execute_reply":"2024-11-24T06:42:21.950555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test_data = test_data.merge(serieses_aggs,on='id',how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.953179Z","iopub.execute_input":"2024-11-24T06:42:21.953505Z","iopub.status.idle":"2024-11-24T06:42:21.968634Z","shell.execute_reply.started":"2024-11-24T06:42:21.953472Z","shell.execute_reply":"2024-11-24T06:42:21.967537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Basic Demographics (Age and Sex)\n# plt.figure(figsize=(12, 5))\n# sns.histplot(train_data['Basic_Demos-Age'], kde=True, bins=30)\n# plt.title('Age Distribution')\n# plt.xlabel('Age')\n# plt.ylabel('Frequency')\n# plt.show()\n\n# plt.figure(figsize=(8, 5))\n# sns.countplot(data=train_data, x='Basic_Demos-Sex', palette='Set2')\n# plt.title('Gender Count')\n# plt.xlabel('Gender')\n# plt.ylabel('Count')\n# plt.show()\n\n# # BMI Analysis\n# plt.figure(figsize=(12, 5))\n# sns.histplot(train_data['Physical-BMI'].dropna(), kde=True, bins=30)\n# plt.title('BMI Distribution')\n# plt.xlabel('BMI')\n# plt.ylabel('Frequency')\n# plt.show()\n\n# # Correlation Matrix\n# plt.figure(figsize=(20, 15))\n# corr_matrix = train_data.corr(numeric_only=True)\n# sns.heatmap(corr_matrix, annot=False, cmap='coolwarm', linewidths=0.5)\n# plt.title('Correlation Heatmap')\n# plt.show()\n\n# # Handling Missing Values\n# # Let's check columns with more than 50% missing values\n# threshold = len(train_data) * 0.5\n# columns_to_drop = missing_values[missing_values > threshold].index\n# df_cleaned = train_data.drop(columns=columns_to_drop)\n\n# print(\"\\nColumns with more than 50% missing values have been dropped:\\n\")\n# print(columns_to_drop)\n\n# # Filling missing values in numerical columns with median\n# for column in df_cleaned.select_dtypes(include=['float64', 'int64']).columns:\n#     df_cleaned[column] = df_cleaned[column].fillna(df_cleaned[column].median())\n\n# # Filling missing values in categorical columns with mode\n# for column in df_cleaned.select_dtypes(include=['object']).columns:\n#     df_cleaned[column] = df_cleaned[column].fillna(df_cleaned[column].mode()[0])\n\n# # Boxplot to identify outliers in some key features\n# plt.figure(figsize=(12, 5))\n# sns.boxplot(data=df_cleaned, x='Physical-Weight')\n# plt.title('Boxplot for Physical Weight')\n# plt.show()\n\n# plt.figure(figsize=(12, 5))\n# sns.boxplot(data=df_cleaned, x='Physical-Height')\n# plt.title('Boxplot for Physical Height')\n# plt.show()\n\n# # Pairplot to observe relationships\n# subset_features = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Weight', 'Physical-Height']\n# sns.pairplot(df_cleaned[subset_features].dropna())\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.970440Z","iopub.execute_input":"2024-11-24T06:42:21.970827Z","iopub.status.idle":"2024-11-24T06:42:21.980559Z","shell.execute_reply.started":"2024-11-24T06:42:21.970767Z","shell.execute_reply":"2024-11-24T06:42:21.979541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AUTO-ML CODE","metadata":{}},{"cell_type":"code","source":"def log_message(level, message):\n    if level.lower() == 'info':\n        print(\"[INFO]: \",message)\n        logging.info(message)\n    elif level.lower() == 'error':\n        print(\"[ERROR]: \",message)\n    elif level.lower() == 'warning':\n        logging.warning(message)\n    elif level.lower() == 'debug':\n        print(\"[DEBUG]: \",message)\n    else:\n        logging.critical('Unsupported logging level: ' + level)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:21.982175Z","iopub.execute_input":"2024-11-24T06:42:21.982673Z","iopub.status.idle":"2024-11-24T06:42:21.996308Z","shell.execute_reply.started":"2024-11-24T06:42:21.982634Z","shell.execute_reply":"2024-11-24T06:42:21.995187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoML:\n    def __init__(self,data:pd.DataFrame,interpretability:float,target_column:str,data_preprocessing:bool=False):\n        self.data = data\n        self.interpretability = interpretability\n        self.target_column = target_column\n        self.data_preprocessing = data_preprocessing\n        if self.target_column not in list(self.data.columns):\n            raise RuntimeError(f\"the target column should be on of the data columns : {list(self.data.columns)}\")\n        self.features_num = []\n        self.features_cat = []\n        self.features_date = []\n        self.features_bool = []\n        self.ml_algorithms = pd.read_csv('/kaggle/input/predict-plus-tool/ml_algorithms.csv')\n        self.ml_algorithms = self.ml_algorithms[~self.ml_algorithms.algorithm.isin(['DecisionTreeClassifier','BaggingClassifier','HistGradientBoostingClassifier','ExtraTreesClassifier','RandomForestClassifier'])]\n        self.ml_algorithms = self.ml_algorithms[self.ml_algorithms.type == 'tree']\n        self.ml_algorithms_parameters = json.load(open('/kaggle/input/predict-plus-tool/ml_algorithms_parameters.json','r'))\n        for col in self.data.columns:\n            if str(self.data[col].dtype) in ['float32','float64','int32','int64']:\n                self.features_num.append(col)\n            elif 'datetime' in str(self.data[col].dtype):\n                self.features_date.append(col)\n            else:\n                self.features_cat.append(col)\n        log_message('debug',self.data.info())\n        self.encoders = {}\n        self.refine_the_data()\n        log_message('debug',self.data.info())\n        if 'float' in str(self.data[self.target_column].dtype) or 'int' in str(self.data[self.target_column].dtype):\n            self.features_num.remove(self.target_column)\n        elif 'bool' in str(self.data[self.target_column].dtype):\n            self.features_bool.remove(self.target_column)\n        else:\n            self.features_cat.remove(self.target_column)\n        self.original_data = self.data\n            \n    def refine_the_data(self):\n        log_message('debug','Start casting to numarical features')\n        # from catagorical data to numarical\n        new_cols = []\n        for col in tqdm(self.features_cat):\n            try:\n                self.data[col] = self.data[col].astype('float64')\n                self.features_num.append(col)\n                new_cols.append(col)\n            except:\n                continue\n        for col in new_cols:\n            self.features_cat.remove(col)\n            \n        # from catagorical data to dates\n        log_message('debug','Start casting to date features')\n        new_cols = []\n        for col in tqdm(self.features_cat):\n            try:\n                warnings.simplefilter(action='ignore', category=UserWarning)\n                self.data[col] = pd.to_datetime(self.data[col])\n                self.features_date.append(col)\n                new_cols.append(col)\n            except Exception as e:\n                continue\n        for col in new_cols:\n            self.features_cat.remove(col)\n\n        # Calculate the n unique\n        cols_nunique = self.data[self.features_cat].nunique()\n        cols_nunique_bool = cols_nunique[cols_nunique==2]\n        cols_nunique_id = cols_nunique[cols_nunique==self.data.shape[0]]\n        cols_nunique_alot = cols_nunique[cols_nunique>100]\n        \n        # from catagorical to bool\n        log_message('debug','Start casting to bool features')\n        for col in tqdm(cols_nunique_bool.index):\n            self.encoders[col] = {}\n            uniques = list(self.data[col].unique())\n            for i,j in enumerate(uniques):\n                self.encoders[col][j] = bool(i)\n            self.data[col] = self.data[col].map(self.encoders[col])\n            self.features_cat.remove(col)\n            self.features_bool.append(col)\n            \n        # from numarical to bool\n        cols_nunique = self.data[self.features_num].nunique()\n        cols_nunique_bool = cols_nunique[cols_nunique==2]\n        for col in tqdm(cols_nunique_bool.index):\n            self.encoders[col] = {}\n            uniques = list(self.data[col].unique())\n            for i,j in enumerate(uniques):\n                self.encoders[col][j] = bool(i)\n            self.data[col] = self.data[col].astype(bool)\n            self.features_num.remove(col)\n            self.features_bool.append(col)\n\n        # drop ids columns and columns with a lot of unique values\n        self.data = self.data.drop(columns=list(cols_nunique_id.index)+list(cols_nunique_alot.index))\n        for col in set(list(cols_nunique_id.index)+list(cols_nunique_alot.index)):\n            self.features_cat.remove(col)\n            \n        # Add date features\n        if self.features_date:\n            log_message('debug','Start creating date features')\n        for col in tqdm(self.features_date):\n            self.data[col+'_'+'day_of_year'] = self.data[col].dt.day_of_year\n            self.features_num.append(col+'_'+'day_of_year')\n            self.data[col+'_'+'quarter'] = self.data[col].dt.quarter\n            self.features_num.append(col+'_'+'quarter')\n            self.data[col+'_'+'day_of_week'] = self.data[col].dt.day_of_week \n            self.features_num.append(col+'_'+'day_of_week')\n            self.data[col+'_'+'days_in_month'] = self.data[col].dt.days_in_month \n            self.features_num.append(col+'_'+'days_in_month')\n            self.data[col+'_'+'day'] = self.data[col].dt.day\n            self.features_num.append(col+'_'+'day')\n            self.data[col+'_'+'month'] = self.data[col].dt.month\n            self.features_num.append(col+'_'+'month')\n            self.data[col+'_'+'year'] = self.data[col].dt.year\n            self.features_num.append(col+'_'+'year')\n            self.data[col+'_'+'hour'] = self.data[col].dt.hour\n            self.features_num.append(col+'_'+'hour')\n            self.data[col+'_'+'day_of_year'+'_sin'] = (np.pi *self.data[col+'_'+'day_of_year'] / 183).apply(lambda x:np.sin(x))\n            self.features_num.append(col+'_'+'day_of_year'+'_sin')\n            self.data[col+'_'+'hour'+'_sin'] = (np.pi *self.data[col+'_'+'hour'] / 12).apply(lambda x:np.sin(x))\n            self.features_num.append(col+'_'+'hour'+'_sin')\n        # remove nan columns\n        nan_columns = list(self.data.isna().all()[self.data.isna().all()].index)\n        self.data = self.data.dropna(axis=1,how='all')\n        for col in nan_columns:\n            if col in self.features_cat:\n                self.features_cat.remove(col)\n            elif col in self.features_num:\n                self.features_num.remove(col)\n            elif col in self.features_bool:\n                self.features_bool.remove(col)\n        self.data = self.data.drop(columns=self.features_date)\n\n        # Fill nan columns\n        # self.data[self.features_cat] = self.data[self.features_cat].fillna('UNK')\n        # self.data[self.features_num] = self.data[self.features_num].fillna(0)\n        # self.data[self.features_bool] = self.data[self.features_bool].fillna(False)\n        \n        # Drop constant columns\n        constant_columns = list(self.data[self.features_num].std(axis=0)[self.data[self.features_num].std(axis=0)==0].index)\n        for col in constant_columns:\n            self.features_num.remove(col)\n        if constant_columns:\n            self.data = self.data.drop(columns=constant_columns)\n        \n        # Define the taask\n        if str(self.data[self.target_column].dtype) == 'bool':\n            self.task = 'binary_classification'\n        elif str(self.data[self.target_column].dtype) == 'object':\n            self.task = 'multi_classification'\n        elif str(self.data[self.target_column].dtype) in ['float64','float32','int64','int32']:\n            self.task = 'regression'\n        log_message('debug',f'The selected task is {self.task}')\n    \n    def preprocess(self,type_num,type_cat,type_cat_target,type_num_target):\n        self.X = self.data.drop(columns=self.target_column)\n        self.y = self.data[self.target_column]\n        X_cat = self.X[self.features_cat].copy()\n        X_num = self.X[self.features_num].copy()\n        X_bool = self.X[self.features_bool].copy().values\n        num_pars = {}\n        if type_num == 'min_max':\n            X_min = np.nanmin(X_num.values,axis=0)\n            X_max = np.nanmax(X_num.values,axis=0)\n            new_X_num = (X_num - X_min)/(X_max - X_min)\n            num_pars = {'X_min':X_min,'X_max':X_max}\n        elif type_num == 'standard':\n            X_mean = np.nanmean(X_num.values,axis=0)\n            X_std = np.nanstd(X_num.values,axis=0)\n            new_X_num = (X_num - X_mean)/X_std\n            num_pars = {'X_mean':X_mean,'X_std':X_std}\n        else:\n            raise RuntimeError(\"wrong preprocessing type\")\n            \n        if type_cat == 'label':\n            for col in self.features_cat:\n                le = LabelEncoder()\n                le.fit(X_cat[col])\n                X_cat.loc[:,col] = le.transform(X_cat[col])\n                self.encoders[col] = le\n            X_cat = X_cat.values\n       \n        elif type_cat == 'one_hot':\n            one_hots = []\n            for col in self.features_cat:\n                le = OneHotEncoder(handle_unknown='ignore')\n                le.fit(X_cat[col].values.reshape(-1, 1))\n                one_hots.append(le.transform(X_cat[col].values.reshape(-1, 1)).toarray())\n                self.encoders[col] = le\n            X_cat = np.concatenate(one_hots,axis=1)\n        else:\n            raise RuntimeError(\"wrong preprocessing type\")\n            \n        if self.task == 'regression':\n            if type_num_target == 'min_max':\n                y_min = self.y.copy().min(axis=0)\n                y_max = self.y.copy().max(axis=0)\n                new_y = (self.y.copy() - y_min)/(y_max - y_min)\n                new_y = new_y.to_numpy()\n                num_pars['y_min']=y_min\n                num_pars['y_max']=y_max\n                \n            elif type_num_target == 'standard':\n                y_mean = self.y.copy().mean(axis=0)\n                y_std = self.y.copy().std(axis=0)\n                new_y = (self.y.copy() - y_mean)/y_std\n                new_y = new_y.to_numpy()\n                num_pars['y_std']=y_std\n                num_pars['y_mean']=y_mean\n\n        elif self.task == 'multi_classification' or self.task == 'binary_classification':\n            if type_cat_target == 'label':\n                le = LabelEncoder()\n                le.fit(self.y.copy())\n                new_y= le.transform(self.y.copy()).reshape(-1, 1)\n                self.encoders['target'] = le\n            elif type_cat_target == 'one_hot':\n                le = OneHotEncoder(handle_unknown='ignore')\n                le.fit(self.y.copy().values.reshape(-1, 1))\n                new_y = le.transform(self.y.copy().values.reshape(-1, 1)).toarray()\n                self.encoders['target'] = le\n        return np.concatenate([X_cat,new_X_num,X_bool],axis=1),new_y,num_pars\n\n    def missing_values_handler(self,drop,handling_type_num='',handling_type_cat=''):\n        log_message(\"debug\",f\"Old data shape is {self.data.shape}\")\n        self.imputer_num = None\n        self.imputer_cat_and_bool = None\n        if drop:\n            self.data = self.data.copy().dropna()\n                 \n        if handling_type_num != 'KNN' and self.features_num:\n            imputer = SimpleImputer(strategy=handling_type_num)\n            self.data.loc[:,self.features_num] = imputer.fit_transform(self.data.loc[:,self.features_num])\n            self.imputer_num = imputer\n\n        if handling_type_cat and self.features_cat + self.features_bool:\n            imputer = SimpleImputer(strategy=handling_type_cat)\n            self.data.loc[:,self.features_cat + self.features_bool] = imputer.fit_transform(self.data.loc[:,self.features_cat + self.features_bool])\n            self.imputer_cat_and_bool = imputer\n\n        if handling_type_num == 'KNN' and self.features_num:\n            imputer = KNNImputer(n_neighbors=7)\n            self.data.loc[:,self.features_num] = imputer.fit_transform(self.data.loc[:,self.features_num])\n            self.imputer_num = imputer\n        temp_data = self.data.copy()\n        le = OneHotEncoder(handle_unknown='ignore')\n        temp_data[['sii']] = temp_data[['sii']].fillna(pd.NA)\n        to_incode = temp_data[~pd.isna(temp_data['sii'])]\n        le.fit(to_incode[['sii']].values.reshape(-1, 1))\n        one_hot_y = le.transform(to_incode[['sii']].values.reshape(-1, 1)).toarray()\n        to_incode = to_incode.drop(columns='sii')\n        to_incode[['_None','Mild','Moderate','Severe']] = one_hot_y\n        # self.data['sii'] = self.data['sii'].apply(lambda x : ssi_dec[x] if not pd.isna(x) else np.nan)\n        temp_data[['_None','Mild','Moderate','Severe']] = np.nan\n        temp_data = temp_data.drop(columns='sii')\n        temp_data.loc[to_incode.index,['_None','Mild','Moderate','Severe']] = to_incode\n        imputer = KNNImputer(n_neighbors=7)\n        temp_data_imp = imputer.fit_transform(temp_data.loc[:,self.features_num + ['_None','Mild','Moderate','Severe']])\n        temp_data.loc[:,self.features_num + ['_None','Mild','Moderate','Severe']] = temp_data_imp\n        for i in ['_None','Mild','Moderate','Severe']:\n            temp_data[i] = temp_data[i].apply(lambda x:round(x))\n        decoded_sii = le.inverse_transform(temp_data[['_None','Mild','Moderate','Severe']].values.astype(int))\n        temp_data['sii'] = decoded_sii.flatten()\n        temp_data = temp_data.drop(columns = ['_None','Mild','Moderate','Severe'])\n        self.data = temp_data\n        # self.data.loc[:,self.features_num + ['sii']] = imputer.fit_transform(self.data.loc[:,self.features_num + ['sii']])\n        # self.data = self.data.drop(columns=to_drop)\n        # features_num = self.features_num.copy()\n        # features_cat = self.features_cat.copy()\n        # for col in features_num:\n        #     if col in to_drop:\n        #         self.features_num.remove(col)\n                \n        # for col in features_cat:\n        #     if col in to_drop:\n        #         self.features_cat.remove(col)\n        # self.data['sii'] = self.data['sii'].apply(lambda x: round(x))\n        # self.data['sii'] = self.data['sii'].apply(lambda x : ssi_enc[x])\n        self.data = self.data.dropna(subset='sii')\n        log_message(\"debug\",f\"New data shape is {self.data.shape}\")\n        log_message(\"debug\",f\"Number of missing vales is {self.data.isna().sum().sum()}\")\n    \n    def postprocess(self,y,params,type_num,type_cat,inference):\n        if self.task == 'regression':\n            if type_num == 'min_max':\n                y_min = params['y_min']\n                y_max = params['y_max']\n                new_y = y * (y_max - y_min) + y_min\n                \n            elif type_num == 'standard':\n                y_mean = params['y_mean']\n                y_std = params['y_std']\n                new_y = (y * y_std) + y_mean\n\n        elif self.task == 'multi_classification' or self.task == 'binary_classification':\n            if inference:\n                if type_cat == 'label':\n                    le = self.encoders['target']\n                    new_y= le.inverse_transform(y)\n\n                elif type_cat == 'one_hot':\n                    le = self.encoders['target']\n                    print(y.reshape(-1, 1).shape)\n                    print(y)\n                    new_y = le.inverse_transform(y.reshape(-1, 1).astype(int))\n            else:\n                new_y = y.copy()\n        return new_y\n    \n    def evaluate(self,y_true,y_pred):\n        if self.task == 'multi_classification':\n            score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n            # score = f1_score(y_true=y_true,y_pred=y_pred,average='weighted')\n        elif self.task == 'binary_classification':\n            score = f1_score(y_true=y_true,y_pred=y_pred)\n        elif self.task == 'regression':\n            score = r2_score(y_true=y_true,y_pred=y_pred)\n        return score\n        \n    def train(self,trial):\n        features_num_old = self.features_num.copy()\n        features_cat_old = self.features_cat.copy()\n        self.data = self.original_data.copy()\n        try:\n            \n            if self.task == 'regression':\n                algorithms_list = self.ml_algorithms[self.ml_algorithms.regression == 1]['algorithm'].tolist()\n                ml_algorithm = trial.suggest_categorical('ml_algorithm', algorithms_list)\n                ml_algorithm_type = self.ml_algorithms[self.ml_algorithms.algorithm==ml_algorithm]['type'].item()\n            elif self.task == 'binary_classification':\n                algorithms_list = self.ml_algorithms[self.ml_algorithms.binary_classification == 1]['algorithm'].tolist()\n                ml_algorithm = trial.suggest_categorical('ml_algorithm', algorithms_list)\n                ml_algorithm_type = self.ml_algorithms[self.ml_algorithms.algorithm==ml_algorithm]['type'].item()\n            elif self.task == 'multi_classification':\n                algorithms_list = self.ml_algorithms[self.ml_algorithms.multi_classification == 1]['algorithm'].tolist()\n                ml_algorithm = trial.suggest_categorical('ml_algorithm', algorithms_list)\n                ml_algorithm_type = self.ml_algorithms[self.ml_algorithms.algorithm==ml_algorithm]['type'].item()\n            \n            if ml_algorithm_type in ['linear','svm','knn']:\n                type_cat = 'one_hot'\n                if self.task == 'binary_classification' or self.task == 'multi_classification':\n                    type_cat_target = 'label'\n                else:\n                    type_cat_target = None\n            else:\n                type_cat = trial.suggest_categorical('type_cat', ['one_hot','label'])\n                if self.task == 'binary_classification' or self.task == 'multi_classification':\n                    type_cat_target = 'label'\n                else:\n                    type_cat_target = None\n\n            if ml_algorithm in ['ComplementNB','MultinomialNB','CategoricalNB']:\n                type_num = 'min_max'\n                if self.task == 'regression':\n                    type_num_target = 'min_max'\n                else:\n                    type_num_target = None\n            else:\n                type_num = trial.suggest_categorical('type_num', ['min_max','standard'])\n                if self.task == 'regression':\n                    type_num_target = trial.suggest_categorical('type_num_target', ['min_max','standard'])\n                else:\n                    type_num_target = None\n\n            if self.data.isnull().sum().sum() > 0:\n                log_message('debug',\"Opps!! there are nulls in your data\")\n\n                if not self.data.copy().dropna().empty:\n                    handling_type_drop = trial.suggest_categorical('handling_type_drop', [True,False])\n                else:\n                    handling_type_drop = False\n                \n                if not handling_type_drop:\n                    if self.features_num:\n                        handling_type_num = trial.suggest_categorical('handling_type_num', ['mean','median','most_frequent','KNN'])\n                    else:\n                        handling_type_num = ''\n                    if self.features_cat:\n                        handling_type_cat = trial.suggest_categorical('handling_type_cat', ['most_frequent',None])\n                    else:\n                        handling_type_cat = ''\n                else:\n                    handling_type_num = ''\n                    handling_type_cat = ''\n\n                self.missing_values_handler(handling_type_cat=handling_type_cat,handling_type_num=handling_type_num,drop=handling_type_drop)\n\n            X,y,pars = self.preprocess(type_num=type_num,\n                                       type_cat=type_cat,\n                                       type_cat_target=type_cat_target,\n                                       type_num_target=type_num_target)  \n            if X.shape[0] > 5000:\n                log_message('info','The data is huge, we will train on a subset of the data')\n                n = 5000\n                indexes = np.random.choice(X.shape[0], n, replace=False)  \n                X = X[indexes,:]\n                y = y[indexes]\n            parameters = self.ml_algorithms_parameters[ml_algorithm]\n            trial_parameters = {}\n            for key,val in parameters.items():\n                if ml_algorithm == 'GradientBoostingClassifier' and key == 'loss' and self.task == 'multi_classification':\n                    trial_parameters[key] = 'log_loss'\n                    continue\n                if ml_algorithm == 'LinearSVC' and key=='penalty' and trial_parameters['loss'] == 'hinge':\n                    trial_parameters[key] = 'l2'\n                    continue\n                if key == 'oob_score' and trial_parameters['bootstrap']==False:\n                    trial_parameters[key] = False  \n                    continue                                  \n                if key == 'n_jobs':\n                    trial_parameters[key] = val\n                    continue\n                if ml_algorithm == 'TweedieRegressor' and key=='power':\n                    trial_parameters[key] = 0\n                    continue\n                if ml_algorithm == 'ExtraTreesRegressor' and type_num == 'standard' and key == 'criterion':\n                    trial_parameters[key] = trial.suggest_categorical(ml_algorithm+\"_\"+key,[\"squared_error\", \"absolute_error\", \"friedman_mse\"])\n                    continue\n                if isinstance(val[0],bool):\n                    trial_parameters[key] = trial.suggest_categorical(ml_algorithm+\"_\"+key,val)\n                elif isinstance(val[0],int):\n                    trial_parameters[key] = trial.suggest_int(ml_algorithm+\"_\"+key,val[0],val[1])\n                elif isinstance(val[0],float):\n                    trial_parameters[key] = trial.suggest_float(ml_algorithm+\"_\"+key,val[0],val[1])\n                else:\n                    trial_parameters[key] = trial.suggest_categorical(ml_algorithm+\"_\"+key,val)\n            \n            kf = KFold(n_splits=5)\n            scores = []\n            log_message(\"debug\",\"Fit \"+ml_algorithm)\n            log_message(\"debug\",\"Trial_parameters: \"+str(trial_parameters))\n            for i, (train_index, test_index) in enumerate(kf.split(X=X,y=y)):\n                X_train, X_test, y_train, y_test = X[train_index],X[test_index],y[train_index],y[test_index]\n\n                smote = SMOTE(random_state=42)\n                X_train, y_train = smote.fit_resample(X_train, y_train)\n                model = eval(ml_algorithm)\n                model = model(**trial_parameters)\n            \n                model.fit(X_train,y_train)  \n                y_pred = model.predict(X_test)\n                y_pred = self.postprocess(y_pred,pars,type_num_target,type_cat_target,False)\n                y_test = self.postprocess(y_test,pars,type_num_target,type_cat_target,False)\n                score = self.evaluate(y_pred=y_pred,y_true=y_test)\n                scores.append(score) \n            score = sum(scores)/len(scores)\n            log_message(\"debug\",\"Fited \"+ml_algorithm)\n            log_message(\"info\",\"Score \"+str(score))\n            self.features_num = features_num_old\n            self.features_cat = features_cat_old\n            return score\n        except Exception as e:\n            log_message('error',e)\n            log_message('error',trial_parameters)\n            return 0\n    \n    def optimize(self,n_trials):\n        for i in range(n_trials):\n            print(\"ask\")\n            trial = self.study.ask()\n            print(\"train\")  # Generate a trial suggestion\n            value = self.train(trial)  # Evaluate the objective function\n            print(\"tell\")\n            self.study.tell(trial, value)\n            print(\"finish\")\n            print(\"=\"*10)\n            gc.collect()\n            yield i\n    \n    def init_study(self):\n        log_message('debug',f'The optimization phase started')\n        self.study = optuna.create_study(direction='maximize')\n\n    def final_training(self):\n        self.data = self.original_data.copy()\n        self.best_trial = self.study.best_trial\n        self.best_params = self.study.best_params\n        log_message('info',f\"The best parameters:\\n{self.best_params}\")\n        self.score = self.study.best_value\n        log_message('info',f'The optimized model achieved {self.score} score')\n        log_message('debug',f'The fitting phase started')\n        temp_parameters = self.best_params\n        ml_algorithm = temp_parameters['ml_algorithm']\n        ml_algorithm_type = self.ml_algorithms[self.ml_algorithms.algorithm==ml_algorithm]['type'].item()\n        if ml_algorithm_type in ['linear','svm','knn']:\n            self.type_cat = 'one_hot'\n            if self.task == 'binary_classification' or self.task == 'multi_classification':\n                self.type_cat_target = 'label'\n            else:\n                self.type_cat_target = None\n        else:\n            self.type_cat = temp_parameters['type_cat']\n            del temp_parameters['type_cat']\n            if self.task == 'binary_classification' or self.task == 'multi_classification':\n                self.type_cat_target = 'label'\n            else:\n                self.type_cat_target = None   \n            \n        if ml_algorithm in ['ComplementNB','MultinomialNB','CategoricalNB']:\n            self.type_num = 'min_max'\n            if self.task == 'regression':\n                self.type_num_target = 'min_max'\n            else:\n                self.type_num_target = None\n        else:\n            self.type_num = temp_parameters['type_num']\n            del temp_parameters['type_num']\n            if self.task == 'regression':\n                self.type_num_target = temp_parameters['type_num_target']\n                del temp_parameters['type_num_target']\n            else:\n                self.type_num_target = None\n                   \n        del temp_parameters['ml_algorithm'] \n        temp_parameters2 = {}\n        for key,val in temp_parameters.items():\n            temp_parameters2[key.replace(ml_algorithm+\"_\",\"\").strip()] = val\n        temp_parameters = temp_parameters2\n        self.handling_type_num = temp_parameters.get('handling_type_num')\n        self.handling_type_cat = temp_parameters.get('handling_type_cat')\n        self.handling_type_drop = temp_parameters.get('handling_type_drop')\n        if self.handling_type_num!='':\n            del temp_parameters['handling_type_num']\n\n        if self.handling_type_cat!='':\n            del temp_parameters['handling_type_cat']\n            \n        if self.handling_type_drop:\n            del temp_parameters['handling_type_drop']\n        self.missing_values_handler(handling_type_cat=self.handling_type_cat,\n                                    handling_type_num=self.handling_type_num,\n                                    drop=self.handling_type_drop)\n        features_num = self.features_num.copy()\n        features_cat = self.features_cat.copy()\n        for col in features_num:\n            if col in to_drop:\n                self.features_num.remove(col)\n                \n        for col in features_cat:\n            if col in to_drop:\n                self.features_cat.remove(col)\n        X,y,pars = self.preprocess(type_num=self.type_num,\n                                   type_cat=self.type_cat,\n                                   type_cat_target=self.type_cat_target,\n                                   type_num_target=self.type_num_target)\n\n        self.numarical_preprocessing_parameters = pars\n        smote = SMOTE(random_state=42)\n        X, y = smote.fit_resample(X, y)\n\n\n        base_model = eval(ml_algorithm)\n        base_model = base_model(**temp_parameters)\n        model = BaggingClassifier(estimator=base_model)\n        model.fit(X,y)\n        self.model = model\n        \n        # kf = KFold(n_splits=5)\n        # models = []\n        # for i, (train_index, test_index) in enumerate(kf.split(X)):\n        #     base_model = eval(ml_algorithm)\n        #     base_model = model(**temp_parameters)\n        #     model = BaggingClassifier(estimator=base_model)\n        #     model.fit(X[train_index],y[train_index])\n        #     y_pred = model.predict(X[test_index])\n        #     y_pred = self.postprocess(y_pred,pars,self.type_num_target,self.type_cat_target,False)\n        #     y_test = self.postprocess(y[test_index],pars,self.type_num_target,self.type_cat_target,False)\n        #     score = self.evaluate(y_pred=y_pred,y_true=y_test)\n        #     log_message('info',f'Fold {i} with score {score}')\n        #     models.append(model)\n        # models = [ (f\"model_{i}\",j) for i,j in enumerate(models)]\n        # voting_model = VotingClassifier(estimators=models, voting='soft')\n        # voting_model.fit(X, y)\n        # self.model = voting_model\n        return vis.plot_optimization_history(self.study)\n           \n    def save(self,model_name):\n        if not os.path.exists(\"Models\"):\n            os.mkdir(\"Models\")\n        with open(f\"Models/{model_name}_model.pkl\", 'wb') as f:\n            pickle.dump(self.model,f)\n        with open(f\"Models/{model_name}_tuner.pkl\", 'wb') as f:\n            pickle.dump(self,f)\n        log_message('debug',f'The model has been saved successfuly, model path is {model_name}_model/tuner.pkl')\n\nclass Module():\n    def __init__(self,automl:AutoML):\n        self.preprocessing_parameters = automl.numarical_preprocessing_parameters\n        self.model = automl.model\n        self.features_cat = automl.features_cat\n        self.features_num = automl.features_num\n        self.features_bool = automl.features_bool\n        self.features_date = automl.features_date\n        self.encoders = automl.encoders\n        self.type_num = automl.type_num\n        self.type_cat = automl.type_cat\n        self.type_cat_target = automl.type_cat_target\n        self.type_num_target = automl.type_num_target\n        self.task = automl.task\n        self.imputer_num = automl.imputer_num\n        self.imputer_cat_and_bool = automl.imputer_cat_and_bool\n    \n    def predict(self,data:pd.DataFrame):\n        output=None\n        try:\n            for col in tqdm(self.features_date ):\n                data[col+'_'+'day_of_year'] = data[col].dt.day_of_year\n                data[col+'_'+'quarter'] = data[col].dt.quarter\n                data[col+'_'+'day_of_week'] = data[col].dt.day_of_week \n                data[col+'_'+'days_in_month'] = data[col].dt.days_in_month \n                data[col+'_'+'day'] = data[col].dt.day\n                data[col+'_'+'month'] = data[col].dt.month\n                data[col+'_'+'year'] = data[col].dt.year\n                data[col+'_'+'hour'] = data[col].dt.hour\n                data[col+'_'+'day_of_year'+'_sin'] = (np.pi *data[col+'_'+'day_of_year'] / 183).apply(lambda x:np.sin(x))\n                data[col+'_'+'hour'+'_sin'] = (np.pi *data[col+'_'+'hour'] / 12).apply(lambda x:np.sin(x))\n            \n            for col in self.features_bool:\n                data[col] = data[col].map(self.encoders[col])\n\n            if self.imputer_cat_and_bool != None :\n                data.loc[:,self.features_cat+self.features_bool] = self.imputer_cat_and_bool.transform(data.loc[:,self.features_cat+self.features_bool])\n                \n            if self.imputer_num != None:\n                data.loc[:,self.features_num] = self.imputer_num.transform(data.loc[:,self.features_num])\n            X_cat = data[self.features_cat].copy()\n            X_num = data[self.features_num].copy()\n            X_bool = data[self.features_bool].copy().values\n            if self.type_num == 'min_max':\n                X_min = self.preprocessing_parameters['X_min']\n                X_max = self.preprocessing_parameters['X_max']\n                new_X_num = (X_num - X_min)/(X_max - X_min)\n    \n            elif self.type_num == 'standard':\n                X_mean = self.preprocessing_parameters['X_mean']\n                X_std = self.preprocessing_parameters['X_std']\n                new_X_num = (X_num - X_mean)/X_std\n        \n            else:\n                raise RuntimeError(\"wrong preprocessing type\")\n                \n            if self.type_cat == 'label':\n                for col in self.features_cat:\n                    le = self.encoders[col]\n                    X_cat.loc[:,col] = le.transform(X_cat[col])\n\n                X_cat = X_cat.values\n        \n            elif self.type_cat == 'one_hot':\n                one_hots = []\n                for col in self.features_cat:\n                    le = self.encoders[col]\n                    one_hots.append(le.transform(X_cat[col].values.reshape(-1, 1)).toarray())\n                X_cat = np.concatenate(one_hots,axis=1)\n            else:\n                raise RuntimeError(\"wrong preprocessing type\")\n            input_data = np.concatenate([X_cat,new_X_num,X_bool],axis=1)\n            output = self.model.predict(input_data)\n            if self.task == 'regression':\n                if self.type_num_target == 'min_max':\n                    y_min = self.preprocessing_parameters['y_min']\n                    y_max = self.preprocessing_parameters['y_max']\n                    new_output = output * (y_max - y_min) + y_min\n                    \n                elif self.type_num_target == 'standard':\n                    y_mean = self.preprocessing_parameters['y_mean']\n                    y_std = self.preprocessing_parameters['y_std']\n                    new_output = (output * y_std ) + y_mean\n\n            elif self.task == 'multi_classification' or self.task == 'binary_classification':\n                if self.type_cat_target == 'label':\n                    le = self.encoders['target']\n                    new_output = le.inverse_transform(output)\n\n                elif self.type_cat_target == 'one_hot':\n                    le = self.encoders['target']\n                    new_output = le.inverse_transform(output.reshape(-1, 1))\n\n            return new_output\n        except Exception as e:\n            log_message('error',e)\n            log_message('error',data.isna().sum()[data.isna().sum()>0])\n            if output:\n                log_message('error',output)\n                log_message('error',self.encoders['target'].classes_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T07:18:54.619604Z","iopub.execute_input":"2024-11-24T07:18:54.620039Z","iopub.status.idle":"2024-11-24T07:18:54.728168Z","shell.execute_reply.started":"2024-11-24T07:18:54.619994Z","shell.execute_reply":"2024-11-24T07:18:54.726946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# APPLYING AUTO-ML","metadata":{}},{"cell_type":"code","source":"train = train.sample(frac=1)\ntuner = AutoML(data=train,data_preprocessing=True,target_column='sii',interpretability=1) \ntuner.init_study()\nlog_message('debug',\"study init\")\nfor i in tuner.optimize(n_trials=N_TRAILS):\n    pass\nfig = tuner.final_training()\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T07:18:55.380760Z","iopub.execute_input":"2024-11-24T07:18:55.381166Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MAKE PREDICTION ON TEST DATA","metadata":{}},{"cell_type":"code","source":"trained_model = Module(tuner)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.013620Z","iopub.status.idle":"2024-11-24T06:42:23.014029Z","shell.execute_reply.started":"2024-11-24T06:42:23.013848Z","shell.execute_reply":"2024-11-24T06:42:23.013867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = trained_model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.016409Z","iopub.status.idle":"2024-11-24T06:42:23.017017Z","shell.execute_reply.started":"2024-11-24T06:42:23.016692Z","shell.execute_reply":"2024-11-24T06:42:23.016742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(prediction)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.018278Z","iopub.status.idle":"2024-11-24T06:42:23.018824Z","shell.execute_reply.started":"2024-11-24T06:42:23.018542Z","shell.execute_reply":"2024-11-24T06:42:23.018567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ssi_dec = {val:key for key,val in ssi_enc.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.020599Z","iopub.status.idle":"2024-11-24T06:42:23.021169Z","shell.execute_reply.started":"2024-11-24T06:42:23.020884Z","shell.execute_reply":"2024-11-24T06:42:23.020914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['id'] = test['id']\nsubmission['sii'] = prediction\nsubmission['sii'] = submission['sii'].apply(lambda x : ssi_dec[x])\nsubmission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.022434Z","iopub.status.idle":"2024-11-24T06:42:23.022818Z","shell.execute_reply.started":"2024-11-24T06:42:23.022617Z","shell.execute_reply":"2024-11-24T06:42:23.022634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T06:42:23.024339Z","iopub.status.idle":"2024-11-24T06:42:23.024712Z","shell.execute_reply.started":"2024-11-24T06:42:23.024539Z","shell.execute_reply":"2024-11-24T06:42:23.024558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}