{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":202213835,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-24T11:03:48.521199Z","iopub.execute_input":"2024-10-24T11:03:48.521684Z","iopub.status.idle":"2024-10-24T11:03:51.652925Z","shell.execute_reply.started":"2024-10-24T11:03:48.521634Z","shell.execute_reply":"2024-10-24T11:03:51.651753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ['KAGGLE_CONFIG_DIR'] = os.path.expanduser(\"~/.kaggle\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.655153Z","iopub.execute_input":"2024-10-24T11:03:51.656177Z","iopub.status.idle":"2024-10-24T11:03:51.661239Z","shell.execute_reply.started":"2024-10-24T11:03:51.656123Z","shell.execute_reply":"2024-10-24T11:03:51.660098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_sub = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\npath_data_dict ='/kaggle/input//child-mind-institute-problematic-internet-use/data_dictionary.csv'\npath_train = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\npath_test = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.663338Z","iopub.execute_input":"2024-10-24T11:03:51.66416Z","iopub.status.idle":"2024-10-24T11:03:51.675029Z","shell.execute_reply.started":"2024-10-24T11:03:51.664109Z","shell.execute_reply":"2024-10-24T11:03:51.673683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv (path_sub)\ndata_dict_df =pd.read_csv (path_data_dict)\ntrain_df =pd.read_csv (path_train)\ntest_df =pd.read_csv(path_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.677726Z","iopub.execute_input":"2024-10-24T11:03:51.678244Z","iopub.status.idle":"2024-10-24T11:03:51.773272Z","shell.execute_reply.started":"2024-10-24T11:03:51.678187Z","shell.execute_reply":"2024-10-24T11:03:51.772085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.774622Z","iopub.execute_input":"2024-10-24T11:03:51.774975Z","iopub.status.idle":"2024-10-24T11:03:51.795494Z","shell.execute_reply.started":"2024-10-24T11:03:51.774937Z","shell.execute_reply":"2024-10-24T11:03:51.794277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.796999Z","iopub.execute_input":"2024-10-24T11:03:51.797388Z","iopub.status.idle":"2024-10-24T11:03:51.814017Z","shell.execute_reply.started":"2024-10-24T11:03:51.797332Z","shell.execute_reply":"2024-10-24T11:03:51.812192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.815794Z","iopub.execute_input":"2024-10-24T11:03:51.816298Z","iopub.status.idle":"2024-10-24T11:03:51.856807Z","shell.execute_reply.started":"2024-10-24T11:03:51.816243Z","shell.execute_reply":"2024-10-24T11:03:51.855687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.858579Z","iopub.execute_input":"2024-10-24T11:03:51.858969Z","iopub.status.idle":"2024-10-24T11:03:51.886352Z","shell.execute_reply.started":"2024-10-24T11:03:51.858927Z","shell.execute_reply":"2024-10-24T11:03:51.884949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.887857Z","iopub.execute_input":"2024-10-24T11:03:51.888268Z","iopub.status.idle":"2024-10-24T11:03:51.917858Z","shell.execute_reply.started":"2024-10-24T11:03:51.888228Z","shell.execute_reply":"2024-10-24T11:03:51.916379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.922048Z","iopub.execute_input":"2024-10-24T11:03:51.922458Z","iopub.status.idle":"2024-10-24T11:03:51.945069Z","shell.execute_reply.started":"2024-10-24T11:03:51.922415Z","shell.execute_reply":"2024-10-24T11:03:51.943721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  list object based feature columns\n\nobject_cols = [col for col in train_df.columns if train_df[col].dtype == 'object']\nprint(object_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.946693Z","iopub.execute_input":"2024-10-24T11:03:51.947185Z","iopub.status.idle":"2024-10-24T11:03:51.959403Z","shell.execute_reply.started":"2024-10-24T11:03:51.947123Z","shell.execute_reply":"2024-10-24T11:03:51.958186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  list number based feature columns\nnumber_cols = [col for col in train_df.columns if train_df[col].dtype != 'object']\nprint(number_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.960905Z","iopub.execute_input":"2024-10-24T11:03:51.96143Z","iopub.status.idle":"2024-10-24T11:03:51.975852Z","shell.execute_reply.started":"2024-10-24T11:03:51.96138Z","shell.execute_reply":"2024-10-24T11:03:51.974733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PCIAT = ['PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n         'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08',\n         'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n         'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16',\n         'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20',\n         'PCIAT-PCIAT_Total', 'PCIAT-Season']\nprint(len('PCIAT-PCIAT_01'))\nprint(len('PCIAT-PCIAT_02'))\nprint(len('PCIAT-PCIAT_03'))\nprint(len('PCIAT-PCIAT_04'))\nprint(len('PCIAT-PCIAT_05'))\nprint(len('PCIAT-PCIAT_06'))\nprint(len('PCIAT-PCIAT_07'))\nprint(len('PCIAT-PCIAT_08'))\nprint(len('PCIAT-PCIAT_09'))\nprint(len('PCIAT-PCIAT_10'))\nprint(len('PCIAT-PCIAT_11'))\nprint(len('PCIAT-PCIAT_12'))\nprint(len('PCIAT-PCIAT_13'))\nprint(len('PCIAT-PCIAT_14'))\nprint(len('PCIAT-PCIAT_15'))\nprint(len('PCIAT-PCIAT_16'))\nprint(len('PCIAT-PCIAT_17'))\nprint(len('PCIAT-PCIAT_18'))\nprint(len('PCIAT-PCIAT_19'))\nprint(len('PCIAT-PCIAT_20'))\nprint(len('PCIAT-PCIAT_Total'))\nprint(len('PCIAT-Season'))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.977237Z","iopub.execute_input":"2024-10-24T11:03:51.977642Z","iopub.status.idle":"2024-10-24T11:03:51.988738Z","shell.execute_reply.started":"2024-10-24T11:03:51.977603Z","shell.execute_reply":"2024-10-24T11:03:51.987635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #### Data Wrangling\n# PCIAT not present in test hence have to drop in train\ntrain_df.drop(PCIAT, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:51.990344Z","iopub.execute_input":"2024-10-24T11:03:51.990817Z","iopub.status.idle":"2024-10-24T11:03:52.009732Z","shell.execute_reply.started":"2024-10-24T11:03:51.990768Z","shell.execute_reply":"2024-10-24T11:03:52.008466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Preprocessing\nprint(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:52.011125Z","iopub.execute_input":"2024-10-24T11:03:52.011544Z","iopub.status.idle":"2024-10-24T11:03:52.024827Z","shell.execute_reply.started":"2024-10-24T11:03:52.011505Z","shell.execute_reply":"2024-10-24T11:03:52.02356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# Display counts of classes\nsns.catplot(x = 'sii', kind = \"count\", data = train_df, height = 6)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:52.026429Z","iopub.execute_input":"2024-10-24T11:03:52.026788Z","iopub.status.idle":"2024-10-24T11:03:53.283818Z","shell.execute_reply.started":"2024-10-24T11:03:52.026751Z","shell.execute_reply":"2024-10-24T11:03:53.282737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data into classes\ndf_0 = train_df[train_df['sii'] == 0.0]\ndf_1 = train_df[train_df['sii'] == 1.0]\ndf_2 = train_df[train_df['sii'] == 2.0]\ndf_3 = train_df[train_df['sii'] == 3.0]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.285083Z","iopub.execute_input":"2024-10-24T11:03:53.285571Z","iopub.status.idle":"2024-10-24T11:03:53.296136Z","shell.execute_reply.started":"2024-10-24T11:03:53.285534Z","shell.execute_reply":"2024-10-24T11:03:53.295121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Resample using \"Bootstrapping\" method to regenerate samples by upsampling for each class.\nfrom sklearn.utils import resample\ndf_0_upsample = resample(df_0, n_samples = 2000, replace = True, random_state = 123)\ndf_1_upsample = resample(df_1, n_samples = 2000, replace = True, random_state = 123)\ndf_2_upsample = resample(df_2, n_samples = 2000, replace = True, random_state = 123)\ndf_3_upsample = resample(df_3, n_samples = 2000, replace = True, random_state = 123)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.297666Z","iopub.execute_input":"2024-10-24T11:03:53.298558Z","iopub.status.idle":"2024-10-24T11:03:53.358684Z","shell.execute_reply.started":"2024-10-24T11:03:53.298516Z","shell.execute_reply":"2024-10-24T11:03:53.357455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Merge all dataframes to create new train samples\ntrain_df = pd.concat([df_0_upsample, df_1_upsample, df_2_upsample, df_3_upsample])\n\ntrain_df['sii'].value_counts()\n\nplt.style.use('ggplot')\nplt.figure(figsize=(10,10))\nmy_circle = plt.Circle((0,0), 0.7, color = 'white')\nplt.pie(train_df['sii'].value_counts(), labels = ['None','Mild','Moderate','Severe'], colors = ['#ff9999','#66b3ff','#99ff99'\n                                                  ], autopct = '%0.0f%%')\np = plt.gcf()\np.gca().add_artist(my_circle)\nplt.show()\n\ntrain_df = train_df.dropna(subset=['sii'])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.360184Z","iopub.execute_input":"2024-10-24T11:03:53.360566Z","iopub.status.idle":"2024-10-24T11:03:53.624483Z","shell.execute_reply.started":"2024-10-24T11:03:53.360529Z","shell.execute_reply":"2024-10-24T11:03:53.623246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Multiple columns have missing data so for outcome fast Simple Imputer is used like from sklearn.impute import SimpleImputer strategy = 'median', 'mean', 'most_frequent' impute = SimpleImputer(strategy = 'median') data_array = impute.fit_transform(data_raw)\n\ndef data_preprocessing(train, test):\n    x = train_df.drop(['id'], axis=1)\n    x_test = test_df\n    y= train_df['sii']\n\n\n    cat_train_feature = list(train_df.select_dtypes(include=['object']).columns)\n    num_train_feature = train_df.select_dtypes(include=['number']).columns.tolist()\n\n    cat_test_feature = list(test_df.select_dtypes(include=['object']).columns)\n    num_test_feature = test_df.select_dtypes(include=['number']).columns.tolist()\n\n    for feature in cat_train_feature:\n      train[feature] = train_df[feature].fillna(train_df[feature].mode()[0])\n      test[feature] = test_df[feature].fillna(test_df[feature].mode()[0])\n\n    # Get common numerical columns for imputation\n    common_num_cols = list(set(num_train_feature).intersection(num_test_feature))\n\n    imputer = SimpleImputer(strategy='mean')\n    # Impute on common numerical columns\n    train_df[common_num_cols] = imputer.fit_transform(train_df[common_num_cols])\n    test_df[common_num_cols] = imputer.transform(test_df[common_num_cols])\n\n\n    # Scale on common numerical columns\n    scaler = StandardScaler()\n    train_df[common_num_cols] = scaler.fit_transform(train_df[common_num_cols])\n    test_df[common_num_cols] = scaler.transform(test_df[common_num_cols])\n\n    encode = LabelEncoder()\n    cat_columns = train_df.select_dtypes(include=['object']).columns\n\n    # Handle unseen labels in test data\n    for feature in cat_columns:\n        # Fit on combined unique values from train and test\n        all_values = pd.concat([train_df[feature], test_df[feature]]).unique()\n        encode.fit(all_values)\n        train_df[feature] = encode.transform(train_df[feature])\n        test_df[feature] = encode.transform(test_df[feature])\n\n    return train_df.to_numpy(), test_df.to_numpy()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.626086Z","iopub.execute_input":"2024-10-24T11:03:53.626596Z","iopub.status.idle":"2024-10-24T11:03:53.63895Z","shell.execute_reply.started":"2024-10-24T11:03:53.626543Z","shell.execute_reply":"2024-10-24T11:03:53.637693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.640339Z","iopub.execute_input":"2024-10-24T11:03:53.640734Z","iopub.status.idle":"2024-10-24T11:03:53.679672Z","shell.execute_reply.started":"2024-10-24T11:03:53.640699Z","shell.execute_reply":"2024-10-24T11:03:53.678486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.680888Z","iopub.execute_input":"2024-10-24T11:03:53.681276Z","iopub.status.idle":"2024-10-24T11:03:53.688931Z","shell.execute_reply.started":"2024-10-24T11:03:53.681237Z","shell.execute_reply":"2024-10-24T11:03:53.687897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import LabelEncoder\nx, x_test = data_preprocessing(train_df, test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:53.690664Z","iopub.execute_input":"2024-10-24T11:03:53.691223Z","iopub.status.idle":"2024-10-24T11:03:54.164428Z","shell.execute_reply.started":"2024-10-24T11:03:53.691164Z","shell.execute_reply":"2024-10-24T11:03:54.163177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.165839Z","iopub.execute_input":"2024-10-24T11:03:54.166188Z","iopub.status.idle":"2024-10-24T11:03:54.173623Z","shell.execute_reply.started":"2024-10-24T11:03:54.166153Z","shell.execute_reply":"2024-10-24T11:03:54.172403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.176473Z","iopub.execute_input":"2024-10-24T11:03:54.176979Z","iopub.status.idle":"2024-10-24T11:03:54.190797Z","shell.execute_reply.started":"2024-10-24T11:03:54.176927Z","shell.execute_reply":"2024-10-24T11:03:54.189509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y= train_df['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.192201Z","iopub.execute_input":"2024-10-24T11:03:54.193048Z","iopub.status.idle":"2024-10-24T11:03:54.201125Z","shell.execute_reply.started":"2024-10-24T11:03:54.193003Z","shell.execute_reply":"2024-10-24T11:03:54.200075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.202445Z","iopub.execute_input":"2024-10-24T11:03:54.202853Z","iopub.status.idle":"2024-10-24T11:03:54.215881Z","shell.execute_reply.started":"2024-10-24T11:03:54.202816Z","shell.execute_reply":"2024-10-24T11:03:54.214783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.224867Z","iopub.execute_input":"2024-10-24T11:03:54.225267Z","iopub.status.idle":"2024-10-24T11:03:54.238642Z","shell.execute_reply.started":"2024-10-24T11:03:54.225228Z","shell.execute_reply":"2024-10-24T11:03:54.237264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.239912Z","iopub.execute_input":"2024-10-24T11:03:54.240332Z","iopub.status.idle":"2024-10-24T11:03:54.248682Z","shell.execute_reply.started":"2024-10-24T11:03:54.240281Z","shell.execute_reply":"2024-10-24T11:03:54.247425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_val.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.250075Z","iopub.execute_input":"2024-10-24T11:03:54.25053Z","iopub.status.idle":"2024-10-24T11:03:54.260249Z","shell.execute_reply.started":"2024-10-24T11:03:54.250493Z","shell.execute_reply":"2024-10-24T11:03:54.259219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import to_categorical # to_categorical has been moved to keras.utils","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:03:54.26185Z","iopub.execute_input":"2024-10-24T11:03:54.262404Z","iopub.status.idle":"2024-10-24T11:04:08.482142Z","shell.execute_reply.started":"2024-10-24T11:03:54.262337Z","shell.execute_reply":"2024-10-24T11:04:08.480835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = to_categorical(y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.483566Z","iopub.execute_input":"2024-10-24T11:04:08.484191Z","iopub.status.idle":"2024-10-24T11:04:08.490491Z","shell.execute_reply.started":"2024-10-24T11:04:08.484141Z","shell.execute_reply":"2024-10-24T11:04:08.48927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val= to_categorical(y_val)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.492056Z","iopub.execute_input":"2024-10-24T11:04:08.492504Z","iopub.status.idle":"2024-10-24T11:04:08.634567Z","shell.execute_reply.started":"2024-10-24T11:04:08.49246Z","shell.execute_reply":"2024-10-24T11:04:08.633168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train[:, :-1]\nx_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.636241Z","iopub.execute_input":"2024-10-24T11:04:08.63674Z","iopub.status.idle":"2024-10-24T11:04:08.650716Z","shell.execute_reply.started":"2024-10-24T11:04:08.636689Z","shell.execute_reply":"2024-10-24T11:04:08.649455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_val = x_val[:, :-1]\nx_val.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.652113Z","iopub.execute_input":"2024-10-24T11:04:08.652595Z","iopub.status.idle":"2024-10-24T11:04:08.664084Z","shell.execute_reply.started":"2024-10-24T11:04:08.652547Z","shell.execute_reply":"2024-10-24T11:04:08.662467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = x_test[:, :-1]\nx_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.665666Z","iopub.execute_input":"2024-10-24T11:04:08.666256Z","iopub.status.idle":"2024-10-24T11:04:08.678939Z","shell.execute_reply.started":"2024-10-24T11:04:08.6662Z","shell.execute_reply":"2024-10-24T11:04:08.677436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # Import numpy for array operations\n\nx_train = x_train.reshape(x_train.shape[0],x_train.shape[1], 1) # Reshape directly without using .values\nx_val = x_val.reshape(x_val.shape[0],x_val.shape[1], 1) # Reshape directly without using .values\nx_test = x_test.reshape(x_test.shape[0],x_test.shape[1], 1) # Reshape directly without using .values","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.680746Z","iopub.execute_input":"2024-10-24T11:04:08.681168Z","iopub.status.idle":"2024-10-24T11:04:08.692823Z","shell.execute_reply.started":"2024-10-24T11:04:08.681127Z","shell.execute_reply":"2024-10-24T11:04:08.691591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.694189Z","iopub.execute_input":"2024-10-24T11:04:08.694591Z","iopub.status.idle":"2024-10-24T11:04:08.709911Z","shell.execute_reply.started":"2024-10-24T11:04:08.694552Z","shell.execute_reply":"2024-10-24T11:04:08.708474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_val.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.711568Z","iopub.execute_input":"2024-10-24T11:04:08.711972Z","iopub.status.idle":"2024-10-24T11:04:08.724054Z","shell.execute_reply.started":"2024-10-24T11:04:08.71192Z","shell.execute_reply":"2024-10-24T11:04:08.72289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.726326Z","iopub.execute_input":"2024-10-24T11:04:08.727321Z","iopub.status.idle":"2024-10-24T11:04:08.741339Z","shell.execute_reply.started":"2024-10-24T11:04:08.727276Z","shell.execute_reply":"2024-10-24T11:04:08.739841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train [0]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.742867Z","iopub.execute_input":"2024-10-24T11:04:08.743362Z","iopub.status.idle":"2024-10-24T11:04:08.756254Z","shell.execute_reply.started":"2024-10-24T11:04:08.74332Z","shell.execute_reply":"2024-10-24T11:04:08.755004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test[0]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.757696Z","iopub.execute_input":"2024-10-24T11:04:08.758075Z","iopub.status.idle":"2024-10-24T11:04:08.772992Z","shell.execute_reply.started":"2024-10-24T11:04:08.758027Z","shell.execute_reply":"2024-10-24T11:04:08.771717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense\n\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dropout\nfrom tensorflow.keras.optimizers import Adam\n# Avoid Overfitting of NN by Normalizing the samples\nfrom tensorflow.keras.layers import BatchNormalization\n# Import regularizers\nfrom tensorflow.keras import regularizers # Import regularizers\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.774316Z","iopub.execute_input":"2024-10-24T11:04:08.774673Z","iopub.status.idle":"2024-10-24T11:04:08.866743Z","shell.execute_reply.started":"2024-10-24T11:04:08.774637Z","shell.execute_reply":"2024-10-24T11:04:08.865144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    # Filters = No. of Neurons\n    # Padding = 'same' : Zero Padding; Padding = 'valid' : valid padding\n    model.add(Conv1D(filters = 64, kernel_size =5, activation = 'relu', padding = 'same', input_shape = (58, 1), kernel_regularizer=regularizers.l2(0.001)))\n    # BatchNormalization to avoid overfitting\n    model.add(BatchNormalization())\n    # Pooling\n    model.add(MaxPooling1D(pool_size=(2), strides=(2), padding='same'))\n\n    # Conv Layer - II\n    model.add(Conv1D(filters = 64, kernel_size = 5, activation = 'relu', padding = 'same', kernel_regularizer=regularizers.l2(0.001)))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.5))\n    model.add(MaxPooling1D(pool_size=(2), strides=(2), padding='same'))\n\n    # Conv Layer - III\n    model.add(Conv1D(filters = 64, kernel_size = 5, activation = 'relu', padding = 'same', kernel_regularizer=regularizers.l2(0.001)))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.5))\n    model.add(MaxPooling1D(pool_size=(2), strides=(2), padding='same'))\n\n\n    # Flatten\n    model.add(Flatten())\n\n    # Fully Connected Layer (FC - Layer)\n    model.add(Dense(units = 64, activation='relu', kernel_regularizer=regularizers.l2(0.001)))\n    model.add(Dropout(0.7))\n    model.add(Dense(units = 64, activation='relu', kernel_regularizer=regularizers.l2(0.001)))\n    model.add(Dropout(0.7))\n\n\n    # Output Layer\n    model.add(Dense(units = 4, activation='softmax'))\n\n    # Define the learning rate schedule\n    initial_learning_rate = 0.00005\n    lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n        initial_learning_rate =0.001,\n        decay_steps=100,  # Decay every 100 steps\n        decay_rate=0.9,    # Decay rate\n        staircase=False     # Use continuous decay\n    )\n\n    optimizer = Adam( decay=lr_schedule)\n\n    # loss = 'categorical_crossentropy'\n    model.compile(optimizer = 'Adam', loss = 'categorical_crossentropy', metrics = ['accuracy'])\n\n    return model\n\nmodel = build_model()\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:08.868617Z","iopub.execute_input":"2024-10-24T11:04:08.86899Z","iopub.status.idle":"2024-10-24T11:04:09.204047Z","shell.execute_reply.started":"2024-10-24T11:04:08.868951Z","shell.execute_reply":"2024-10-24T11:04:09.202971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save best model\nfrom tensorflow.keras import callbacks\nfilepath ='/kaggle/working/CMI_Model.keras'\n\ncheckpoint = callbacks.ModelCheckpoint(filepath, monitor='val_loss', save_best_only=True,\n                                       mode = 'min', verbose = 1)\ncheckpoint","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:09.205523Z","iopub.execute_input":"2024-10-24T11:04:09.205862Z","iopub.status.idle":"2024-10-24T11:04:09.214515Z","shell.execute_reply.started":"2024-10-24T11:04:09.205827Z","shell.execute_reply":"2024-10-24T11:04:09.213029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport datetime\nfrom tensorflow import keras\nlogdir = os.path.join(\"/kaggle/working/CMI_Model_logs\", datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\"))\ntensorboard_callback = keras.callbacks.TensorBoard(logdir)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:09.215851Z","iopub.execute_input":"2024-10-24T11:04:09.216482Z","iopub.status.idle":"2024-10-24T11:04:09.232477Z","shell.execute_reply.started":"2024-10-24T11:04:09.216439Z","shell.execute_reply":"2024-10-24T11:04:09.231437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = callbacks.EarlyStopping(monitor='val_loss', patience=30, restore_best_weights=True)\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\n\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=10, min_lr=1e-14)\n\n#  calculate class weights\n\nfrom sklearn.utils import class_weight\n\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(y),\n    y=y\n)\nclass_weights = dict(enumerate(class_weights))\nprint(class_weights)\nimport time # \n\nstart_time = time.time()\n\n#  model.fit\n\nhistory = model.fit(x_train, y_train, epochs = 400, batch_size = 32, validation_data = (x_val, y_val), callbacks = [checkpoint, early_stopping, reduce_lr, tensorboard_callback], class_weight=class_weights)\nend_time = time.time()\n\ntotal_time = end_time - start_time\nprint(f\"Total training time: {total_time:.2f} seconds\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:04:09.233878Z","iopub.execute_input":"2024-10-24T11:04:09.23427Z","iopub.status.idle":"2024-10-24T11:22:08.643563Z","shell.execute_reply.started":"2024-10-24T11:04:09.234233Z","shell.execute_reply":"2024-10-24T11:22:08.641932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(history.history)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:08.645452Z","iopub.execute_input":"2024-10-24T11:22:08.645942Z","iopub.status.idle":"2024-10-24T11:22:08.664809Z","shell.execute_reply.started":"2024-10-24T11:22:08.64589Z","shell.execute_reply":"2024-10-24T11:22:08.663338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(history.history)[['accuracy','val_accuracy']].plot(figsize = (7,6))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:08.666423Z","iopub.execute_input":"2024-10-24T11:22:08.666917Z","iopub.status.idle":"2024-10-24T11:22:09.003667Z","shell.execute_reply.started":"2024-10-24T11:22:08.666864Z","shell.execute_reply":"2024-10-24T11:22:09.002434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(history.history)[['loss','val_loss']].plot()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:09.005308Z","iopub.execute_input":"2024-10-24T11:22:09.005808Z","iopub.status.idle":"2024-10-24T11:22:09.306131Z","shell.execute_reply.started":"2024-10-24T11:22:09.005757Z","shell.execute_reply":"2024-10-24T11:22:09.304721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(x_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:09.307585Z","iopub.execute_input":"2024-10-24T11:22:09.307965Z","iopub.status.idle":"2024-10-24T11:22:09.582406Z","shell.execute_reply.started":"2024-10-24T11:22:09.307927Z","shell.execute_reply":"2024-10-24T11:22:09.581271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Classification Matrix","metadata":{}},{"cell_type":"code","source":"x_val_pred=x_val.reshape(x_val.shape[0],x_val.shape[1], 1)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:09.583799Z","iopub.execute_input":"2024-10-24T11:22:09.584152Z","iopub.status.idle":"2024-10-24T11:22:09.589543Z","shell.execute_reply.started":"2024-10-24T11:22:09.584116Z","shell.execute_reply":"2024-10-24T11:22:09.588216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val_pred = model.predict(x_val_pred)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:09.590941Z","iopub.execute_input":"2024-10-24T11:22:09.591455Z","iopub.status.idle":"2024-10-24T11:22:10.107129Z","shell.execute_reply.started":"2024-10-24T11:22:09.591405Z","shell.execute_reply":"2024-10-24T11:22:10.105975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# 1. Print the shape of x_val\nprint(\"Shape of x_val:\", x_val.shape)\n\n# 2. Check for empty dimensions\nif 0 in x_val.shape:\n    print(\"Error: x_val has a dimension with size 0.\")\n    # Investigate why x_val has a dimension with size 0 and fix the data loading process.\n\n# 3. Compare with the model's input shape\n# Assuming your model's input shape is (59, 1) based on the global variables\nexpected_input_shape = (59, 1)\nif x_val.shape[1:] != expected_input_shape:\n    print(\"Error: x_val shape does not match the model's input shape.\")\n    print(\"Expected input shape:\", expected_input_shape)\n    print(\"Actual input shape:\", x_val.shape[1:])\n    # Reshape x_val to match the expected input shape using np.reshape\n    # x_val = np.reshape(x_val, (x_val.shape[0],) + expected_input_shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:10.109325Z","iopub.execute_input":"2024-10-24T11:22:10.10975Z","iopub.status.idle":"2024-10-24T11:22:10.118174Z","shell.execute_reply.started":"2024-10-24T11:22:10.109704Z","shell.execute_reply":"2024-10-24T11:22:10.116759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x_val, y_val evaluation, classification report confusion matrix\n\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport numpy as np\nimport tensorflow as tf  # Import TensorFlow\n\n# Assuming you have y_val_pred from your model predictions\n\n# Check the shape of x_val\nprint(\"Shape of x_val:\", x_val.shape)\n\n# If x_val has a dimension with size 0, reshape it to remove that dimension\nif 0 in x_val.shape:\n    x_val = x_val.reshape(x_val.shape[0], x_val.shape[1])\n    print(\"Reshaped x_val:\", x_val.shape)\n\n# Reshape the output of Flatten layer to be compatible with the next layer\n# The issue was with the reshaping of x_val for the prediction.\n# The original code was creating a shape of (1280, 1, 59)\n# which was not compatible with the model's input shape.\n# We need to reshape x_val to match the original input shape of the model,\n# which was likely (None, 59, 1) based on your provided information.\n\nx_val_reshaped = x_val.reshape((-1, x_val.shape[1], 1)) # Reshape to (None, 59, 1)\n\ny_val_pred = np.argmax(model.predict(x_val_reshaped), axis=1)  # Get predicted class labels\ny_val_true = np.argmax(y_val, axis=1)  # Get true class labels\n\n# Classification report\nprint(classification_report(y_val_true, y_val_pred))\n\n# Confusion matrix\ncm = confusion_matrix(y_val_true, y_val_pred)\nprint(\"Confusion Matrix:\")\nprint(cm)\n\n# You can visualize the confusion matrix using seaborn\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['None', 'Mild', 'Moderate', 'Severe'], yticklabels=['None', 'Mild', 'Moderate', 'Severe'])\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:10.120056Z","iopub.execute_input":"2024-10-24T11:22:10.120571Z","iopub.status.idle":"2024-10-24T11:22:10.742276Z","shell.execute_reply.started":"2024-10-24T11:22:10.120518Z","shell.execute_reply":"2024-10-24T11:22:10.740804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  calculate cohen_kappa_score\n\nfrom sklearn.metrics import cohen_kappa_score\n\n# Assuming y_val_true and y_val_pred are already defined as in your previous code:\n# y_val_pred = np.argmax(model.predict(x_val_reshaped), axis=1)\n# y_val_true = np.argmax(y_val, axis=1)\n\nkappa_score = cohen_kappa_score(y_val_true, y_val_pred)\nprint(\"Cohen's Kappa Score:\", kappa_score)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:10.743698Z","iopub.execute_input":"2024-10-24T11:22:10.744259Z","iopub.status.idle":"2024-10-24T11:22:10.754419Z","shell.execute_reply.started":"2024-10-24T11:22:10.744204Z","shell.execute_reply":"2024-10-24T11:22:10.753087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = (x_test)\ntest_data","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:23:27.272101Z","iopub.execute_input":"2024-10-24T11:23:27.272616Z","iopub.status.idle":"2024-10-24T11:23:27.282221Z","shell.execute_reply.started":"2024-10-24T11:23:27.272571Z","shell.execute_reply":"2024-10-24T11:23:27.280889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = model.predict(test_data)\nyhat = np.round(predict)\nyhat","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:30:19.063438Z","iopub.execute_input":"2024-10-24T11:30:19.063868Z","iopub.status.idle":"2024-10-24T11:30:19.202116Z","shell.execute_reply.started":"2024-10-24T11:30:19.063822Z","shell.execute_reply":"2024-10-24T11:30:19.20035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat=display(predict)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.542454Z","iopub.status.idle":"2024-10-24T11:22:11.542865Z","shell.execute_reply.started":"2024-10-24T11:22:11.542677Z","shell.execute_reply":"2024-10-24T11:22:11.542697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yhat = np.round(predict)\nyhat","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.544683Z","iopub.status.idle":"2024-10-24T11:22:11.545112Z","shell.execute_reply.started":"2024-10-24T11:22:11.544889Z","shell.execute_reply":"2024-10-24T11:22:11.544911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Assuming 'y_hat' is already defined as in your provided code\n# If not, replace this with the actual code to generate y_hat\n\n# Create a DataFrame with the 'SII' predictions\nsubmission_df = pd.DataFrame({'SII': y_hat})\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('sii_predictions.csv', index=False)\n\n# Download the file\nfrom google.colab import files\nfiles.download('sii_predictions.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.546382Z","iopub.status.idle":"2024-10-24T11:22:11.546799Z","shell.execute_reply.started":"2024-10-24T11:22:11.546606Z","shell.execute_reply":"2024-10-24T11:22:11.546627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"   import torch\n   import numpy as np\n   import pandas as pd\n   from sklearn.preprocessing import StandardScaler, LabelEncoder\n   from sklearn.impute import SimpleImputer\n\n   def data_preprocessing(test):\n       x_test = (test_df)  \n\n       cat_test_feature = list(test.select_dtypes(include=['object']).columns)\n       num_test_feature = test.select_dtypes(include=['number']).columns.tolist()\n\n       for feature in cat_test_feature:\n           test[feature] = test[feature].fillna(test[feature].mode()[0])\n\n       num_train_feature = test.select_dtypes(include=['number']).columns.tolist()\n\n       common_num_cols = list(set(num_train_feature).intersection(num_test_feature))\n\n       imputer = SimpleImputer(strategy='mean')\n       imputer.fit(test[common_num_cols])\n       test[common_num_cols] = imputer.transform(test[common_num_cols])\n\n       scaler = StandardScaler()\n       scaler.fit(test[common_num_cols])\n       test[common_num_cols] = scaler.transform(test[common_num_cols])\n\n       encode = LabelEncoder()\n       cat_columns = test.select_dtypes(include=['object']).columns\n\n       for feature in cat_columns:\n           test_values = test[feature].unique()\n           encode.fit(test_values)\n           test[feature] = encode.transform(test[feature])\n\n       x_test_np = test.to_numpy()  # Convert to NumPy array\n       x_test_np = x_test_np.astype(np.float32)  # Convert to float32\n\n       return x_test_np\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.54853Z","iopub.status.idle":"2024-10-24T11:22:11.549084Z","shell.execute_reply.started":"2024-10-24T11:22:11.5488Z","shell.execute_reply":"2024-10-24T11:22:11.548829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Assuming test_data is your pandas dataframe for test data\nx_test_processed = data_preprocessing(test_data) #Fixed: Passing test_data to data_preprocessing\n\n# Reshape if necessary based on the expected input shape of the model (59, 1)\nx_test_processed = x_test_processed.reshape(x_test_processed.shape[0], 59, 1)\n\npredict = model.predict(x_test_processed)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.550549Z","iopub.status.idle":"2024-10-24T11:22:11.551094Z","shell.execute_reply.started":"2024-10-24T11:22:11.550814Z","shell.execute_reply":"2024-10-24T11:22:11.550843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distributed probability to discrete class\nyhat = np.argmax(predict, axis = 1)\nyhat","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.552242Z","iopub.status.idle":"2024-10-24T11:22:11.55282Z","shell.execute_reply.started":"2024-10-24T11:22:11.552536Z","shell.execute_reply":"2024-10-24T11:22:11.55257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save file as submission.csv\n\nsubmission_df = pd.DataFrame({'id': test_df['id'], 'sii': yhat})\nprint (submission_df)\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Submission saved to submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.555315Z","iopub.status.idle":"2024-10-24T11:22:11.555766Z","shell.execute_reply.started":"2024-10-24T11:22:11.555563Z","shell.execute_reply":"2024-10-24T11:22:11.555585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  predicted yhat and it's timing\n\nimport time\n\n# Assuming 'X_test' is your test data and 'model' is your trained model\nstart_time = time.time()\nyhat = np.argmax(predict)\nend_time = time.time()\n\n# Convert predicted probabilities to class labels\nyhat = np.argmax(yhat)\n\nprint(\"Predicted yhat:\", yhat)\nprint(\"Prediction time:\", end_time - start_time, \"seconds\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.556822Z","iopub.status.idle":"2024-10-24T11:22:11.557322Z","shell.execute_reply.started":"2024-10-24T11:22:11.557108Z","shell.execute_reply":"2024-10-24T11:22:11.557142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras.layers import BatchNormalization, Dropout\nloaded_model = keras.models.load_model('/kaggle/working/CMI_Model.keras', custom_objects={'BatchNormalization': BatchNormalization, 'Dropout': Dropout})\n\nimport tensorflow as tf\n\ndef calculate_flops_gmacs(model, input_shape):\n    \"\"\"Calculates the floating point operations (FLOPs) and Giga Multiply-Accumulates (GMACs)\n       for a TensorFlow model.\n\n    Args:\n        model: The TensorFlow model.\n        input_shape: The shape of the input tensor.\n\n    Returns:\n        A tuple containing the FLOPs and GMACs.\n    \"\"\"\n\n    concrete_func = tf.function(lambda x: model(x)).get_concrete_function(\n        tf.TensorSpec(input_shape, dtype=tf.float32)\n    )\n    run_meta = tf.compat.v1.RunMetadata()\n    opts = tf.compat.v1.profiler.ProfileOptionBuilder.float_operation()\n    flops = tf.compat.v1.profiler.profile(\n        graph=concrete_func.graph,\n        run_meta=run_meta,\n        cmd='op',\n        options=opts\n    ).total_float_ops\n\n    gmacs = flops / (10**9 * 2)  # Divide by 2 to account for Multiply and Accumulate\n\n    return flops, gmacs\n\n# Example usage:\ninput_shape = (1, 59, 1)  # Replace with the actual input shape of your model\nflops, gmacs = calculate_flops_gmacs(loaded_model, input_shape)\n\nprint(f\"FLOPs: {flops}\")\nprint(f\"GMACs: {gmacs}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.55883Z","iopub.status.idle":"2024-10-24T11:22:11.559438Z","shell.execute_reply.started":"2024-10-24T11:22:11.559097Z","shell.execute_reply":"2024-10-24T11:22:11.559137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nsample_sub_data = pd.read_csv (path_sub)\nsample_sub_data","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.560708Z","iopub.status.idle":"2024-10-24T11:22:11.561217Z","shell.execute_reply.started":"2024-10-24T11:22:11.560968Z","shell.execute_reply":"2024-10-24T11:22:11.560997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub_data['sii'].y_pred = predict  # Assuming 'predict' holds the model's predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.562716Z","iopub.status.idle":"2024-10-24T11:22:11.563196Z","shell.execute_reply.started":"2024-10-24T11:22:11.562988Z","shell.execute_reply":"2024-10-24T11:22:11.563012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  sub_data['sii'].y_pred round off\n\nsample_sub_data['sii'] = np.round(sample_sub_data['sii']).astype(int)\nsample_sub_data['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.564525Z","iopub.status.idle":"2024-10-24T11:22:11.564936Z","shell.execute_reply.started":"2024-10-24T11:22:11.564741Z","shell.execute_reply":"2024-10-24T11:22:11.564763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the predictions from the 'sii' column\npredictions = sample_sub_data['sii']\n\n# Find the class with the highest probability for each sample\nyhat1 = np.argmax(predictions, axis=0)  # Changed axis to 0\n\nprint(yhat1)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.567655Z","iopub.status.idle":"2024-10-24T11:22:11.568079Z","shell.execute_reply.started":"2024-10-24T11:22:11.567873Z","shell.execute_reply":"2024-10-24T11:22:11.567893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  classification report and confusion matrix between y_true.value_counts and predicted sub_data['sii']\n\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np # Import numpy\n\n# Assuming 'y_true' is not in sub_data, you'll need to add it\n# Replace this with how 'y_true' should be derived\n# For example, if 'y_true' was a list of ground truth labels:\n# sub_data['y_true'] = y_true_list\n# Where y_true_list is the list containing actual labels.\n# If y_true_list is not defined, you might have it saved in a variable.\n# Replace with the appropriate variable or method to generate it.\n# If you're unsure how y_true was previously defined, you need to review your earlier code\n# Or provide more context on how it should be created\n\n# Creating y_true_list as an example (replace if you have a different way of obtaining it)\n# Assuming you have 20 rows in sub_data with evenly distributed labels (0,1,2,3)\ny_true_list = np.repeat([0, 1, 2, 3], 5)\n\n# Add y_true column to sub_data\nsample_sub_data['y_true'] = y_true_list\n\n\n# Assuming y_true and y_pred are defined from your code\ny_true = sample_sub_data['y_true']\ny_pred = sample_sub_data['sii']\n\n# Print classification report\nprint(classification_report(y_true, y_pred))\n\n# Print confusion matrix\ncm = confusion_matrix(y_true, y_pred)\nprint(\"Confusion Matrix:\")\nprint(cm)\n\n# Visualize confusion matrix using seaborn\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=['None', 'Mild', 'Moderate', 'Severe'], yticklabels=['None', 'Mild', 'Moderate', 'Severe'])\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.570308Z","iopub.status.idle":"2024-10-24T11:22:11.57092Z","shell.execute_reply.started":"2024-10-24T11:22:11.570632Z","shell.execute_reply":"2024-10-24T11:22:11.570664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  save file to submission1.csv\n\nimport pandas as pd\n\n# Assuming you have your predictions in a variable called 'yhat'\nsubmission1_df = pd.DataFrame({'id': test_data['id'], 'sii': yhat})\nprint(submission1_df)\n# Save the DataFrame to a CSV file named 'submission.csv'\nsubmission1_df.to_csv('submission1.csv', index=False)\n\nprint(\"Submission saved to submission1.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.57218Z","iopub.status.idle":"2024-10-24T11:22:11.572781Z","shell.execute_reply.started":"2024-10-24T11:22:11.572475Z","shell.execute_reply":"2024-10-24T11:22:11.572505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras.layers import BatchNormalization\nfrom tensorflow.keras.layers import Dropout\n# Load the model architecture\nloaded_model = keras.models.load_model('/kaggle/working/CMI_Model.keras', custom_objects={'BatchNormalization': BatchNormalization, 'Dropout': Dropout})\n# Compile the loaded model\nloaded_model.compile(optimizer = 'Adam', loss = 'categorical_crossentropy', metrics = ['accuracy'])\nimport tensorflow as tf\n# Convert the Keras model to TensorFlow Lite format\nconverter = tf.lite.TFLiteConverter.from_keras_model(loaded_model)\ntflite_model = converter.convert()\n\n# Save the TensorFlow Lite model to a file\n# Changed the file path to include a filename 'CMI_Model.tflite' to avoid writing to a directory.\nwith open('/kaggle/working/CMI_Model.tflite', 'wb') as f:\n  f.write(tflite_model)\n\n\nimport tensorflow as tf\n#  how to load this model in android /ios mobile app\n\n# Convert the Keras model to TensorFlow Lite format\nconverter = tf.lite.TFLiteConverter.from_keras_model(loaded_model)\ntflite_model = converter.convert()\n\n# Save the TensorFlow Lite model to a file\nwith open('/kaggle/working/CMI_Model.tflite', 'wb') as f:\n  f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:22:11.575208Z","iopub.status.idle":"2024-10-24T11:22:11.575811Z","shell.execute_reply.started":"2024-10-24T11:22:11.575511Z","shell.execute_reply":"2024-10-24T11:22:11.575542Z"},"trusted":true},"execution_count":null,"outputs":[]}]}