{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:10.279184Z","iopub.execute_input":"2024-12-19T18:26:10.279578Z","iopub.status.idle":"2024-12-19T18:26:10.284492Z","shell.execute_reply.started":"2024-12-19T18:26:10.279538Z","shell.execute_reply":"2024-12-19T18:26:10.283309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport missingno as msno # visualizing missing values\n\n# Data processing\nfrom sklearn.semi_supervised import LabelPropagation\nfrom sklearn.semi_supervised import LabelSpreading\nfrom sklearn.semi_supervised import SelfTrainingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import StandardScaler\n\n# Models and algorithms\nfrom xgboost import XGBClassifier, XGBRegressor\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.model_selection import StratifiedKFold\n\n\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\n\n# Evaluation\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import cross_val_score\nimport time\n\nimport warnings\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\n#np.seterr(all='raise')\npd.set_option('display.max_columns', None)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:10.285565Z","iopub.execute_input":"2024-12-19T18:26:10.285955Z","iopub.status.idle":"2024-12-19T18:26:14.693380Z","shell.execute_reply.started":"2024-12-19T18:26:10.285920Z","shell.execute_reply":"2024-12-19T18:26:14.692141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_data_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndf_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_sample_sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n#df_parquet1 = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/id=00115b9f/part-0.parquet')","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.694559Z","iopub.execute_input":"2024-12-19T18:26:14.695385Z","iopub.status.idle":"2024-12-19T18:26:14.765489Z","shell.execute_reply.started":"2024-12-19T18:26:14.695352Z","shell.execute_reply":"2024-12-19T18:26:14.764158Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analyze the data\n\n* Relation between features and SII\n* Missing values\n* Extreme values","metadata":{}},{"cell_type":"code","source":"# Check the train dataset\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.766582Z","iopub.execute_input":"2024-12-19T18:26:14.767195Z","iopub.status.idle":"2024-12-19T18:26:14.772160Z","shell.execute_reply.started":"2024-12-19T18:26:14.767157Z","shell.execute_reply":"2024-12-19T18:26:14.770990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the type of data\n\n# Get the numeric columns\nnumeric_cols = df_train.select_dtypes(np.number).columns\nprint(\"Numerical columns : \", numeric_cols)\nprint(\"Number of numerical columns : \", len(numeric_cols))\n\n# Get the categorical columns\ncategorical_cols =  df_train.select_dtypes(include=['object']).columns# Categorical column name\nprint(\"Categorical columns : \", categorical_cols)\nprint(\"Number of categorical columns : \", len(categorical_cols))\n\n\n# Get the label type\nprint(\"Total number of columns : \", len(df_train.columns))\n\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.775490Z","iopub.execute_input":"2024-12-19T18:26:14.775873Z","iopub.status.idle":"2024-12-19T18:26:14.794723Z","shell.execute_reply.started":"2024-12-19T18:26:14.775834Z","shell.execute_reply":"2024-12-19T18:26:14.793592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the instruments\ninstruments = df_data_dictionary.loc[:, 'Instrument'].unique()\ninstruments","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.797685Z","iopub.execute_input":"2024-12-19T18:26:14.798136Z","iopub.status.idle":"2024-12-19T18:26:14.814463Z","shell.execute_reply.started":"2024-12-19T18:26:14.798095Z","shell.execute_reply":"2024-12-19T18:26:14.813206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# What are the demographic data used ?\ndf_data_dictionary.loc[(df_data_dictionary['Instrument'] == 'Demographics')]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.815712Z","iopub.execute_input":"2024-12-19T18:26:14.816110Z","iopub.status.idle":"2024-12-19T18:26:14.835199Z","shell.execute_reply.started":"2024-12-19T18:26:14.816072Z","shell.execute_reply":"2024-12-19T18:26:14.833834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_data_dictionary.iloc[:54, :]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.836773Z","iopub.execute_input":"2024-12-19T18:26:14.837200Z","iopub.status.idle":"2024-12-19T18:26:14.855986Z","shell.execute_reply.started":"2024-12-19T18:26:14.837154Z","shell.execute_reply":"2024-12-19T18:26:14.854616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check missing values matrix\nmsno.matrix(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.857288Z","iopub.execute_input":"2024-12-19T18:26:14.857743Z","iopub.status.idle":"2024-12-19T18:26:14.874759Z","shell.execute_reply.started":"2024-12-19T18:26:14.857697Z","shell.execute_reply":"2024-12-19T18:26:14.873381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check missing values percentage\nmsno.bar(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.876205Z","iopub.execute_input":"2024-12-19T18:26:14.876716Z","iopub.status.idle":"2024-12-19T18:26:14.894239Z","shell.execute_reply.started":"2024-12-19T18:26:14.876645Z","shell.execute_reply":"2024-12-19T18:26:14.892789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check missing values percentage per feature\ndf_missing_value =  pd.DataFrame(df_train.isna().sum().sort_values(ascending=False))\ndf_missing_value.columns = ['Missing values']\ndf_train_size = len(df_train)\n\ndf_missing_value_ratio = df_missing_value.assign(ratio = (df_missing_value['Missing values'] / df_train_size)*100)\n\nprint(df_missing_value_ratio.iloc[:50])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.895536Z","iopub.execute_input":"2024-12-19T18:26:14.895977Z","iopub.status.idle":"2024-12-19T18:26:14.917060Z","shell.execute_reply.started":"2024-12-19T18:26:14.895932Z","shell.execute_reply":"2024-12-19T18:26:14.915599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the test dataset\ndf_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.918281Z","iopub.execute_input":"2024-12-19T18:26:14.918719Z","iopub.status.idle":"2024-12-19T18:26:14.932256Z","shell.execute_reply.started":"2024-12-19T18:26:14.918679Z","shell.execute_reply":"2024-12-19T18:26:14.931011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the difference between train and test\nprint(df_train.columns)\nprint(len(df_test.columns))\ndf_test.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.933495Z","iopub.execute_input":"2024-12-19T18:26:14.933968Z","iopub.status.idle":"2024-12-19T18:26:14.956939Z","shell.execute_reply.started":"2024-12-19T18:26:14.933924Z","shell.execute_reply":"2024-12-19T18:26:14.955196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Columns difference between the train and test dataset\ndiff_cols = set(df_train.columns) - set(df_test.columns)\ndiff_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.958339Z","iopub.execute_input":"2024-12-19T18:26:14.958879Z","iopub.status.idle":"2024-12-19T18:26:14.982111Z","shell.execute_reply.started":"2024-12-19T18:26:14.958830Z","shell.execute_reply":"2024-12-19T18:26:14.980812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sii value given the PCAT Total\nparent_Child_Question_fields = df_data_dictionary.loc[(df_data_dictionary['Instrument'] == 'Parent-Child Internet Addiction Test')].Field.values\nparent_Child_Question_fields = np.append(parent_Child_Question_fields, 'sii')\n\ndf_pciat_sii = df_train[parent_Child_Question_fields]\nsns.scatterplot(x=df_pciat_sii['sii'], y=df_pciat_sii['PCIAT-PCIAT_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:14.983628Z","iopub.execute_input":"2024-12-19T18:26:14.984082Z","iopub.status.idle":"2024-12-19T18:26:15.004466Z","shell.execute_reply.started":"2024-12-19T18:26:14.984039Z","shell.execute_reply":"2024-12-19T18:26:15.003158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check pciat values where sii is null\ndf_pciat_sii = df_pciat_sii.loc[df_pciat_sii['sii'].isnull()]\nmsno.matrix(df_pciat_sii)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.006030Z","iopub.execute_input":"2024-12-19T18:26:15.006381Z","iopub.status.idle":"2024-12-19T18:26:15.022734Z","shell.execute_reply.started":"2024-12-19T18:26:15.006347Z","shell.execute_reply":"2024-12-19T18:26:15.021385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age and sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.023972Z","iopub.execute_input":"2024-12-19T18:26:15.024331Z","iopub.status.idle":"2024-12-19T18:26:15.042616Z","shell.execute_reply.started":"2024-12-19T18:26:15.024269Z","shell.execute_reply":"2024-12-19T18:26:15.041257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sex and sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.043915Z","iopub.execute_input":"2024-12-19T18:26:15.044409Z","iopub.status.idle":"2024-12-19T18:26:15.061428Z","shell.execute_reply.started":"2024-12-19T18:26:15.044365Z","shell.execute_reply":"2024-12-19T18:26:15.060042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age and sex and sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.070510Z","iopub.execute_input":"2024-12-19T18:26:15.071106Z","iopub.status.idle":"2024-12-19T18:26:15.081035Z","shell.execute_reply.started":"2024-12-19T18:26:15.071057Z","shell.execute_reply":"2024-12-19T18:26:15.079784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Physical activity and sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.083790Z","iopub.execute_input":"2024-12-19T18:26:15.084129Z","iopub.status.idle":"2024-12-19T18:26:15.103787Z","shell.execute_reply.started":"2024-12-19T18:26:15.084100Z","shell.execute_reply":"2024-12-19T18:26:15.102455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Age, sex, physical activity and sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.105088Z","iopub.execute_input":"2024-12-19T18:26:15.105497Z","iopub.status.idle":"2024-12-19T18:26:15.121387Z","shell.execute_reply.started":"2024-12-19T18:26:15.105455Z","shell.execute_reply":"2024-12-19T18:26:15.120094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seasons\nseason_cols = [col for col in df_train.columns if 'Season' in col]\n\nprint(season_cols)\nprint(len(season_cols))\n\nseasons = df_train[season_cols]\n\n# Number of different unique values per row\nprint(seasons.stack().groupby(level=0).apply(lambda x: len(x.unique().tolist())).unique())\n\n# Season features correlation matrix\nseasons_corr = seasons.apply(lambda x : pd.factorize(x)[0]).corr(method='pearson', min_periods=1)\n\n# Show the correlation heatmap\nplt.matshow(seasons_corr)\nplt.show()\n\n# Show the correlation matrix\nseasons_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.122680Z","iopub.execute_input":"2024-12-19T18:26:15.123126Z","iopub.status.idle":"2024-12-19T18:26:15.142055Z","shell.execute_reply.started":"2024-12-19T18:26:15.123095Z","shell.execute_reply":"2024-12-19T18:26:15.140549Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analyze the time series","metadata":{}},{"cell_type":"code","source":"#df_parquet1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.143307Z","iopub.execute_input":"2024-12-19T18:26:15.143773Z","iopub.status.idle":"2024-12-19T18:26:15.160536Z","shell.execute_reply.started":"2024-12-19T18:26:15.143731Z","shell.execute_reply":"2024-12-19T18:26:15.159321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_parquet1.relative_date_PCIAT.unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.161800Z","iopub.execute_input":"2024-12-19T18:26:15.162198Z","iopub.status.idle":"2024-12-19T18:26:15.180185Z","shell.execute_reply.started":"2024-12-19T18:26:15.162153Z","shell.execute_reply":"2024-12-19T18:26:15.178812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Fourier","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary of analysis\n\n\n---\n**Missing labels : around 30.9% of missing labels**\n\n\nAlmost a third of the labels are missing. I think it's too much to use deletion. \n\nA few solutions can be considered :\n* Imputation\n* Inference\n* Propagation\n* Semi-supervised\n\nGiven the context in the data presentation, it is said that The target sii for this competition is derived from the 'Parent-Child Internet Addiction Test' (PCIAT). And after analyzing it, there's indeed a strong correlation between them so i thought of using imputation / inference / or propagation to fill the label with values based on the PCIAT. It turns out we don't have them either when the label is missing so we can't.\n\nI think semi-supervised learning should be the way to go here. I'll be using an auto-encoder\n\n---\n**Difference between training set and test set : the instrument 'Parent-Child Internet Addiction Test' is not included in the test set**\n\n\n\nThe majority of other instruments are about the physical condition and physical activity of the child. The only data related to internet use in the test dataset concerns the number of hours using internet. \n\nShould the instrument 'Parent-Child Internet Addiction Test' be removed from the model training since it won't be available in real use case ?\n\n\n---\n**Features importance :**\n\n\nSince the majority of features include physical condition and activity, we can presume that it will have an impact on the sii score, in relation to the time spent on computer/internet.\nSome feature engineering could be done on the physical condition and activity.\n\n\n---\n**Categorical data : Only Season features**\n\nWe can see that all the categorical datas concern seasons. It is known that seasons influence mental health so my first take was to keep them.\n\nHowever it seems that the seasons presents in the differents features do not have the same values for a given input. Sometimes the Season features of one input can contain the four different seasons.\n\nThe question here is, do all Season features influence the outcome ? If any ? My concern is that there are 11 season columns. If we do one hot encoding for each one of them, it will add up to 55 features (11 x (4 values + 1 Nan)), increasing the time and computation and it might not even help the model.\n\nI searched for correlation between the different season features, with the idea of grouping the season features sharing the same values and trim down to only keep a few selected. It turns out there is not much correlation between the values provided in the different season features. The highest correlation (0.71) is between the Basic_Demos-Enroll_Season from the 'Demographic' instrument and the PreInt_EduHx-Season of the 'Internet Use' instrument.\n\nI think we should drop all of the seasons feature except for those two, or even only one of those two since they're \"higly\" correlated.\n\nSince the goal of the prediction is to determine the mental health regarding to the usage of internet, the PreInt_EduHx-Season feature would be my go-to Season feaure to keep.\n\n\n---\n**Missing values :**\n\nThe missing value in Physical Activity Questionnaire (Adolescents) is high (around 88%) but it's only because Physical Activity Questionnaire (Adolescents) is complementary of Physical Activity Questionnaire (Children) and there is a higher percentage of children compared to adolescent. The column will still be useful so we should keep it.\n\n\n---\n**Actigraphy**\nThe actigraphy monitors the pyshical activity, taking into accounts if it's day or night and the days of the week of the children. It does it with a time tracker relative to the PCIAT.\n\nThe data of the actigraphy might show a correlation between physical activity and PCIAT, and as we know SII is closely correlated to PCIAT.\nSi it's a really interesting data to have in the features, nonetheless there is too much data as it is know.\nWe should reduce the data to find the latent representation\n","metadata":{}},{"cell_type":"markdown","source":"# Process the data","metadata":{}},{"cell_type":"code","source":"df_data_dictionary.iloc[:54, :]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.181820Z","iopub.execute_input":"2024-12-19T18:26:15.182292Z","iopub.status.idle":"2024-12-19T18:26:15.226095Z","shell.execute_reply.started":"2024-12-19T18:26:15.182248Z","shell.execute_reply":"2024-12-19T18:26:15.224678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_data_dictionary.iloc[76:, :]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.227268Z","iopub.execute_input":"2024-12-19T18:26:15.227744Z","iopub.status.idle":"2024-12-19T18:26:15.244589Z","shell.execute_reply.started":"2024-12-19T18:26:15.227701Z","shell.execute_reply":"2024-12-19T18:26:15.243383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering\ndef feature_engineering(df):\n    \n    # BMI * with Internet Hour Use\n    df['BMI_InternetHourUse'] = df['Physical-BMI'] * (df['PreInt_EduHx-computerinternet_hoursday'])\n    \n    # BMI * with Age\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n\n    # HeartRate / Age\n    df['HeartRate_Age_Ratio'] = df['Physical-HeartRate'] / df['Basic_Demos-Age']\n    \n    # Curl  Up Fitness Zone  / Age\n    df['FGC_CU_Zone_Age_Ratio'] = df['FGC-FGC_CU_Zone'] / df['Basic_Demos-Age']\n\n    # Activity level * age\n    df['BIA_Activity_Level_num_Age'] = df['BIA-BIA_Activity_Level_num'] * df['Basic_Demos-Age']\n\n    # (Maximum stage reached / Age )* with Internet Hour Use\n    df['Fitness_Endurance-Max_Stage_Age_Ratio_InternetHourUse'] = (df['BIA-BIA_Activity_Level_num'] / df['Basic_Demos-Age']) * df['PreInt_EduHx-computerinternet_hoursday']\n    \n    # Regroup Grip Strength total and Grip Strength fitness zone\n    df['FGC-FGC_GS'] = df['FGC-FGC_GSD'] + df['FGC-FGC_GSND']\n    df['FGC-FGC_GS_Zone'] = df['FGC-FGC_GSND_Zone'] + df['FGC-FGC_GSD_Zone']\n    \n    # Regroup Sit & Reach total and Sit & Reach fitness zone\n    df['FGC-FGC_SR'] = df['FGC-FGC_SRR'] + df['FGC-FGC_SRL']\n    df['FGC-FGC_SR_Zone'] = df['FGC-FGC_SRL_Zone'] + df['FGC-FGC_SRR_Zone']\n    \n    # Sleep with Internet Hour Use\n    df['SDS-SDS_Total_T'] * ( df['PreInt_EduHx-computerinternet_hoursday'] + 1 )\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.246116Z","iopub.execute_input":"2024-12-19T18:26:15.246558Z","iopub.status.idle":"2024-12-19T18:26:15.265965Z","shell.execute_reply.started":"2024-12-19T18:26:15.246515Z","shell.execute_reply":"2024-12-19T18:26:15.264593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = feature_engineering(df_train)\ndf_test = feature_engineering(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.267399Z","iopub.execute_input":"2024-12-19T18:26:15.267936Z","iopub.status.idle":"2024-12-19T18:26:15.302427Z","shell.execute_reply.started":"2024-12-19T18:26:15.267894Z","shell.execute_reply":"2024-12-19T18:26:15.301008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.303695Z","iopub.execute_input":"2024-12-19T18:26:15.304135Z","iopub.status.idle":"2024-12-19T18:26:15.427895Z","shell.execute_reply.started":"2024-12-19T18:26:15.304095Z","shell.execute_reply":"2024-12-19T18:26:15.426763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:26:15.428919Z","iopub.execute_input":"2024-12-19T18:26:15.429339Z","iopub.status.idle":"2024-12-19T18:27:42.511043Z","shell.execute_reply.started":"2024-12-19T18:26:15.429300Z","shell.execute_reply":"2024-12-19T18:27:42.509788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.LeakyReLU(0.2)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=1):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    criterion = F.smooth_l1_loss\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:42.512220Z","iopub.execute_input":"2024-12-19T18:27:42.512534Z","iopub.status.idle":"2024-12-19T18:27:42.523251Z","shell.execute_reply.started":"2024-12-19T18:27:42.512507Z","shell.execute_reply":"2024-12-19T18:27:42.522070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train_ts.drop('id', axis=1)\ntest = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\ntrain_ts = train_ts_encoded\ntest_ts = test_ts_encoded\n\ndf_train = pd.merge(df_train, train_ts_encoded, how=\"left\", on='id')\ndf_test = pd.merge(df_test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:42.524574Z","iopub.execute_input":"2024-12-19T18:27:42.525010Z","iopub.status.idle":"2024-12-19T18:27:53.766775Z","shell.execute_reply.started":"2024-12-19T18:27:42.524971Z","shell.execute_reply":"2024-12-19T18:27:53.765702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:53.767830Z","iopub.execute_input":"2024-12-19T18:27:53.768480Z","iopub.status.idle":"2024-12-19T18:27:53.916028Z","shell.execute_reply.started":"2024-12-19T18:27:53.768440Z","shell.execute_reply":"2024-12-19T18:27:53.914564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:53.917195Z","iopub.execute_input":"2024-12-19T18:27:53.917562Z","iopub.status.idle":"2024-12-19T18:27:53.924314Z","shell.execute_reply.started":"2024-12-19T18:27:53.917514Z","shell.execute_reply":"2024-12-19T18:27:53.923334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_columns = ['Basic_Demos-Age',\n       'Basic_Demos-Sex', 'CGAS-CGAS_Score',\n       'Physical-BMI', 'Physical-Height',\n       'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate',\n       'Physical-Systolic_BP',\n       'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n       'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU',\n       'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone',\n       'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone',\n       'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n       'PAQ_C-PAQ_C_Total', 'PreInt_EduHx-Season',\n       'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', \n       'PreInt_EduHx-computerinternet_hoursday', 'sii',\n       'BMI_InternetHourUse', 'BMI_Age', 'HeartRate_Age_Ratio',\n       'FGC_CU_Zone_Age_Ratio', 'BIA_Activity_Level_num_Age',\n       'Fitness_Endurance-Max_Stage_Age_Ratio_InternetHourUse',\n       'FGC-FGC_GS', 'FGC-FGC_GS_Zone', 'FGC-FGC_SR', 'FGC-FGC_SR_Zone']\n\nfeature_columns += time_series_cols\n\n#'Basic_Demos-Enroll_Season', 'PreInt_EduHx-Season',","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:53.925286Z","iopub.execute_input":"2024-12-19T18:27:53.925645Z","iopub.status.idle":"2024-12-19T18:27:53.941619Z","shell.execute_reply.started":"2024-12-19T18:27:53.925610Z","shell.execute_reply":"2024-12-19T18:27:53.940403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_train = df_train.dropna(subset=['sii'])\ndf_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:53.942876Z","iopub.execute_input":"2024-12-19T18:27:53.943214Z","iopub.status.idle":"2024-12-19T18:27:54.094527Z","shell.execute_reply.started":"2024-12-19T18:27:53.943187Z","shell.execute_reply":"2024-12-19T18:27:54.093323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_features = df_train[feature_columns]\nX_train_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.095636Z","iopub.execute_input":"2024-12-19T18:27:54.096101Z","iopub.status.idle":"2024-12-19T18:27:54.204571Z","shell.execute_reply.started":"2024-12-19T18:27:54.096061Z","shell.execute_reply":"2024-12-19T18:27:54.203528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_features.select_dtypes(include=['object']).columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.205797Z","iopub.execute_input":"2024-12-19T18:27:54.206111Z","iopub.status.idle":"2024-12-19T18:27:54.213247Z","shell.execute_reply.started":"2024-12-19T18:27:54.206084Z","shell.execute_reply":"2024-12-19T18:27:54.212185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(df, scaler):\n    # Make a copy of the dataframe\n    df = df.copy()\n    print(df.shape)\n    \n    categorical_cols =  df.select_dtypes(include=['object']).columns # Categorical column name\n    \n    #One hot encoding categorical values and handling missing values by adding NA\n    df_categorical = pd.get_dummies(df[categorical_cols], dummy_na = True)\n\n    # Get the numerical and categorical columns\n    numerical_cols = df.select_dtypes(np.number).columns\n    \n    if 'sii' in numerical_cols:\n        is_training = True\n        numerical_cols = numerical_cols.drop('sii')\n    else:\n        is_training = False\n    \n    # Normalizing the numerical values\n    #df_numerical = df[numerical_cols].apply(lambda x: (x - x.mean()) / x.std() )\n    df_numerical_scaled = scaler.fit_transform(df[numerical_cols])\n    \n    # Fill missing numerical values with the mean 0\n    df_numerical_scaled = pd.DataFrame(df_numerical_scaled).fillna(0)\n    df_numerical_scaled = df_numerical_scaled.set_axis(numerical_cols, axis=1)\n        \n    \n    # Concat numerical and categorical values\n    if is_training:\n        # Get the target back\n        df_target = df['sii']\n\n        # Concatenate the numerical values with categories and the target\n        df = pd.concat([df_numerical_scaled, df_categorical, df_target], axis=1)\n    else:\n        df = pd.concat([df_numerical_scaled, df_categorical], axis=1)\n\n    print(df.shape)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.214319Z","iopub.execute_input":"2024-12-19T18:27:54.214704Z","iopub.status.idle":"2024-12-19T18:27:54.232283Z","shell.execute_reply.started":"2024-12-19T18:27:54.214642Z","shell.execute_reply":"2024-12-19T18:27:54.230881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Process the data\n#df_train_processed = preprocess(df_train)\n#df_train_processed.head()\nscaler = StandardScaler()\n    \nX_train_features_processed = preprocess(X_train_features, scaler)\nX_train_features_processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.233477Z","iopub.execute_input":"2024-12-19T18:27:54.233892Z","iopub.status.idle":"2024-12-19T18:27:54.389916Z","shell.execute_reply.started":"2024-12-19T18:27:54.233850Z","shell.execute_reply":"2024-12-19T18:27:54.388554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get features and target\ndf_train_processed = X_train_features_processed\nX = df_train_processed.iloc[:,:-1]\ny = df_train_processed.iloc[:, -1]\ndf_train_processed.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.391071Z","iopub.execute_input":"2024-12-19T18:27:54.391418Z","iopub.status.idle":"2024-12-19T18:27:54.399949Z","shell.execute_reply.started":"2024-12-19T18:27:54.391364Z","shell.execute_reply":"2024-12-19T18:27:54.398774Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Handle missing labels","metadata":{}},{"cell_type":"code","source":"# Sklearn Label Propagation\nlabel_prop_model = LabelPropagation()\n\ny_prop = y.copy()\ny_prop = y_prop.fillna(-1)\n\nlabel_prop_model.fit(X, y_prop)\ny_prop = label_prop_model.predict(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:27:54.401179Z","iopub.execute_input":"2024-12-19T18:27:54.401712Z","iopub.status.idle":"2024-12-19T18:28:19.943174Z","shell.execute_reply.started":"2024-12-19T18:27:54.401647Z","shell.execute_reply":"2024-12-19T18:28:19.941551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sklearn Label Spreading\nlabel_spread_model = LabelSpreading()\n\ny_spread = y.copy()\ny_spread = y_spread.fillna(-1)\nlabel_spread_model.fit(X, y_spread)\ny_spread = label_spread_model.predict(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:19.944416Z","iopub.execute_input":"2024-12-19T18:28:19.945003Z","iopub.status.idle":"2024-12-19T18:28:21.458261Z","shell.execute_reply.started":"2024-12-19T18:28:19.944951Z","shell.execute_reply":"2024-12-19T18:28:21.456804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sklearn Self Training\nsvc = SVC(probability=True, gamma=\"auto\")\nself_training_model = SelfTrainingClassifier(svc)\n\ny_self_train = y.copy()\ny_self_train = y_self_train.fillna(-1)\nself_training_model.fit(X, y_self_train)\ny_st = self_training_model.predict(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:21.459369Z","iopub.execute_input":"2024-12-19T18:28:21.459853Z","iopub.status.idle":"2024-12-19T18:28:56.013038Z","shell.execute_reply.started":"2024-12-19T18:28:21.459811Z","shell.execute_reply":"2024-12-19T18:28:56.012046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_processed_sii_null = df_train_processed.loc[df_train_processed['sii'].isna()]\n\nX_nsii = df_processed_sii_null.iloc[:, :-1]\ny_nsii = df_processed_sii_null.iloc[:, -1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:56.014189Z","iopub.execute_input":"2024-12-19T18:28:56.014586Z","iopub.status.idle":"2024-12-19T18:28:56.023145Z","shell.execute_reply.started":"2024-12-19T18:28:56.014549Z","shell.execute_reply":"2024-12-19T18:28:56.021917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting rid of the missing labels for the semi-supervised training\ndf_train_processed = df_train_processed.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:56.024323Z","iopub.execute_input":"2024-12-19T18:28:56.024632Z","iopub.status.idle":"2024-12-19T18:28:56.047488Z","shell.execute_reply.started":"2024-12-19T18:28:56.024607Z","shell.execute_reply":"2024-12-19T18:28:56.046209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_del = df_train_processed.iloc[:,:-1]\ny_del = df_train_processed.iloc[:, -1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:56.048789Z","iopub.execute_input":"2024-12-19T18:28:56.049086Z","iopub.status.idle":"2024-12-19T18:28:56.069036Z","shell.execute_reply.started":"2024-12-19T18:28:56.049061Z","shell.execute_reply":"2024-12-19T18:28:56.067820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data\n# Split using label propagation\nX_train_prop, X_val_prop, y_train_prop, y_val_prop = train_test_split(X, y_prop, random_state=42)\n\n# Split using label spreading\nX_train_spread, X_val_spread, y_train_spread, y_val_spread = train_test_split(X, y_spread, random_state=42)\n\n# Split using Self training\nX_train_st, X_val_st, y_train_st, y_val_st = train_test_split(X, y_st, random_state=42)\n\n# Split using the data without null labels\nX_train_del, X_val_del, y_train_del, y_val_del = train_test_split(X_del, y_del, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:56.070369Z","iopub.execute_input":"2024-12-19T18:28:56.070844Z","iopub.status.idle":"2024-12-19T18:28:56.103912Z","shell.execute_reply.started":"2024-12-19T18:28:56.070800Z","shell.execute_reply":"2024-12-19T18:28:56.102963Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models\n\nWe'll try with different algorithms and compare the results.\n\nHere is the list of the algorithms we will try:\n* XGBoost\n* LightGBM\n* CatBoost\n* TabNet\n* TabPFN\n* Random Forest\n\nFor each algorithm, we'll create multiple models using the different label predictions we made.\nSo we'll have 30 models to test.","metadata":{}},{"cell_type":"markdown","source":"**Combinations**\n* X\n  * y_prop\n  * y_spread\n  * y_self_train\n* X_shortened and y_shortened\n\nFor each algorigthm:\n1. Train a model on every combination\n2. Use the model trained on X_shortened and y_shortened to predict y_null_sii given X_processed_sii_null\n3. Concatenate X_shortened with X_null_sii, y_shortened with y_null_sii and train a new model on it","metadata":{}},{"cell_type":"code","source":"# XGBoost\n\ndef train_xgb(X_train, y_train, X_val, y_val, xgb_clf):\n    xgb_clf_m = xgb_clf.fit(X_train, y_train)\n    y_predict = xgb_clf_m.predict(X_val)\n    kappa_prop = cohen_kappa_score(y_val, y_predict, weights='quadratic')\n    return y_predict, kappa_prop, xgb_clf_m\n    \n\nxgb_clf = XGBClassifier()\ny_clf_prop_predict, kappa_prop, xgb_prop = train_xgb(X_train_prop, y_train_prop, X_val_prop, y_val_prop, xgb_clf)\ny_clf_spread_predict, kappa_spread, xgb_spread = train_xgb(X_train_spread, y_train_spread, X_val_spread, y_val_spread, xgb_clf)\ny_clf_st_predict, kappa_st, xgb_st = train_xgb(X_train_st, y_train_st, X_val_st, y_val_st, xgb_clf)\ny_clf_del_predict, kappa_del, xgb_del = train_xgb(X_train_del, y_train_del, X_val_del, y_val_del.values.ravel(), xgb_clf)\n\n\nprint('XGB Prop Kappa : ', kappa_prop)\nprint('XGB spread Kappa : ', kappa_spread)\nprint('XGB st Kappa : ', kappa_st)\nprint('XGB del Kappa : ', kappa_del)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:28:56.105025Z","iopub.execute_input":"2024-12-19T18:28:56.105403Z","iopub.status.idle":"2024-12-19T18:29:01.739643Z","shell.execute_reply.started":"2024-12-19T18:28:56.105363Z","shell.execute_reply":"2024-12-19T18:29:01.738783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use the XGBoost trained on the data without the null labeled to predict the null label\ny_nsii_pred = xgb_del.predict(X_nsii)\n\n# Concat both datasets\nX_smsp_xgb = pd.concat([X_train_del, X_nsii])\n\ny_nsii_pred = pd.DataFrame(y_nsii_pred)\ny_nsii_pred.columns = ['sii']\ny_train_del = pd.DataFrame(y_train_del)\ny_train_del.columns = ['sii']\ny_smsp_xgb = pd.concat([y_train_del, y_nsii_pred])\n\n# Split the data set\nX_train_smsp_xgb, X_val_smsp_xgb, y_train_smsp_xgb, y_val_smsp_xgb = train_test_split(X_smsp_xgb, y_smsp_xgb, random_state=42)\n\n# Train the model on the concatenated data\ny_clf_smsp_predict_xgb, kappa_smsp_xgb, xgb_smsp = train_xgb(X_train_smsp_xgb, y_train_smsp_xgb, X_val_smsp_xgb, y_val_smsp_xgb, xgb_clf)\nprint('XGB smsp Kappa : ', kappa_smsp_xgb)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:29:01.740405Z","iopub.execute_input":"2024-12-19T18:29:01.740697Z","iopub.status.idle":"2024-12-19T18:29:03.100069Z","shell.execute_reply.started":"2024-12-19T18:29:01.740670Z","shell.execute_reply":"2024-12-19T18:29:03.099225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LightGBM\n# specify your configurations as a dict\nlgb_params = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"multiclass\",\n    \"num_class\": 4,\n    \"metric\": \"multi_logloss\",\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"verbose\": -1,\n}\n\nnum_train, num_feature = X_train_del.shape\n\n# generate feature names\nfeature_name = [f\"feature_{col}\" for col in range(num_feature)]\n\nprint(\"Starting training...\") \n# train\n\ndef train_pred_lgb(lgb, X_train, y_train, X_val, y_val, feature_name, lgb_params):\n    # Create dataset for lightgbm\n    lgb_train = lgb.Dataset(X_train, y_train, feature_name=feature_name)\n    lgb_eval = lgb.Dataset(X_val, y_val, reference=lgb_train)\n\n    # Train the model\n    gbm = lgb.train(lgb_params, lgb_train, num_boost_round=20, valid_sets=lgb_eval, callbacks=[lgb.early_stopping(stopping_rounds=5)])\n\n    # Predict\n    y_pred = gbm.predict(X_val, num_iteration=gbm.best_iteration)\n\n    y_pred = np.argmax(y_pred, axis=1)\n    # feature importances\n    #print(f\"Feature importances: {list(gbm.feature_importance())}\")\n    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n\n    return y_pred, kappa, gbm\n    \n\ny_prop_gbm_pred, kappa_prop_gbm, gbm_prop = train_pred_lgb(lgb, X_train_prop, y_train_prop, X_val_prop, y_val_prop, feature_name, lgb_params)\ny_spread_gbm_pred, kappa_spread_gbm, gbm_spread = train_pred_lgb(lgb, X_train_spread, y_train_spread, X_val_spread, y_val_spread, feature_name, lgb_params)\ny_st_gbm_pred, kappa_st_gbm, gbm_st = train_pred_lgb(lgb, X_train_st, y_train_st, X_val_st, y_val_st, feature_name, lgb_params)\ny_del_gbm_pred, kappa_del_gbm, gbm_del = train_pred_lgb(lgb, X_train_del, y_train_del, X_val_del, y_val_del.values.ravel(), feature_name, lgb_params)\n\n\nprint(\"LightGBM Propagation quadratic: \", kappa_prop_gbm)\nprint(\"LightGBM Spreading quadratic: \", kappa_spread_gbm)\nprint(\"LightGBM ST quadratic: \", kappa_st_gbm)\nprint(\"LightGBM Del quadratic: \", kappa_del_gbm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:29:03.100768Z","iopub.execute_input":"2024-12-19T18:29:03.101056Z","iopub.status.idle":"2024-12-19T18:29:04.947567Z","shell.execute_reply.started":"2024-12-19T18:29:03.101031Z","shell.execute_reply":"2024-12-19T18:29:04.946728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#LightGBM Classifier\n\nlgb_clf = lgb.LGBMClassifier(boosting_type= \"gbdt\",\n    objective = \"multiclass\",\n    metric = \"multi_logloss\",\n    num_leaves= 31,\n    learning_rate = 0.05,\n    feature_fraction = 0.9,\n    bagging_fraction = 0.8,\n    bagging_freq = 5,\n    verbose = -1,)\n\ndef train_pred_lgb_clf(lgb_clf, X_train, y_train, X_val, y_val):\n    lgb_clf_m = lgb_clf.fit(X_train, y_train)\n    y_predict = lgb_clf_m.predict(X_val)\n    kappa_prop = cohen_kappa_score(y_val, y_predict, weights='quadratic')\n    return y_predict, kappa_prop, lgb_clf_m\n\ny_prop_gbm_clf_pred, kappa_prop_gbm_clf, gbm_clf_prop = train_pred_lgb_clf(lgb_clf, X_train_prop, y_train_prop, X_val_prop, y_val_prop)\ny_spread_gbm_clf_pred, kappa_spread_gbm_clf, gbm_clf_spread = train_pred_lgb_clf(lgb_clf, X_train_spread, y_train_spread, X_val_spread, y_val_spread)\ny_st_gbm_clf_pred, kappa_st_gbm_clf, gbm_clf_st = train_pred_lgb_clf(lgb_clf, X_train_st, y_train_st, X_val_st, y_val_st)\ny_del_gbm_clf_pred, kappa_del_gbm_clf, gbm_clf_del = train_pred_lgb_clf(lgb_clf, X_train_del, y_train_del, X_val_del, y_val_del.values.ravel())\n\nprint(\"LightGBM Propagation quadratic: \", kappa_prop_gbm_clf)\nprint(\"LightGBM Spreading quadratic: \", kappa_spread_gbm_clf)\nprint(\"LightGBM ST quadratic: \", kappa_st_gbm_clf)\nprint(\"LightGBM Del quadratic: \", kappa_del_gbm_clf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:29:04.948230Z","iopub.execute_input":"2024-12-19T18:29:04.948502Z","iopub.status.idle":"2024-12-19T18:29:13.837334Z","shell.execute_reply.started":"2024-12-19T18:29:04.948477Z","shell.execute_reply":"2024-12-19T18:29:13.836271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Catboost\ncboost_clf = CatBoostClassifier(\n    iterations=50,\n    random_seed=42,\n    learning_rate=0.5,\n    #custom_loss=['AUC', 'Accuracy']\n)\n\ndef train_pred_catboost(cboost_clf, X_train, y_train, X_val, y_val):\n    cat_model = cboost_clf.fit(X_train, y_train, eval_set=(X_val, y_val), verbose=False, plot=False)\n    y_pred = cat_model.predict(X_val)\n    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n   \n    return y_pred, kappa, cat_model\n\n#y_prop_cboost_pred, kappa_prop_cboost, cboost_prop = train_pred_catboost(cboost_clf, X_train_prop, y_train_prop, X_val_prop, y_val_prop)\n#y_spread_cboost_pred, kappa_spread_cboost, cboost_spread = train_pred_catboost(cboost_clf, X_train_spread, y_train_spread, X_val_spread, y_val_spread)\ny_st_cboost_pred, kappa_st_cboost, cboost_st = train_pred_catboost(cboost_clf, X_train_st, y_train_st, X_val_st, y_val_st)\ny_del_cboost_pred, kappa_del_cboost, cboost_del = train_pred_catboost(cboost_clf, X_train_del, y_train_del.values.ravel(), X_val_del, y_val_del)\n\n#print('Catboost Prop Kappa : ', kappa_prop_cboost)\n#print('Catboost Spread Kappa : ', kappa_spread_cboost)\nprint('Catboost ST Kappa : ', kappa_st_cboost)\nprint('Catboost Del Kappa : ', kappa_del_cboost)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:29:13.838347Z","iopub.execute_input":"2024-12-19T18:29:13.838637Z","iopub.status.idle":"2024-12-19T18:29:17.847236Z","shell.execute_reply.started":"2024-12-19T18:29:13.838612Z","shell.execute_reply":"2024-12-19T18:29:17.845947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forest\ndef train_pred_rf(rnd_clf, X_train, y_train, X_val, y_val):\n    model = rnd_clf.fit(X_train, y_train)\n    y_pred = model.predict(X_val)\n    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    return y_pred, kappa, model\n\n\nrnd_clf = RandomForestClassifier()\n\n#y_prop_rf, kappa_prop_rf, rf_prop = train_pred_rf(rnd_clf, X_train_prop, y_train_prop, X_val_prop, y_val_prop)\n#y_spread_rf, kappa_spread_rf, rf_spread = train_pred_rf(rnd_clf, X_train_spread, y_train_spread, X_val_spread, y_val_spread)\ny_st_rf, kappa_st_rf, rf_st = train_pred_rf(rnd_clf, X_train_st, y_train_st, X_val_st, y_val_st)\ny_del_rf, kappa_del_rf, rf_del = train_pred_rf(rnd_clf, X_train_del, y_train_del.values.ravel(), X_val_del, y_val_del)\n\n#print('RF Prop Kappa : ', kappa_prop_rf)\n#print('RF Spread Kappa : ', kappa_spread_rf)\nprint('RF ST Kappa : ', kappa_st_rf)\nprint('RF Del Kappa : ', kappa_del_rf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:29:17.848603Z","iopub.execute_input":"2024-12-19T18:29:17.849071Z","iopub.status.idle":"2024-12-19T18:29:20.157552Z","shell.execute_reply.started":"2024-12-19T18:29:17.849027Z","shell.execute_reply":"2024-12-19T18:29:20.156234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier, StackingClassifier\nfrom sklearn.base import clone\n\ndef train_pred_ensemble(clf, X_train, y_train, X_val, y_val):\n    model = clf.fit(X_train, y_train)\n    y_pred = model.predict(X_val)\n    kappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    return y_pred, kappa, model\n    \n#clf = clf.fit(X_train, y_train).score(X_test, y_test)\n\nxgb_clf_ens = XGBClassifier()\n\n\ncboost_clf_ens = CatBoostClassifier(\n    iterations=50,\n    random_seed=42,\n    learning_rate=0.5,\n    #custom_loss=['AUC', 'Accuracy']\n)\n\nlgb_clf_ens = lgb.LGBMClassifier(boosting_type= \"gbdt\",\n    objective = \"multiclass\",\n    metric = \"multi_logloss\",\n    num_leaves= 31,\n    learning_rate = 0.05,\n    feature_fraction = 0.9,\n    bagging_fraction = 0.8,\n    bagging_freq = 5,\n    verbose = -1,)\n\nrnd_clf_ens = RandomForestClassifier()\n\nestimators = [\n    ('lightgbm', lgb_clf_ens),\n    ('xgboost', xgb_clf_ens),\n    ('catboost', cboost_clf_ens),\n    ('rf', rnd_clf_ens),\n]\n\n\n# VotingRegressor\nvoting_model = VotingClassifier(\n    estimators=estimators,\n    weights=[4.0, 4.0, 5.0, 4.0] \n)\n\n# StackingRegressor\nstacking_model = StackingClassifier(\n    estimators=estimators\n)\n\nensemble_model = VotingClassifier(\n    estimators=[\n        ('stacking', stacking_model)\n    ]\n)\n\n#y_ens_prop, kappa_ens_prop, ens_prop = train_pred_ensemble(ensemble_model, X_train_prop, y_train_prop, X_val_prop, y_val_prop)\n#y_ens_spread, kappa_ens_spread, ens_spread = train_pred_ensemble(ensemble_model, X_train_spread, y_train_spread, X_val_spread, y_val_spread)\n#y_ens_st, kappa_ens_st, ens_st = train_pred_ensemble(ensemble_model, X_train_st, y_train_st, X_val_st, y_val_st)\ny_ens_del, kappa_ens_del, ens_del = train_pred_ensemble(ensemble_model, X_train_del, y_train_del, X_val_del, y_val_del)\n\n#print('RF Prop Kappa : ', kappa_ens_prop)\n#print('RF Spread Kappa : ', kappa_ens_spread)\n#print('RF ST Kappa : ', kappa_ens_st)\nprint('RF Del Kappa : ', kappa_ens_del)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:49:16.677105Z","iopub.execute_input":"2024-12-19T18:49:16.677694Z","iopub.status.idle":"2024-12-19T18:49:56.580415Z","shell.execute_reply.started":"2024-12-19T18:49:16.677621Z","shell.execute_reply":"2024-12-19T18:49:56.578887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use the XGBoost trained on the data without the null labeled to predict the null label\ny_nsii_ens_pred = ens_del.predict(X_nsii)\n\n# Concat both datasets\nX_smsp_ens = pd.concat([X_train_del, X_nsii])\n\ny_nsii_ens_pred = pd.DataFrame(y_nsii_ens_pred)\ny_nsii_pred.columns = ['sii']\n\ny_smsp_ens = pd.concat([y_train_del, y_nsii_pred])\n\n# Split the data set\nX_train_smsp_ens, X_val_smsp_ens, y_train_smsp_ens, y_val_smsp_ens = train_test_split(X_smsp_ens, y_smsp_ens, random_state=42)\n\n# Train the model on the concatenated data\ny_clf_smsp_predict_ens, kappa_smsp_ens, ens_smsp = train_pred_ensemble(ensemble_model, X_train_smsp_ens, y_train_smsp_ens, X_val_smsp_ens, y_val_smsp_ens)\nprint('XGB smsp Kappa : ', kappa_smsp_ens)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:50:01.877698Z","iopub.execute_input":"2024-12-19T18:50:01.878151Z","iopub.status.idle":"2024-12-19T18:50:37.998222Z","shell.execute_reply.started":"2024-12-19T18:50:01.878114Z","shell.execute_reply":"2024-12-19T18:50:37.996991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:50:41.603109Z","iopub.execute_input":"2024-12-19T18:50:41.603485Z","iopub.status.idle":"2024-12-19T18:50:41.698070Z","shell.execute_reply.started":"2024-12-19T18:50:41.603456Z","shell.execute_reply":"2024-12-19T18:50:41.696951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the features\nfeature_columns.remove(\"sii\")\nX_test_features = df_test[feature_columns]\n\n# Process the data\ndf_test_processed = preprocess(X_test_features, scaler)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:50:44.355543Z","iopub.execute_input":"2024-12-19T18:50:44.355976Z","iopub.status.idle":"2024-12-19T18:50:44.382605Z","shell.execute_reply.started":"2024-12-19T18:50:44.355944Z","shell.execute_reply":"2024-12-19T18:50:44.381111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make the predictions using the models\n\n#y_test_st_ens_pred = ens_st.predict(df_test_processed)\n#y_test_st_xgb_pred = xgb_st.predict(df_test_processed)\n\n#y_test_st_xgb_pred = xgb_st.predict(df_test_processed)\n#y_test_st_cboost_pred = cboost_st.predict(df_test_processed)\n\n#y_test_prop_xgb_pred = xgb_prop.predict(df_test_processed)\n#y_test_prop_cboost_pred = cboost_prop.predict(df_test_processed)\n\ny_test_smsp_ens_pred = ens_smsp.predict(df_test_processed)\n#y_test_smsp_xgb_pred = xgb_smsp.predict(df_test_processed)\n#y_test_del_xgb_pred = xgb_del.predict(X_test_features_processed)\n#y_test_del_xgb_pred_rounded = round_predictions(y_test_del_xgb_pred, thresholds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:52:40.388526Z","iopub.execute_input":"2024-12-19T18:52:40.389028Z","iopub.status.idle":"2024-12-19T18:52:40.435632Z","shell.execute_reply.started":"2024-12-19T18:52:40.388987Z","shell.execute_reply":"2024-12-19T18:52:40.434688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Make a dataframe with the ids and the prediction\n\n#prediction_submission_st_ens = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_st_ens_pred})\n#prediction_submission_st_xgb = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_st_xgb_pred})\n#prediction_submission_st_cboost = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': np.squeeze(y_test_st_cboost_pred)})\n\n#prediction_submission_prop_xgb = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_prop_xgb_pred})\n#prediction_submission_prop_cboost = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': np.squeeze(y_test_prop_cboost_pred)})\n\nprediction_submission_smsp_ens = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_smsp_ens_pred})\n#prediction_submission_smsp_xgb = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_smsp_xgb_pred})\n#prediction_submission_del_xgb = pd.DataFrame({'id' : df_test.loc[:, 'id'], 'sii': y_test_del_xgb_pred_rounded})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:52:41.968969Z","iopub.execute_input":"2024-12-19T18:52:41.969321Z","iopub.status.idle":"2024-12-19T18:52:41.974911Z","shell.execute_reply.started":"2024-12-19T18:52:41.969294Z","shell.execute_reply":"2024-12-19T18:52:41.973896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize the prediction\nprediction_submission_smsp_ens['sii'] = prediction_submission_smsp_ens['sii'].astype(int)\nprediction_submission_smsp_ens","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:52:45.276333Z","iopub.execute_input":"2024-12-19T18:52:45.276936Z","iopub.status.idle":"2024-12-19T18:52:45.287873Z","shell.execute_reply.started":"2024-12-19T18:52:45.276878Z","shell.execute_reply":"2024-12-19T18:52:45.286635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the prediction in a csv to submit\n#prediction_submission_st_xgb.to_csv('submission.csv', index=False)\n#prediction_submission_st_cboost.to_csv('submission_st_cboost.csv', index=False)\n\n#prediction_submission_prop_xgb.to_csv('submission_prop_xgb.csv', index=False)\n#prediction_submission_prop_cboost.to_csv('submission_prop_cboost.csv', index=False)\n\nprediction_submission_smsp_ens.to_csv('submission.csv', index=False)\n\n#prediction_submission_smsp_xgb.to_csv('submission.csv', index=False)\n#prediction_submission_del_xgb.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:53:19.019867Z","iopub.execute_input":"2024-12-19T18:53:19.020229Z","iopub.status.idle":"2024-12-19T18:53:19.029157Z","shell.execute_reply.started":"2024-12-19T18:53:19.020201Z","shell.execute_reply":"2024-12-19T18:53:19.028009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}