{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"text-align: left; font-family:Comic Sans MS ; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 70px;border-style: solid;border-color: dark green; font-weight: bold;\">Problematic Internet Use</div>\n\n\n# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">1. Importing libraries</p>","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()\nplt.rcParams['figure.facecolor'] = '#FFEDD8'\n\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score, make_scorer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.base import clone\n\nfrom scipy.optimize import minimize\n\nimport lightgbm as lgb\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras.layers import Dense, Dropout, BatchNormalization\n\nimport optuna","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-13T08:52:49.831956Z","iopub.execute_input":"2024-10-13T08:52:49.832255Z","iopub.status.idle":"2024-10-13T08:53:07.164761Z","shell.execute_reply.started":"2024-10-13T08:52:49.832222Z","shell.execute_reply":"2024-10-13T08:53:07.163607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">2. Loading the data</p>","metadata":{}},{"cell_type":"code","source":"train_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntest_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:53:07.166779Z","iopub.execute_input":"2024-10-13T08:53:07.167657Z","iopub.status.idle":"2024-10-13T08:53:07.172034Z","shell.execute_reply.started":"2024-10-13T08:53:07.167611Z","shell.execute_reply":"2024-10-13T08:53:07.170961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv', index_col = 'id')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:53:07.173198Z","iopub.execute_input":"2024-10-13T08:53:07.173556Z","iopub.status.idle":"2024-10-13T08:53:07.294026Z","shell.execute_reply.started":"2024-10-13T08:53:07.173513Z","shell.execute_reply":"2024-10-13T08:53:07.293083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train dataset has {train_df.shape[0]} samples and {train_df.shape[1]} features\")","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:53:20.428325Z","iopub.execute_input":"2024-10-13T08:53:20.429029Z","iopub.status.idle":"2024-10-13T08:53:20.434374Z","shell.execute_reply.started":"2024-10-13T08:53:20.428989Z","shell.execute_reply":"2024-10-13T08:53:20.433228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv', index_col = 'id')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:33.527972Z","iopub.execute_input":"2024-10-13T08:57:33.528373Z","iopub.status.idle":"2024-10-13T08:57:33.559664Z","shell.execute_reply.started":"2024-10-13T08:57:33.528337Z","shell.execute_reply":"2024-10-13T08:57:33.558753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Test dataset has {test_df.shape[0]} samples and {test_df.shape[1]} features\")","metadata":{"execution":{"iopub.status.busy":"2024-10-13T04:36:26.083904Z","iopub.execute_input":"2024-10-13T04:36:26.084260Z","iopub.status.idle":"2024-10-13T04:36:26.089318Z","shell.execute_reply.started":"2024-10-13T04:36:26.084229Z","shell.execute_reply":"2024-10-13T04:36:26.088188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndata_dict.tail()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:36.771369Z","iopub.execute_input":"2024-10-13T08:57:36.772227Z","iopub.status.idle":"2024-10-13T08:57:36.789144Z","shell.execute_reply.started":"2024-10-13T08:57:36.772189Z","shell.execute_reply":"2024-10-13T08:57:36.788292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:40.396770Z","iopub.execute_input":"2024-10-13T08:57:40.397531Z","iopub.status.idle":"2024-10-13T08:57:40.424736Z","shell.execute_reply.started":"2024-10-13T08:57:40.397491Z","shell.execute_reply":"2024-10-13T08:57:40.423831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:49.791880Z","iopub.execute_input":"2024-10-13T08:57:49.792256Z","iopub.status.idle":"2024-10-13T08:57:49.804678Z","shell.execute_reply.started":"2024-10-13T08:57:49.792219Z","shell.execute_reply":"2024-10-13T08:57:49.803686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total number of missing training values: \", train_df.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2024-10-13T02:02:01.666809Z","iopub.execute_input":"2024-10-13T02:02:01.667182Z","iopub.status.idle":"2024-10-13T02:02:01.677335Z","shell.execute_reply.started":"2024-10-13T02:02:01.667145Z","shell.execute_reply":"2024-10-13T02:02:01.676419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n## <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 40px;border-style: solid;border-color: dark green; font-weight: bold;\">2.1. Feature description</p>\n","metadata":{}},{"cell_type":"code","source":"data_dict['Instrument'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:52.665438Z","iopub.execute_input":"2024-10-13T08:57:52.666210Z","iopub.status.idle":"2024-10-13T08:57:52.672717Z","shell.execute_reply.started":"2024-10-13T08:57:52.666170Z","shell.execute_reply":"2024-10-13T08:57:52.671818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"|    **Intrument**      |    **Description**          |\n|------------------------|----------------------------------------|\n|    **Demographics**     |     Some basic demographic data, like age and sex of participants    |\n|   **Children's Global Assessment Scale (CGAS)**      |   Numeric scale used by mental health clinicians to assess he overall psychological, social, and functioning level of youths under the age of 18.    |\n|    **Physical Measures**     |   Collection of BMI, height, weight, waist, blood pressure (diastolic and systolic)), hearth rate and hip measurements.    |\n|    **FitnessGram Vitals and Treadmill**      |       Measurements of cardiovascular fitness assessed using the NHANES treadmill protocol. |\n|    **FitnessGram Child**      |      A youth-oriented fitness assessment program designed to evaluate and improve the physical health of children and adolescents, by measuring five different parameters including aerobic capacity, muscular strength, muscular endurance, flexibility, and body composition.   |\n|    **Bio-electric Impedance Analysis (BIA)**   |     Measure of key body composition elements, including BMI, fat, muscle, and water content.    |\n|    **Physical Activity Questionnaire (PAQ)**     |    Information about participation in vigorous activities over the last 7 days.     |\n|    **Parent-Child Internet Addiction Test (PCIAT)**     |     A standardized questionnaire designed to help assess the potential for internet addiction in children, as reported by their parents or guardians.   |\n|    **Sleep Disturbance Scale (SDS)**     |   Scale to categorize sleep disorders in children.    |\n|     **Internet Use**    |    Daily internet or computer usage time.   |\n|    **Season**      |   for each set of measurements there is a 'season' feature which gives the season of the year when the measurements were carried out. These are the only predictive categorical features in the dataset and can be easily preprocessed.      |\n\n- Target features: **Severity Impairment Index**(sii)","metadata":{}},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">3. Exploratory Data Analysis</p>\n\n## <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 40px;border-style: solid;border-color: dark green; font-weight: bold;\">3.1. Numerical features</p>\n","metadata":{}},{"cell_type":"code","source":"numerical_columns = [col for col in train_df.columns if train_df[col].dtype != 'O']","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:56.479200Z","iopub.execute_input":"2024-10-13T08:57:56.480093Z","iopub.status.idle":"2024-10-13T08:57:56.485066Z","shell.execute_reply.started":"2024-10-13T08:57:56.480051Z","shell.execute_reply":"2024-10-13T08:57:56.484134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=len(numerical_columns)//3 + 1, ncols=3, figsize=(30, 90))\naxes = axes.flatten()\n\nfor i, col in enumerate(numerical_columns):\n    axes[i].hist(train_df[col], bins=50, color = '#387ADF', edgecolor = 'black', linewidth = 0.5)\n    axes[i].set_title(col, fontsize=25, fontweight = 'bold')\n    axes[i].set_xlabel('Value', fontsize=10, fontweight = 'bold')\n    axes[i].set_ylabel('Frequency', fontsize=10, fontweight = 'bold')\n    \nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:57:58.013849Z","iopub.execute_input":"2024-10-13T08:57:58.014217Z","iopub.status.idle":"2024-10-13T08:58:23.731149Z","shell.execute_reply.started":"2024-10-13T08:57:58.014184Z","shell.execute_reply":"2024-10-13T08:58:23.730128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_palette = ['#0D92F4', '#FF8911'] ","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:58:44.095447Z","iopub.execute_input":"2024-10-13T08:58:44.095863Z","iopub.status.idle":"2024-10-13T08:58:44.100305Z","shell.execute_reply.started":"2024-10-13T08:58:44.095818Z","shell.execute_reply":"2024-10-13T08:58:44.099446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **3.1 PCIAT Features**\n\n","metadata":{}},{"cell_type":"code","source":"PCIAT_cols = [val for val in train_df.columns[train_df.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:58:46.460510Z","iopub.execute_input":"2024-10-13T08:58:46.460916Z","iopub.status.idle":"2024-10-13T08:58:46.467593Z","shell.execute_reply.started":"2024-10-13T08:58:46.460877Z","shell.execute_reply":"2024-10-13T08:58:46.466596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nscatter = plt.scatter( train_df['PCIAT-PCIAT_Total'], train_df['sii'], c=train_df['sii'], cmap='viridis')\n\nplt.title('PCIAT Total and SII', fontweight = 'bold', fontsize=15)\nplt.xlabel('PCIAT-PCIAT_Total')\nplt.ylabel('SII')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T08:59:10.622499Z","iopub.execute_input":"2024-10-13T08:59:10.623257Z","iopub.status.idle":"2024-10-13T08:59:11.016949Z","shell.execute_reply.started":"2024-10-13T08:59:10.623214Z","shell.execute_reply":"2024-10-13T08:59:11.016010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that `SII` is just devired from the `Total PCIAT`:\n- `0-30` gives `sii = 0`.\n- `31-49` gives `sii = 1`.\n- `50-79` gives `sii = 2`.\n- `80-100` gives `sii = 3`.\n\nThat's why we don't see any `PCIAT` features in the test dataset. Also, the task can be convert from a classification task to regression task by using the `PCIAT` as the target features.\n\nWe will also remove all other `PCIAT` columns, just keep the  `PCIAT total`.","metadata":{}},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total')\ntrain_df = train_df.drop(columns = PCIAT_cols)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T10:34:15.550071Z","iopub.execute_input":"2024-10-13T10:34:15.550396Z","iopub.status.idle":"2024-10-13T10:34:15.578897Z","shell.execute_reply.started":"2024-10-13T10:34:15.550360Z","shell.execute_reply":"2024-10-13T10:34:15.577958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **3.2 Severity Impairment Index**\n\n#### **Count of sii**","metadata":{}},{"cell_type":"code","source":"sns.countplot(train_df, x = 'sii').set_title('Count of sii', fontweight='bold', fontsize=15);","metadata":{"execution":{"iopub.status.busy":"2024-10-13T09:10:11.961937Z","iopub.execute_input":"2024-10-13T09:10:11.962331Z","iopub.status.idle":"2024-10-13T09:10:12.296292Z","shell.execute_reply.started":"2024-10-13T09:10:11.962294Z","shell.execute_reply":"2024-10-13T09:10:12.295412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **SII and Internet use**\n","metadata":{}},{"cell_type":"code","source":"\ncontingency_table = train_df.pivot_table(\n    index='PreInt_EduHx-computerinternet_hoursday', \n    columns='sii', \n    aggfunc='size', \n    fill_value=0\n)\n\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(contingency_table, annot=True, fmt=\"d\", cmap=\"YlGnBu\", cbar_kws={'label': 'Count'})\nplt.title('Distribution of Internet Use Hours by SII Levels')\nplt.xlabel('SII Levels')\nplt.ylabel('Internet Use Hours per Day')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T09:26:26.234866Z","iopub.execute_input":"2024-10-13T09:26:26.235544Z","iopub.status.idle":"2024-10-13T09:26:26.628630Z","shell.execute_reply.started":"2024-10-13T09:26:26.235503Z","shell.execute_reply":"2024-10-13T09:26:26.627712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From this plot, we can have some observations:\n- With the `low SII levels`, most participants spend less hours in the Internet. This indicates that the mininal Internet usage may be associated with low SII levels\n- When the `SII levels` increase, the percentage of participants with high level of internet usage also increase.","metadata":{}},{"cell_type":"markdown","source":"#### **SII and Age**","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=train_df, y=train_df['Basic_Demos-Age'], x=train_df['sii'], hue=train_df['Basic_Demos-Sex'],  palette = custom_palette)\nplt.title('Age vs Severity Impairment Index by gender', fontsize = 15, fontweight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T09:49:08.329271Z","iopub.execute_input":"2024-10-13T09:49:08.329669Z","iopub.status.idle":"2024-10-13T09:49:08.875293Z","shell.execute_reply.started":"2024-10-13T09:49:08.329627Z","shell.execute_reply":"2024-10-13T09:49:08.874392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From this plot, we can see that the people who has higher impairment index are generally older.\n\nFor lower SII scores, both groups of gender tend to have a relatively similar age range. The median age for higher SII scores may differ between genders, which could indicate slight gender-based differences in how SII relates to age.","metadata":{}},{"cell_type":"markdown","source":"### **3.3 Age and Internet use**","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def search_data_dict(column_name):\n    return data_dict[data_dict['Field'] == column_name]","metadata":{"execution":{"iopub.status.busy":"2024-10-13T05:51:38.809842Z","iopub.execute_input":"2024-10-13T05:51:38.810769Z","iopub.status.idle":"2024-10-13T05:51:38.817585Z","shell.execute_reply.started":"2024-10-13T05:51:38.810711Z","shell.execute_reply":"2024-10-13T05:51:38.816618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_data_dict('PreInt_EduHx-computerinternet_hoursday')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T05:51:44.473078Z","iopub.execute_input":"2024-10-13T05:51:44.473906Z","iopub.status.idle":"2024-10-13T05:51:44.486287Z","shell.execute_reply.started":"2024-10-13T05:51:44.473866Z","shell.execute_reply":"2024-10-13T05:51:44.485328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_data_dict('Basic_Demos-Sex')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T05:51:48.135473Z","iopub.execute_input":"2024-10-13T05:51:48.136307Z","iopub.status.idle":"2024-10-13T05:51:48.147637Z","shell.execute_reply.started":"2024-10-13T05:51:48.136267Z","shell.execute_reply":"2024-10-13T05:51:48.146757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(data=train_df, y=train_df['Basic_Demos-Age'], x=train_df['PreInt_EduHx-computerinternet_hoursday'], hue=train_df['Basic_Demos-Sex'], palette = custom_palette)\nplt.title(\"Age vs Hours on the Internet per day by gender\", fontweight = 'bold', fontsize = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:39:39.239541Z","iopub.execute_input":"2024-10-10T13:39:39.239924Z","iopub.status.idle":"2024-10-10T13:39:39.742467Z","shell.execute_reply.started":"2024-10-10T13:39:39.239889Z","shell.execute_reply":"2024-10-10T13:39:39.741566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This plot indicates that the amount of time we spend for the internet a day tends to increase when we get older, showed by the higher age range of users with high internet usage frequency.","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=train_df, y=train_df['Basic_Demos-Age'], x=train_df['PreInt_EduHx-computerinternet_hoursday'], hue=train_df['sii'])\nplt.title(\"Age vs Hours on the Internet per day by SII\", fontweight = 'bold', fontsize = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:11:08.588876Z","iopub.execute_input":"2024-10-10T14:11:08.589756Z","iopub.status.idle":"2024-10-10T14:11:09.321939Z","shell.execute_reply.started":"2024-10-10T14:11:08.589718Z","shell.execute_reply":"2024-10-10T14:11:09.320929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Within each level of internet usage per day, we can see that the person with higher SII also tends to have higher age.","metadata":{}},{"cell_type":"markdown","source":"### **3.3 Correlation**","metadata":{}},{"cell_type":"code","source":"numerical_df = train_df.select_dtypes(include=['number'])\ncorrelation_matrix = numerical_df.corr()\ncorr = correlation_matrix['PCIAT-PCIAT_Total'].sort_values(ascending=False)\ncorr_df = pd.DataFrame(corr)\ncorr_df","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:31.542320Z","iopub.execute_input":"2024-10-13T13:17:31.543171Z","iopub.status.idle":"2024-10-13T13:17:31.583024Z","shell.execute_reply.started":"2024-10-13T13:17:31.543131Z","shell.execute_reply":"2024-10-13T13:17:31.582129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will filter some feature, which is significantly correlated with `PCIAT Total`","metadata":{}},{"cell_type":"code","source":"selection = corr_df[(corr_df['PCIAT-PCIAT_Total']>.1) | (corr_df['PCIAT-PCIAT_Total']<-.1)]\nselection = [val for val in selection.index]\nselection.remove('PCIAT-PCIAT_Total')\nselection.remove('sii')\nselection.remove('Physical-BMI')\nselection.remove('SDS-SDS_Total_Raw')\nselection","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:33.838036Z","iopub.execute_input":"2024-10-13T13:17:33.838836Z","iopub.status.idle":"2024-10-13T13:17:33.847714Z","shell.execute_reply.started":"2024-10-13T13:17:33.838775Z","shell.execute_reply":"2024-10-13T13:17:33.846715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 40px;border-style: solid;border-color: dark green; font-weight: bold;\">3.2. Categorical features</p>","metadata":{}},{"cell_type":"code","source":"categorical_columns = [col for col in train_df.columns if col not in numerical_columns]","metadata":{"execution":{"iopub.status.busy":"2024-10-13T11:43:11.289823Z","iopub.execute_input":"2024-10-13T11:43:11.290699Z","iopub.status.idle":"2024-10-13T11:43:11.295183Z","shell.execute_reply.started":"2024-10-13T11:43:11.290657Z","shell.execute_reply":"2024-10-13T11:43:11.294123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns","metadata":{"execution":{"iopub.status.busy":"2024-10-13T11:43:15.987201Z","iopub.execute_input":"2024-10-13T11:43:15.988046Z","iopub.status.idle":"2024-10-13T11:43:15.994056Z","shell.execute_reply.started":"2024-10-13T11:43:15.988006Z","shell.execute_reply":"2024-10-13T11:43:15.993015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All of the categorical features is represented for the season","metadata":{}},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">4. Preprocessing</p>","metadata":{}},{"cell_type":"markdown","source":"### **4.1. Missing values**\n\nThere is a large number of missing values remaining in the dataset, let's check them.","metadata":{}},{"cell_type":"code","source":"train_df.shape[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:29:42.636367Z","iopub.execute_input":"2024-10-13T12:29:42.636771Z","iopub.status.idle":"2024-10-13T12:29:42.643058Z","shell.execute_reply.started":"2024-10-13T12:29:42.636733Z","shell.execute_reply":"2024-10-13T12:29:42.642190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum() / train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:35:11.835029Z","iopub.execute_input":"2024-10-13T12:35:11.835770Z","iopub.status.idle":"2024-10-13T12:35:11.848750Z","shell.execute_reply.started":"2024-10-13T12:35:11.835733Z","shell.execute_reply":"2024-10-13T12:35:11.847817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Nearly all of the features contain missing values, even the the target features like `sii` or `PCIAT-PCIAT_Total`, with 11 features have the missing ratio over 50%, we can drop them.","metadata":{}},{"cell_type":"code","source":"half_missing = [val for val in train_df.columns[train_df.isnull().sum()>len(train_df)/2]]\nhalf_missing","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:43.717271Z","iopub.execute_input":"2024-10-13T13:17:43.717642Z","iopub.status.idle":"2024-10-13T13:17:43.729726Z","shell.execute_reply.started":"2024-10-13T13:17:43.717605Z","shell.execute_reply":"2024-10-13T13:17:43.728846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selection = [i for i in selection if i not in half_missing]\nselection","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:45.811890Z","iopub.execute_input":"2024-10-13T13:17:45.812851Z","iopub.status.idle":"2024-10-13T13:17:45.819712Z","shell.execute_reply.started":"2024-10-13T13:17:45.812777Z","shell.execute_reply":"2024-10-13T13:17:45.818819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can easily fill the missing value in the categorical columns and encode it:","metadata":{}},{"cell_type":"code","source":"for season in categorical_columns:\n    train_df[season] = train_df[season].fillna(0)\n    train_df[season] = train_df[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})\n    \n    test_df[season] = test_df[season].fillna(0)\n    test_df[season] = test_df[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:50.735048Z","iopub.execute_input":"2024-10-13T13:17:50.735940Z","iopub.status.idle":"2024-10-13T13:17:50.797019Z","shell.execute_reply.started":"2024-10-13T13:17:50.735899Z","shell.execute_reply":"2024-10-13T13:17:50.795669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selection = selection + categorical_columns\nselection","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:17:48.061962Z","iopub.execute_input":"2024-10-13T13:17:48.062848Z","iopub.status.idle":"2024-10-13T13:17:48.069293Z","shell.execute_reply.started":"2024-10-13T13:17:48.062779Z","shell.execute_reply":"2024-10-13T13:17:48.068211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dropna(inplace = True, subset = 'sii')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:27:23.566166Z","iopub.execute_input":"2024-10-13T13:27:23.567055Z","iopub.status.idle":"2024-10-13T13:27:23.574703Z","shell.execute_reply.started":"2024-10-13T13:27:23.567012Z","shell.execute_reply":"2024-10-13T13:27:23.573709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **4.1 Series test data**","metadata":{}},{"cell_type":"code","source":"def process_file(fname, drname):\n    df = pd.read_parquet(os.path.join(drname, fname, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), fname.split('=')[1]\n\ndef load_data(drname):\n    ids = os.listdir(drname)\n    \n    with ThreadPoolExecutor() as executor:\n        result = list(tqdm(executor.map(lambda fname: process_file(fname, drname), ids), total=len(ids)))\n        \n    stats, indices = zip(*result)\n    \n    df = pd.DataFrame(stats, columns=[f'stats_{i}' for i in range(len(stats[0]))])\n    df['id'] = indices\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:10:44.600999Z","iopub.execute_input":"2024-10-13T12:10:44.601376Z","iopub.status.idle":"2024-10-13T12:10:44.611526Z","shell.execute_reply.started":"2024-10-13T12:10:44.601341Z","shell.execute_reply":"2024-10-13T12:10:44.610648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = load_data(train_dir)\ntest_ds = load_data(test_dir)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:10:46.930523Z","iopub.execute_input":"2024-10-13T12:10:46.930939Z","iopub.status.idle":"2024-10-13T12:12:06.989253Z","shell.execute_reply.started":"2024-10-13T12:10:46.930902Z","shell.execute_reply":"2024-10-13T12:12:06.988175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:46:45.126534Z","iopub.execute_input":"2024-10-13T12:46:45.126925Z","iopub.status.idle":"2024-10-13T12:46:45.154944Z","shell.execute_reply.started":"2024-10-13T12:46:45.126890Z","shell.execute_reply":"2024-10-13T12:46:45.154047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:17:51.763561Z","iopub.execute_input":"2024-10-13T12:17:51.763971Z","iopub.status.idle":"2024-10-13T12:17:51.796014Z","shell.execute_reply.started":"2024-10-13T12:17:51.763932Z","shell.execute_reply":"2024-10-13T12:17:51.795087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged = pd.merge(train_df, train_ds, how='left', on='id')\n# test_merged = pd.merge(test_df, test_ds, how='left', on='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:47:43.490831Z","iopub.execute_input":"2024-10-10T14:47:43.491203Z","iopub.status.idle":"2024-10-10T14:47:43.509305Z","shell.execute_reply.started":"2024-10-10T14:47:43.491169Z","shell.execute_reply":"2024-10-10T14:47:43.508468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_cols = [col for col in train_merged.columns if not 'PCIAT-' in col]\n# train_merged = train_merged[train_cols]\n# train_cols.remove('sii')\n# test_merged = test_merged[train_cols]\n\n# print(train_merged.shape, test_merged.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:49:46.497850Z","iopub.execute_input":"2024-10-10T14:49:46.498316Z","iopub.status.idle":"2024-10-10T14:49:46.508450Z","shell.execute_reply.started":"2024-10-10T14:49:46.498258Z","shell.execute_reply":"2024-10-10T14:49:46.507412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged.drop(columns='id', inplace=True)\n# test_merged.drop(columns='id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:50:05.806372Z","iopub.execute_input":"2024-10-10T14:50:05.807085Z","iopub.status.idle":"2024-10-10T14:50:05.814938Z","shell.execute_reply.started":"2024-10-10T14:50:05.807048Z","shell.execute_reply":"2024-10-10T14:50:05.813897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged_num_columns = [col for col in train_merged.columns if train_merged[col].dtype != 'O']\n# train_merged_cat_columns = [col for col in train_merged.columns if col not in train_merged_num_columns]\n# train_merged_corr = train_merged[train_merged_num_columns].corr()['sii']\n# train_merged_corr_2 = train_merged_corr[train_merged_corr > 0.2].sort_values(ascending=False).to_dict()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:50:07.894407Z","iopub.execute_input":"2024-10-10T14:50:07.895153Z","iopub.status.idle":"2024-10-10T14:50:08.032302Z","shell.execute_reply.started":"2024-10-10T14:50:07.895113Z","shell.execute_reply":"2024-10-10T14:50:08.031499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in train_merged_cat_columns:\n#     print(i, '--', train_merged[i].unique())","metadata":{"execution":{"iopub.status.busy":"2024-10-13T12:50:36.787073Z","iopub.execute_input":"2024-10-13T12:50:36.787986Z","iopub.status.idle":"2024-10-13T12:50:36.791783Z","shell.execute_reply.started":"2024-10-13T12:50:36.787944Z","shell.execute_reply":"2024-10-13T12:50:36.790737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in train_merged_cat_columns:\n#     train_merged[col] = train_merged[col].fillna('Missing')\n#     train_merged[col] = train_merged[col].astype('category')\n    \n#     test_merged[col] = test_merged[col].fillna('Missing')\n#     test_merged[col] = test_merged[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:50:52.632338Z","iopub.execute_input":"2024-10-10T14:50:52.633008Z","iopub.status.idle":"2024-10-10T14:50:52.667151Z","shell.execute_reply.started":"2024-10-10T14:50:52.632954Z","shell.execute_reply":"2024-10-10T14:50:52.666385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged_encoded = pd.get_dummies(train_merged, dtype='int')\n# test_merged_encoded = pd.get_dummies(test_merged, dtype='int')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:52:16.293655Z","iopub.execute_input":"2024-10-10T14:52:16.294303Z","iopub.status.idle":"2024-10-10T14:52:16.319669Z","shell.execute_reply.started":"2024-10-10T14:52:16.294265Z","shell.execute_reply":"2024-10-10T14:52:16.318731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged_encoded.shape, test_merged_encoded.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:52:41.669997Z","iopub.execute_input":"2024-10-10T14:52:41.670850Z","iopub.status.idle":"2024-10-10T14:52:41.676944Z","shell.execute_reply.started":"2024-10-10T14:52:41.670808Z","shell.execute_reply":"2024-10-10T14:52:41.676011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_cols = set(train_merged_encoded.columns)\n# test_cols = set(test_merged_encoded.columns)\n# train_cols - test_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:52:51.710701Z","iopub.execute_input":"2024-10-10T14:52:51.711820Z","iopub.status.idle":"2024-10-10T14:52:51.718578Z","shell.execute_reply.started":"2024-10-10T14:52:51.711765Z","shell.execute_reply":"2024-10-10T14:52:51.717641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_merged_encoded_reindexed = test_merged_encoded.reindex(columns=train_merged_encoded.columns, fill_value=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:52:56.244589Z","iopub.execute_input":"2024-10-10T14:52:56.245463Z","iopub.status.idle":"2024-10-10T14:52:56.251105Z","shell.execute_reply.started":"2024-10-10T14:52:56.245421Z","shell.execute_reply":"2024-10-10T14:52:56.250160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_merged_encoded_reindexed.drop('sii', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:52:59.075527Z","iopub.execute_input":"2024-10-10T14:52:59.076174Z","iopub.status.idle":"2024-10-10T14:52:59.082071Z","shell.execute_reply.started":"2024-10-10T14:52:59.076134Z","shell.execute_reply":"2024-10-10T14:52:59.081047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #filtering data having sii\n# train_merged_encoded = train_merged_encoded.loc[train_merged_encoded['sii'].notna()]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:53:01.493402Z","iopub.execute_input":"2024-10-10T14:53:01.493785Z","iopub.status.idle":"2024-10-10T14:53:01.500960Z","shell.execute_reply.started":"2024-10-10T14:53:01.493750Z","shell.execute_reply":"2024-10-10T14:53:01.500130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_merged_encoded.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:53:04.789592Z","iopub.execute_input":"2024-10-10T14:53:04.790477Z","iopub.status.idle":"2024-10-10T14:53:04.796192Z","shell.execute_reply.started":"2024-10-10T14:53:04.790437Z","shell.execute_reply":"2024-10-10T14:53:04.795242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">5. Modelling</p>","metadata":{}},{"cell_type":"code","source":"def convert(scores):\n    scores = np.array(scores)*1.25\n    bins = np.zeros_like(scores)\n    bins[scores <= 30] = 0\n    bins[(scores > 30) & (scores < 50)] = 1\n    bins[(scores >= 50) & (scores < 80)] = 2\n    bins[scores >= 80] = 3\n    return bins\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:32:35.877528Z","iopub.execute_input":"2024-10-13T13:32:35.877898Z","iopub.status.idle":"2024-10-13T13:32:35.883631Z","shell.execute_reply.started":"2024-10-13T13:32:35.877864Z","shell.execute_reply":"2024-10-13T13:32:35.882696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df[selection]\ny = train_df['PCIAT-PCIAT_Total']\n\ntrain_x, val_x, train_y, val_y = train_test_split(X, y, random_state=42, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:27:55.572674Z","iopub.execute_input":"2024-10-13T13:27:55.573055Z","iopub.status.idle":"2024-10-13T13:27:55.585199Z","shell.execute_reply.started":"2024-10-13T13:27:55.573019Z","shell.execute_reply":"2024-10-13T13:27:55.584033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X), len(y), train_x.shape, val_x.shape, train_y.shape, val_y.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:27:57.407238Z","iopub.execute_input":"2024-10-13T13:27:57.407666Z","iopub.status.idle":"2024-10-13T13:27:57.415287Z","shell.execute_reply.started":"2024-10-13T13:27:57.407617Z","shell.execute_reply":"2024-10-13T13:27:57.414272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def kappa_scorer( y_true, y_pred):\n    y_pred_converted = convert(y_pred)\n    y_true_converted = convert(y_true)\n    QWK_kappa_score = -cohen_kappa_score(y_true_converted, y_pred_converted, weights='quadratic')\n    return QWK_kappa_score\n\n\ndef objective(trial):\n    param = {\n        'num_leaves': trial.suggest_int('num_leaves', 5, 50),\n        'max_depth': trial.suggest_int('max_depth', -1, 25),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.15, step=0.01),\n        'random_state': 42,\n        'n_jobs': -1,\n        'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n        'min_child_weight': trial.suggest_float('min_child_weight', 1e-3, 1e-1, log=True),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0, step=0.1),\n        'subsample_freq': trial.suggest_int('subsample_freq', 0, 10),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0, step=0.1),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 1.0, log=True),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 1.0, log=True),\n    }\n\n    kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    kappa_scores = []\n    \n    for train_index, valid_index in kf.split(X, y):\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n        \n        model = lgb.LGBMRegressor(**param)\n        model.fit(X_train, y_train)\n\n        y_pred = model.predict(X_valid)\n        kappa_score = kappa_scorer(y_valid, y_pred)\n        kappa_scores.append(kappa_score)\n\n    return np.mean(kappa_scores)\n\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=50)\n\nprint(\"Best Score:\", study.best_value)\nprint(\"Best Hyperparameters:\", study.best_params)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-13T13:37:29.212350Z","iopub.execute_input":"2024-10-13T13:37:29.212763Z","iopub.status.idle":"2024-10-13T13:37:52.995561Z","shell.execute_reply.started":"2024-10-13T13:37:29.212724Z","shell.execute_reply":"2024-10-13T13:37:52.994628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:38:26.724226Z","iopub.execute_input":"2024-10-13T13:38:26.724619Z","iopub.status.idle":"2024-10-13T13:38:26.731277Z","shell.execute_reply.started":"2024-10-13T13:38:26.724582Z","shell.execute_reply":"2024-10-13T13:38:26.730433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = study.best_params\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:38:29.503218Z","iopub.execute_input":"2024-10-13T13:38:29.504149Z","iopub.status.idle":"2024-10-13T13:38:29.508668Z","shell.execute_reply.started":"2024-10-13T13:38:29.504105Z","shell.execute_reply":"2024-10-13T13:38:29.507689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nfinal_model = lgb.LGBMRegressor(**best_params)\nfinal_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:38:38.634560Z","iopub.execute_input":"2024-10-13T13:38:38.635277Z","iopub.status.idle":"2024-10-13T13:38:38.679198Z","shell.execute_reply.started":"2024-10-13T13:38:38.635239Z","shell.execute_reply":"2024-10-13T13:38:38.678221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">6. Predicting on test data</p>","metadata":{}},{"cell_type":"code","source":"test_df = test_df[selection]\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:41:06.993262Z","iopub.execute_input":"2024-10-13T13:41:06.993658Z","iopub.status.idle":"2024-10-13T13:41:07.021432Z","shell.execute_reply.started":"2024-10-13T13:41:06.993618Z","shell.execute_reply":"2024-10-13T13:41:07.020472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = final_model.predict(test_df)\nprediction","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:41:30.049885Z","iopub.execute_input":"2024-10-13T13:41:30.050280Z","iopub.status.idle":"2024-10-13T13:41:30.059281Z","shell.execute_reply.started":"2024-10-13T13:41:30.050245Z","shell.execute_reply":"2024-10-13T13:41:30.058390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_prediction = convert(prediction)\nfinal_prediction","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:41:50.588540Z","iopub.execute_input":"2024-10-13T13:41:50.589294Z","iopub.status.idle":"2024-10-13T13:41:50.595522Z","shell.execute_reply.started":"2024-10-13T13:41:50.589254Z","shell.execute_reply":"2024-10-13T13:41:50.594503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">7. Submission</p>","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:41:54.400384Z","iopub.execute_input":"2024-10-13T13:41:54.401152Z","iopub.status.idle":"2024-10-13T13:41:54.432011Z","shell.execute_reply.started":"2024-10-13T13:41:54.401110Z","shell.execute_reply":"2024-10-13T13:41:54.431229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id': sample['id'], 'sii': final_prediction})","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:42:21.150920Z","iopub.execute_input":"2024-10-13T13:42:21.151866Z","iopub.status.idle":"2024-10-13T13:42:21.471706Z","shell.execute_reply.started":"2024-10-13T13:42:21.151824Z","shell.execute_reply":"2024-10-13T13:42:21.470306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T13:42:25.401446Z","iopub.execute_input":"2024-10-13T13:42:25.401861Z","iopub.status.idle":"2024-10-13T13:42:25.442974Z","shell.execute_reply.started":"2024-10-13T13:42:25.401821Z","shell.execute_reply":"2024-10-13T13:42:25.441790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"text-align: left; font-family:Comic Sans MS; padding: 12px; line-height:2; border-radius:1px; margin-bottom: 0em; text-align: center; font-size: 50px;border-style: solid;border-color: dark green; font-weight: bold;\">8. References</p>\n- [Exploring Problematic Internet Usage](https://www.kaggle.com/code/gkitchen/exploring-problematic-internet-usage)","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}