{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from gc import collect\nfrom IPython.display import clear_output, HTML\n\n# !pip install feature_engine\n\n# collect();\n# clear_output();","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:19.026586Z","iopub.execute_input":"2024-10-01T18:52:19.027052Z","iopub.status.idle":"2024-10-01T18:52:19.038951Z","shell.execute_reply.started":"2024-10-01T18:52:19.027002Z","shell.execute_reply":"2024-10-01T18:52:19.037672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install scikit-learn==1.4.0\n\n# collect();\n# clear_output();","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:19.041607Z","iopub.execute_input":"2024-10-01T18:52:19.042057Z","iopub.status.idle":"2024-10-01T18:52:19.053679Z","shell.execute_reply.started":"2024-10-01T18:52:19.041981Z","shell.execute_reply":"2024-10-01T18:52:19.052411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing\nfrom sklearn.model_selection import train_test_split, GridSearchCV, RepeatedStratifiedKFold, StratifiedKFold\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import VotingClassifier, StackingClassifier\nfrom sklearn.preprocessing import StandardScaler\n# from feature_engine.selection import DropCorrelatedFeatures, DropConstantFeatures\nfrom imblearn.pipeline import make_pipeline as make_imblearn_pipeline\nfrom sklearn.compose import ColumnTransformer, make_column_selector, make_column_transformer\nfrom sklearn.feature_selection import VarianceThreshold\nimport pandas as pd\nfrom sklearn.impute import KNNImputer\n\n# Target encoding/decoding\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.preprocessing import LabelEncoder\nimport category_encoders as ce\nfrom category_encoders import TargetEncoder\n\n# Metrics\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix, auc, roc_curve, log_loss\nfrom sklearn.model_selection import cross_val_score\n\n# Models\nfrom xgboost import XGBClassifier\n# from lightgbm import LGBMClassifier, plot_importance\nfrom catboost import CatBoostClassifier\nfrom imblearn.over_sampling import SMOTE\n\n# Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sn\n\n# Math and DataFrame\nimport numpy as np\nfrom scipy import stats\nfrom scipy import special\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nfrom colorama import Fore, Style\nfrom scipy.optimize import minimize\n\n# Feature clustering\nfrom scipy.cluster import hierarchy\nfrom scipy.spatial.distance import squareform\n\n# Hyperparameter tuning\nimport optuna\n\n# Warnings ignore\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Polars library\n# import polars as pl\n# import polars.selectors as cs","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:19.055423Z","iopub.execute_input":"2024-10-01T18:52:19.055928Z","iopub.status.idle":"2024-10-01T18:52:20.451120Z","shell.execute_reply.started":"2024-10-01T18:52:19.055883Z","shell.execute_reply":"2024-10-01T18:52:20.449562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA\n\nExample notebook: https://www.kaggle.com/code/ambrosm/piu-eda-which-makes-sense","metadata":{}},{"cell_type":"markdown","source":"**The competition data is compiled into two sources, parquet files containing the accelerometer (actigraphy) series and csv files containing the remaining tabular data. The majority of measures are missing for most participants. In particular, the target sii is missing for a portion of the participants in the training set. You may wish to apply non-supervised learning techniques to this data. The sii value is present for all instances in the test set.**","metadata":{}},{"cell_type":"markdown","source":"The tabular data in train.csv and test.csv comprises measurements from a variety of instruments. The fields within each instrument are described in data_dictionary.csv. These instruments are:\n\n* Demographics - Information about age and sex of participants.\n\n* Internet Use - Number of hours of using computer/internet per day.\n\n* Children's Global Assessment Scale - Numeric scale used by mental health clinicians to rate the general functioning of youths under the age of 18.\n\n* Physical Measures - Collection of blood pressure, heart rate, height, weight and waist, and hip measurements.\n\n* FitnessGram Vitals and Treadmill - Measurements of cardiovascular fitness assessed using the NHANES treadmill protocol.\n\n* FitnessGram Child - Health related physical fitness assessment measuring five different parameters including aerobic capacity, muscular strength, muscular endurance, flexibility, and body composition.\n\n* Bio-electric Impedance Analysis - Measure of key body composition elements, including BMI, fat, muscle, and water content.\n\n* Physical Activity Questionnaire - Information about children's participation in vigorous activities over the last 7 days.\n\n* Sleep Disturbance Scale - Scale to categorize sleep disorders in children.\nActigraphy - Objective measure of ecological physical activity through a research-grade biotracker.\n\n* Parent-Child Internet Addiction Test - 20-item scale that measures characteristics and behaviors associated with compulsive use of the Internet including compulsivity, escapism, and dependency.\n\nNote in particular the field PCIAT-PCIAT_Total. The target sii for this competition is derived from this field as described in the data dictionary: 0 for None, 1 for Mild, 2 for Moderate, and 3 for Severe. Additionally, each participant has been assigned a unique identifier id.\n","metadata":{}},{"cell_type":"code","source":"# Read the training dataset, then cast all columns ending with 'Season' to the custom 'season_dtype'\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')  # Read the train CSV file\n\n\n# Read the test dataset, then cast all columns ending with 'Season' to the custom 'season_dtype'\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')  # Read the test CSV file\n\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:20.453543Z","iopub.execute_input":"2024-10-01T18:52:20.454083Z","iopub.status.idle":"2024-10-01T18:52:20.550637Z","shell.execute_reply.started":"2024-10-01T18:52:20.454034Z","shell.execute_reply":"2024-10-01T18:52:20.549548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:20.552134Z","iopub.execute_input":"2024-10-01T18:52:20.553154Z","iopub.status.idle":"2024-10-01T18:52:20.562098Z","shell.execute_reply.started":"2024-10-01T18:52:20.553100Z","shell.execute_reply":"2024-10-01T18:52:20.560658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"During their participation in the HBN study, some participants were given an accelerometer to wear for up to 30 days continually while at home and going about their regular daily lives.\n\nseries_{train|test}.parquet/id={id} - Series to be used as training data, partitioned by id. Each series is a continuous recording of accelerometer data for a single subject spanning many days.\n\n* id - The patient identifier corresponding to the id field in train/test.csv.\n\n* step - An integer timestep for each observation within a series.\n\n* X, Y, Z - Measure of acceleration, in g, experienced by the wrist-worn watch along each standard axis.\n\n* enmo - As calculated and described by the wristpy package, ENMO is the Euclidean Norm Minus One of all accelerometer signals (along each of the x-, y-, and z-axis, measured in g-force) with negative values rounded to zero. Zero values are indicative of periods of no motion. While no standard measure of acceleration exists in this space, this is one of the several commonly computed features\n\n* anglez - As calculated and described by the wristpy package, Angle-Z is a metric derived from individual accelerometer components and refers to the angle of the arm relative to the horizontal plane.\n\n* non-wear_flag - A flag (0: watch is being worn, 1: the watch is not worn) to help determine periods when the watch has been removed, based on the GGIR definition, which uses the standard deviation and range of the accelerometer data.\n\n* light - Measure of ambient light in lux.\n\n* battery_voltage - A measure of the battery voltage in mV.\n\n* time_of_day - Time of day representing the start of a 5s window that the data has been sampled over, with format %H:%M:%S.%9f.\n\n* weekday - The day of the week, coded as an integer with 1 being Monday and 7 being Sunday.\nquarter - The quarter of the year, an integer from 1 to 4.\n\n* relative_date_PCIAT - The number of days (integer) since the PCIAT test was administered (negative days indicate that the actigraphy data has been collected before the test was administered).","metadata":{}},{"cell_type":"markdown","source":"Let's read parquest.file for id 00115b9f","metadata":{}},{"cell_type":"code","source":"# Define the file path\nfile_path_00115b9f = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet'\n\n# Read the parquet file\ndf_00115b9f = pd.read_parquet(file_path_00115b9f)\n\ndf_00115b9f.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:20.564354Z","iopub.execute_input":"2024-10-01T18:52:20.564905Z","iopub.status.idle":"2024-10-01T18:52:20.623973Z","shell.execute_reply.started":"2024-10-01T18:52:20.564852Z","shell.execute_reply":"2024-10-01T18:52:20.622786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate missing values percentage per column\ndef missing_values_by_column(df):\n    # Calculate the percentage of missing values for each column\n    null_percentage = (df.isna().sum() / len(df)) * 100\n    return null_percentage\n\n# Example of using the function to get missing percentage per column\nmissing_percentage = missing_values_by_column(train)\n\n# Plotting the missing percentage per column\nplt.figure(figsize=(10, 20))\nmissing_percentage.sort_values().plot(kind='barh', color='skyblue')  # Horizontal bar plot\nplt.title('Percentage of Missing Values per Column')\nplt.xlabel('Percentage of Missing Values (%)')\nplt.ylabel('Columns')\nplt.grid(True, axis='x')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:20.625470Z","iopub.execute_input":"2024-10-01T18:52:20.626299Z","iopub.status.idle":"2024-10-01T18:52:21.756037Z","shell.execute_reply.started":"2024-10-01T18:52:20.626248Z","shell.execute_reply":"2024-10-01T18:52:21.754642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some columns, apart target, are messed in test","metadata":{}},{"cell_type":"code","source":"print('Columns missing in test:')\nprint([f for f in train.columns if f not in test.columns])\n\ncolumns_to_drop = ['id','PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:21.757421Z","iopub.execute_input":"2024-10-01T18:52:21.757779Z","iopub.status.idle":"2024-10-01T18:52:21.770773Z","shell.execute_reply.started":"2024-10-01T18:52:21.757743Z","shell.execute_reply":"2024-10-01T18:52:21.769292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the distribution of classes in the 'sii' column\nplt.figure(figsize=(10, 6))  # Setting the size of the plot\nsn.countplot(x='sii', data=train)  # Plotting the count of each class in 'sii'\n\n# Adding titles and labels\nplt.title('Distribution of Classes in the \"sii\" Column')  # Title of the plot\nplt.xlabel('Classes')  # X-axis label\nplt.ylabel('Count')  # Y-axis label\n\n# Display the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:21.774629Z","iopub.execute_input":"2024-10-01T18:52:21.775518Z","iopub.status.idle":"2024-10-01T18:52:22.041016Z","shell.execute_reply.started":"2024-10-01T18:52:21.775460Z","shell.execute_reply":"2024-10-01T18:52:22.039746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = train.dropna(subset=['sii'])\n\ntrain1 = train1.drop(columns=columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.042594Z","iopub.execute_input":"2024-10-01T18:52:22.043112Z","iopub.status.idle":"2024-10-01T18:52:22.055109Z","shell.execute_reply.started":"2024-10-01T18:52:22.043060Z","shell.execute_reply":"2024-10-01T18:52:22.053466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subtrain = train[train['sii'].isna()].drop(columns=columns_to_drop)\n\nsubtrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.056632Z","iopub.execute_input":"2024-10-01T18:52:22.057024Z","iopub.status.idle":"2024-10-01T18:52:22.070416Z","shell.execute_reply.started":"2024-10-01T18:52:22.056984Z","shell.execute_reply":"2024-10-01T18:52:22.069226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\n\nBase trained model on the rough data.","metadata":{}},{"cell_type":"code","source":"target_cols = ['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.071808Z","iopub.execute_input":"2024-10-01T18:52:22.072201Z","iopub.status.idle":"2024-10-01T18:52:22.079193Z","shell.execute_reply.started":"2024-10-01T18:52:22.072164Z","shell.execute_reply":"2024-10-01T18:52:22.077909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train1[target_cols] # target remain unchanged from the initial dataset\n\nX_train = train1.drop(target_cols, axis=1)\n\nX_subtrain = subtrain.drop(target_cols, axis=1)\n\nX_test = test.drop(columns='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.080725Z","iopub.execute_input":"2024-10-01T18:52:22.081217Z","iopub.status.idle":"2024-10-01T18:52:22.094118Z","shell.execute_reply.started":"2024-10-01T18:52:22.081165Z","shell.execute_reply":"2024-10-01T18:52:22.092820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.095741Z","iopub.execute_input":"2024-10-01T18:52:22.096343Z","iopub.status.idle":"2024-10-01T18:52:22.105368Z","shell.execute_reply.started":"2024-10-01T18:52:22.096288Z","shell.execute_reply":"2024-10-01T18:52:22.103990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_subtrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.106894Z","iopub.execute_input":"2024-10-01T18:52:22.107340Z","iopub.status.idle":"2024-10-01T18:52:22.118699Z","shell.execute_reply.started":"2024-10-01T18:52:22.107290Z","shell.execute_reply":"2024-10-01T18:52:22.117258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.120294Z","iopub.execute_input":"2024-10-01T18:52:22.120818Z","iopub.status.idle":"2024-10-01T18:52:22.129872Z","shell.execute_reply.started":"2024-10-01T18:52:22.120764Z","shell.execute_reply":"2024-10-01T18:52:22.128598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_columns = X_train.select_dtypes(include=['object', 'category']).columns.tolist()\n\n# cat_columns = test.columns.tolist()\n\ncat_columns","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.131367Z","iopub.execute_input":"2024-10-01T18:52:22.131757Z","iopub.status.idle":"2024-10-01T18:52:22.142443Z","shell.execute_reply.started":"2024-10-01T18:52:22.131708Z","shell.execute_reply":"2024-10-01T18:52:22.141056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_binary_numeric_cols = X_train.select_dtypes(include=['int64', 'float64']).nunique() > 2\n\nnum_columns = non_binary_numeric_cols[non_binary_numeric_cols].index.tolist()\n\n# num_columns = []\n\nnum_columns","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.144286Z","iopub.execute_input":"2024-10-01T18:52:22.144672Z","iopub.status.idle":"2024-10-01T18:52:22.166848Z","shell.execute_reply.started":"2024-10-01T18:52:22.144634Z","shell.execute_reply":"2024-10-01T18:52:22.165702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\n\n# poly = PolynomialFeatures()\n\n# drop_correlation = DropCorrelatedFeatures(threshold=0.9)\n\nt_encoder = TargetEncoder()\n\nnum_pipeline = make_imblearn_pipeline(scaler,\n                                      KNNImputer(),\n                                      # poly,\n                                      # drop_correlation\n                                      )\n\ncat_pipeline = make_imblearn_pipeline(t_encoder,\n                                      scaler,\n                                      KNNImputer(),\n                                      # poly\n                                      # drop_correlation\n                                      )\n                                      \n\ncolumn_transformer = make_column_transformer((num_pipeline, num_columns),\n                                             (cat_pipeline, cat_columns),\n                                              remainder='passthrough')\n\ncolumn_transformer","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.168381Z","iopub.execute_input":"2024-10-01T18:52:22.168887Z","iopub.status.idle":"2024-10-01T18:52:22.210351Z","shell.execute_reply.started":"2024-10-01T18:52:22.168814Z","shell.execute_reply":"2024-10-01T18:52:22.209228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params_lgbm2000 = {'learning_rate': 0.03,\n                    'max_depth': 10, 'n_estimators': 2000, 'num_leaves': 15}\n\nbest_params_lgbm3000 = {'learning_rate': 0.05,\n                    'max_depth': 6, 'n_estimators': 3000, 'num_leaves': 15}\n\nbest_params_xgb = {'n_estimators': 2000, 'max_depth': 10, \n                    'learning_rate': 0.05, 'subsample': 0.8}\n\nbest_params_catboost2300 = {'iterations': 2300, 'depth': 10, 'learning_rate': 0.03, \n                        'random_strength': 1, 'bagging_temperature': 0.3} \n\nbest_params_catboost3500 = {'iterations': 3500, 'depth': 6, 'learning_rate': 0.02, \n                        'random_strength': 7, 'bagging_temperature': 0.5} ","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.212301Z","iopub.execute_input":"2024-10-01T18:52:22.212796Z","iopub.status.idle":"2024-10-01T18:52:22.219646Z","shell.execute_reply.started":"2024-10-01T18:52:22.212743Z","shell.execute_reply":"2024-10-01T18:52:22.218528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = XGBClassifier(**best_params_xgb, random_state=102030)\n\ncatboost2300 = CatBoostClassifier(**best_params_catboost2300, random_state=102030, \n                             verbose=False)\n\ncatboost3500 = CatBoostClassifier(**best_params_catboost3500, random_state=102030, \n                             verbose=False)\n\nens_pipeline = make_imblearn_pipeline(column_transformer,\n#                                  FunctionSampler(func=outlier_detector_), \n#                                  DropConstantFeatures(),\n                                  VotingClassifier(estimators=[('catboost2300', catboost2300),\n                                                               ('catboost3500', catboost3500),\n                                                              # ('lgbm2000', lgbm2000),\n                                                              # ('lgbm3000', lgbm3000),\n                                                              # ('ridge', ridge),\n                                                               ('xgb', xgb), \n                                                              ], voting='soft',\n                                                              #  weights = [0.5, 0.1, 0.4]\n                                                              ))\n                           \n\n# adding of ensemble name\nens_pipeline.steps[-1] = ('ensemble', ens_pipeline.steps[-1][1])","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.221533Z","iopub.execute_input":"2024-10-01T18:52:22.222065Z","iopub.status.idle":"2024-10-01T18:52:22.232922Z","shell.execute_reply.started":"2024-10-01T18:52:22.221991Z","shell.execute_reply":"2024-10-01T18:52:22.231706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nens_pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T18:52:22.234520Z","iopub.execute_input":"2024-10-01T18:52:22.235923Z","iopub.status.idle":"2024-10-01T19:00:43.657094Z","shell.execute_reply.started":"2024-10-01T18:52:22.235853Z","shell.execute_reply":"2024-10-01T19:00:43.655792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npredictions_subtrain = ens_pipeline.predict(X_subtrain)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:00:43.658637Z","iopub.execute_input":"2024-10-01T19:00:43.659103Z","iopub.status.idle":"2024-10-01T19:00:45.886736Z","shell.execute_reply.started":"2024-10-01T19:00:43.659053Z","shell.execute_reply":"2024-10-01T19:00:45.885524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_indices = train[train['sii'].isna()].index\n\ntrain2 = train.copy()\n\ntrain2.loc[nan_indices, 'sii'] = predictions_subtrain\n\ntrain2 = train2.drop(columns=columns_to_drop)\n\ntrain2['sii'].isna().sum() # we got datafreim without NaN in target values","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:00:45.892074Z","iopub.execute_input":"2024-10-01T19:00:45.892861Z","iopub.status.idle":"2024-10-01T19:00:45.906708Z","shell.execute_reply.started":"2024-10-01T19:00:45.892795Z","shell.execute_reply":"2024-10-01T19:00:45.905500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train2 = train2[target_cols] # target remain unchanged from the initial dataset\n\nX_train2 = train2.drop(target_cols, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:00:45.908113Z","iopub.execute_input":"2024-10-01T19:00:45.908578Z","iopub.status.idle":"2024-10-01T19:00:45.916127Z","shell.execute_reply.started":"2024-10-01T19:00:45.908527Z","shell.execute_reply":"2024-10-01T19:00:45.914990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nens_pipeline.fit(X_train2, y_train2)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:00:45.917483Z","iopub.execute_input":"2024-10-01T19:00:45.917878Z","iopub.status.idle":"2024-10-01T19:09:48.363462Z","shell.execute_reply.started":"2024-10-01T19:00:45.917811Z","shell.execute_reply":"2024-10-01T19:09:48.362267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npredictions = ens_pipeline.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:09:48.365125Z","iopub.execute_input":"2024-10-01T19:09:48.365929Z","iopub.status.idle":"2024-10-01T19:09:48.526202Z","shell.execute_reply.started":"2024-10-01T19:09:48.365873Z","shell.execute_reply":"2024-10-01T19:09:48.524953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nsample.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:09:48.527912Z","iopub.execute_input":"2024-10-01T19:09:48.528336Z","iopub.status.idle":"2024-10-01T19:09:48.546265Z","shell.execute_reply.started":"2024-10-01T19:09:48.528289Z","shell.execute_reply":"2024-10-01T19:09:48.544931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare submission\n\ntest_ids = test['id']\n\nsubmission = pd.DataFrame({\n    'id': test_ids.values,\n    'sii': predictions\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:09:48.547643Z","iopub.execute_input":"2024-10-01T19:09:48.548114Z","iopub.status.idle":"2024-10-01T19:09:48.557104Z","shell.execute_reply.started":"2024-10-01T19:09:48.548075Z","shell.execute_reply":"2024-10-01T19:09:48.555906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T19:09:48.558737Z","iopub.execute_input":"2024-10-01T19:09:48.559866Z","iopub.status.idle":"2024-10-01T19:09:48.571809Z","shell.execute_reply.started":"2024-10-01T19:09:48.559786Z","shell.execute_reply":"2024-10-01T19:09:48.570442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here is rough prediction. Next step is feature engineering, cleaning of DS and understanding of information in parquet.files.","metadata":{}}]}