{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:10:57.509711Z","iopub.execute_input":"2024-12-07T05:10:57.510071Z","iopub.status.idle":"2024-12-07T05:10:58.080115Z","shell.execute_reply.started":"2024-12-07T05:10:57.510037Z","shell.execute_reply":"2024-12-07T05:10:58.078999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport math\n\nfrom plotly.subplots import make_subplots\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nimport xgboost as xgb\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import  LabelEncoder ,StandardScaler, OneHotEncoder,RobustScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import confusion_matrix , accuracy_score , recall_score , f1_score , classification_report\nfrom collections import Counter\nfrom sklearn.svm import SVC","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:09:31.624562Z","iopub.execute_input":"2024-12-07T05:09:31.625166Z","iopub.status.idle":"2024-12-07T05:09:32.666335Z","shell.execute_reply.started":"2024-12-07T05:09:31.625116Z","shell.execute_reply":"2024-12-07T05:09:32.664858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:09:32.668064Z","iopub.execute_input":"2024-12-07T05:09:32.669513Z","iopub.status.idle":"2024-12-07T05:09:32.763042Z","shell.execute_reply.started":"2024-12-07T05:09:32.669459Z","shell.execute_reply":"2024-12-07T05:09:32.761713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom  concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n# Custom functions\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data\n\ndic= pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n# Load time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntrain_ts.head()\n\ntime_series_cols = train_ts.columns.tolist()\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\nid=test['id']\ntrain = train.drop('id', axis=1) \ntest = test.drop('id', axis=1)\ntrain = train.dropna(subset=['sii'])\ntrain.head()\n\nnum_feature = train.select_dtypes(include=['number']).columns.to_numpy()\nprint(num_feature)\nTARGET_COLS = [\n    \"PCIAT-Season\", \"PCIAT-PCIAT_01\", \"PCIAT-PCIAT_02\", \"PCIAT-PCIAT_03\", \"PCIAT-PCIAT_04\", \"PCIAT-PCIAT_05\", \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\", \"PCIAT-PCIAT_08\", \"PCIAT-PCIAT_09\", \"PCIAT-PCIAT_10\", \"PCIAT-PCIAT_11\", \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\", \"PCIAT-PCIAT_14\", \"PCIAT-PCIAT_15\", \"PCIAT-PCIAT_16\", \"PCIAT-PCIAT_17\", \"PCIAT-PCIAT_18\", \n    \"PCIAT-PCIAT_19\", \"PCIAT-PCIAT_20\", \"PCIAT-PCIAT_Total\"]\n\ntrain= train.drop(TARGET_COLS,axis=1)\n\ntrain.info()\n\ntest.info()\n\ny = train['sii']\nx = train.drop('sii', axis=1)\n\ny.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:09:32.994755Z","iopub.execute_input":"2024-12-07T05:09:32.995761Z","iopub.status.idle":"2024-12-07T05:10:56.048802Z","shell.execute_reply.started":"2024-12-07T05:09:32.995704Z","shell.execute_reply":"2024-12-07T05:10:56.047630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\n\ndef data_preprocessing(train, test):\n    # Identifying categorical and numerical features\n    cat_train_features = train.select_dtypes(include=['object']).columns.tolist()\n    num_train_features = train.select_dtypes(include=['number']).columns.tolist()\n    \n    cat_test_features = test.select_dtypes(include=['object']).columns.tolist()\n    num_test_features = test.select_dtypes(include=['number']).columns.tolist()\n\n    # Filling missing values in categorical columns\n    for feature in cat_train_features:\n        train[feature] = train[feature].fillna('Missing').astype('category')\n        test[feature] = test[feature].fillna('Missing').astype('category')\n\n    # Imputing missing values for numerical columns\n    imputer = SimpleImputer(strategy='mean')\n    train[num_train_features] = imputer.fit_transform(train[num_train_features])\n    test[num_train_features] = imputer.transform(test[num_train_features])\n\n    # Scaling numerical columns\n    scaler = StandardScaler()\n    train[num_train_features] = scaler.fit_transform(train[num_train_features])\n    test[num_train_features] = scaler.transform(test[num_train_features])\n\n    # Encoding categorical columns\n    encoder = LabelEncoder()\n    for feature in cat_train_features:\n        train[feature] = encoder.fit_transform(train[feature])\n        test[feature] = encoder.transform(test[feature])\n\n    return train, test\n\n# Example usage of the function\nx_processed, x_test_processed = data_preprocessing(x, test)\n\n# Displaying the first few rows of the processed data\nprint(x_processed.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:12:57.596422Z","iopub.execute_input":"2024-12-07T05:12:57.597092Z","iopub.status.idle":"2024-12-07T05:12:57.779654Z","shell.execute_reply.started":"2024-12-07T05:12:57.597023Z","shell.execute_reply":"2024-12-07T05:12:57.778482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier, StackingClassifier, RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier \nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nimport optuna","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:59:03.506079Z","iopub.execute_input":"2024-12-07T04:59:03.506557Z","iopub.status.idle":"2024-12-07T04:59:03.513973Z","shell.execute_reply.started":"2024-12-07T04:59:03.506520Z","shell.execute_reply":"2024-12-07T04:59:03.512477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective_xgb(trial):\n    param = {\n        'subsample': trial.suggest_float('subsample', 0.1, 0.9),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 10.0),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 10.0),\n        'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.4),\n        'gamma': trial.suggest_float('gamma', 1e-8, 10.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.2, 1.0)\n    }\n    \n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    \n\n    kappa_scores = []\n    for train_idx, val_idx in skf.split(x, y):\n        x_train_fold, x_val_fold = x.iloc[train_idx], x.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n        model = XGBClassifier(**param, tree_method='hist', device='cuda')\n        model.fit(x_train_fold, y_train_fold)\n        y_pred = model.predict(x_val_fold)\n        kappa = cohen_kappa_score(y_val_fold, y_pred, weights='quadratic')\n        kappa_scores.append(kappa)\n    \n    print(f'{np.mean(kappa_scores)} : {kappa_scores}')\n    \n    return np.mean(kappa_scores)\n\nbest_param_xgb = {'subsample': 0.5704784142330617, 'reg_lambda': 9.659566361955893, 'reg_alpha': 0.019505444210683898, 'n_estimators': 249, 'min_child_weight': 10, 'max_depth': 6, 'learning_rate': 0.08954604150212105, 'gamma': 2.119887304669434, 'colsample_bytree': 0.9553505917703975}\nxgb_model = XGBClassifier(**best_param_xgb, tree_method='hist', device='cuda')\nxgb_model.fit(x, y) #0.39670540254141295\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:59:05.443092Z","iopub.execute_input":"2024-12-07T04:59:05.443602Z","iopub.status.idle":"2024-12-07T04:59:11.014881Z","shell.execute_reply.started":"2024-12-07T04:59:05.443562Z","shell.execute_reply":"2024-12-07T04:59:11.011200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure that both train and test data have the same columns\nx_processed, x_test_processed = data_preprocessing(x, test)\n\n# Re-add the 'id' column from the test set to the x_test_processed\nx_test_processed['id'] = id\n\n# Predicting on the test data using the trained XGBoost model\ny_pred_test = xgb_model.predict(x_test_processed.drop('id', axis=1))  # Drop 'id' during prediction\n\n# Create a DataFrame for the submission\nsubmission = pd.DataFrame({\n    'id': id,  # Using 'id' from the test set\n    'sii': y_pred_test  # Predicted values for 'sii'\n})\n\n# Saving the submission file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file has been saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T05:31:04.831900Z","iopub.execute_input":"2024-12-07T05:31:04.832838Z","iopub.status.idle":"2024-12-07T05:31:04.980000Z","shell.execute_reply.started":"2024-12-07T05:31:04.832793Z","shell.execute_reply":"2024-12-07T05:31:04.977213Z"}},"outputs":[],"execution_count":null}]}