{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:45:16.896393Z","iopub.execute_input":"2024-12-05T12:45:16.896733Z","iopub.status.idle":"2024-12-05T12:45:21.408657Z","shell.execute_reply.started":"2024-12-05T12:45:16.896697Z","shell.execute_reply":"2024-12-05T12:45:21.407620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport math\n\nfrom plotly.subplots import make_subplots\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nimport xgboost as xgb\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import  LabelEncoder ,StandardScaler, OneHotEncoder,RobustScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import confusion_matrix , accuracy_score , recall_score , f1_score , classification_report\nfrom collections import Counter\nfrom sklearn.svm import SVC\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:45:21.410537Z","iopub.execute_input":"2024-12-05T12:45:21.410997Z","iopub.status.idle":"2024-12-05T12:45:22.931811Z","shell.execute_reply.started":"2024-12-05T12:45:21.410949Z","shell.execute_reply":"2024-12-05T12:45:22.930859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\ntrain.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:45:22.932903Z","iopub.execute_input":"2024-12-05T12:45:22.933346Z","iopub.status.idle":"2024-12-05T12:45:23.040592Z","shell.execute_reply.started":"2024-12-05T12:45:22.933315Z","shell.execute_reply":"2024-12-05T12:45:23.039475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom  concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n# Custom functions\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data\n\ndic= pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n# Load time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntrain_ts.head()\n\ntime_series_cols = train_ts.columns.tolist()\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\nid=test['id']\ntrain = train.drop('id', axis=1) \ntest = test.drop('id', axis=1)\ntrain = train.dropna(subset=['sii'])\ntrain.head()\n\nnum_feature = train.select_dtypes(include=['number']).columns.to_numpy()\nprint(num_feature)\nTARGET_COLS = [\n    \"PCIAT-Season\",\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",    \n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n    \"PCIAT-PCIAT_Total\"]\n\ntrain= train.drop(TARGET_COLS,axis=1)\n\ntrain.info()\n\ntest.info()\n\ny = train['sii']\nx = train.drop('sii', axis=1)\n\ny.value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:45:23.042784Z","iopub.execute_input":"2024-12-05T12:45:23.043114Z","iopub.status.idle":"2024-12-05T12:46:40.932654Z","shell.execute_reply.started":"2024-12-05T12:45:23.043080Z","shell.execute_reply":"2024-12-05T12:46:40.931633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:46:40.934137Z","iopub.execute_input":"2024-12-05T12:46:40.934836Z","iopub.status.idle":"2024-12-05T12:46:40.943867Z","shell.execute_reply.started":"2024-12-05T12:46:40.934788Z","shell.execute_reply":"2024-12-05T12:46:40.942941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def data_preprocessing(train, test):\n    \n    \n    cat_train_feature = list(train.select_dtypes(include=['object']).columns)\n    num_train_feature = train.select_dtypes(include=['number']).columns.tolist()\n    \n    cat_test_feature = list(test.select_dtypes(include=['object']).columns)\n    num_test_feature = test.select_dtypes(include=['number']).columns.tolist()\n\n    for feature in cat_train_feature:\n        train[feature] = train[feature].fillna('Missing')\n        train[feature] = train[feature].astype('category')\n        test[feature] = test[feature].fillna('Missing')\n        test[feature] = test[feature].astype('category')\n\n    imputer = SimpleImputer(strategy='mean')\n    num_columns = train.select_dtypes(include=['number']).columns\n    train[num_columns] = imputer.fit_transform(train[num_columns])\n    test[num_columns] = imputer.transform(test[num_columns])\n    \n    scaler = StandardScaler()\n    train[num_columns] = scaler.fit_transform(train[num_columns])\n    test[num_columns] = scaler.transform(test[num_columns])\n    \n    encode = LabelEncoder()\n    cat_columns = train.select_dtypes(include=['category']).columns\n    \n    for feature in cat_columns:\n        train[feature] = encode.fit_transform(train[feature])\n        test[feature] = encode.transform(test[feature])\n    \n    return train, test\n\nx, x_test = data_preprocessing(x, test)\n\nx.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:46:40.945326Z","iopub.execute_input":"2024-12-05T12:46:40.946113Z","iopub.status.idle":"2024-12-05T12:46:41.102271Z","shell.execute_reply.started":"2024-12-05T12:46:40.946066Z","shell.execute_reply":"2024-12-05T12:46:41.101107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier, StackingClassifier, RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier \nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nimport optuna\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:46:41.103831Z","iopub.execute_input":"2024-12-05T12:46:41.104619Z","iopub.status.idle":"2024-12-05T12:46:42.371894Z","shell.execute_reply.started":"2024-12-05T12:46:41.104569Z","shell.execute_reply":"2024-12-05T12:46:42.370975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective_xgb(trial):\n    param = {\n        'subsample': trial.suggest_float('subsample', 0.1, 0.9),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 10.0),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 10.0),\n        'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.4),\n        'gamma': trial.suggest_float('gamma', 1e-8, 10.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.2, 1.0)\n    }\n    \n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    \n\n    kappa_scores = []\n    for train_idx, val_idx in skf.split(x, y):\n        x_train_fold, x_val_fold = x.iloc[train_idx], x.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n        model = XGBClassifier(**param, tree_method='hist', device='cuda')\n        model.fit(x_train_fold, y_train_fold)\n        y_pred = model.predict(x_val_fold)\n        kappa = cohen_kappa_score(y_val_fold, y_pred, weights='quadratic')\n        kappa_scores.append(kappa)\n    \n    print(f'{np.mean(kappa_scores)} : {kappa_scores}')\n    \n    return np.mean(kappa_scores)\n\nbest_param_xgb = {'subsample': 0.5704784142330617, 'reg_lambda': 9.659566361955893, 'reg_alpha': 0.019505444210683898, 'n_estimators': 249, 'min_child_weight': 10, 'max_depth': 6, 'learning_rate': 0.08954604150212105, 'gamma': 2.119887304669434, 'colsample_bytree': 0.9553505917703975}\nxgb_model = XGBClassifier(**best_param_xgb, tree_method='hist', device='cuda')\nxgb_model.fit(x, y) #0.39670540254141295\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:46:42.372932Z","iopub.execute_input":"2024-12-05T12:46:42.373390Z","iopub.status.idle":"2024-12-05T12:46:44.160765Z","shell.execute_reply.started":"2024-12-05T12:46:42.373360Z","shell.execute_reply":"2024-12-05T12:46:44.159727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_predict = xgb_model.predict(x_test) \n#print(y_test_predict[:, 0])\nsubmission = pd.DataFrame({\n    'id': id, \n    'sii': y_test_predict.astype(int)\n})\nsubmission.to_csv('submission.csv',index=False)\n\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:46:44.161839Z","iopub.execute_input":"2024-12-05T12:46:44.162207Z","iopub.status.idle":"2024-12-05T12:46:44.196391Z","shell.execute_reply.started":"2024-12-05T12:46:44.162175Z","shell.execute_reply":"2024-12-05T12:46:44.195348Z"}},"outputs":[],"execution_count":null}]}