{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\n\nimport lightgbm as lgb\nimport optuna\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nUSE_OPTUNA = False","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-30T15:22:49.439240Z","iopub.execute_input":"2024-09-30T15:22:49.439785Z","iopub.status.idle":"2024-09-30T15:22:49.447643Z","shell.execute_reply.started":"2024-09-30T15:22:49.439733Z","shell.execute_reply":"2024-09-30T15:22:49.446350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:49.868546Z","iopub.execute_input":"2024-09-30T15:22:49.869010Z","iopub.status.idle":"2024-09-30T15:22:49.923297Z","shell.execute_reply.started":"2024-09-30T15:22:49.868969Z","shell.execute_reply":"2024-09-30T15:22:49.922099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[~train_df['sii'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:50.095352Z","iopub.execute_input":"2024-09-30T15:22:50.095812Z","iopub.status.idle":"2024-09-30T15:22:50.103759Z","shell.execute_reply.started":"2024-09-30T15:22:50.095768Z","shell.execute_reply":"2024-09-30T15:22:50.102464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_column = test_df['id']\nX = train_df.drop(['sii','id'], axis=1)\ny = train_df['sii']\nX_test = test_df.drop(['id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:50.257426Z","iopub.execute_input":"2024-09-30T15:22:50.257870Z","iopub.status.idle":"2024-09-30T15:22:50.267379Z","shell.execute_reply.started":"2024-09-30T15:22:50.257822Z","shell.execute_reply":"2024-09-30T15:22:50.266028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_to_delete = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n       'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n       'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10',\n       'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n       'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n       'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']\n\nX.drop(col_to_delete,axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:50.460527Z","iopub.execute_input":"2024-09-30T15:22:50.460982Z","iopub.status.idle":"2024-09-30T15:22:50.469581Z","shell.execute_reply.started":"2024-09-30T15:22:50.460941Z","shell.execute_reply":"2024-09-30T15:22:50.468314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined = pd.concat([X, X_test], keys=['train', 'test'])","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:50.677135Z","iopub.execute_input":"2024-09-30T15:22:50.677576Z","iopub.status.idle":"2024-09-30T15:22:50.688662Z","shell.execute_reply.started":"2024-09-30T15:22:50.677538Z","shell.execute_reply":"2024-09-30T15:22:50.687270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical = combined.select_dtypes(include='object').columns\nnumerical = combined.select_dtypes(include=['int64','float64']).columns","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:50.843780Z","iopub.execute_input":"2024-09-30T15:22:50.844284Z","iopub.status.idle":"2024-09-30T15:22:50.854552Z","shell.execute_reply.started":"2024-09-30T15:22:50.844242Z","shell.execute_reply":"2024-09-30T15:22:50.853367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in categorical:\n    if combined[col].isna().sum() > 0:\n        mode = combined[col].mode()[0]\n        combined[col].fillna(mode, inplace=True)\n\ndef encoding(variables):\n    encoders = {}\n    for var in variables:\n        le = LabelEncoder()\n        combined[var] = le.fit_transform(combined[var])\n        encoders[var] = le\n    return encoders\n\nencoders = encoding(categorical)\n\nfor col in numerical:\n    if combined[col].isna().sum() > 0:\n        median = combined[col].mean()\n        combined[col].fillna(median, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:51.068186Z","iopub.execute_input":"2024-09-30T15:22:51.068667Z","iopub.status.idle":"2024-09-30T15:22:51.132784Z","shell.execute_reply.started":"2024-09-30T15:22:51.068609Z","shell.execute_reply":"2024-09-30T15:22:51.131502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = combined.xs('train')\nX_test = combined.xs('test')","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:51.323248Z","iopub.execute_input":"2024-09-30T15:22:51.323717Z","iopub.status.idle":"2024-09-30T15:22:51.334619Z","shell.execute_reply.started":"2024-09-30T15:22:51.323674Z","shell.execute_reply":"2024-09-30T15:22:51.333300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_objective(trial):\n    \n    params = {\n        'objective': 'multiclass',\n        'metric': 'multi_logloss',\n        'boosting_type': 'gbdt',\n        'num_class': 4,\n        'verbosity': -1,\n        'learning_rate': trial.suggest_float('learning_rate', 0.001, 0.1, log=True),\n        'n_estimators': trial.suggest_int('n_estimators', 20, 2000),\n        'max_depth': trial.suggest_int('max_depth', 1, 50),\n        'num_leaves': trial.suggest_int('num_leaves', 32, 2048),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n        'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.6, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n    }\n\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    qwk_scores = []\n\n    for train_idx, val_idx in skf.split(X, y):\n        X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n\n        train_data = lgb.Dataset(X_train_fold, label=y_train_fold)\n        val_data = lgb.Dataset(X_val_fold, label=y_val_fold, reference=train_data)\n\n        model = lgb.train(params, train_data, valid_sets=[val_data])\n\n        y_pred = model.predict(X_val_fold)\n        y_pred_labels = y_pred.argmax(axis=1)\n\n        qwk = cohen_kappa_score(y_val_fold, y_pred_labels, weights='quadratic')\n        qwk_scores.append(qwk)\n\n    mean_qwk = sum(qwk_scores) / len(qwk_scores)\n    return mean_qwk  ","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:51.567107Z","iopub.execute_input":"2024-09-30T15:22:51.567519Z","iopub.status.idle":"2024-09-30T15:22:51.581246Z","shell.execute_reply.started":"2024-09-30T15:22:51.567483Z","shell.execute_reply":"2024-09-30T15:22:51.579880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_OPTUNA:\n    lgb_study = optuna.create_study(direction='maximize') \n    lgb_study.optimize(lgb_objective, n_trials=50)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:52.238796Z","iopub.execute_input":"2024-09-30T15:22:52.239286Z","iopub.status.idle":"2024-09-30T15:22:52.245542Z","shell.execute_reply.started":"2024-09-30T15:22:52.239243Z","shell.execute_reply":"2024-09-30T15:22:52.244094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"basic = {'objective': 'multiclass',\n        'metric': 'multi_logloss',\n        'boosting_type': 'gbdt',\n        'num_class': 4,\n        'verbosity': -1}\n\nparams = {'learning_rate': 0.052844122375780005, 'n_estimators': 1988, 'max_depth': 31, 'num_leaves': 632, 'feature_fraction': 0.8861326280057349, 'bagging_fraction': 0.6031319034359963, 'bagging_freq': 8, 'lambda_l1': 9.069172656380543, 'lambda_l2': 0.06989596654622207}\n\nparams.update(basic)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:53.453559Z","iopub.execute_input":"2024-09-30T15:22:53.454020Z","iopub.status.idle":"2024-09-30T15:22:53.462032Z","shell.execute_reply.started":"2024-09-30T15:22:53.453978Z","shell.execute_reply":"2024-09-30T15:22:53.460563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_model = lgb.LGBMRegressor(**params, verbose=-1)\nlgb_model.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:22:55.255373Z","iopub.execute_input":"2024-09-30T15:22:55.256296Z","iopub.status.idle":"2024-09-30T15:23:07.684736Z","shell.execute_reply.started":"2024-09-30T15:22:55.256250Z","shell.execute_reply":"2024-09-30T15:23:07.683527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_lgb = lgb_model.predict(X_test)\ny_pred_lgb = y_pred_lgb.argmax(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:23:18.803117Z","iopub.execute_input":"2024-09-30T15:23:18.803903Z","iopub.status.idle":"2024-09-30T15:23:18.828659Z","shell.execute_reply.started":"2024-09-30T15:23:18.803856Z","shell.execute_reply":"2024-09-30T15:23:18.827480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id': id_column, 'sii': y_pred_lgb})\n\nfile_path = 'submission.csv'\n\nsubmission.to_csv(file_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T15:23:19.626664Z","iopub.execute_input":"2024-09-30T15:23:19.627105Z","iopub.status.idle":"2024-09-30T15:23:19.638469Z","shell.execute_reply.started":"2024-09-30T15:23:19.627065Z","shell.execute_reply":"2024-09-30T15:23:19.637196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}