{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nimport re\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nimport lightgbm as lgb\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score, make_scorer\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.870484Z","iopub.execute_input":"2024-10-05T11:09:43.871198Z","iopub.status.idle":"2024-10-05T11:09:43.876970Z","shell.execute_reply.started":"2024-10-05T11:09:43.871158Z","shell.execute_reply":"2024-10-05T11:09:43.875962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load train data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.890728Z","iopub.execute_input":"2024-10-05T11:09:43.891206Z","iopub.status.idle":"2024-10-05T11:09:43.955943Z","shell.execute_reply.started":"2024-10-05T11:09:43.891174Z","shell.execute_reply":"2024-10-05T11:09:43.954931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove id column\ntrain = train.drop(columns = ['id'])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.958328Z","iopub.execute_input":"2024-10-05T11:09:43.958664Z","iopub.status.idle":"2024-10-05T11:09:43.963652Z","shell.execute_reply.started":"2024-10-05T11:09:43.958632Z","shell.execute_reply":"2024-10-05T11:09:43.962775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop all PCIAT cols\npciat_cols = [col for col in train.columns if re.match('^(PCIAT)', col)]\ntrain = train.drop(columns = pciat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.964821Z","iopub.execute_input":"2024-10-05T11:09:43.965119Z","iopub.status.idle":"2024-10-05T11:09:43.974539Z","shell.execute_reply.started":"2024-10-05T11:09:43.965088Z","shell.execute_reply":"2024-10-05T11:09:43.973694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove rows where sii is null\ntrain = train[~train['sii'].isnull()]\n\n# print train shape\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.976709Z","iopub.execute_input":"2024-10-05T11:09:43.977036Z","iopub.status.idle":"2024-10-05T11:09:43.986105Z","shell.execute_reply.started":"2024-10-05T11:09:43.977003Z","shell.execute_reply":"2024-10-05T11:09:43.985069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change dtype of sii column as object\ntrain['sii'] = train['sii'].astype('object')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.987613Z","iopub.execute_input":"2024-10-05T11:09:43.988039Z","iopub.status.idle":"2024-10-05T11:09:43.994519Z","shell.execute_reply.started":"2024-10-05T11:09:43.987997Z","shell.execute_reply":"2024-10-05T11:09:43.993807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify and fill categorical columns\nimputer = SimpleImputer(strategy='most_frequent')\ncat_cols = list(train.select_dtypes(include='object').columns)\ntrain[cat_cols] = imputer.fit_transform(train[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:43.995749Z","iopub.execute_input":"2024-10-05T11:09:43.996153Z","iopub.status.idle":"2024-10-05T11:09:44.016835Z","shell.execute_reply.started":"2024-10-05T11:09:43.996114Z","shell.execute_reply":"2024-10-05T11:09:44.015774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify and fill numerical columns\nimputer = SimpleImputer(strategy='mean')\nnum_cols = list(train.select_dtypes(exclude='object').columns)\ntrain[num_cols] = imputer.fit_transform(train[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.018076Z","iopub.execute_input":"2024-10-05T11:09:44.018496Z","iopub.status.idle":"2024-10-05T11:09:44.037504Z","shell.execute_reply.started":"2024-10-05T11:09:44.018448Z","shell.execute_reply":"2024-10-05T11:09:44.036678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale all numeric columns\nscaler = StandardScaler()\ntrain[num_cols] = scaler.fit_transform(train[num_cols])\ntrain[num_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.038712Z","iopub.execute_input":"2024-10-05T11:09:44.039022Z","iopub.status.idle":"2024-10-05T11:09:44.165002Z","shell.execute_reply.started":"2024-10-05T11:09:44.038991Z","shell.execute_reply":"2024-10-05T11:09:44.163985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode categorical columns\nencoder = LabelEncoder()\nfor col in cat_cols:\n    train[col] = encoder.fit_transform(train[col])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.168216Z","iopub.execute_input":"2024-10-05T11:09:44.168680Z","iopub.status.idle":"2024-10-05T11:09:44.183683Z","shell.execute_reply.started":"2024-10-05T11:09:44.168646Z","shell.execute_reply":"2024-10-05T11:09:44.182799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split into train and validation\ntrain_data, val_data = train_test_split(train, train_size=0.8, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.184649Z","iopub.execute_input":"2024-10-05T11:09:44.184947Z","iopub.status.idle":"2024-10-05T11:09:44.198620Z","shell.execute_reply.started":"2024-10-05T11:09:44.184908Z","shell.execute_reply":"2024-10-05T11:09:44.197910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define X and y\nX_train = train_data.drop(columns = ['sii'])\ny_train = train_data['sii']\n\nX_val = val_data.drop(columns = ['sii'])\ny_val = val_data['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.199817Z","iopub.execute_input":"2024-10-05T11:09:44.200124Z","iopub.status.idle":"2024-10-05T11:09:44.212590Z","shell.execute_reply.started":"2024-10-05T11:09:44.200066Z","shell.execute_reply":"2024-10-05T11:09:44.211709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # define lgb model\n\n# # define hyperparameters\n# params = {\n#     'learning_rate': [0.01, 0.02, 0.03, 0.04, 0.05],\n#     'n_estimators': [200, 210, 220, 230, 240, 250],\n#     'num_leaves': [50, 60, 70, 80, 90, 100],  \n#     'max_depth': [4, 6, 8, 10],  \n#     'min_child_samples': [30, 35, 40, 45, 50],  \n#     'subsample': [0.5, 0.6, 0.7, 0.8],\n#     'colsample_bytree': np.linspace(0.5, 0.7, 3),\n#     'reg_alpha': [0.5, 1.0, 1.5, 2.0, 2.5],\n#     'reg_lambda': [0.01, 0.02],\n#     'device': ['gpu'],  \n\n# }\n\n# # define model\n# model = lgb.LGBMClassifier(verbosity=-1)\n\n# # define kappa scorer\n# kappa_scorer = make_scorer(cohen_kappa_score)\n\n# # define grid search\n# grid_search = GridSearchCV(estimator=model, cv=3, param_grid=params, n_jobs=-1, \n#                            scoring=kappa_scorer, verbose=1)\n\n# # fit the data\n# grid_search.fit(X_train, y_train)\n\n# best_params = grid_search.best_params_\n# best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.213733Z","iopub.execute_input":"2024-10-05T11:09:44.214008Z","iopub.status.idle":"2024-10-05T11:09:44.219197Z","shell.execute_reply.started":"2024-10-05T11:09:44.213978Z","shell.execute_reply":"2024-10-05T11:09:44.218351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define best model\nbest_params = {\n 'subsample': 0.7,\n 'reg_lambda': 0.01,\n 'reg_alpha': 1.25,\n 'num_leaves': 60,\n 'n_estimators': 230,\n 'min_child_samples': 45, \n 'max_depth': 4,\n 'learning_rate': 0.05,\n 'device': 'gpu',\n 'colsample_bytree': 0.7}\nmodel = lgb.LGBMClassifier(**best_params, verbosity=-1)\n\n# train model\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:44.220152Z","iopub.execute_input":"2024-10-05T11:09:44.220428Z","iopub.status.idle":"2024-10-05T11:09:45.199761Z","shell.execute_reply.started":"2024-10-05T11:09:44.220397Z","shell.execute_reply":"2024-10-05T11:09:45.198711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores on validation data\ny_val_pred = model.predict(X_val)\nprint('accuracy score:', accuracy_score(y_val, y_val_pred))\nprint('kappa score:', cohen_kappa_score(y_val, y_val_pred, weights='quadratic'))","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:45.201385Z","iopub.execute_input":"2024-10-05T11:09:45.201779Z","iopub.status.idle":"2024-10-05T11:09:45.234795Z","shell.execute_reply.started":"2024-10-05T11:09:45.201734Z","shell.execute_reply":"2024-10-05T11:09:45.233769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define X and y\nX = train.drop(columns = ['sii'])\ny = train['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:45.236620Z","iopub.execute_input":"2024-10-05T11:09:45.237008Z","iopub.status.idle":"2024-10-05T11:09:45.245689Z","shell.execute_reply.started":"2024-10-05T11:09:45.236964Z","shell.execute_reply":"2024-10-05T11:09:45.244673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train model on complete dataset\nmodel = lgb.LGBMClassifier(**best_params, verbosity=-1)\n\n# train model\nmodel.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:45.250398Z","iopub.execute_input":"2024-10-05T11:09:45.250722Z","iopub.status.idle":"2024-10-05T11:09:46.545286Z","shell.execute_reply.started":"2024-10-05T11:09:45.250684Z","shell.execute_reply":"2024-10-05T11:09:46.544265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions on complete  train dataset\ny_pred = model.predict(X)\nprint('accuracy score:', accuracy_score(y, y_pred))\nprint('kappa score:', cohen_kappa_score(y, y_pred, weights='quadratic'))","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.547056Z","iopub.execute_input":"2024-10-05T11:09:46.547455Z","iopub.status.idle":"2024-10-05T11:09:46.655856Z","shell.execute_reply.started":"2024-10-05T11:09:46.547410Z","shell.execute_reply":"2024-10-05T11:09:46.655020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# process test data\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntest.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.657195Z","iopub.execute_input":"2024-10-05T11:09:46.658133Z","iopub.status.idle":"2024-10-05T11:09:46.687198Z","shell.execute_reply.started":"2024-10-05T11:09:46.658093Z","shell.execute_reply":"2024-10-05T11:09:46.686406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare test data\ntest_data = test[X_train.columns]","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.688340Z","iopub.execute_input":"2024-10-05T11:09:46.688911Z","iopub.status.idle":"2024-10-05T11:09:46.693058Z","shell.execute_reply.started":"2024-10-05T11:09:46.688873Z","shell.execute_reply":"2024-10-05T11:09:46.692357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify and fill categorical columns\nimputer = SimpleImputer(strategy='most_frequent')\ncat_cols = list(test_data.select_dtypes(include='object').columns)\ntest_data[cat_cols] = imputer.fit_transform(test_data[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.694181Z","iopub.execute_input":"2024-10-05T11:09:46.694808Z","iopub.status.idle":"2024-10-05T11:09:46.711859Z","shell.execute_reply.started":"2024-10-05T11:09:46.694770Z","shell.execute_reply":"2024-10-05T11:09:46.710942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify and fill numerical columns\nimputer = SimpleImputer(strategy='mean')\nnum_cols = list(test_data.select_dtypes(exclude='object').columns)\ntest_data[num_cols] = imputer.fit_transform(test_data[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.714525Z","iopub.execute_input":"2024-10-05T11:09:46.715446Z","iopub.status.idle":"2024-10-05T11:09:46.737046Z","shell.execute_reply.started":"2024-10-05T11:09:46.715408Z","shell.execute_reply":"2024-10-05T11:09:46.735814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale numerical cols of test data\ntest_data[num_cols] = scaler.transform(test[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.740294Z","iopub.execute_input":"2024-10-05T11:09:46.740813Z","iopub.status.idle":"2024-10-05T11:09:46.752286Z","shell.execute_reply.started":"2024-10-05T11:09:46.740777Z","shell.execute_reply":"2024-10-05T11:09:46.751610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode categorical columns\nencoder = LabelEncoder()\nfor col in cat_cols:\n    test_data[col] = encoder.fit_transform(test_data[col])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.753552Z","iopub.execute_input":"2024-10-05T11:09:46.754195Z","iopub.status.idle":"2024-10-05T11:09:46.763774Z","shell.execute_reply.started":"2024-10-05T11:09:46.754154Z","shell.execute_reply":"2024-10-05T11:09:46.762985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions on test data\ntest_pred = model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.764965Z","iopub.execute_input":"2024-10-05T11:09:46.765511Z","iopub.status.idle":"2024-10-05T11:09:46.776293Z","shell.execute_reply.started":"2024-10-05T11:09:46.765475Z","shell.execute_reply":"2024-10-05T11:09:46.775423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare submit df\nsubmit_df = pd.DataFrame({\n    'id': test['id'],\n    'sii': test_pred\n})\n\nsubmit_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T11:09:46.777653Z","iopub.execute_input":"2024-10-05T11:09:46.778281Z","iopub.status.idle":"2024-10-05T11:09:46.786646Z","shell.execute_reply.started":"2024-10-05T11:09:46.778236Z","shell.execute_reply":"2024-10-05T11:09:46.785637Z"},"trusted":true},"execution_count":null,"outputs":[]}]}