{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport warnings\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\n\nfrom scipy.optimize import minimize\n\nfrom xgboost import XGBRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-19T18:22:25.315268Z","iopub.execute_input":"2024-10-19T18:22:25.315721Z","iopub.status.idle":"2024-10-19T18:22:25.322101Z","shell.execute_reply.started":"2024-10-19T18:22:25.315675Z","shell.execute_reply":"2024-10-19T18:22:25.320817Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:22:26.672024Z","iopub.execute_input":"2024-10-19T18:22:26.672476Z","iopub.status.idle":"2024-10-19T18:22:26.677539Z","shell.execute_reply.started":"2024-10-19T18:22:26.672433Z","shell.execute_reply":"2024-10-19T18:22:26.676192Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    TRAIN_CSV = \"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\"\n    TEST_CSV = \"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\"\n    DATA_DICTIONARY = \"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\"\n    \n    FEATURE_COLS = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score',\n                'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n                'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n                'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',\n                'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday']\n    \n    CATEGORY_COLS = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season',\n         'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n    GROUD_TRUTH_COL = 'sii'\n    SPLITS = 20\n    MODEL_PARAMS = {}\n    MODEL = \"xgb\"","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:41:14.939566Z","iopub.execute_input":"2024-10-19T18:41:14.940078Z","iopub.status.idle":"2024-10-19T18:41:14.948883Z","shell.execute_reply.started":"2024-10-19T18:41:14.940027Z","shell.execute_reply":"2024-10-19T18:41:14.947542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Preprocessor:\n    def __init__(self, config: Config):\n        self._validate_config(config)\n        self.CONFIG = config\n        self.TRAIN_DF = pd.read_csv(self.CONFIG.TRAIN_CSV)\n        self.TEST_DF = pd.read_csv(self.CONFIG.TEST_CSV)\n        self.COLUMNS = self.TRAIN_DF.columns\n        self.TRAIN_DATA = None\n        self.TEST_DATA = None\n        \n    def get_train_df(self):\n        return self.TRAIN_DF\n    \n    def get_test_df(self):\n        return self.TEST_DF\n    \n    def processed_data(self):\n        self.TRAIN_DATA = self.TRAIN_DF.copy(deep=True)\n        self.TEST_DATA = self.TEST_DF.copy(deep=True)\n        \n        self._fill_missing_category_value()\n        self._encode_categorical_cols()\n        \n        self.TRAIN_DATA = self.TRAIN_DATA.dropna(subset=[self.CONFIG.GROUD_TRUTH_COL])\n        train_ids = self.TRAIN_DATA['id']\n        train_y = self.TRAIN_DATA[self.CONFIG.GROUD_TRUTH_COL]\n        test_ids = self.TEST_DATA['id']\n\n        self.TRAIN_DATA = self.TRAIN_DATA[self.CONFIG.FEATURE_COLS]\n        self.TEST_DATA = self.TEST_DATA[self.CONFIG.FEATURE_COLS]\n        self._imputer()\n        \n        return self.TRAIN_DATA, self.TEST_DATA, train_ids.values, train_y.values, test_ids.values\n        \n    def _validate_config(self, config):\n        required_keys = ['TRAIN_CSV', 'TEST_CSV', 'DATA_DICTIONARY', 'CATEGORY_COLS']\n        for key in required_keys:\n            if key not in dir(config):\n                raise AssertionError(f\"Required config memeber: {key} not provided\")\n        assert True, \"Bad Config\"\n        \n    def create_mapping(self, column):\n        unique_values_train = self.TRAIN_DATA[column].unique()\n        unique_values_test = self.TEST_DATA[column].unique()\n\n        train_cat_dict = {value: idx for idx, value in enumerate(unique_values_train)}\n        train_cat_dict.update({value: idx for idx, value in enumerate(unique_values_test)})\n        \n        return train_cat_dict\n    \n    def _fill_missing_category_value(self):\n        for col in self.CONFIG.CATEGORY_COLS:\n            self.TRAIN_DATA[col] = self.TRAIN_DATA[col].fillna('Missing')\n            self.TRAIN_DATA[col] = self.TRAIN_DATA[col].astype('category')\n            \n            self.TEST_DATA[col] = self.TEST_DATA[col].fillna('Missing')\n            self.TEST_DATA[col] = self.TEST_DATA[col].astype('category')\n        return True\n    \n    def _encode_categorical_cols(self):\n        for col in self.CONFIG.CATEGORY_COLS:\n            mapping = self.create_mapping(col)\n            self.TRAIN_DATA[col] = self.TRAIN_DATA[col].replace(mapping).astype(int)\n            self.TEST_DATA[col] = self.TEST_DATA[col].replace(mapping).astype(int)\n            \n    def _imputer(self):\n        imputer = SimpleImputer(strategy='mean')\n        self.TRAIN_DATA = imputer.fit_transform(self.TRAIN_DATA)\n        self.TEST_DATA = imputer.transform(self.TEST_DATA)\n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:41:15.522106Z","iopub.execute_input":"2024-10-19T18:41:15.522600Z","iopub.status.idle":"2024-10-19T18:41:15.540437Z","shell.execute_reply.started":"2024-10-19T18:41:15.522552Z","shell.execute_reply":"2024-10-19T18:41:15.539130Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Model:\n    \n    def __init__(self, config: Config):\n        model, params = self._model_dispatcher(config.MODEL)\n        self.MODEL_PARAMS = {**params, **config.MODEL_PARAMS}\n        self.MODEL = model\n\n    def _model_dispatcher(self, model: str):\n        XGB_PARAMS = {\n            'learning_rate': 0.05,\n            'max_depth': 10,\n            'n_estimators': 200,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'reg_alpha': 1,\n            'reg_lambda': 5,\n            'random_state': 42,\n            'tree_method': 'hist',\n            'device': 'cuda'\n        }\n        if model == 'xgb':\n            return XGBRegressor, XGB_PARAMS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T18:41:16.052300Z","iopub.execute_input":"2024-10-19T18:41:16.052718Z","iopub.status.idle":"2024-10-19T18:41:16.060204Z","shell.execute_reply.started":"2024-10-19T18:41:16.052668Z","shell.execute_reply":"2024-10-19T18:41:16.058926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PostProcessor:\n\n    def __init__(self):\n        self.STEP = \"POST PROCESSING\"\n        self.THRESHOLDS = [0.5, 1.5, 2.5]\n\n\n    def optimize_threshold(self, y_true, y_pred_non_rounded):\n        KappaOPtimizer = minimize(self._evaluate_predictions,\n                                x0=self.THRESHOLDS, args=(y_true, y_pred_non_rounded), \n                                method='Nelder-Mead')\n\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        self.THRESHOLDS = KappaOPtimizer.x\n        \n        return True\n\n\n    def quadratic_weighted_kappa(self, y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    \n    def threshold_rounder(self, y_pred_non_rounded):\n        return np.where(y_pred_non_rounded < self.THRESHOLDS[0], 0,\n                        np.where(y_pred_non_rounded < self.THRESHOLDS[1], 1,\n                                 np.where(y_pred_non_rounded < self.THRESHOLDS[2], 2, 3)))\n    \n    # this is to optimize threshold\n    def _evaluate_predictions(self,THRESHOLDS, y_true, y_pred_non_rounded):\n        rounded_p = self.threshold_rounder(y_pred_non_rounded, THRESHOLDS)\n        return -self.quadratic_weighted_kappa(y_true, rounded_p)\n\n    def evaluate_predictions(self, y_true, y_pred_non_rounded):\n        rounded_p = self.threshold_rounder(y_pred_non_rounded)\n        return self.quadratic_weighted_kappa(y_true, rounded_p)\n\n\n    # Internal function used to optimize threshold\n    def _threshold_rounder(self, y_pred_non_rounded, THRESHOLDS):\n        return np.where(y_pred_non_rounded < THRESHOLDS[0], 0,\n                        np.where(y_pred_non_rounded < THRESHOLDS[1], 1,\n                                 np.where(y_pred_non_rounded < THRESHOLDS[2], 2, 3)))\n    \n    # this is to optimize threshold\n    def _evaluate_predictions(self,THRESHOLDS, y_true, y_pred_non_rounded):\n        rounded_p = self._threshold_rounder(y_pred_non_rounded, THRESHOLDS)\n        return -self.quadratic_weighted_kappa(y_true, rounded_p)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T18:41:16.535444Z","iopub.execute_input":"2024-10-19T18:41:16.535855Z","iopub.status.idle":"2024-10-19T18:41:16.548203Z","shell.execute_reply.started":"2024-10-19T18:41:16.535815Z","shell.execute_reply":"2024-10-19T18:41:16.546856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass Trainer(Preprocessor, Model, PostProcessor):\n    \n    def __init__(self, config: Config):\n        Preprocessor.__init__(self, config)\n        Model.__init__(self, config)\n        PostProcessor.__init__(self)\n        \n    def train_and_prepare_submission(self):\n        train_x, test_x, train_ids, train_y, test_ids = self.processed_data()\n        \n        SKF = StratifiedKFold(n_splits=self.CONFIG.SPLITS, shuffle=True, random_state=42)\n        \n        best_val_score = 0,\n        best_model = None\n\n        y_pred_non_rounded = np.zeros(len(train_y), dtype=float) \n        val_scores = []\n        \n        for fold, (train_idx, val_idx) in enumerate(SKF.split(train_x, train_y)):\n            X_train, X_val = train_x[train_idx], train_x[val_idx]\n            y_train, y_val = train_y[train_idx], train_y[val_idx]\n            \n            model = self.MODEL(**self.MODEL_PARAMS)\n            model.fit(X_train, y_train)\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            y_pred_non_rounded[val_idx] = y_val_pred\n            \n            y_val_pred_rounded = np.round(y_val_pred).astype(int)\n\n            train_kappa = self.quadratic_weighted_kappa(y_train, np.round(y_train_pred).astype(int))\n            val_kappa = self.quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n            val_scores.append(val_kappa)\n            \n            if val_kappa > best_val_score:\n                best_val_score = val_kappa\n                best_model = model\n\n            print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            \n        print(f\"Current Thresholds: {self.THRESHOLDS}\")\n        print(f\"Current QWK: {np.mean(val_scores)}\")\n        self.optimize_threshold(train_y, y_pred_non_rounded)\n        print(f\"Optimized Thresholds: {self.THRESHOLDS}\")\n        print(f\"Optimized QWK: {self.evaluate_predictions(train_y, y_pred_non_rounded)}\")\n        \n        print(\"\\n\\n\\n=======================\\n\\n\\n\")\n        test_pred = best_model.predict(test_x)\n        # test_pred_rounded = np.round(test_pred).astype(int)\n        test_pred_rounded = self.threshold_rounder(test_pred)\n        submission = pd.DataFrame({\n            'id': test_ids,\n            'sii': test_pred_rounded\n        })\n        submission.to_csv('submission.csv', index=False)\n        print(submission['sii'].value_counts())\n        return True\n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:41:17.031943Z","iopub.execute_input":"2024-10-19T18:41:17.032395Z","iopub.status.idle":"2024-10-19T18:41:17.046366Z","shell.execute_reply.started":"2024-10-19T18:41:17.032348Z","shell.execute_reply":"2024-10-19T18:41:17.044737Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"config = Config()\ntrainer = Trainer(config)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:41:17.865203Z","iopub.execute_input":"2024-10-19T18:41:17.865706Z","iopub.status.idle":"2024-10-19T18:41:17.913454Z","shell.execute_reply.started":"2024-10-19T18:41:17.865657Z","shell.execute_reply":"2024-10-19T18:41:17.912294Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.train_and_prepare_submission()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T18:41:18.993643Z","iopub.execute_input":"2024-10-19T18:41:18.994067Z","iopub.status.idle":"2024-10-19T18:42:09.054134Z","shell.execute_reply.started":"2024-10-19T18:41:18.994025Z","shell.execute_reply":"2024-10-19T18:42:09.052989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.train_and_prepare_submission()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:09:57.262072Z","iopub.execute_input":"2024-10-19T18:09:57.263129Z","iopub.status.idle":"2024-10-19T18:10:47.025738Z","shell.execute_reply.started":"2024-10-19T18:09:57.263080Z","shell.execute_reply":"2024-10-19T18:10:47.024410Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}