{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T11:07:21.656898Z","iopub.execute_input":"2024-10-16T11:07:21.657387Z","iopub.status.idle":"2024-10-16T11:07:22.573333Z","shell.execute_reply.started":"2024-10-16T11:07:21.657343Z","shell.execute_reply":"2024-10-16T11:07:22.571934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modules**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split \nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler \nimport xgboost as xgb\nfrom sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:31.777080Z","iopub.execute_input":"2024-10-16T11:07:31.777542Z","iopub.status.idle":"2024-10-16T11:07:31.783247Z","shell.execute_reply.started":"2024-10-16T11:07:31.777498Z","shell.execute_reply":"2024-10-16T11:07:31.782035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load data and get_summary**","metadata":{}},{"cell_type":"code","source":"%%time\n#submission path\nsubmits = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n#training path\ntrn = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n#testing path\ntst = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nclass get_summary:\n    def __init__(self, x):\n        self.x = x\n    def data_set(self):\n        #checks for duplicate\n        duplicate = self.x.duplicated().any()\n        #drop duplicates \n        if duplicate == True:\n            self.x.drop_duplicates(inplace=True)\n            self.x.reset_index(drop=True)\n        #checks for empty values\n        null = self.x.isna().sum().any()\n        #missing values\n        total_missing = self.x.isnull().sum().sum()\n        #data types\n        data_type = self.x.dtypes\n        #shape\n        shapes = self.x.shape\n        return f\"Duplicate: {duplicate}\\nNull: {null}\\nMissing_value: {total_missing}\\nTypes:\\n{data_type}\\nShape: {shapes}\"\n    \nprint(f\"Training dataset:\\n{get_summary(trn).data_set()}\\nTest dataset:\\n{get_summary(tst).data_set()}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:36.996729Z","iopub.execute_input":"2024-10-16T11:07:36.997169Z","iopub.status.idle":"2024-10-16T11:07:37.095239Z","shell.execute_reply.started":"2024-10-16T11:07:36.997127Z","shell.execute_reply":"2024-10-16T11:07:37.093912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:41.205441Z","iopub.execute_input":"2024-10-16T11:07:41.205902Z","iopub.status.idle":"2024-10-16T11:07:41.233199Z","shell.execute_reply.started":"2024-10-16T11:07:41.205860Z","shell.execute_reply":"2024-10-16T11:07:41.231960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data cleaning and engineering**","metadata":{}},{"cell_type":"code","source":"%%time\ndef find_missing(x):\n    missing = trn.isnull().sum()\n    missing_cols = missing[missing > 0].to_dict()\n    \n    print(f\"total of {len(missing_cols)} columns contains {sum(missing)} missing values\")\n    return missing_cols\nfind_missing(trn)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:48.165594Z","iopub.execute_input":"2024-10-16T11:07:48.166067Z","iopub.status.idle":"2024-10-16T11:07:48.183357Z","shell.execute_reply.started":"2024-10-16T11:07:48.166023Z","shell.execute_reply":"2024-10-16T11:07:48.181886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thats a whole lot of missing values","metadata":{}},{"cell_type":"code","source":"tst.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:52.914171Z","iopub.execute_input":"2024-10-16T11:07:52.915498Z","iopub.status.idle":"2024-10-16T11:07:52.940948Z","shell.execute_reply.started":"2024-10-16T11:07:52.915442Z","shell.execute_reply":"2024-10-16T11:07:52.939746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#dropping the columns that are not in tst\ntrn_copy = [col for col in trn.columns if col in tst.columns]\n#adding the target column 'sii'\ntrain = pd.concat([trn[trn_copy], trn[['sii']]], axis=1)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:54.914235Z","iopub.execute_input":"2024-10-16T11:07:54.914756Z","iopub.status.idle":"2024-10-16T11:07:54.929978Z","shell.execute_reply.started":"2024-10-16T11:07:54.914704Z","shell.execute_reply":"2024-10-16T11:07:54.928827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:07:56.994428Z","iopub.execute_input":"2024-10-16T11:07:56.994897Z","iopub.status.idle":"2024-10-16T11:07:57.021434Z","shell.execute_reply.started":"2024-10-16T11:07:56.994854Z","shell.execute_reply":"2024-10-16T11:07:57.020112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the datasets is balance, now to some engineering...","metadata":{}},{"cell_type":"code","source":"%%time\n#this function fills any missing values in the dataframe.\ndef fill_missing(data):\n    #columns with missing values\n    empty_cols = [i for i in data.columns[data.isna().any()]]\n    for col in empty_cols:\n        #check if the column is object\n        if data[col].dtype == 'object':\n            data[col] = data[col].fillna('Unknown')\n        #check if the column is int or float\n        elif data[col].dtype in ['int', 'float']:\n            col_mean = data[col].mean()\n            data[col] = data[col].fillna(col_mean)\n\nfill_missing(train)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:01.255004Z","iopub.execute_input":"2024-10-16T11:08:01.256441Z","iopub.status.idle":"2024-10-16T11:08:01.300517Z","shell.execute_reply.started":"2024-10-16T11:08:01.256353Z","shell.execute_reply":"2024-10-16T11:08:01.298936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fill_missing(tst)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:04.254407Z","iopub.execute_input":"2024-10-16T11:08:04.254911Z","iopub.status.idle":"2024-10-16T11:08:04.287487Z","shell.execute_reply.started":"2024-10-16T11:08:04.254862Z","shell.execute_reply":"2024-10-16T11:08:04.286235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{train.head(3)}\\n{train.isna().sum().any()}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:06.705713Z","iopub.execute_input":"2024-10-16T11:08:06.706232Z","iopub.status.idle":"2024-10-16T11:08:06.734395Z","shell.execute_reply.started":"2024-10-16T11:08:06.706184Z","shell.execute_reply":"2024-10-16T11:08:06.733038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{tst.head(3)}\\n{tst.isna().sum().any()}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:08.936041Z","iopub.execute_input":"2024-10-16T11:08:08.937319Z","iopub.status.idle":"2024-10-16T11:08:08.958960Z","shell.execute_reply.started":"2024-10-16T11:08:08.937238Z","shell.execute_reply":"2024-10-16T11:08:08.957469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preprocessing**","metadata":{}},{"cell_type":"code","source":"def make_integer(train):\n    make_int = [val for val in train.columns if train[val].dtype == 'float']\n    train[make_int] = train[make_int].astype(int)\n    return train.head(2)\nmake_integer(train)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:16.275387Z","iopub.execute_input":"2024-10-16T11:08:16.275849Z","iopub.status.idle":"2024-10-16T11:08:16.316137Z","shell.execute_reply.started":"2024-10-16T11:08:16.275807Z","shell.execute_reply":"2024-10-16T11:08:16.314683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"make_integer(tst)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:19.035651Z","iopub.execute_input":"2024-10-16T11:08:19.036108Z","iopub.status.idle":"2024-10-16T11:08:19.075308Z","shell.execute_reply.started":"2024-10-16T11:08:19.036068Z","shell.execute_reply":"2024-10-16T11:08:19.073803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoder\nenc = OrdinalEncoder()\n#encode function\ndef encode(w):\n    for cols in w.columns:\n        #encoding all object types\n        if w[cols].dtype == 'object':\n            w[cols] = enc.fit_transform(w[[cols]])\n    return w.head(4)\n    \nencode(train)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:21.055255Z","iopub.execute_input":"2024-10-16T11:08:21.055777Z","iopub.status.idle":"2024-10-16T11:08:21.130139Z","shell.execute_reply.started":"2024-10-16T11:08:21.055731Z","shell.execute_reply":"2024-10-16T11:08:21.128777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode(tst)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:23.704373Z","iopub.execute_input":"2024-10-16T11:08:23.704787Z","iopub.status.idle":"2024-10-16T11:08:23.752737Z","shell.execute_reply.started":"2024-10-16T11:08:23.704747Z","shell.execute_reply":"2024-10-16T11:08:23.751376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split and scale**","metadata":{}},{"cell_type":"code","source":"X = train.drop(['id', 'sii'], axis=1)\ny = train['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:30.195590Z","iopub.execute_input":"2024-10-16T11:08:30.196049Z","iopub.status.idle":"2024-10-16T11:08:30.207073Z","shell.execute_reply.started":"2024-10-16T11:08:30.196007Z","shell.execute_reply":"2024-10-16T11:08:30.205727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = tst.drop('id', axis=1)\nX_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:35.015554Z","iopub.execute_input":"2024-10-16T11:08:35.016049Z","iopub.status.idle":"2024-10-16T11:08:35.027616Z","shell.execute_reply.started":"2024-10-16T11:08:35.016001Z","shell.execute_reply":"2024-10-16T11:08:35.026331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:37.334638Z","iopub.execute_input":"2024-10-16T11:08:37.335117Z","iopub.status.idle":"2024-10-16T11:08:37.344934Z","shell.execute_reply.started":"2024-10-16T11:08:37.335070Z","shell.execute_reply":"2024-10-16T11:08:37.343590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale = StandardScaler()\nX = pd.DataFrame(scale.fit_transform(X))\nX.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:40.125698Z","iopub.execute_input":"2024-10-16T11:08:40.126708Z","iopub.status.idle":"2024-10-16T11:08:40.162797Z","shell.execute_reply.started":"2024-10-16T11:08:40.126655Z","shell.execute_reply":"2024-10-16T11:08:40.161547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.DataFrame(scale.fit_transform(X_test))\nX_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:46.711463Z","iopub.execute_input":"2024-10-16T11:08:46.712196Z","iopub.status.idle":"2024-10-16T11:08:46.740851Z","shell.execute_reply.started":"2024-10-16T11:08:46.712150Z","shell.execute_reply":"2024-10-16T11:08:46.739779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.25, shuffle=True, random_state=2)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:08:49.014546Z","iopub.execute_input":"2024-10-16T11:08:49.015737Z","iopub.status.idle":"2024-10-16T11:08:49.025999Z","shell.execute_reply.started":"2024-10-16T11:08:49.015684Z","shell.execute_reply":"2024-10-16T11:08:49.024790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**models**","metadata":{}},{"cell_type":"code","source":"parameters = {'max_depth' : 5,\n              'n_estimators' : 100,\n              'learning_rate' : 0.05,\n              'eta' : 1.0,\n              'n_jobs' : -1,\n              'objective' : 'Multiclass'}\n\nxgb_model = xgb.XGBClassifier(verbosity=0, **parameters, random_state=2, tree_method='hist', device='cuda')\nxgb_model.fit(X_train, y_train)\n\ny_pred = xgb_model.predict(X_val)\nkappa = cohen_kappa_score(y_val, y_pred, weights='quadratic')\nprint(f\"kappa_score: {kappa:.2f}%\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:09:26.412561Z","iopub.execute_input":"2024-10-16T11:09:26.413004Z","iopub.status.idle":"2024-10-16T11:09:27.099530Z","shell.execute_reply.started":"2024-10-16T11:09:26.412961Z","shell.execute_reply":"2024-10-16T11:09:27.098470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test**","metadata":{}},{"cell_type":"code","source":"predicts = xgb_model.predict(X_test)\npredicts","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:09:37.255173Z","iopub.execute_input":"2024-10-16T11:09:37.255667Z","iopub.status.idle":"2024-10-16T11:09:37.277932Z","shell.execute_reply.started":"2024-10-16T11:09:37.255622Z","shell.execute_reply":"2024-10-16T11:09:37.276967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submits\nsubmission['sii'] = predicts\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T11:09:39.195412Z","iopub.execute_input":"2024-10-16T11:09:39.195966Z","iopub.status.idle":"2024-10-16T11:09:39.204619Z","shell.execute_reply.started":"2024-10-16T11:09:39.195919Z","shell.execute_reply":"2024-10-16T11:09:39.203245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}