{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-06T09:43:05.879062Z","iopub.execute_input":"2024-10-06T09:43:05.879537Z","iopub.status.idle":"2024-10-06T09:43:06.876066Z","shell.execute_reply.started":"2024-10-06T09:43:05.879490Z","shell.execute_reply":"2024-10-06T09:43:06.874800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:06.878430Z","iopub.execute_input":"2024-10-06T09:43:06.879546Z","iopub.status.idle":"2024-10-06T09:43:06.940487Z","shell.execute_reply.started":"2024-10-06T09:43:06.879479Z","shell.execute_reply":"2024-10-06T09:43:06.939447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:07.044856Z","iopub.execute_input":"2024-10-06T09:43:07.045740Z","iopub.status.idle":"2024-10-06T09:43:07.053458Z","shell.execute_reply.started":"2024-10-06T09:43:07.045688Z","shell.execute_reply":"2024-10-06T09:43:07.051849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_COLS = [\n    \"PCIAT-Season\",\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",    \n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n    \"PCIAT-PCIAT_Total\"\n]","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:07.859949Z","iopub.execute_input":"2024-10-06T09:43:07.861001Z","iopub.status.idle":"2024-10-06T09:43:07.867633Z","shell.execute_reply.started":"2024-10-06T09:43:07.860950Z","shell.execute_reply":"2024-10-06T09:43:07.866213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_df.drop(TARGET_COLS,axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:08.653425Z","iopub.execute_input":"2024-10-06T09:43:08.653882Z","iopub.status.idle":"2024-10-06T09:43:08.661816Z","shell.execute_reply.started":"2024-10-06T09:43:08.653837Z","shell.execute_reply":"2024-10-06T09:43:08.660612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:08.936185Z","iopub.execute_input":"2024-10-06T09:43:08.936673Z","iopub.status.idle":"2024-10-06T09:43:08.979721Z","shell.execute_reply.started":"2024-10-06T09:43:08.936625Z","shell.execute_reply":"2024-10-06T09:43:08.978450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nids = test_df['id']","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:28.064471Z","iopub.execute_input":"2024-10-06T09:43:28.064935Z","iopub.status.idle":"2024-10-06T09:43:28.077687Z","shell.execute_reply.started":"2024-10-06T09:43:28.064890Z","shell.execute_reply":"2024-10-06T09:43:28.076473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, LabelEncoder\n\ndef preprocess_data(df,train_data=False):\n    # Handle numerical columns\n    scaler = StandardScaler()\n    num_cols = df.select_dtypes(include=np.number).columns\n    df[num_cols] = df[num_cols].fillna(df[num_cols].median())\n    \n    # Handle categorical columns\n    cat_cols = df.select_dtypes(include='object').columns\n    for col in cat_cols:\n        df[col] = df[col].fillna(df[col].mode()[0])  # Fill missing with the most frequent value\n        df[col] = LabelEncoder().fit_transform(df[col].astype(str))\n    \n    if train_data:\n        y = list(df['sii'])\n        X = scaler.fit_transform(df.drop(['sii'],axis=1))\n        return X,y\n    scaled_df = scaler.fit_transform(df)\n    \n    \n    return scaled_df","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:32.401246Z","iopub.execute_input":"2024-10-06T09:43:32.401745Z","iopub.status.idle":"2024-10-06T09:43:32.414400Z","shell.execute_reply.started":"2024-10-06T09:43:32.401698Z","shell.execute_reply":"2024-10-06T09:43:32.413057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX,y = preprocess_data(train_data,train_data=True)\ntest_data = preprocess_data(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:33.853989Z","iopub.execute_input":"2024-10-06T09:43:33.854467Z","iopub.status.idle":"2024-10-06T09:43:34.010080Z","shell.execute_reply.started":"2024-10-06T09:43:33.854423Z","shell.execute_reply":"2024-10-06T09:43:34.008946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nmy_model = XGBClassifier()\n# Add silent=True to avoid printing out updates with each cycle\nmy_model.fit(X, y, verbose=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:34.593720Z","iopub.execute_input":"2024-10-06T09:43:34.594825Z","iopub.status.idle":"2024-10-06T09:43:35.651250Z","shell.execute_reply.started":"2024-10-06T09:43:34.594755Z","shell.execute_reply":"2024-10-06T09:43:35.650259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_result = my_model.predict(test_data).astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:42.149626Z","iopub.execute_input":"2024-10-06T09:43:42.150102Z","iopub.status.idle":"2024-10-06T09:43:42.161367Z","shell.execute_reply.started":"2024-10-06T09:43:42.150059Z","shell.execute_reply":"2024-10-06T09:43:42.160349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.DataFrame(columns=['id','sii'])\nsample_submission['id'] = ids\nsample_submission['sii'] = test_result","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:52.861377Z","iopub.execute_input":"2024-10-06T09:43:52.861850Z","iopub.status.idle":"2024-10-06T09:43:52.870165Z","shell.execute_reply.started":"2024-10-06T09:43:52.861806Z","shell.execute_reply":"2024-10-06T09:43:52.868814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:53.946578Z","iopub.execute_input":"2024-10-06T09:43:53.947663Z","iopub.status.idle":"2024-10-06T09:43:53.960263Z","shell.execute_reply.started":"2024-10-06T09:43:53.947610Z","shell.execute_reply":"2024-10-06T09:43:53.958787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-06T09:43:57.400045Z","iopub.execute_input":"2024-10-06T09:43:57.401002Z","iopub.status.idle":"2024-10-06T09:43:57.408180Z","shell.execute_reply.started":"2024-10-06T09:43:57.400952Z","shell.execute_reply":"2024-10-06T09:43:57.406889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}