{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:39:38.721265Z","iopub.execute_input":"2024-10-23T09:39:38.722134Z","iopub.status.idle":"2024-10-23T09:39:43.477053Z","shell.execute_reply.started":"2024-10-23T09:39:38.722070Z","shell.execute_reply":"2024-10-23T09:39:43.475826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:39:58.511348Z","iopub.execute_input":"2024-10-23T09:39:58.511973Z","iopub.status.idle":"2024-10-23T09:39:58.517118Z","shell.execute_reply.started":"2024-10-23T09:39:58.511928Z","shell.execute_reply":"2024-10-23T09:39:58.516183Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:39:59.256407Z","iopub.execute_input":"2024-10-23T09:39:59.256839Z","iopub.status.idle":"2024-10-23T09:39:59.395224Z","shell.execute_reply.started":"2024-10-23T09:39:59.256798Z","shell.execute_reply":"2024-10-23T09:39:59.394049Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:39:59.556512Z","iopub.execute_input":"2024-10-23T09:39:59.557546Z","iopub.status.idle":"2024-10-23T09:39:59.590359Z","shell.execute_reply.started":"2024-10-23T09:39:59.557495Z","shell.execute_reply":"2024-10-23T09:39:59.589173Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:39:59.787171Z","iopub.execute_input":"2024-10-23T09:39:59.787618Z","iopub.status.idle":"2024-10-23T09:39:59.804481Z","shell.execute_reply.started":"2024-10-23T09:39:59.787575Z","shell.execute_reply":"2024-10-23T09:39:59.803267Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Droping column which is not in text data","metadata":{}},{"cell_type":"code","source":"df_train.drop(columns = ['PCIAT-Season',\n             'PCIAT-PCIAT_01',\n             'PCIAT-PCIAT_02',\n             'PCIAT-PCIAT_03',\n             'PCIAT-PCIAT_04',\n             'PCIAT-PCIAT_05',\n             'PCIAT-PCIAT_06',\n             'PCIAT-PCIAT_07',\n             'PCIAT-PCIAT_08',\n             'PCIAT-PCIAT_09',\n             'PCIAT-PCIAT_10',\n             'PCIAT-PCIAT_11',\n             'PCIAT-PCIAT_12',\n             'PCIAT-PCIAT_13',\n             'PCIAT-PCIAT_14',\n             'PCIAT-PCIAT_15',\n             'PCIAT-PCIAT_16',\n             'PCIAT-PCIAT_17',\n             'PCIAT-PCIAT_18',\n             'PCIAT-PCIAT_19',\n             'PCIAT-PCIAT_20',\n             'PCIAT-PCIAT_Total'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:00.527775Z","iopub.execute_input":"2024-10-23T09:40:00.528627Z","iopub.status.idle":"2024-10-23T09:40:00.538121Z","shell.execute_reply.started":"2024-10-23T09:40:00.528581Z","shell.execute_reply":"2024-10-23T09:40:00.536698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:00.564387Z","iopub.execute_input":"2024-10-23T09:40:00.564958Z","iopub.status.idle":"2024-10-23T09:40:00.572571Z","shell.execute_reply.started":"2024-10-23T09:40:00.564899Z","shell.execute_reply":"2024-10-23T09:40:00.571358Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:00.596326Z","iopub.execute_input":"2024-10-23T09:40:00.596772Z","iopub.status.idle":"2024-10-23T09:40:00.604362Z","shell.execute_reply.started":"2024-10-23T09:40:00.596712Z","shell.execute_reply":"2024-10-23T09:40:00.603103Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# bmi is calculated with the help of height and the weight so we can remove them\ndf_train.drop(columns = ['Physical-Height' , 'Physical-Weight'] , inplace = True)\ndf_test.drop(columns = ['Physical-Height' , 'Physical-Weight'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:00.857234Z","iopub.execute_input":"2024-10-23T09:40:00.857687Z","iopub.status.idle":"2024-10-23T09:40:00.866838Z","shell.execute_reply.started":"2024-10-23T09:40:00.857637Z","shell.execute_reply":"2024-10-23T09:40:00.865603Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.367733Z","iopub.execute_input":"2024-10-23T09:40:01.368200Z","iopub.status.idle":"2024-10-23T09:40:01.389542Z","shell.execute_reply.started":"2024-10-23T09:40:01.368157Z","shell.execute_reply":"2024-10-23T09:40:01.388212Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(df_train.isnull().sum()/len(df_train))*100","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.401011Z","iopub.execute_input":"2024-10-23T09:40:01.401442Z","iopub.status.idle":"2024-10-23T09:40:01.420071Z","shell.execute_reply.started":"2024-10-23T09:40:01.401397Z","shell.execute_reply":"2024-10-23T09:40:01.418846Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### If over half the data is missing, consider whether the column is essential for your analysis.\n###In many cases, it may be better to drop such columns, especially if they don't provide crucial information.","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.432192Z","iopub.execute_input":"2024-10-23T09:40:01.432629Z","iopub.status.idle":"2024-10-23T09:40:01.438003Z","shell.execute_reply.started":"2024-10-23T09:40:01.432586Z","shell.execute_reply":"2024-10-23T09:40:01.436679Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop(columns = ['Fitness_Endurance-Time_Mins', 'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Time_Sec', \n                        'FGC-FGC_GSND_Zone', 'PAQ_A-PAQ_A_Total', 'Fitness_Endurance-Season', \n                        'FGC-FGC_GSND', 'FGC-FGC_GSD_Zone', 'PAQ_A-Season', 'FGC-FGC_GSD', \n                        'PAQ_C-Season', 'Physical-Waist_Circumference', 'Fitness_Endurance-Max_Stage'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.465461Z","iopub.execute_input":"2024-10-23T09:40:01.465925Z","iopub.status.idle":"2024-10-23T09:40:01.474257Z","shell.execute_reply.started":"2024-10-23T09:40:01.465882Z","shell.execute_reply":"2024-10-23T09:40:01.473111Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.drop(columns = ['Fitness_Endurance-Time_Mins', 'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Time_Sec', \n                        'FGC-FGC_GSND_Zone', 'PAQ_A-PAQ_A_Total', 'Fitness_Endurance-Season', \n                        'FGC-FGC_GSND', 'FGC-FGC_GSD_Zone', 'PAQ_A-Season', 'FGC-FGC_GSD', \n                        'PAQ_C-Season', 'Physical-Waist_Circumference', 'Fitness_Endurance-Max_Stage'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.493000Z","iopub.execute_input":"2024-10-23T09:40:01.493439Z","iopub.status.idle":"2024-10-23T09:40:01.500804Z","shell.execute_reply.started":"2024-10-23T09:40:01.493388Z","shell.execute_reply":"2024-10-23T09:40:01.499609Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.527828Z","iopub.execute_input":"2024-10-23T09:40:01.528261Z","iopub.status.idle":"2024-10-23T09:40:01.534613Z","shell.execute_reply.started":"2024-10-23T09:40:01.528218Z","shell.execute_reply":"2024-10-23T09:40:01.533377Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:01.752314Z","iopub.execute_input":"2024-10-23T09:40:01.752791Z","iopub.status.idle":"2024-10-23T09:40:01.772547Z","shell.execute_reply.started":"2024-10-23T09:40:01.752728Z","shell.execute_reply":"2024-10-23T09:40:01.771124Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop(columns = ['id'] , inplace = True)\ndf_test.drop(columns = ['id'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:02.054816Z","iopub.execute_input":"2024-10-23T09:40:02.055266Z","iopub.status.idle":"2024-10-23T09:40:02.064235Z","shell.execute_reply.started":"2024-10-23T09:40:02.055224Z","shell.execute_reply":"2024-10-23T09:40:02.062848Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Basic_Demos-Enroll_Season'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:02.287709Z","iopub.execute_input":"2024-10-23T09:40:02.288223Z","iopub.status.idle":"2024-10-23T09:40:02.298968Z","shell.execute_reply.started":"2024-10-23T09:40:02.288166Z","shell.execute_reply":"2024-10-23T09:40:02.297843Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['CGAS-Season'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:02.502997Z","iopub.execute_input":"2024-10-23T09:40:02.504009Z","iopub.status.idle":"2024-10-23T09:40:02.513500Z","shell.execute_reply.started":"2024-10-23T09:40:02.503951Z","shell.execute_reply":"2024-10-23T09:40:02.512300Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Physical-Season'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:02.707965Z","iopub.execute_input":"2024-10-23T09:40:02.708437Z","iopub.status.idle":"2024-10-23T09:40:02.719145Z","shell.execute_reply.started":"2024-10-23T09:40:02.708394Z","shell.execute_reply":"2024-10-23T09:40:02.717486Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## removing all the season column because of not more impact on the data\ndf_train.drop(columns = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season',\n                        'FGC-Season','BIA-Season','SDS-Season','PreInt_EduHx-Season'] , inplace = True)\n\ndf_test.drop(columns = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season',\n                        'FGC-Season','BIA-Season','SDS-Season','PreInt_EduHx-Season'] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:02.974919Z","iopub.execute_input":"2024-10-23T09:40:02.975395Z","iopub.status.idle":"2024-10-23T09:40:02.984056Z","shell.execute_reply.started":"2024-10-23T09:40:02.975348Z","shell.execute_reply":"2024-10-23T09:40:02.982621Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:03.232195Z","iopub.execute_input":"2024-10-23T09:40:03.232650Z","iopub.status.idle":"2024-10-23T09:40:03.239198Z","shell.execute_reply.started":"2024-10-23T09:40:03.232599Z","shell.execute_reply":"2024-10-23T09:40:03.237863Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(df_train.isnull().sum()/len(df_train))*100","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:03.448213Z","iopub.execute_input":"2024-10-23T09:40:03.448732Z","iopub.status.idle":"2024-10-23T09:40:03.462140Z","shell.execute_reply.started":"2024-10-23T09:40:03.448683Z","shell.execute_reply":"2024-10-23T09:40:03.460845Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=10)\ndf_train = pd.DataFrame(imputer.fit_transform(df_train), columns=df_train.columns)\nprint(df_train.isnull().sum()) ","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:03.679933Z","iopub.execute_input":"2024-10-23T09:40:03.681082Z","iopub.status.idle":"2024-10-23T09:40:08.767706Z","shell.execute_reply.started":"2024-10-23T09:40:03.681032Z","shell.execute_reply":"2024-10-23T09:40:08.766371Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=10)\ndf_test = pd.DataFrame(imputer.fit_transform(df_test), columns=df_test.columns)\nprint(df_test.isnull().sum()) ","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:08.770177Z","iopub.execute_input":"2024-10-23T09:40:08.770935Z","iopub.status.idle":"2024-10-23T09:40:08.802597Z","shell.execute_reply.started":"2024-10-23T09:40:08.770876Z","shell.execute_reply":"2024-10-23T09:40:08.801271Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:08.804048Z","iopub.execute_input":"2024-10-23T09:40:08.804416Z","iopub.status.idle":"2024-10-23T09:40:08.817597Z","shell.execute_reply.started":"2024-10-23T09:40:08.804370Z","shell.execute_reply":"2024-10-23T09:40:08.816263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.combine import SMOTETomek\nfrom collections import Counter\n\ndf_train['sii'] = df_train['sii'].astype(int)\n\nX = df_train.drop(columns=['sii'])\ny = df_train['sii']\n\nsmt = SMOTETomek(random_state=42)\nX, y = smt.fit_resample(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:08.820237Z","iopub.execute_input":"2024-10-23T09:40:08.820645Z","iopub.status.idle":"2024-10-23T09:40:09.475934Z","shell.execute_reply.started":"2024-10-23T09:40:08.820605Z","shell.execute_reply":"2024-10-23T09:40:09.474832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nss = StandardScaler()\n\nX = ss.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:09.477308Z","iopub.execute_input":"2024-10-23T09:40:09.477827Z","iopub.status.idle":"2024-10-23T09:40:09.496314Z","shell.execute_reply.started":"2024-10-23T09:40:09.477785Z","shell.execute_reply":"2024-10-23T09:40:09.494883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = ss.transform(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:09.497679Z","iopub.execute_input":"2024-10-23T09:40:09.498068Z","iopub.status.idle":"2024-10-23T09:40:09.505872Z","shell.execute_reply.started":"2024-10-23T09:40:09.498026Z","shell.execute_reply":"2024-10-23T09:40:09.504582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train , X_test , y_train , y_test = train_test_split(X , y , test_size = 0.1 , random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:09.507327Z","iopub.execute_input":"2024-10-23T09:40:09.507781Z","iopub.status.idle":"2024-10-23T09:40:09.521223Z","shell.execute_reply.started":"2024-10-23T09:40:09.507681Z","shell.execute_reply":"2024-10-23T09:40:09.519975Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import  HistGradientBoostingRegressor\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:09.523041Z","iopub.execute_input":"2024-10-23T09:40:09.523640Z","iopub.status.idle":"2024-10-23T09:40:09.529626Z","shell.execute_reply.started":"2024-10-23T09:40:09.523584Z","shell.execute_reply":"2024-10-23T09:40:09.528425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = HistGradientBoostingRegressor(\n    learning_rate=0.1145911305574919,\n    max_iter=493,\n    max_depth=15,\n    random_state=42\n)\n\n# Train the best model on the full training set\nbest_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = best_model.predict(X_test)\n\n# Evaluate the model's performance\nmse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\n\nprint(f\"Test RMSE: {rmse}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:09.531653Z","iopub.execute_input":"2024-10-23T09:40:09.532080Z","iopub.status.idle":"2024-10-23T09:40:13.112696Z","shell.execute_reply.started":"2024-10-23T09:40:09.532041Z","shell.execute_reply":"2024-10-23T09:40:13.111643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = best_model.predict(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-23T09:40:13.119441Z","iopub.execute_input":"2024-10-23T09:40:13.120233Z","iopub.status.idle":"2024-10-23T09:40:13.140267Z","shell.execute_reply.started":"2024-10-23T09:40:13.120183Z","shell.execute_reply":"2024-10-23T09:40:13.139265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsubmission = df_test[['id']]\nsubmission['sii'] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:13.142020Z","iopub.execute_input":"2024-10-23T09:40:13.142781Z","iopub.status.idle":"2024-10-23T09:40:13.159430Z","shell.execute_reply.started":"2024-10-23T09:40:13.142716Z","shell.execute_reply":"2024-10-23T09:40:13.158542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv' , index = None)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T09:40:13.160895Z","iopub.execute_input":"2024-10-23T09:40:13.161501Z","iopub.status.idle":"2024-10-23T09:40:13.168913Z","shell.execute_reply.started":"2024-10-23T09:40:13.161459Z","shell.execute_reply":"2024-10-23T09:40:13.167821Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}