{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30749,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.475930Z","iopub.execute_input":"2024-11-13T10:13:12.476313Z","iopub.status.idle":"2024-11-13T10:13:12.483826Z","shell.execute_reply.started":"2024-11-13T10:13:12.476285Z","shell.execute_reply":"2024-11-13T10:13:12.482784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Importing the training data\ntrain= pd.read_csv('../input/child-mind-institute-problematic-internet-use/train.csv')\ntest= pd.read_csv('../input/child-mind-institute-problematic-internet-use/test.csv')\ntrain.head()\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.498235Z","iopub.execute_input":"2024-11-13T10:13:12.498629Z","iopub.status.idle":"2024-11-13T10:13:12.569288Z","shell.execute_reply.started":"2024-11-13T10:13:12.498600Z","shell.execute_reply":"2024-11-13T10:13:12.568169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.571118Z","iopub.execute_input":"2024-11-13T10:13:12.571454Z","iopub.status.idle":"2024-11-13T10:13:12.590089Z","shell.execute_reply.started":"2024-11-13T10:13:12.571427Z","shell.execute_reply":"2024-11-13T10:13:12.589115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.591881Z","iopub.execute_input":"2024-11-13T10:13:12.592275Z","iopub.status.idle":"2024-11-13T10:13:12.744874Z","shell.execute_reply.started":"2024-11-13T10:13:12.592240Z","shell.execute_reply":"2024-11-13T10:13:12.743927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check for missing values:\nnas=train.isnull().sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.747136Z","iopub.execute_input":"2024-11-13T10:13:12.747465Z","iopub.status.idle":"2024-11-13T10:13:12.755694Z","shell.execute_reply.started":"2024-11-13T10:13:12.747437Z","shell.execute_reply":"2024-11-13T10:13:12.754671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking duplicate values \ntrain.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.756854Z","iopub.execute_input":"2024-11-13T10:13:12.757170Z","iopub.status.idle":"2024-11-13T10:13:12.784740Z","shell.execute_reply.started":"2024-11-13T10:13:12.757137Z","shell.execute_reply":"2024-11-13T10:13:12.783763Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Exploratory Data Analysis**","metadata":{}},{"cell_type":"code","source":"sii_counts = train['sii'].value_counts()\n\n# Using Matplotlib to create a count plot\nplt.figure(figsize=(8, 6))\nplt.bar(sii_counts.index,sii_counts, color='blue')\nplt.title('Count Plot of Severity impairment index')\nplt.xlabel('Severity impairment index')\nplt.ylabel('Count')\nplt.show()\n#most participants have no severity impariment","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:12.786040Z","iopub.execute_input":"2024-11-13T10:13:12.786344Z","iopub.status.idle":"2024-11-13T10:13:13.024346Z","shell.execute_reply.started":"2024-11-13T10:13:12.786319Z","shell.execute_reply":"2024-11-13T10:13:13.023297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Histogram of age\nplt.hist(train['Basic_Demos-Age'], color='blue')\nplt.title('Distribution of Age')\nplt.xlabel('Age')\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:13.025689Z","iopub.execute_input":"2024-11-13T10:13:13.026020Z","iopub.status.idle":"2024-11-13T10:13:13.230410Z","shell.execute_reply.started":"2024-11-13T10:13:13.025992Z","shell.execute_reply":"2024-11-13T10:13:13.229337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_var_train = train[['sii','Basic_Demos-Age', 'Basic_Demos-Sex' ,'Physical-BMI', 'Physical-Height',\n                             'Physical-Weight','Physical-HeartRate','Physical-Diastolic_BP', \n                             'Physical-Waist_Circumference', 'Physical-Systolic_BP','Fitness_Endurance-Time_Mins',\n                             'PAQ_C-PAQ_C_Total','PAQ_A-PAQ_A_Total', 'FGC-FGC_PU','FGC-FGC_GSND','FGC-FGC_GSD',\n                             'PCIAT-PCIAT_Total','FGC-FGC_CU','Fitness_Endurance-Max_Stage','SDS-SDS_Total_T',\n                             'PreInt_EduHx-computerinternet_hoursday']] \n# Find the pearson correlations matrix\ncorr = numerical_var_train.corr(method = 'pearson')\ncorr\n\nplt.figure(figsize=(15, 10))\n\n# Using Seaborn to create a heatmap\nsns.heatmap(corr, annot=True, fmt='.2f', cmap='Pastel2', linewidths=2)\n\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:13.231763Z","iopub.execute_input":"2024-11-13T10:13:13.232103Z","iopub.status.idle":"2024-11-13T10:13:14.608959Z","shell.execute_reply.started":"2024-11-13T10:13:13.232075Z","shell.execute_reply":"2024-11-13T10:13:14.607861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data Analysis**","metadata":{}},{"cell_type":"code","source":"# Extract feature and target arrays\nX, y = train.drop(['id','sii'], axis=1), train[['sii']]\n\nfrom sklearn.preprocessing import OrdinalEncoder\nimport xgboost as xgb\n\n# Encode y to numeric\ny_encoded = OrdinalEncoder().fit_transform(y)\n\n# Extract text features\ncats = X.select_dtypes(exclude=np.number).columns.tolist()\n\n# Convert to pd.Categorical\nfor col in cats:\n   X[col] = X[col].astype('category')\n\n# Replace NaN values in y with the median \ny = y.fillna(y.median())\ny.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:14.610189Z","iopub.execute_input":"2024-11-13T10:13:14.610522Z","iopub.status.idle":"2024-11-13T10:13:14.640082Z","shell.execute_reply.started":"2024-11-13T10:13:14.610494Z","shell.execute_reply":"2024-11-13T10:13:14.639110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:14.645321Z","iopub.execute_input":"2024-11-13T10:13:14.645680Z","iopub.status.idle":"2024-11-13T10:13:14.652176Z","shell.execute_reply.started":"2024-11-13T10:13:14.645651Z","shell.execute_reply":"2024-11-13T10:13:14.651143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Transforming the X variables X = X.drop(columns=cats)\nX= X.drop(X.loc[:, 'PCIAT-PCIAT_01':'PCIAT-PCIAT_Total'].columns, axis=1)\nX=X.drop('PCIAT-Season', axis=1)\nX.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:14.653523Z","iopub.execute_input":"2024-11-13T10:13:14.653895Z","iopub.status.idle":"2024-11-13T10:13:14.675401Z","shell.execute_reply.started":"2024-11-13T10:13:14.653862Z","shell.execute_reply":"2024-11-13T10:13:14.674406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport xgboost as xgb\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size= 0.2, random_state=42, stratify=y)\n#Creating an XGBoost classifier\nmodel = xgb.XGBClassifier(objective='multi:softprob',\n    num_class=4, enable_categorical=True)\n\n#Training the model on the training data\nmodel.fit(X_train, y_train)\n\n#Making predictions on the test set\npredictions = model.predict(X_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:14.676550Z","iopub.execute_input":"2024-11-13T10:13:14.676827Z","iopub.status.idle":"2024-11-13T10:13:17.544606Z","shell.execute_reply.started":"2024-11-13T10:13:14.676804Z","shell.execute_reply":"2024-11-13T10:13:17.543757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test.drop(['id'], axis=1)\ncats_test = test.select_dtypes(exclude=np.number).columns.tolist()\n# Convert to pd.Categorical\nfor col in cats_test:\n   test[col] = test[col].astype('category')\n# Extract text features test1 = test.drop(columns=cats_test)\n\n\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.545771Z","iopub.execute_input":"2024-11-13T10:13:17.546108Z","iopub.status.idle":"2024-11-13T10:13:17.572809Z","shell.execute_reply.started":"2024-11-13T10:13:17.546078Z","shell.execute_reply":"2024-11-13T10:13:17.571722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred=model.predict(test)\npred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.574137Z","iopub.execute_input":"2024-11-13T10:13:17.576069Z","iopub.status.idle":"2024-11-13T10:13:17.596500Z","shell.execute_reply.started":"2024-11-13T10:13:17.576040Z","shell.execute_reply":"2024-11-13T10:13:17.595496Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Sudmission**","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nsub['sii'] = pred\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.597719Z","iopub.execute_input":"2024-11-13T10:13:17.598101Z","iopub.status.idle":"2024-11-13T10:13:17.607276Z","shell.execute_reply.started":"2024-11-13T10:13:17.598065Z","shell.execute_reply":"2024-11-13T10:13:17.606311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Ensemble Models**","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\n# Model parameters for LightGBM\nparams = {\n    'boosting': 'gbdt',\n    'objective': 'multiclass',\n    'num_leaves': 45,\n    'learning_rate': 0.1,\n    'max_depth': 5\n}\nlgb_classifier = LGBMClassifier()\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200\n}\nxgb_classifier = xgb.XGBClassifier(**XGB_Params, objective='multi:softmax',\n    num_class=4, enable_categorical=True)\n\nfrom catboost import CatBoostClassifier\ncatboost_params = {\n    'eval_metric': 'AUC',\n    'learning_rate': 0.05,\n    'iterations': 1000,\n    'depth': 6,\n    'random_strength':0,\n    'l2_leaf_reg': 0.7047064221215757,\n    'task_type':'CPU',\n    'random_seed':42,\n    'verbose':False    \n}\ncats_train = X_train.select_dtypes(exclude=np.number).columns.tolist()\nmodel_catboost = CatBoostClassifier(**catboost_params,cat_features= cats_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.608485Z","iopub.execute_input":"2024-11-13T10:13:17.608782Z","iopub.status.idle":"2024-11-13T10:13:17.617263Z","shell.execute_reply.started":"2024-11-13T10:13:17.608756Z","shell.execute_reply":"2024-11-13T10:13:17.616301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cats_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.618228Z","iopub.execute_input":"2024-11-13T10:13:17.618570Z","iopub.status.idle":"2024-11-13T10:13:17.628507Z","shell.execute_reply.started":"2024-11-13T10:13:17.618540Z","shell.execute_reply":"2024-11-13T10:13:17.627651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cats_train:\n    # Add 'None' as a new category\n    X_train[col] = X_train[col].cat.add_categories(['None'])\n    # Now fill NaN values with 'None'\n    X_train[col] = X_train[col].fillna('None')\n    \nfor col in cats_train:\n    # Add 'None' as a new category\n    X_test[col] = X_test[col].cat.add_categories(['None'])\n    # Now fill NaN values with 'None'\n    X_test[col] = X_test[col].fillna('None')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.629605Z","iopub.execute_input":"2024-11-13T10:13:17.629891Z","iopub.status.idle":"2024-11-13T10:13:17.650090Z","shell.execute_reply.started":"2024-11-13T10:13:17.629867Z","shell.execute_reply":"2024-11-13T10:13:17.648899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create an ensemble using VotingClassifier\nfrom sklearn.ensemble import VotingClassifier\nensemble_classifier = VotingClassifier(estimators= [('xgb1', xgb_classifier),('lgb', lgb_classifier),('CatBoost', model_catboost)], voting ='soft')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.651242Z","iopub.execute_input":"2024-11-13T10:13:17.651564Z","iopub.status.idle":"2024-11-13T10:13:17.659001Z","shell.execute_reply.started":"2024-11-13T10:13:17.651537Z","shell.execute_reply":"2024-11-13T10:13:17.658081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit the ensemble model\nmodel_ensemble = ensemble_classifier.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:13:17.660211Z","iopub.execute_input":"2024-11-13T10:13:17.661152Z","iopub.status.idle":"2024-11-13T10:14:31.626275Z","shell.execute_reply.started":"2024-11-13T10:13:17.661123Z","shell.execute_reply":"2024-11-13T10:14:31.625418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Predictions\ny_predictions = ensemble_classifier.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.627613Z","iopub.execute_input":"2024-11-13T10:14:31.627921Z","iopub.status.idle":"2024-11-13T10:14:31.743804Z","shell.execute_reply.started":"2024-11-13T10:14:31.627895Z","shell.execute_reply":"2024-11-13T10:14:31.742708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate accuracy\nfrom sklearn.metrics import accuracy_score \naccuracy = accuracy_score(y_test,y_predictions)\nprint(f'Ensemble Accuracy: {accuracy}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.747319Z","iopub.execute_input":"2024-11-13T10:14:31.747671Z","iopub.status.idle":"2024-11-13T10:14:31.757753Z","shell.execute_reply.started":"2024-11-13T10:14:31.747642Z","shell.execute_reply":"2024-11-13T10:14:31.756776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test\ncats_test = test.select_dtypes(exclude=np.number).columns.tolist()\n# Convert to pd.Categorical\nfor col in cats_test:\n   test[col] = test[col].astype('category')\nfor col in cats_test:\n    # Add 'None' as a new category\n    test[col] = test[col].cat.add_categories(['None'])\n    # Now fill NaN values with 'None'\n    test[col] = test[col].fillna('None')\n\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.758886Z","iopub.execute_input":"2024-11-13T10:14:31.759172Z","iopub.status.idle":"2024-11-13T10:14:31.784270Z","shell.execute_reply.started":"2024-11-13T10:14:31.759149Z","shell.execute_reply":"2024-11-13T10:14:31.783143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_ensemble=model_ensemble.predict(test)\npred_ensemble","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.785670Z","iopub.execute_input":"2024-11-13T10:14:31.785972Z","iopub.status.idle":"2024-11-13T10:14:31.820556Z","shell.execute_reply.started":"2024-11-13T10:14:31.785946Z","shell.execute_reply":"2024-11-13T10:14:31.819515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nsub['sii'] = pred_ensemble\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.822036Z","iopub.execute_input":"2024-11-13T10:14:31.822345Z","iopub.status.idle":"2024-11-13T10:14:31.831807Z","shell.execute_reply.started":"2024-11-13T10:14:31.822319Z","shell.execute_reply":"2024-11-13T10:14:31.830764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:14:31.833408Z","iopub.execute_input":"2024-11-13T10:14:31.833796Z","iopub.status.idle":"2024-11-13T10:14:31.848394Z","shell.execute_reply.started":"2024-11-13T10:14:31.833760Z","shell.execute_reply":"2024-11-13T10:14:31.847535Z"}},"outputs":[],"execution_count":null}]}