{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:48.346422Z","iopub.execute_input":"2024-10-01T05:24:48.346986Z","iopub.status.idle":"2024-10-01T05:24:49.645666Z","shell.execute_reply.started":"2024-10-01T05:24:48.346927Z","shell.execute_reply":"2024-10-01T05:24:49.644395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Importing the data**","metadata":{}},{"cell_type":"code","source":"#Importing the training data\ntrain= pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest= pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntrain.head()\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:49.647629Z","iopub.execute_input":"2024-10-01T05:24:49.648110Z","iopub.status.idle":"2024-10-01T05:24:49.761834Z","shell.execute_reply.started":"2024-10-01T05:24:49.648069Z","shell.execute_reply":"2024-10-01T05:24:49.760712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for missing values\nnas=train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:49.763098Z","iopub.execute_input":"2024-10-01T05:24:49.763439Z","iopub.status.idle":"2024-10-01T05:24:49.772544Z","shell.execute_reply.started":"2024-10-01T05:24:49.763403Z","shell.execute_reply":"2024-10-01T05:24:49.771538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model Training**","metadata":{}},{"cell_type":"code","source":"# Extract feature and target arrays\nX, y = train.drop('sii', axis=1), train[['sii']]\n\nfrom sklearn.preprocessing import OrdinalEncoder\nimport xgboost as xgb\n\n# Encode y to numeric\ny_encoded = OrdinalEncoder().fit_transform(y)\n\n# Extract text features\ncats = X.select_dtypes(exclude=np.number).columns.tolist()\n\n# Convert to pd.Categorical\nfor col in cats:\n   X[col] = X[col].astype('category')\n\n# Replace NaN values in y with the median \ny = y.fillna(y.median())","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:49.775333Z","iopub.execute_input":"2024-10-01T05:24:49.776202Z","iopub.status.idle":"2024-10-01T05:24:50.208753Z","shell.execute_reply.started":"2024-10-01T05:24:49.776153Z","shell.execute_reply":"2024-10-01T05:24:50.207401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Transforming the X variables\nX = X.drop(columns=cats)\nX= X.drop(X.loc[:, 'PCIAT-PCIAT_01':'PCIAT-PCIAT_Total'].columns, axis=1)\nX.info()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.210290Z","iopub.execute_input":"2024-10-01T05:24:50.210713Z","iopub.status.idle":"2024-10-01T05:24:50.240703Z","shell.execute_reply.started":"2024-10-01T05:24:50.210671Z","shell.execute_reply":"2024-10-01T05:24:50.239578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport xgboost as xgb\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size= 0.2, random_state=42, stratify=y)\n#Creating an XGBoost classifier\nparams = {\n    'tree_method': 'approx',\n    'objective': 'multi:softprob',\n}\nnum_boost_round = 10\n\nclf = xgb.XGBClassifier(n_estimators=num_boost_round, **params,enable_categorical=True)\nmodel= clf.fit(X_train, y_train)\ny_pred = clf.predict(X_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.242705Z","iopub.execute_input":"2024-10-01T05:24:50.243169Z","iopub.status.idle":"2024-10-01T05:24:50.766673Z","shell.execute_reply.started":"2024-10-01T05:24:50.243115Z","shell.execute_reply":"2024-10-01T05:24:50.765694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics \n\nmetrics.accuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.767902Z","iopub.execute_input":"2024-10-01T05:24:50.768868Z","iopub.status.idle":"2024-10-01T05:24:50.780680Z","shell.execute_reply.started":"2024-10-01T05:24:50.768824Z","shell.execute_reply":"2024-10-01T05:24:50.779558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metrics.classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.782018Z","iopub.execute_input":"2024-10-01T05:24:50.782368Z","iopub.status.idle":"2024-10-01T05:24:50.799680Z","shell.execute_reply.started":"2024-10-01T05:24:50.782331Z","shell.execute_reply":"2024-10-01T05:24:50.798535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Predictions**","metadata":{}},{"cell_type":"code","source":"test\ncats_test = test.select_dtypes(exclude=np.number).columns.tolist()\n# Convert to pd.Categorical\nfor col in cats_test:\n    test[col] = test[col].astype('category')\n# Extract text features \ntest = test.drop(columns=cats_test)\n\n\n\ntest.info()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.801200Z","iopub.execute_input":"2024-10-01T05:24:50.801619Z","iopub.status.idle":"2024-10-01T05:24:50.828502Z","shell.execute_reply.started":"2024-10-01T05:24:50.801572Z","shell.execute_reply":"2024-10-01T05:24:50.827251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**","metadata":{}},{"cell_type":"code","source":"pred=clf.predict(test)\npred","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.832666Z","iopub.execute_input":"2024-10-01T05:24:50.833047Z","iopub.status.idle":"2024-10-01T05:24:50.849869Z","shell.execute_reply.started":"2024-10-01T05:24:50.833008Z","shell.execute_reply":"2024-10-01T05:24:50.848788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nsub['sii'] = pred\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.851257Z","iopub.execute_input":"2024-10-01T05:24:50.851687Z","iopub.status.idle":"2024-10-01T05:24:50.866545Z","shell.execute_reply.started":"2024-10-01T05:24:50.851649Z","shell.execute_reply":"2024-10-01T05:24:50.865376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature Importance**","metadata":{}},{"cell_type":"code","source":"from sklearn.inspection import permutation_importance\n\n# Calculate permutation feature importance\nresult = permutation_importance(\n    model, X_test, y_test, scoring='neg_log_loss', n_repeats=10, random_state=42\n)\n\n# Get the feature importances and sort them in descending order\nfeature_importances = pd.Series(result.importances_mean, index=X.columns).sort_values(ascending=False)\n\n# Print the feature importances\nprint(feature_importances)\n\n# Plot the feature importances\nfeature_importances.sort_values(ascending=True)[-10:].plot.barh()\nplt.title('Permutation Importance on Out-of-Sample Set')\nplt.xlabel('change in log likelihood');","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:50.868011Z","iopub.execute_input":"2024-10-01T05:24:50.868491Z","iopub.status.idle":"2024-10-01T05:24:59.867446Z","shell.execute_reply.started":"2024-10-01T05:24:50.868422Z","shell.execute_reply":"2024-10-01T05:24:59.866230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model Training with important features**","metadata":{}},{"cell_type":"code","source":"features=X[['SDS-SDS_Total_Raw' ,'PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age','CGAS-CGAS_Score','FGC-FGC_CU','Physical-Weight',\n'Physical-Height',  'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total', 'Physical-BMI','BIA-BIA_BMR', 'BIA-BIA_FFMI','Basic_Demos-Sex',\n'BIA-BIA_LDM', 'Physical-HeartRate', 'FGC-FGC_SRR', 'FGC-FGC_PU_Zone','FGC-FGC_TL','Fitness_Endurance-Time_Mins']]\n\nfeatures      ","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:59.869207Z","iopub.execute_input":"2024-10-01T05:24:59.869707Z","iopub.status.idle":"2024-10-01T05:24:59.904830Z","shell.execute_reply.started":"2024-10-01T05:24:59.869654Z","shell.execute_reply":"2024-10-01T05:24:59.903563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport xgboost as xgb\nX_train1, X_test1, y_train1, y_test1 = train_test_split(features, y, test_size= 0.2, random_state=42, stratify=y)\n#Creating an XGBoost classifier\nparams = {\n    'tree_method': 'approx',\n    'objective': 'multi:softprob',\n}\nnum_boost_round = 10\n\nclf1 = xgb.XGBClassifier(n_estimators=num_boost_round, **params,enable_categorical=True)\nmodel= clf1.fit(X_train1, y_train1)\ny_pred1 = clf1.predict(X_test1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:24:59.906253Z","iopub.execute_input":"2024-10-01T05:24:59.906650Z","iopub.status.idle":"2024-10-01T05:25:00.126209Z","shell.execute_reply.started":"2024-10-01T05:24:59.906611Z","shell.execute_reply":"2024-10-01T05:25:00.125177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics \n\nmetrics.accuracy_score(y_test1, y_pred1)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:00.127892Z","iopub.execute_input":"2024-10-01T05:25:00.128593Z","iopub.status.idle":"2024-10-01T05:25:00.138960Z","shell.execute_reply.started":"2024-10-01T05:25:00.128547Z","shell.execute_reply":"2024-10-01T05:25:00.137718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test\ntest1=test[['SDS-SDS_Total_Raw' ,'PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age','CGAS-CGAS_Score','FGC-FGC_CU','Physical-Weight',\n'Physical-Height',  'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total', 'Physical-BMI','BIA-BIA_BMR', 'BIA-BIA_FFMI','Basic_Demos-Sex',\n'BIA-BIA_LDM', 'Physical-HeartRate', 'FGC-FGC_SRR', 'FGC-FGC_PU_Zone','FGC-FGC_TL','Fitness_Endurance-Time_Mins']]\n\n\n\ntest1.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:00.140728Z","iopub.execute_input":"2024-10-01T05:25:00.141172Z","iopub.status.idle":"2024-10-01T05:25:00.157054Z","shell.execute_reply.started":"2024-10-01T05:25:00.141123Z","shell.execute_reply":"2024-10-01T05:25:00.155843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=clf1.predict(test1)\npred","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:00.158648Z","iopub.execute_input":"2024-10-01T05:25:00.159094Z","iopub.status.idle":"2024-10-01T05:25:00.185407Z","shell.execute_reply.started":"2024-10-01T05:25:00.159046Z","shell.execute_reply":"2024-10-01T05:25:00.183996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nsub['sii'] = pred\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:00.186825Z","iopub.execute_input":"2024-10-01T05:25:00.187194Z","iopub.status.idle":"2024-10-01T05:25:00.198004Z","shell.execute_reply.started":"2024-10-01T05:25:00.187154Z","shell.execute_reply":"2024-10-01T05:25:00.196959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Ensemble models**","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nparams = {\n    'tree_method': 'approx',\n    'objective': 'multi:softprob',\n}\nnum_boost_round = 10\n\nxgb_classifier = xgb.XGBClassifier(n_estimators=num_boost_round, **params,enable_categorical=True, \n                                   learning_rate = 0.1, max_depth = 8, min_child_weight = 5)\nlgb_classifier = LGBMClassifier()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:00.199455Z","iopub.execute_input":"2024-10-01T05:25:00.199913Z","iopub.status.idle":"2024-10-01T05:25:01.383540Z","shell.execute_reply.started":"2024-10-01T05:25:00.199864Z","shell.execute_reply":"2024-10-01T05:25:01.382201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an ensemble using VotingClassifier\nfrom sklearn.ensemble import VotingClassifier\nensemble_classifier = VotingClassifier(estimators= [('xgb1', xgb_classifier),('lgb', lgb_classifier)], voting ='soft')","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:01.384892Z","iopub.execute_input":"2024-10-01T05:25:01.385433Z","iopub.status.idle":"2024-10-01T05:25:01.392097Z","shell.execute_reply.started":"2024-10-01T05:25:01.385394Z","shell.execute_reply":"2024-10-01T05:25:01.390889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the ensemble model\n# model_ensemble= ensemble_classifier.fit(X_train,y_train)\nmodel_ensemble= ensemble_classifier.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:01.393881Z","iopub.execute_input":"2024-10-01T05:25:01.394348Z","iopub.status.idle":"2024-10-01T05:25:04.660053Z","shell.execute_reply.started":"2024-10-01T05:25:01.394213Z","shell.execute_reply":"2024-10-01T05:25:04.658655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predictions\ny_predictions = ensemble_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.661933Z","iopub.execute_input":"2024-10-01T05:25:04.662931Z","iopub.status.idle":"2024-10-01T05:25:04.713762Z","shell.execute_reply.started":"2024-10-01T05:25:04.662874Z","shell.execute_reply":"2024-10-01T05:25:04.712406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate accuracy\nfrom sklearn.metrics import accuracy_score\naccuracy = accuracy_score(y_test,y_predictions)\nprint(f'Ensemble Accuracy: {accuracy}')","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.715388Z","iopub.execute_input":"2024-10-01T05:25:04.715887Z","iopub.status.idle":"2024-10-01T05:25:04.727939Z","shell.execute_reply.started":"2024-10-01T05:25:04.715835Z","shell.execute_reply":"2024-10-01T05:25:04.726552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions on the test dataset\ntest\ncats_test = test.select_dtypes(exclude=np.number).columns.tolist()\n# Convert to pd.Categorical\nfor col in cats_test:\n    test[col] = test[col].astype('category')\n# Extract text features \ntest = test.drop(columns=cats_test)\n\n\n\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.729268Z","iopub.execute_input":"2024-10-01T05:25:04.729621Z","iopub.status.idle":"2024-10-01T05:25:04.749929Z","shell.execute_reply.started":"2024-10-01T05:25:04.729586Z","shell.execute_reply":"2024-10-01T05:25:04.748570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions=model_ensemble.predict(test)\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.751280Z","iopub.execute_input":"2024-10-01T05:25:04.751671Z","iopub.status.idle":"2024-10-01T05:25:04.777896Z","shell.execute_reply.started":"2024-10-01T05:25:04.751631Z","shell.execute_reply":"2024-10-01T05:25:04.776527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_ensemble = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nsub_ensemble['sii'] = predictions\nsub_ensemble.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.779612Z","iopub.execute_input":"2024-10-01T05:25:04.780706Z","iopub.status.idle":"2024-10-01T05:25:04.790751Z","shell.execute_reply.started":"2024-10-01T05:25:04.780649Z","shell.execute_reply":"2024-10-01T05:25:04.789596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_ensemble","metadata":{"execution":{"iopub.status.busy":"2024-10-01T05:25:04.792201Z","iopub.execute_input":"2024-10-01T05:25:04.792814Z","iopub.status.idle":"2024-10-01T05:25:04.812951Z","shell.execute_reply.started":"2024-10-01T05:25:04.792761Z","shell.execute_reply":"2024-10-01T05:25:04.811606Z"},"trusted":true},"execution_count":null,"outputs":[]}]}