{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n'''  IMPORTING IMPORTANT LIBRARIES '''\n\nimport numpy as np \nimport pandas as pd \n# import kagglehub \nimport seaborn as sns\nimport matplotlib.pyplot as plt \nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestClassifier,RandomForestRegressor \nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.metrics import accuracy_score\nimport xgboost as xgb\nimport lightgbm as lgb \nfrom sklearn.ensemble import VotingClassifier \n\n\n''' Supress warnings '''\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-13T13:09:07.167087Z","iopub.execute_input":"2024-10-13T13:09:07.167551Z","iopub.status.idle":"2024-10-13T13:09:12.378626Z","shell.execute_reply.started":"2024-10-13T13:09:07.167506Z","shell.execute_reply":"2024-10-13T13:09:12.377457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' important -> For hyperparameter tunign , do Validation '''\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:17.952450Z","iopub.execute_input":"2024-10-12T06:53:17.952994Z","iopub.status.idle":"2024-10-12T06:53:17.961139Z","shell.execute_reply.started":"2024-10-12T06:53:17.952898Z","shell.execute_reply":"2024-10-12T06:53:17.959961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LOADING THE DATASET","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:17.962368Z","iopub.execute_input":"2024-10-12T06:53:17.962716Z","iopub.status.idle":"2024-10-12T06:53:18.031739Z","shell.execute_reply.started":"2024-10-12T06:53:17.962680Z","shell.execute_reply":"2024-10-12T06:53:18.030804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data.columns\nid_col = test_data['id']\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:18.033834Z","iopub.execute_input":"2024-10-12T06:53:18.034223Z","iopub.status.idle":"2024-10-12T06:53:18.039021Z","shell.execute_reply.started":"2024-10-12T06:53:18.034186Z","shell.execute_reply":"2024-10-12T06:53:18.037953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols = ['sii']\ndropped_pciat_cols = [\n       'PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n       'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n       'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10',\n       'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n       'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n       'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:18.040562Z","iopub.execute_input":"2024-10-12T06:53:18.040998Z","iopub.status.idle":"2024-10-12T06:53:18.048369Z","shell.execute_reply.started":"2024-10-12T06:53:18.040951Z","shell.execute_reply":"2024-10-12T06:53:18.047210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.columns ","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:18.049730Z","iopub.execute_input":"2024-10-12T06:53:18.050319Z","iopub.status.idle":"2024-10-12T06:53:18.059862Z","shell.execute_reply.started":"2024-10-12T06:53:18.050277Z","shell.execute_reply":"2024-10-12T06:53:18.058682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OBSERVING THE CORRELATION OF 'PCIAT-TOTAL' WITH REST OF BATCH OF FEATURES","metadata":{}},{"cell_type":"code","source":"\nnum_cols = ['Physical-Weight','Physical-Height','Physical-Waist_Circumference','Physical-Diastolic_BP', \n            'Physical-HeartRate', 'Physical-Systolic_BP',\n            'PCIAT-PCIAT_Total']\n\nnum_data = train_data[ num_cols ]\n\ncorr_mat = num_data.corr()\n\nsns.heatmap(corr_mat )\n\n''' Physicla health\nThough PCIAT ahs high correlation with height , it soes NOT signify 9relate\nwith high internet use . \n'''\ncorr_mat","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:18.061241Z","iopub.execute_input":"2024-10-12T06:53:18.061636Z","iopub.status.idle":"2024-10-12T06:53:18.533264Z","shell.execute_reply.started":"2024-10-12T06:53:18.061592Z","shell.execute_reply":"2024-10-12T06:53:18.532174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nnum_cols = [\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW','PCIAT-PCIAT_Total']\n\nnum_data = train_data[ num_cols ]\n\ncorr_mat = num_data.corr()\n\nsns.heatmap(corr_mat )\n''' Physicla health\n\nwith BIA , PCIAT total its higly correlated \n'''\ncorr_mat\n\n''' Droping BIA-FAT column due to high negative corr with rest and BIA features except PCIAT total '''\n\n''' dropping features with high negaitv ecorr '''\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:18.534825Z","iopub.execute_input":"2024-10-12T06:53:18.535647Z","iopub.status.idle":"2024-10-12T06:53:19.139772Z","shell.execute_reply.started":"2024-10-12T06:53:18.535594Z","shell.execute_reply":"2024-10-12T06:53:19.138714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace higly corrlated variables with a single variable \n\n''' reducing multi-collinearity for better and faster convergence .'''\n\nnum_cols = [\n       'BIA-BIA_BMC' ,'BIA-BIA_BMR' ,'BIA-BIA_ECW' ,'BIA-BIA_FFM']\n\ntrain_data['BIA_net'] = train_data[num_cols].mean(axis=1)\ntest_data['BIA_net'] = test_data[num_cols].mean(axis=1)\n\ntrain_data.drop( num_cols , axis=1 , inplace= True )\ntest_data.drop( num_cols , axis=1 , inplace= True )\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:19.143203Z","iopub.execute_input":"2024-10-12T06:53:19.143550Z","iopub.status.idle":"2024-10-12T06:53:19.158179Z","shell.execute_reply.started":"2024-10-12T06:53:19.143513Z","shell.execute_reply":"2024-10-12T06:53:19.157161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    \nnum_cols = [\n       'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T' ,'PCIAT-PCIAT_Total']\n\nnum_data = train_data[ num_cols ]\n\ncorr_mat = num_data.corr()\n\nsns.heatmap(corr_mat )\n''' Physicla health'''\ncorr_mat\n\n#dropping one ogthe the highly coorelated feature \ntrain_data.drop(['SDS-SDS_Total_T' ] ,axis=1  , inplace =True )","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:19.159485Z","iopub.execute_input":"2024-10-12T06:53:19.159906Z","iopub.status.idle":"2024-10-12T06:53:19.463803Z","shell.execute_reply.started":"2024-10-12T06:53:19.159834Z","shell.execute_reply":"2024-10-12T06:53:19.462846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' By drawing corrlation matrices   ,most important features for PCIAT total \nFGC-FGC_CU  and SDS-SDS_Total_T/SDS-SDS_Total_Raw\n'''\n\n#so i am going to use unsupervised learning \n\nfeatures = ['FGC-FGC_CU' , 'SDS-SDS_Total_Raw' , 'PCIAT-PCIAT_Total']\n\ndummy = train_data[features ]\ndummy.dropna( inplace = True )\n\ntrain_dummy = dummy.drop('PCIAT-PCIAT_Total' ,axis=1 )\ny_dummy = dummy['PCIAT-PCIAT_Total']\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:19.465512Z","iopub.execute_input":"2024-10-12T06:53:19.466506Z","iopub.status.idle":"2024-10-12T06:53:19.477102Z","shell.execute_reply.started":"2024-10-12T06:53:19.466452Z","shell.execute_reply":"2024-10-12T06:53:19.475686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fititng a RANDOM FOREST regression model (b.w the high corrlated fetaure wrt PCIAT total )\n\n\nrand_reg = RandomForestRegressor(n_estimators = 2000  , bootstrap =True )\n\nrand_reg.fit( train_dummy , y_dummy )\n\ntrain_data['pciat_extracted'] = y_dummy\n\nimpute_ = SimpleImputer( strategy = 'median')\nnum_cols = train_dummy.select_dtypes( include=['number']).columns\ntrain_dummy[num_cols] = impute_.fit_transform(train_dummy[num_cols])\n\ntest_data[num_cols] = impute_.transform(test_data[num_cols])\n\ntest_data['pciat_extracted'] = rand_reg.predict( test_data[features[:2]])\n# ppredicting this feature for test Data \n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:19.478479Z","iopub.execute_input":"2024-10-12T06:53:19.478925Z","iopub.status.idle":"2024-10-12T06:53:22.121949Z","shell.execute_reply.started":"2024-10-12T06:53:19.478863Z","shell.execute_reply":"2024-10-12T06:53:22.120940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace higly corrlated variables with a single variable \nnum_cols = [\n       'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL']\n\ntrain_data['FGC_net'] = train_data[num_cols].mean(axis=1)\ntest_data['FGC_net'] = test_data[num_cols].mean(axis=1)\n\ntrain_data.drop( num_cols , axis=1 , inplace= True )\ntest_data.drop( num_cols , axis=1 , inplace= True )\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.133383Z","iopub.execute_input":"2024-10-12T06:53:22.133781Z","iopub.status.idle":"2024-10-12T06:53:22.151393Z","shell.execute_reply.started":"2024-10-12T06:53:22.133738Z","shell.execute_reply":"2024-10-12T06:53:22.150188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[dropped_pciat_cols].info()","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.152572Z","iopub.execute_input":"2024-10-12T06:53:22.152985Z","iopub.status.idle":"2024-10-12T06:53:22.168439Z","shell.execute_reply.started":"2024-10-12T06:53:22.152943Z","shell.execute_reply":"2024-10-12T06:53:22.167198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use clustering to repalce these \n\npciat_cols = [\n        'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n       'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n       'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10',\n       'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n       'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n       'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']\n\n\ntrain_data['sii'].value_counts()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.169767Z","iopub.execute_input":"2024-10-12T06:53:22.170182Z","iopub.status.idle":"2024-10-12T06:53:22.181475Z","shell.execute_reply.started":"2024-10-12T06:53:22.170141Z","shell.execute_reply":"2024-10-12T06:53:22.180389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OVERSAMPLING THE MINORITY CLASS USING SMOTE ","metadata":{}},{"cell_type":"code","source":"# Oversampling the Minority Class \n# TO remove any type of bias in the data \n\ndef manage_imbalance( train_data ) :\n    \n    # Separate your target variable\n    col = train_data['sii']\n \n# for Oversampling using the miority class , give the dataet itself , smote knwos how to oversmaple minority clas \n    \n    \n    # Apply SMOTE only to class 2 and class 3 ( but maintaing  original ration of these classes )\n    smote = SMOTE(sampling_strategy={ 1:1000 , 2: 900, 3: 300 } , random_state =42 )\n    \n    X_sample ,y_sample = smote.fit_resample(  train_data.drop(['sii'] ,axis=1 ) , train_data['sii'])\n        \n    print(col.value_counts())\n    \n    return  X_sample,y_sample\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.183163Z","iopub.execute_input":"2024-10-12T06:53:22.183703Z","iopub.status.idle":"2024-10-12T06:53:22.192869Z","shell.execute_reply.started":"2024-10-12T06:53:22.183646Z","shell.execute_reply":"2024-10-12T06:53:22.191848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def outliers_(data, col):\n    # Define the inter-quartile range\n    q1 = data[col].quantile(0.25)\n    q3 = data[col].quantile(0.75)\n    \n    iqr = q3 - q1\n    \n    # Define lower and upper bounds for outliers\n    lower_bound = q1 - 1.5 * iqr\n    upper_bound = q3 + 1.5 * iqr\n    \n    # Filter out the outliers using element-wise comparison\n    outliers = data[(data[col] < lower_bound) | (data[col] > upper_bound)]\n    \n    # Return the number of outliers and the outlier rows\n    return outliers.shape[0]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.194301Z","iopub.execute_input":"2024-10-12T06:53:22.194708Z","iopub.status.idle":"2024-10-12T06:53:22.203162Z","shell.execute_reply.started":"2024-10-12T06:53:22.194668Z","shell.execute_reply":"2024-10-12T06:53:22.201947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # shuffle the daatset \n\n\ntrain_data = train_data.sample( frac = 1  , random_state = 42 )","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.204651Z","iopub.execute_input":"2024-10-12T06:53:22.205036Z","iopub.status.idle":"2024-10-12T06:53:22.214579Z","shell.execute_reply.started":"2024-10-12T06:53:22.205000Z","shell.execute_reply":"2024-10-12T06:53:22.213393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATA PRE-PROCESSING PIPELINE","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.preprocessing import StandardScaler \n\n\ndef preprocess( train_data  , test_data ) :\n    \n    # remove PCIAT columns to  be dropped (Pciat) from trianing data only \n    train = train_data.drop(dropped_pciat_cols , axis=1 )\n    test = test_data \n    \n    label = train['sii']\n    \n    \n    \n    # object cols \n    obj_cols_train = train.select_dtypes(include =['object']).columns \n    obj_cols_test = test.select_dtypes(include =['object']).columns \n    \n    train.drop( obj_cols_train , axis=1 , inplace = True )\n    \n    test.drop( obj_cols_test , axis=1 , inplace = True )\n    \n        \n    \n    \n   \n    # imputing numerical cols \n    num_cols = train.select_dtypes( include=['number']).columns \n\n    mean_data = []\n    median_data = []\n    \n    for col in num_cols :\n        out = outliers_( train , col)\n        total = train[col].shape[0]\n\n        if (out==0) :\n            mean_data.append(col)\n        else :\n            if (out/total)>0.10 :\n                median_data.append(col)\n            else :\n                mean_data.append(col)\n\n    mean_data = [col for col in mean_data if col != 'sii']\n    median_data = [col for col in median_data if col != 'sii']\n\n    impute_median= SimpleImputer( strategy = 'median')\n    impute_mean = SimpleImputer( strategy = 'mean')\n    \n    train[mean_data] = impute_median.fit_transform(train[ mean_data])\n    train[median_data] = impute_mean.fit_transform(train[ median_data ])\n\n    test[mean_data] = impute_median.transform( test [mean_data])\n    test[median_data] = impute_mean.transform(test[ median_data ])\n    \n    #test[num_cols] = impute_.transform(test[num_cols])\n    \n    train['sii'] = train['sii'].fillna(train['sii'].median())\n\n    # splitting thew dataet \n\n    train['sii']= train['sii'].astype(int)\n    print( train['sii'].unique())\n    \n    \n    # Handling the imbalance \n    X_train ,y_train = manage_imbalance( train)\n\n\n    X_test = test\n    \n    X_test = X_test[X_train.columns]\n    \n    # seperate out categoricla columns \n    cat_cols = []\n    for col in X_train.columns :\n        if X_train[col].nunique()<=5 :\n            cat_cols.append(col)\n    \n    #aalready contians ht enumericla cols with categorcia data(one hot encoded )\n    X_train_ohe = pd.get_dummies( X_train , columns  = cat_cols )\n    X_test_ohe = pd.get_dummies( X_test , columns  = cat_cols )\n    \n     # Reindex test columns to match train columns\n    X_test_ohe = X_test_ohe.reindex(columns=X_train_ohe.columns, fill_value=0)\n    \n     # select nuemrical cols after ohe\n    num_cols_ohe = X_train_ohe.select_dtypes(include=['number']).columns\n    \n    # contains ONLY cat columsn \n    X_train_ohe = X_train_ohe.drop(columns=num_cols_ohe)\n    X_test_ohe = X_test_ohe.drop(columns=num_cols_ohe)\n    \n    #X_test = X_test[X_train.columns]\n   # scaling(normalixzing the data )\n    # scale = StandardScaler()\n    \n    # # do sclaing but ONLY on numerical columns \n   \n    # # contains scaled nuemrical columns \n    # X_train_scale =  pd.DataFrame( scale.fit_transform(X_train[num_cols_ohe] ) , columns = num_cols_ohe )\n    # X_test_scale = pd.DataFrame( scale.transform(X_test[num_cols_ohe]) , columns = num_cols_ohe )\n    \n    # #concatenate them \n    # X_train_final = pd.concat([X_train_scale , X_train_ohe] ,axis=1)\n    # X_test_final = pd.concat([X_test_scale , X_test_ohe] ,axis=1)\n\n    X_train_final = pd.concat([X_train[num_cols_ohe]  , X_train_ohe] ,axis=1)\n    X_test_final = pd.concat( [X_test[num_cols_ohe]  , X_test_ohe] ,axis=1)\n    \n    return X_train_final ,y_train , X_test_final\n    \n    '''\n    To prevent Data Leakege (during validation or testing) , use the feaures /info. extracted from trianing data only . \n    Toc use on tetsing data , lie mean ,std dev from trianing data for imputing /scaling , NOT from test data .\n    '''\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.216334Z","iopub.execute_input":"2024-10-12T06:53:22.216725Z","iopub.status.idle":"2024-10-12T06:53:22.237529Z","shell.execute_reply.started":"2024-10-12T06:53:22.216685Z","shell.execute_reply":"2024-10-12T06:53:22.236308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train ,y_train , X_test = preprocess( train_data , test_data )","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.238833Z","iopub.execute_input":"2024-10-12T06:53:22.239250Z","iopub.status.idle":"2024-10-12T06:53:22.415465Z","shell.execute_reply.started":"2024-10-12T06:53:22.239192Z","shell.execute_reply":"2024-10-12T06:53:22.414314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ENSEMBLE VOTING - WITH (XGBOOST + LGBM + Random Forest)","metadata":{}},{"cell_type":"code","source":"\n\n# class_weights = {\n#     0: 1,   # Class 0 weight\n#     1: 2,   # Class 1 weight\n#     2: 4,   # Class 2 weight (focus more)\n#     3: 8   # Class 3 weight (focus more)\n# }\n\n\nparams_xgb = {'n_estimators': 3000, 'max_depth': 10 ,\n                    'learning_rate': 0.055, 'subsample': 0.90 }\n# Create the XGBClassifier with the provided parameters\nxgb_clf = xgb.XGBClassifier(**params_xgb , seed= 0 )\n\n\n\n# xgb_clf.fit(X_train , y_train)\n\n# #validation\n# y_hat = xgb_clf.predict(X_train)\n# print(accuracy_score(y_train ,y_hat))","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.417185Z","iopub.execute_input":"2024-10-12T06:53:22.417563Z","iopub.status.idle":"2024-10-12T06:53:22.423076Z","shell.execute_reply.started":"2024-10-12T06:53:22.417524Z","shell.execute_reply":"2024-10-12T06:53:22.421818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGradientBoosting (faster than XGBOOSt) - trees grow verticlaly in this ensembe(learnign \n# complex features / patterns . )\n\nparams_lgbm = { 'learning_rate': 0.052844122375780005, 'n_estimators': 2000, 'max_depth': 31, 'num_leaves': 632, \n                'feature_fraction': 0.8861326280057349, 'bagging_fraction': 0.6031319034359963,\n                        'bagging_freq': 8, 'lambda_l1': 9.069172656380543,\n                                'lambda_l2': 0.06989596654622207 }\n\n\n# Create the LGBMClassifier with the provided parameters\nlgb_clf = lgb.LGBMClassifier(**params_lgbm ,verbose = -1 , seed= 0 )\n\n\n\n# lgb_clf.fit( X_train , y_train )\n\n#validation\n# y_hat = lgb_clf.predict(X_train)\n# print(accuracy_score(y_train ,y_hat))","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.424394Z","iopub.execute_input":"2024-10-12T06:53:22.424716Z","iopub.status.idle":"2024-10-12T06:53:22.434026Z","shell.execute_reply.started":"2024-10-12T06:53:22.424682Z","shell.execute_reply":"2024-10-12T06:53:22.432942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from lightgbm import LGBMRegressor\n\n# lgb_reg = LGBMRegressor( n_estimators= 500  , random_state =0  )\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MAnual Hyper-Parameter Tuning for Random Forest","metadata":{}},{"cell_type":"code","source":"# # Hyperparameter Tuning \n\n# from sklearn.ensemble import RandomForestClassifier \n\n# from sklearn.model_selection import GridSearchCV\n\n\n# rand_clf = RandomForestClassifier (\n#                            criterion = 'gini' , \n#                             max_features = 'auto'\n    \n# )\n# # uses a utility function( eg - negative of MSE )\n# gs = GridSearchCV (cv = 5 , error_score = np.nan , estimator = rand_clf  , \n#                    param_grid = \n#                   {'min_samples_leaf':  [10, 20],\n#                     'max_depth':         [3, 4 ],\n#                     'n_estimators':      [ 400 , 800 , 900]})\n# gs.fit(X_train ,y_train)\n\n# the_best_parameters = gs.best_params_\n\n# print(the_best_parameters)","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.439470Z","iopub.execute_input":"2024-10-12T06:53:22.440189Z","iopub.status.idle":"2024-10-12T06:53:22.447014Z","shell.execute_reply.started":"2024-10-12T06:53:22.440146Z","shell.execute_reply":"2024-10-12T06:53:22.445954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params = {'max_depth': 5 , 'min_samples_leaf': 20, 'n_estimators': 600}\n\n# rand_clf = RandomForestClassifier (\n#                           **best_params ,\n#                             random_state  =0 \n                                \n# )\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.448418Z","iopub.execute_input":"2024-10-12T06:53:22.448775Z","iopub.status.idle":"2024-10-12T06:53:22.454645Z","shell.execute_reply.started":"2024-10-12T06:53:22.448737Z","shell.execute_reply":"2024-10-12T06:53:22.453523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.455855Z","iopub.execute_input":"2024-10-12T06:53:22.457954Z","iopub.status.idle":"2024-10-12T06:53:22.464967Z","shell.execute_reply.started":"2024-10-12T06:53:22.457903Z","shell.execute_reply":"2024-10-12T06:53:22.463962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # change the submision dynamically ( not based on sample sbmission only )\n\n# predict_xgb= xgb_clf.predict( X_test ).astype(int)\n# # predict_lgbm = (lgb_clf.predict( X_test)).astype(int)\n# predict_xgb","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.466236Z","iopub.execute_input":"2024-10-12T06:53:22.466601Z","iopub.status.idle":"2024-10-12T06:53:22.475647Z","shell.execute_reply.started":"2024-10-12T06:53:22.466557Z","shell.execute_reply":"2024-10-12T06:53:22.474664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Ensemble Methods (Models) have also to be trained on the - Dataset . \n\n\n\nfinal_model = VotingClassifier( estimators = [ ('xgb' , xgb_clf)   ] )\n\n# final_model.fit(X_train ,y_train)\n\n# Trianing in Batches \n\nbatch_size = 820  \nX_train_batches = np.array_split(X_train, len(X_train) // batch_size)\ny_train_batches = np.array_split(y_train, len(y_train) // batch_size)\n\n# now , train model on each batch\nfor X_batch, y_batch in zip(X_train_batches, y_train_batches):\n    final_model.fit(X_batch, y_batch) \n    \nfinal_predict = final_model.predict(X_test)\n\nprint(final_predict)","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:53:22.478998Z","iopub.execute_input":"2024-10-12T06:53:22.479346Z","iopub.status.idle":"2024-10-12T06:54:14.188850Z","shell.execute_reply.started":"2024-10-12T06:53:22.479309Z","shell.execute_reply":"2024-10-12T06:54:14.187629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# COMPILING THE FINAL RESULTS ","metadata":{}},{"cell_type":"code","source":"\n''' CHANGE THE SUBMISISON DATASET DYNAMICALLY .'''\n\nsubmission_= pd.DataFrame ({\n    'id' : id_col , \n    'sii': final_predict\n})\n\n# submission_['sii'] = submission_['sii'].apply(binning)\n\nsubmission_.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:54:14.190506Z","iopub.execute_input":"2024-10-12T06:54:14.191454Z","iopub.status.idle":"2024-10-12T06:54:14.203302Z","shell.execute_reply.started":"2024-10-12T06:54:14.191398Z","shell.execute_reply":"2024-10-12T06:54:14.202378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:54:14.204544Z","iopub.execute_input":"2024-10-12T06:54:14.205366Z","iopub.status.idle":"2024-10-12T06:54:14.216914Z","shell.execute_reply.started":"2024-10-12T06:54:14.205320Z","shell.execute_reply":"2024-10-12T06:54:14.215614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_.to_csv('submission.csv',index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:54:14.218844Z","iopub.execute_input":"2024-10-12T06:54:14.219538Z","iopub.status.idle":"2024-10-12T06:54:14.226364Z","shell.execute_reply.started":"2024-10-12T06:54:14.219481Z","shell.execute_reply":"2024-10-12T06:54:14.225115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}