{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## TPS July 2022\n\n- Inspired by https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting/notebook, following notebook is to demonstrate different classifiers with Soft voting only using Sklearn\n- Thank you everyone for great ideas (Apology if i did not add anyone specifically)","metadata":{}},{"cell_type":"code","source":"!pip install scikit-lego","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:26:51.420786Z","iopub.execute_input":"2022-07-25T08:26:51.421243Z","iopub.status.idle":"2022-07-25T08:27:05.994528Z","shell.execute_reply.started":"2022-07-25T08:26:51.421209Z","shell.execute_reply":"2022-07-25T08:27:05.993344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport gc,random,os\n\nfrom sklearn.preprocessing import PowerTransformer,RobustScaler\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.metrics import roc_auc_score,accuracy_score\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.ensemble import VotingClassifier,GradientBoostingClassifier,HistGradientBoostingClassifier\n\nfrom sklearn.neural_network import MLPClassifier\n\nfrom sklego.mixture import BayesianGMMClassifier\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T08:27:05.997438Z","iopub.execute_input":"2022-07-25T08:27:05.997955Z","iopub.status.idle":"2022-07-25T08:27:06.013064Z","shell.execute_reply.started":"2022-07-25T08:27:05.997889Z","shell.execute_reply":"2022-07-25T08:27:06.011961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set Static variables\nrandom_state = 0\n#n_folds = 10\nn_components = 7\nverbose = 500\nos.environ['PYTHONHASHSEED'] = str(random_state)\nnp.random.seed(random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:27:10.111996Z","iopub.execute_input":"2022-07-25T08:27:10.112456Z","iopub.status.idle":"2022-07-25T08:27:10.119327Z","shell.execute_reply.started":"2022-07-25T08:27:10.112420Z","shell.execute_reply":"2022-07-25T08:27:10.118256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read training data\ndata = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\",index_col = 'id')\nsample_submission =pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:27:11.444931Z","iopub.execute_input":"2022-07-25T08:27:11.445461Z","iopub.status.idle":"2022-07-25T08:27:13.027829Z","shell.execute_reply.started":"2022-07-25T08:27:11.445417Z","shell.execute_reply":"2022-07-25T08:27:13.026244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = data.dtypes.value_counts().plot.barh(color = ['green','orange'],rot = 0,figsize = (8,3))\nax.set_title('Distribution of Column dtypes')\nax.set_xlabel('Number of columns')\nax.set_ylabel('Column dtype')\n\nfor i in ax.patches:\n    ax.text(i.get_width()+.3, i.get_y()+.38, str(i.get_width()), fontsize=12,\ncolor='dimgrey')\n    \nax.invert_yaxis()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:27:13.030397Z","iopub.execute_input":"2022-07-25T08:27:13.030805Z","iopub.status.idle":"2022-07-25T08:27:13.272872Z","shell.execute_reply.started":"2022-07-25T08:27:13.030768Z","shell.execute_reply":"2022-07-25T08:27:13.271951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using PowerTransformer before Scaling to adjust for Outliers\ndata_scaled = pd.DataFrame(PowerTransformer().fit_transform(data),columns=data.columns)\ndata_scaled = pd.DataFrame(RobustScaler().fit_transform(data_scaled), columns=data_scaled.columns)\n\n#www.kaggle.com/competitions/tabular-playground-series-jul-2022/discussion/334808\nuseful_cols  = ['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']\n\n# Test Data for predictions later\ntest_data = data_scaled[useful_cols].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:27:13.343274Z","iopub.execute_input":"2022-07-25T08:27:13.344111Z","iopub.status.idle":"2022-07-25T08:27:17.301514Z","shell.execute_reply.started":"2022-07-25T08:27:13.344069Z","shell.execute_reply":"2022-07-25T08:27:17.300180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit Bayesian Gaussian Mixture\nprint('Fitting Bayesian Gaussian Mixture..')\nbgm = BayesianGaussianMixture(n_components = n_components,\n                         max_iter = 300, n_init = 5, \n                     random_state = random_state,\n                 verbose = 1,\n             verbose_interval = 100\n                     )\n\nbgm_labels = bgm.fit_predict(data_scaled[useful_cols])\nbgm_proba = bgm.predict_proba(data_scaled[useful_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:27:27.502351Z","iopub.execute_input":"2022-07-25T08:27:27.502818Z","iopub.status.idle":"2022-07-25T08:32:19.284897Z","shell.execute_reply.started":"2022-07-25T08:27:27.502783Z","shell.execute_reply":"2022-07-25T08:32:19.283250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using idea from https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting\n\n# Creating Best data based on predicted probability of BGM model\n\ndata_scaled['predict']= bgm_labels\ndata_scaled['predict_proba']=0\n\nfor n in range(n_components):\n    data_scaled[f'bgm_proba_{n}']= bgm_proba[:,n]\n    data_scaled.loc[data_scaled.predict == n,'bgm_proba']=data_scaled[f'bgm_proba_{n}']\n    \ntrain_index=np.array([])\nfor n in range(n_components):\n    median=data_scaled[data_scaled.predict==n]['bgm_proba'].median()\n\n    # Experiment with different thresholds\n    # Higher thereshold might overfit\n    n_inx=data_scaled[(data_scaled.predict==n) & (data_scaled.bgm_proba > 0.9)].index\n    \n    train_index = np.concatenate((train_index, n_inx))\n    print(f'class:{n}',f'median: {round(median,4)}','Training data:'+str(round(len(n_inx)/len(data_scaled[(data_scaled.predict==n)]),2)*100)+'%')\n    \n    \nprint(f'\\nSize of Training data : {len(train_index)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:32:19.288269Z","iopub.execute_input":"2022-07-25T08:32:19.289272Z","iopub.status.idle":"2022-07-25T08:32:19.457192Z","shell.execute_reply.started":"2022-07-25T08:32:19.289188Z","shell.execute_reply":"2022-07-25T08:32:19.455986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=data_scaled.loc[train_index][useful_cols].to_numpy()\ny=data_scaled.loc[train_index]['predict'].to_numpy()\n\n#xtrain,xvalid, ytrain,yvalid = train_test_split(X,y,random_state = random_state,test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:32:19.458720Z","iopub.execute_input":"2022-07-25T08:32:19.459116Z","iopub.status.idle":"2022-07-25T08:32:19.513304Z","shell.execute_reply.started":"2022-07-25T08:32:19.459082Z","shell.execute_reply":"2022-07-25T08:32:19.511860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XG Boost can also be used\ngbc = GradientBoostingClassifier(n_estimators = 5000,\n                        subsample = 0.6,\n                        random_state=random_state,\n                     max_features = 'auto',\n                    validation_fraction =0.2,\n                    n_iter_no_change = 300,\n                    )\n\n# This is inspired by LGBM. LGBM can also be used directly\nhgbc = HistGradientBoostingClassifier(max_iter = 5000,\n                                      learning_rate = 0.06,\n                                      random_state=random_state,\n                    validation_fraction =0.2,\n                    n_iter_no_change = 300,\n                    l2_regularization=0.1,\n)\n\n# Saving time by using ML classifier. Can experiment with Torch/Keras based NN models as well\nmlpc = MLPClassifier(learning_rate = 'adaptive',\n                    max_iter = 5000,\n                    hidden_layer_sizes = (400,200,100,),\n                    early_stopping = True,\n                    n_iter_no_change = 200,\n                    validation_fraction = 0.2,\n                    )\n\n\nbgmmc = BayesianGMMClassifier(\n        n_components=7,\n        random_state = random_state,\n        tol =1e-3,\n        covariance_type = 'full',\n        max_iter = 300,\n        n_init=10,\n        init_params='kmeans')\n\n# This is simplified version of soft voting as suggested by \n# https://www.kaggle.com/code/pourchot/simple-soft-voting?scriptVersionId=100462375&cellId=6\nvcsft = VotingClassifier(estimators=[('GBC', gbc), ('HGBC', hgbc),('MLPC',mlpc),('BGMMC',bgmmc)], \n                 voting='soft',verbose=verbose,n_jobs = -1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:32:19.516504Z","iopub.execute_input":"2022-07-25T08:32:19.516922Z","iopub.status.idle":"2022-07-25T08:32:19.527914Z","shell.execute_reply.started":"2022-07-25T08:32:19.516881Z","shell.execute_reply":"2022-07-25T08:32:19.526927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vcsft.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:32:19.529567Z","iopub.execute_input":"2022-07-25T08:32:19.530370Z","iopub.status.idle":"2022-07-25T09:09:50.171966Z","shell.execute_reply.started":"2022-07-25T08:32:19.530326Z","shell.execute_reply":"2022-07-25T09:09:50.170389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Predicted = vcsft.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:09:50.173811Z","iopub.execute_input":"2022-07-25T09:09:50.175027Z","iopub.status.idle":"2022-07-25T09:10:47.037384Z","shell.execute_reply.started":"2022-07-25T09:09:50.174978Z","shell.execute_reply":"2022-07-25T09:10:47.035883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission['Predicted']= Predicted\nsample_submission.to_csv(\"submission.csv\",index=False)\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:10:47.038972Z","iopub.execute_input":"2022-07-25T09:10:47.039332Z","iopub.status.idle":"2022-07-25T09:10:47.219346Z","shell.execute_reply.started":"2022-07-25T09:10:47.039303Z","shell.execute_reply":"2022-07-25T09:10:47.218018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- This submission has LB score of 0.61428\n\n#### Future thoughts for experiment:\n- Different combinations of soft voting : Adding Linear Discriminant Analysis, Gaussian Naive Bias for analysis as well\n- Use lower threshold for preparing classification data\n- Tune Hyperparameter for each model before Soft Voting\n- Train with different seed (we have observed change in score with different threshholds)\n- Use Cross validation","metadata":{}}]}