{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# K-Means Intro\n\nHi there! This notebook is just a sketch to spark some thoughts on the challenge proposed in the competition. \nFeel free to give any suggestions, use and share. Upvotes are also appreciated.\n\n**Updated**: using now some hints found in Discussions.\n<br>Links:<br>\n    1. https://www.kaggle.com/competitions/tabular-playground-series-jul-2022/discussion/334875<br>\n    2. https://www.kaggle.com/code/ricopue/tps-jul22-clusters-and-lgb/notebook?scriptVersionId=101237416<br>\n    3. https://www.kaggle.com/code/hiro5299834/tps-jul-2022-unsupervised-and-supervised-learning","metadata":{}},{"cell_type":"markdown","source":"# 0.0 Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler, PowerTransformer\nfrom sklearn.cluster       import KMeans, MiniBatchKMeans\nfrom sklearn.datasets      import make_blobs\nfrom sklearn.mixture       import GaussianMixture, BayesianGaussianMixture\n\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T19:21:17.254021Z","iopub.execute_input":"2022-07-20T19:21:17.254680Z","iopub.status.idle":"2022-07-20T19:21:20.814442Z","shell.execute_reply.started":"2022-07-20T19:21:17.254581Z","shell.execute_reply":"2022-07-20T19:21:20.813271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 0.1 Helper Function","metadata":{}},{"cell_type":"code","source":"\n# Set Matplotlib defaults\ndef jupyter_settings():\n    %matplotlib inline\n\n    plt.style.use('bmh')\n    plt.rcParams['figure.figsize'] = [25, 12]\n    plt.rcParams['font.size'] = 24\n\n#     display(HTML('<style>.conteiner{width:100% !important;}</style>'))\n\n    pd.options.display.max_columns = None\n    pd.options.display.max_rows = None\n    pd.set_option('display.expand_frame_repr', False)\n    # configura o pandas para quantidade de casas decimais\n    pd.set_option('display.float_format', lambda x: '%.4f' % x)\n\n    sns.set()\njupyter_settings()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-20T19:21:20.816834Z","iopub.execute_input":"2022-07-20T19:21:20.817312Z","iopub.status.idle":"2022-07-20T19:21:20.833379Z","shell.execute_reply.started":"2022-07-20T19:21:20.817265Z","shell.execute_reply":"2022-07-20T19:21:20.832122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv', index_col='id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:21:20.835532Z","iopub.execute_input":"2022-07-20T19:21:20.836129Z","iopub.status.idle":"2022-07-20T19:21:22.285371Z","shell.execute_reply.started":"2022-07-20T19:21:20.836082Z","shell.execute_reply":"2022-07-20T19:21:22.283942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.0 Saving some memory","metadata":{}},{"cell_type":"code","source":"# Saving some memory (- 16mb)\n\nfor col in df.columns:\n    if df[col].dtype == 'float64':\n        df[col] = df[col].astype('float16')\n\nfor col in df.columns:\n    if df[col].dtype == 'int64':\n        df[col] = df[col].astype('int16')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:21:22.288589Z","iopub.execute_input":"2022-07-20T19:21:22.289016Z","iopub.status.idle":"2022-07-20T19:21:22.389673Z","shell.execute_reply.started":"2022-07-20T19:21:22.288980Z","shell.execute_reply":"2022-07-20T19:21:22.388776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.0 EDA - Visualizing Further","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:21:22.391708Z","iopub.execute_input":"2022-07-20T19:21:22.392177Z","iopub.status.idle":"2022-07-20T19:21:22.919336Z","shell.execute_reply.started":"2022-07-20T19:21:22.392131Z","shell.execute_reply":"2022-07-20T19:21:22.918216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:21:22.920662Z","iopub.execute_input":"2022-07-20T19:21:22.921015Z","iopub.status.idle":"2022-07-20T19:21:22.950553Z","shell.execute_reply.started":"2022-07-20T19:21:22.920984Z","shell.execute_reply":"2022-07-20T19:21:22.949421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.hist(bins=30);","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:22:04.578844Z","iopub.execute_input":"2022-07-20T19:22:04.579739Z","iopub.status.idle":"2022-07-20T19:22:10.846682Z","shell.execute_reply.started":"2022-07-20T19:22:04.579686Z","shell.execute_reply":"2022-07-20T19:22:10.845675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.0 Pre-processing Data","metadata":{}},{"cell_type":"code","source":"features = df.columns.to_list()\n\nbest_features =['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28'] #reference link 1\n\ndrop_features = [col for col in features if col not in best_features]\n\ninteger_features = [\"f_07\", \"f_08\", \"f_09\", \"f_10\", \"f_11\", \"f_12\", \"f_13\"]\n\nfloat_features = [col for col in best_features if col not in integer_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:49:44.809979Z","iopub.execute_input":"2022-07-20T19:49:44.810447Z","iopub.status.idle":"2022-07-20T19:49:44.818664Z","shell.execute_reply.started":"2022-07-20T19:49:44.810412Z","shell.execute_reply":"2022-07-20T19:49:44.817457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(drop_features,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:51:33.501561Z","iopub.execute_input":"2022-07-20T19:51:33.502048Z","iopub.status.idle":"2022-07-20T19:51:33.511757Z","shell.execute_reply.started":"2022-07-20T19:51:33.502008Z","shell.execute_reply":"2022-07-20T19:51:33.510725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1 Applying RobustScaler to float_features, PowerTransform and MinMax to best_features","metadata":{}},{"cell_type":"code","source":"def iqr_outliers(df,col_list):\n    for col in col_list:\n        q1 = df[col].quantile(0.25)\n        q3 = df[col].quantile(0.75)\n        iqr = q3-q1\n        Lower_tail = q1 - 2 * iqr\n        Upper_tail = q3 + 2 * iqr\n        df.loc[df[col] > Upper_tail,col ]=Upper_tail\n        df.loc[df[col] < Lower_tail,col]=Lower_tail\n    return df\n\n\nX_scaled=iqr_outliers(X,float_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:56:41.470761Z","iopub.execute_input":"2022-07-20T19:56:41.471370Z","iopub.status.idle":"2022-07-20T19:56:41.582395Z","shell.execute_reply.started":"2022-07-20T19:56:41.471310Z","shell.execute_reply":"2022-07-20T19:56:41.580782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_scaled[best_features] = PowerTransformer().fit_transform(X_scaled[best_features])\nX_scaled[best_features] = MinMaxScaler().fit_transform(X_scaled[best_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:56:44.546533Z","iopub.execute_input":"2022-07-20T19:56:44.547008Z","iopub.status.idle":"2022-07-20T19:56:47.255915Z","shell.execute_reply.started":"2022-07-20T19:56:44.546969Z","shell.execute_reply":"2022-07-20T19:56:47.254338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.0 Unsupervised Modeling ","metadata":{}},{"cell_type":"markdown","source":"## 4.1 GaussianMixture","metadata":{}},{"cell_type":"code","source":"%%time\n\nK = 7\nTIMES = 10\nITER = 200\n\n# km = KMeans(n_clusters=K, n_init = TIMES, max_iter=ITER, random_state=42)\n# pred2 = km.fit_predict(X_test)\n\nGM = GaussianMixture(n_components=K, covariance_type = 'full', n_init=TIMES, max_iter=ITER, random_state=42)\nGM.fit(X_scaled[best_features])\n\npredict=GM.predict(X_scaled[best_features])\nproba=GM.predict_proba(X_scaled[best_features])\n\nX_scaled['predict']=predict\nX_scaled['predict_proba']=np.max(proba, axis=1) \n\ntrain_index=X_scaled[X_scaled.predict_proba > 0.7].index\nprint(round(len(train_index)/len(X_scaled),3))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:06:04.735609Z","iopub.execute_input":"2022-07-20T20:06:04.736103Z","iopub.status.idle":"2022-07-20T20:07:40.979280Z","shell.execute_reply.started":"2022-07-20T20:06:04.736059Z","shell.execute_reply":"2022-07-20T20:07:40.977950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(GM.score(X_scaled[best_features]))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:07:40.981448Z","iopub.execute_input":"2022-07-20T20:07:40.982241Z","iopub.status.idle":"2022-07-20T20:07:41.232343Z","shell.execute_reply.started":"2022-07-20T20:07:40.982187Z","shell.execute_reply":"2022-07-20T20:07:41.231017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2 BayesianGaussianMixture","metadata":{}},{"cell_type":"code","source":"%%time\n\nK = 7\nTIMES = 10\nITER = 200\nSEED = 42\nN_FOLDS = 10\n\n# km = KMeans(n_clusters=K, n_init = TIMES, max_iter=ITER, random_state=42)\n# pred2 = km.fit_predict(X_test)\n\nBGM = BayesianGaussianMixture(n_components=K, covariance_type = 'full', n_init=TIMES, max_iter=ITER, random_state=SEED)\nBGM.fit(X_scaled[best_features])\n\npredict=GM.predict(X_scaled[best_features])\nproba=GM.predict_proba(X_scaled[best_features])\n\nX_scaled['predict']=predict\nX_scaled['predict_proba']=np.max(proba, axis=1) \n\ntrain_index=X_scaled[X_scaled.predict_proba > 0.7].index\nprint(round(len(train_index)/len(X_scaled),3))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:13:55.896340Z","iopub.execute_input":"2022-07-20T20:13:55.896800Z","iopub.status.idle":"2022-07-20T20:20:09.165070Z","shell.execute_reply.started":"2022-07-20T20:13:55.896762Z","shell.execute_reply":"2022-07-20T20:20:09.163820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(BGM.score(X_scaled[best_features]))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:20:09.167715Z","iopub.execute_input":"2022-07-20T20:20:09.172901Z","iopub.status.idle":"2022-07-20T20:20:09.410661Z","shell.execute_reply.started":"2022-07-20T20:20:09.172813Z","shell.execute_reply":"2022-07-20T20:20:09.409366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5.0 Supervised Modeling","metadata":{}},{"cell_type":"markdown","source":"## 5.0.1 Parameters\n\n* Reference:<br>\n    https://www.kaggle.com/code/hiro5299834/tps-jul-2022-unsupervised-and-supervised-learning\n\nThanks,<a href='https://www.kaggle.com/hiro5299834'>@bizen</a>","metadata":{}},{"cell_type":"code","source":"params_lgb = {\n    'objective': 'multiclass',\n    'boosting': 'gbdt',\n    'learning_rate': 4e-2,\n    'verbosity': -1,\n    'n_jobs': -1,\n    'num_classes': K,\n    'random_state': SEED\n}\n\nparams_xgb = {\n    'booster': 'gbtree',\n    'objective': 'multi:softprob',\n    'learning_rate': 4e-2,\n    'num_class': K,\n    'seed': SEED,\n    'gpu_id': 0,\n    'tree_method': 'gpu_hist',\n    'predictor': 'gpu_predictor'\n    }\n\nparams_ctb = {\n    'objective': 'MultiClass',\n    'bootstrap_type': 'Poisson',\n    #'boosting_type': 'Ordered',  # or 'Plain'\n    'classes_count': SEED,\n    'num_boost_round': 20000,\n    'learning_rate': 4e-1,\n    'random_seed': SEED,\n    'task_type': 'GPU'\n    \n}","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:29:25.179103Z","iopub.execute_input":"2022-07-20T20:29:25.180225Z","iopub.status.idle":"2022-07-20T20:29:25.190465Z","shell.execute_reply.started":"2022-07-20T20:29:25.180182Z","shell.execute_reply":"2022-07-20T20:29:25.189190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.1 LGBM","metadata":{}},{"cell_type":"code","source":"X_test = X_scaled.loc[train_index][best_features]\ny_test = X_scaled.loc[train_index]['predict']\n\n# params_lgb = {'learning_rate': 0.07,'objective': 'multiclass','boosting': 'gbdt','verbosity': -1,'n_jobs': -1, 'num_classes':N_cluster} \n\nmodel_list=[]\n\ngkf = StratifiedKFold(N_FOLDS)\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X_test,y_test)):   \n    print(f\"===== fold{fold} =====\")\n    tr_dataset = lgb.Dataset(X_test.iloc[train_idx],y_test.iloc[train_idx],feature_name = best_features)\n    vl_dataset = lgb.Dataset(X_test.iloc[valid_idx],y_test.iloc[valid_idx],feature_name = best_features)\n    \n    model = lgb.train(params = params_lgb, \n                train_set = tr_dataset, \n                valid_sets =  vl_dataset, \n                num_boost_round = 5000, \n                callbacks=[ lgb.early_stopping(stopping_rounds=500, verbose=True), lgb.log_evaluation(period=200)])  \n    \n    model_list.append(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:35:19.221603Z","iopub.execute_input":"2022-07-20T20:35:19.222084Z","iopub.status.idle":"2022-07-20T20:46:41.482103Z","shell.execute_reply.started":"2022-07-20T20:35:19.222046Z","shell.execute_reply":"2022-07-20T20:46:41.480707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_preds=0\nfor model in model_list:\n    lgb_preds+=model.predict(X_scaled[best_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:46:41.484677Z","iopub.execute_input":"2022-07-20T20:46:41.485204Z","iopub.status.idle":"2022-07-20T20:54:09.350839Z","shell.execute_reply.started":"2022-07-20T20:46:41.485155Z","shell.execute_reply":"2022-07-20T20:54:09.349936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2 XGB","metadata":{}},{"cell_type":"code","source":"# X_test = X_scaled.loc[train_index][best_features]\n# y_test = X_scaled.loc[train_index]['predict']\n\n# # params_lgb = {'learning_rate': 0.07,'objective': 'multiclass','boosting': 'gbdt','verbosity': -1,'n_jobs': -1, 'num_classes':N_cluster} \n\n# model_list=[]\n\n# gkf = StratifiedKFold(N_FOLDS)\n# for fold, (train_idx, valid_idx) in enumerate(gkf.split(X,y)):   \n#     print(f\"===== fold{fold} =====\")\n#     tr_dataset = lgb.Dataset(X.iloc[train_idx],y.iloc[train_idx],feature_name = best_cols)\n#     vl_dataset = lgb.Dataset(X.iloc[valid_idx],y.iloc[valid_idx],feature_name = best_cols)\n    \n#     model = xgb.train(params_xgb,\n#                       dtrain=xgb_train,\n#                       evals=[(xgb_train, 'train'),(xgb_valid, 'eval')],\n#                       verbose_eval=False,\n#                       num_boost_round=20000,\n#                       early_stopping_rounds=500,\n#                      )    \n#     model_list.append(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:54:09.355842Z","iopub.execute_input":"2022-07-20T20:54:09.358179Z","iopub.status.idle":"2022-07-20T20:54:09.364899Z","shell.execute_reply.started":"2022-07-20T20:54:09.358133Z","shell.execute_reply":"2022-07-20T20:54:09.363926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xgb_preds=0\n# for model in model_list:\n#     xgb_preds+=model.predict(train_scaled[best_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:54:09.367434Z","iopub.execute_input":"2022-07-20T20:54:09.367757Z","iopub.status.idle":"2022-07-20T20:54:09.382677Z","shell.execute_reply.started":"2022-07-20T20:54:09.367730Z","shell.execute_reply":"2022-07-20T20:54:09.381378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"## Submission\n\nsub_df = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\n\nsub_df[\"Predicted\"] = np.argmax(lgb_preds, axis=1)\nsub_df.to_csv('submission.csv', index=False)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:54:09.384884Z","iopub.execute_input":"2022-07-20T20:54:09.385378Z","iopub.status.idle":"2022-07-20T20:54:09.597716Z","shell.execute_reply.started":"2022-07-20T20:54:09.385339Z","shell.execute_reply":"2022-07-20T20:54:09.596950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}