{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc,random,os\nfrom sklearn.preprocessing import PowerTransformer,MinMaxScaler,RobustScaler\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn import metrics\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:23:17.953775Z","iopub.execute_input":"2022-07-19T11:23:17.954354Z","iopub.status.idle":"2022-07-19T11:23:20.401945Z","shell.execute_reply.started":"2022-07-19T11:23:17.954225Z","shell.execute_reply":"2022-07-19T11:23:20.400691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=2022):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed=2020\nseed_everything(seed)\n\nN_FOLDS =10\nN_cluster=7","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:23:20.404489Z","iopub.execute_input":"2022-07-19T11:23:20.405391Z","iopub.status.idle":"2022-07-19T11:23:20.413010Z","shell.execute_reply.started":"2022-07-19T11:23:20.405341Z","shell.execute_reply":"2022-07-19T11:23:20.411526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\n\nall_cols=[col for col in train.columns]\nbest_cols =['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']\ndrop_cols=[col for col in all_cols if col not in best_cols]\n\ntrain=train.drop(drop_cols,axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:23:20.416334Z","iopub.execute_input":"2022-07-19T11:23:20.417026Z","iopub.status.idle":"2022-07-19T11:23:21.786248Z","shell.execute_reply.started":"2022-07-19T11:23:20.416991Z","shell.execute_reply":"2022-07-19T11:23:21.784777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int_cols=['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13']\nfloat_cols=[col for col in best_cols if col not in int_cols]\n\ndef iqr_outliers(df,col_list):\n    for col in col_list:\n        q1 = df[col].quantile(0.25)\n        q3 = df[col].quantile(0.75)\n        iqr = q3-q1\n        Lower_tail = q1 - 2 * iqr\n        Upper_tail = q3 + 2 * iqr\n        df.loc[df[col] > Upper_tail,col ]=Upper_tail\n        df.loc[df[col] < Lower_tail,col]=Lower_tail\n    return df\n\n\ntrain=iqr_outliers(train,float_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:23:21.789314Z","iopub.execute_input":"2022-07-19T11:23:21.790932Z","iopub.status.idle":"2022-07-19T11:23:21.852682Z","shell.execute_reply.started":"2022-07-19T11:23:21.790881Z","shell.execute_reply":"2022-07-19T11:23:21.851465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_scaled=train.copy()\ntrain_scaled[best_cols]= PowerTransformer().fit_transform(train_scaled[best_cols])\ntrain_scaled[best_cols]= MinMaxScaler().fit_transform(train_scaled[best_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:23:21.854225Z","iopub.execute_input":"2022-07-19T11:23:21.854694Z","iopub.status.idle":"2022-07-19T11:23:23.628253Z","shell.execute_reply.started":"2022-07-19T11:23:21.854650Z","shell.execute_reply":"2022-07-19T11:23:23.627142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BGM = BayesianGaussianMixture(n_components=N_cluster,covariance_type='full', max_iter=300, random_state=1,n_init = 5)\nBGM.fit(train_scaled[best_cols])\n\npredict=BGM.predict(train_scaled[best_cols])\nproba=BGM.predict_proba(train_scaled[best_cols])\n\ntrain_scaled['predict']=predict\ntrain_scaled['predict_proba']=np.max(proba, axis=1) \n\ntrain_index=train_scaled[train_scaled.predict_proba > 0.8].index\nprint(round(len(train_index)/len(train),3))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:01.915160Z","iopub.execute_input":"2022-07-19T11:28:01.915704Z","iopub.status.idle":"2022-07-19T11:30:41.201696Z","shell.execute_reply.started":"2022-07-19T11:28:01.915654Z","shell.execute_reply":"2022-07-19T11:30:41.197745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX=train_scaled.loc[train_index][best_cols]\ny=train_scaled.loc[train_index]['predict']\n\nparams_lgb = {'learning_rate': 0.07,'objective': 'multiclass','boosting': 'gbdt','verbosity': -1,'n_jobs': -1, 'num_classes':N_cluster} \n\nmodel_list=[]\n\ngkf = StratifiedKFold(N_FOLDS)\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X,y)):   \n\n    tr_dataset = lgb.Dataset(X.iloc[train_idx],y.iloc[train_idx],feature_name = best_cols)\n    vl_dataset = lgb.Dataset(X.iloc[valid_idx],y.iloc[valid_idx],feature_name = best_cols)\n    \n    model = lgb.train(params = params_lgb, \n                train_set = tr_dataset, \n                valid_sets =  vl_dataset, \n                num_boost_round = 5000, \n                callbacks=[ lgb.early_stopping(stopping_rounds=300, verbose=True), lgb.log_evaluation(period=200)])  \n    \n    model_list.append(model) \n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:42:30.060528Z","iopub.execute_input":"2022-07-19T11:42:30.060935Z","iopub.status.idle":"2022-07-19T11:49:02.627870Z","shell.execute_reply.started":"2022-07-19T11:42:30.060903Z","shell.execute_reply":"2022-07-19T11:49:02.626799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_preds=0\nfor model in model_list:\n    lgb_preds+=model.predict(train_scaled[best_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:04:17.742363Z","iopub.execute_input":"2022-07-19T12:04:17.742809Z","iopub.status.idle":"2022-07-19T12:07:39.395597Z","shell.execute_reply.started":"2022-07-19T12:04:17.742772Z","shell.execute_reply":"2022-07-19T12:07:39.394593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nss['Predicted']=np.argmax(lgb_preds, axis=1)\nss.to_csv(\"submission.csv\",index=False)\nss.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:07:39.397670Z","iopub.execute_input":"2022-07-19T12:07:39.398360Z","iopub.status.idle":"2022-07-19T12:07:39.607047Z","shell.execute_reply.started":"2022-07-19T12:07:39.398322Z","shell.execute_reply":"2022-07-19T12:07:39.605704Z"},"trusted":true},"execution_count":null,"outputs":[]}]}