{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install sklego","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:16.391553Z","iopub.execute_input":"2022-07-29T08:10:16.392006Z","iopub.status.idle":"2022-07-29T08:10:25.789842Z","shell.execute_reply.started":"2022-07-29T08:10:16.391966Z","shell.execute_reply":"2022-07-29T08:10:25.788713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom scipy.stats import shapiro\nfrom termcolor import colored\nfrom scipy import stats\nfrom sklearn.metrics import silhouette_score\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom tqdm import tqdm\nfrom sklego.mixture import BayesianGMMClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T08:10:25.792793Z","iopub.execute_input":"2022-07-29T08:10:25.793564Z","iopub.status.idle":"2022-07-29T08:10:25.804240Z","shell.execute_reply.started":"2022-07-29T08:10:25.793519Z","shell.execute_reply":"2022-07-29T08:10:25.803047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:25.805948Z","iopub.execute_input":"2022-07-29T08:10:25.806548Z","iopub.status.idle":"2022-07-29T08:10:26.359156Z","shell.execute_reply.started":"2022-07-29T08:10:25.806510Z","shell.execute_reply":"2022-07-29T08:10:26.358198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(25,25)})\nfor i, column in enumerate(list(df.columns), 1):\n    plt.subplot(5,6,i)\n    p=sns.histplot(x=column,data=df.sample(1000),stat='count',kde=True,color='green')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:26.362465Z","iopub.execute_input":"2022-07-29T08:10:26.362894Z","iopub.status.idle":"2022-07-29T08:10:32.646159Z","shell.execute_reply.started":"2022-07-29T08:10:26.362852Z","shell.execute_reply":"2022-07-29T08:10:32.645102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(40, 25))\n# sns.set_style('white')\n# mask = np.triu(np.ones_like(df.corr(), dtype=np.bool))\n# heatmap = sns.heatmap(df.corr(), mask=mask,annot=True, cmap='BrBG', linewidths = 2)\n# heatmap.set_title('Triangle Correlation Heatmap', fontdict={'fontsize':30}, pad=16);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-29T08:10:32.647744Z","iopub.execute_input":"2022-07-29T08:10:32.648177Z","iopub.status.idle":"2022-07-29T08:10:32.652567Z","shell.execute_reply.started":"2022-07-29T08:10:32.648135Z","shell.execute_reply":"2022-07-29T08:10:32.651740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f7, 8 9 10 11 12 22 23 24 25","metadata":{}},{"cell_type":"code","source":"df = df.drop(['id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:32.653917Z","iopub.execute_input":"2022-07-29T08:10:32.654875Z","iopub.status.idle":"2022-07-29T08:10:32.670933Z","shell.execute_reply.started":"2022-07-29T08:10:32.654840Z","shell.execute_reply":"2022-07-29T08:10:32.669978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Univariate normality test\nbest_features = []\nfor col in df.columns:\n    stat, p_value = shapiro(df[col])\n    alpha = 0.05    # significance level\n    if p_value <= alpha: \n        best_features.append(col)\nprint(best_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:32.672418Z","iopub.execute_input":"2022-07-29T08:10:32.673487Z","iopub.status.idle":"2022-07-29T08:10:32.895244Z","shell.execute_reply.started":"2022-07-29T08:10:32.673450Z","shell.execute_reply":"2022-07-29T08:10:32.894264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Outlier processing\n# tmp_df = df\n# plt.figure(figsize=(20,10)) \n# sns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmp_df)).set_title('Boxplot of each feature',size=15)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:32.896530Z","iopub.execute_input":"2022-07-29T08:10:32.898292Z","iopub.status.idle":"2022-07-29T08:10:32.902514Z","shell.execute_reply.started":"2022-07-29T08:10:32.898251Z","shell.execute_reply":"2022-07-29T08:10:32.901610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing Outliers**\nscore drops by 0.02","metadata":{}},{"cell_type":"markdown","source":"### Normalize data\n","metadata":{}},{"cell_type":"code","source":"#and here are what we got\ndff = df.copy()\ndffs = PowerTransformer().fit_transform(dff)\ndffs = RobustScaler().fit_transform(dffs)\n\npower_features =[ 'f_07','f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24','f_26','f_27','f_26','f_27','f_28']\n\ndffs_ = pd.DataFrame(dffs, columns=dff.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:32.904114Z","iopub.execute_input":"2022-07-29T08:10:32.904804Z","iopub.status.idle":"2022-07-29T08:10:36.152977Z","shell.execute_reply.started":"2022-07-29T08:10:32.904766Z","shell.execute_reply":"2022-07-29T08:10:36.152046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dffs_","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:36.158572Z","iopub.execute_input":"2022-07-29T08:10:36.159360Z","iopub.status.idle":"2022-07-29T08:10:36.194225Z","shell.execute_reply.started":"2022-07-29T08:10:36.159322Z","shell.execute_reply":"2022-07-29T08:10:36.193097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_data = dffs_[power_features]\ntest_data = scaled_data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:36.195840Z","iopub.execute_input":"2022-07-29T08:10:36.196211Z","iopub.status.idle":"2022-07-29T08:10:36.206152Z","shell.execute_reply.started":"2022-07-29T08:10:36.196175Z","shell.execute_reply":"2022-07-29T08:10:36.205097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = [0,1,2,3,4,5,6]\npred_test = pd.DataFrame(np.zeros((scaled_data.shape[0],7)), columns = values)\n\nfor seed in tqdm(range(2)):\n    \n    data = pd.DataFrame(index = dff.index)\n    gmm = BayesianGaussianMixture(\n            n_components=7,\n            random_state = seed,\n            tol = 0.01,\n            covariance_type = 'full',\n            max_iter = 100,\n            n_init=3\n          )\n    \n    # fitting and probability prediction\n    gmm.fit(scaled_data)\n    pred_seed = gmm.predict_proba(scaled_data) # predict_proba for probabilities\n    \n    # the clusters prediction for the current seed :\n    MAX = np.argmax(pred_seed, axis=1)\n    data[f'pred_{seed}'] = MAX\n    \n    # Sort of the prediction by same value of cluster (for addition of every seed)\n    pred_keys = data[f'pred_{seed}'].value_counts().index.tolist()\n    pred_dict = dict(zip(pred_keys, values))\n    data[f'pred_{seed}'] = data[f'pred_{seed}'].map(pred_dict)\n\n    pred_new = pd.DataFrame(pred_seed).rename(columns = pred_dict)\n    pred_new = pred_new.reindex(sorted(pred_new.columns), axis=1)\n    pred_test += pred_new # Soft voting by probabiliy addition\n\npredictions = np.argmax(np.array(pred_test), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:10:36.207885Z","iopub.execute_input":"2022-07-29T08:10:36.208680Z","iopub.status.idle":"2022-07-29T08:13:55.595623Z","shell.execute_reply.started":"2022-07-29T08:10:36.208641Z","shell.execute_reply":"2022-07-29T08:13:55.594591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:43:16.399934Z","iopub.execute_input":"2022-07-29T08:43:16.400426Z","iopub.status.idle":"2022-07-29T08:43:16.413177Z","shell.execute_reply.started":"2022-07-29T08:43:16.400384Z","shell.execute_reply":"2022-07-29T08:43:16.412281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def best_class(df):\n    new_df = df.copy()\n    new_df[\"highest_prob\"] = df.max(axis=1)\n    new_df[\"best_class\"] = df.idxmax(axis=1)\n    return new_df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:13:55.601283Z","iopub.execute_input":"2022-07-29T08:13:55.604742Z","iopub.status.idle":"2022-07-29T08:13:55.613784Z","shell.execute_reply.started":"2022-07-29T08:13:55.604696Z","shell.execute_reply":"2022-07-29T08:13:55.612399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#normalise so rows sums to 1\npredictions_df = pred_test.div(pred_test.sum(axis=1), axis=0)\npredictions_df = best_class(predictions_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:44:45.660806Z","iopub.execute_input":"2022-07-29T08:44:45.661222Z","iopub.status.idle":"2022-07-29T08:44:45.861195Z","shell.execute_reply.started":"2022-07-29T08:44:45.661190Z","shell.execute_reply":"2022-07-29T08:44:45.860267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:44:54.450968Z","iopub.execute_input":"2022-07-29T08:44:54.451574Z","iopub.status.idle":"2022-07-29T08:44:54.473423Z","shell.execute_reply.started":"2022-07-29T08:44:54.451537Z","shell.execute_reply":"2022-07-29T08:44:54.472505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Using idea from: https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting\n\n# Creating Best data based on predicted probability of BGM model\nn_components = 7\nscaled_data[\"predict\"] = predictions\nscaled_data[\"predict_proba\"] = 0\ntrain_index = np.array([])\n\nfor n in range(n_components):\n    scaled_data[f\"bgm_proba_{n}\"] = predictions_df[n]\n    scaled_data.loc[scaled_data.predict == n, \"bgm_proba\"] = scaled_data[\n        f\"bgm_proba_{n}\"\n    ]\n\nfor n in range(n_components):\n    median = scaled_data[scaled_data.predict == n][\"bgm_proba\"].median()\n\n    # Experiment with different thresholds\n    # Higher thereshold might overfit\n    n_inx = scaled_data[\n        (scaled_data.predict == n) & (scaled_data.bgm_proba >= 0.8)\n    ].index\n\n    train_index = np.concatenate((train_index, n_inx))\n    print(\n        f\"class:{n}\",\n        f\"median: {round(median,4)}\",\n        \"Training data:\"\n        + str(round(len(n_inx) / len(scaled_data[(scaled_data.predict == n)]), 2) * 100)\n        + \"%\",\n    )\n\n\nprint(f\"\\nSize of Training data : {len(train_index)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:51:14.815870Z","iopub.execute_input":"2022-07-29T08:51:14.816478Z","iopub.status.idle":"2022-07-29T08:51:14.955842Z","shell.execute_reply.started":"2022-07-29T08:51:14.816440Z","shell.execute_reply":"2022-07-29T08:51:14.953712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_data","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:53:24.264703Z","iopub.execute_input":"2022-07-29T08:53:24.265148Z","iopub.status.idle":"2022-07-29T08:53:24.310855Z","shell.execute_reply.started":"2022-07-29T08:53:24.265111Z","shell.execute_reply":"2022-07-29T08:53:24.309895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = scaled_data.loc[train_index][power_features]\ny = scaled_data.loc[train_index][\"predict\"].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:54:55.238246Z","iopub.execute_input":"2022-07-29T08:54:55.238738Z","iopub.status.idle":"2022-07-29T08:54:55.296940Z","shell.execute_reply.started":"2022-07-29T08:54:55.238694Z","shell.execute_reply":"2022-07-29T08:54:55.295840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.reset_index(drop=True)\ny.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:55:24.681619Z","iopub.execute_input":"2022-07-29T08:55:24.682152Z","iopub.status.idle":"2022-07-29T08:55:24.703411Z","shell.execute_reply.started":"2022-07-29T08:55:24.682119Z","shell.execute_reply":"2022-07-29T08:55:24.701691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# https://www.kaggle.com/code/karlcini/bayesiangmmclassifier\n\nbgm = BayesianGMMClassifier(\n    n_components=7,\n    random_state=42,\n    # tol =1e-3,\n    covariance_type=\"full\",\n    max_iter=500,\n    n_init=7,\n    init_params=\"kmeans\", # you can use k-means++\n)\nbgm.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T08:55:32.512037Z","iopub.execute_input":"2022-07-29T08:55:32.512674Z","iopub.status.idle":"2022-07-29T09:05:04.921778Z","shell.execute_reply.started":"2022-07-29T08:55:32.512618Z","shell.execute_reply":"2022-07-29T09:05:04.920711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\ny_pred = bgm.predict(X)\naccuracy_score(y, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:06:23.771082Z","iopub.execute_input":"2022-07-29T09:06:23.771636Z","iopub.status.idle":"2022-07-29T09:06:24.862225Z","shell.execute_reply.started":"2022-07-29T09:06:23.771600Z","shell.execute_reply":"2022-07-29T09:06:24.860790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_bgm = bgm.predict(test_data[power_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:06:52.043320Z","iopub.execute_input":"2022-07-29T09:06:52.043675Z","iopub.status.idle":"2022-07-29T09:06:53.361968Z","shell.execute_reply.started":"2022-07-29T09:06:52.043645Z","shell.execute_reply":"2022-07-29T09:06:53.360872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pl = sns.countplot(x=pred_bgm)\npl.set_title(\"Distribution of clusters\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:07:08.465805Z","iopub.execute_input":"2022-07-29T09:07:08.466367Z","iopub.status.idle":"2022-07-29T09:07:08.854052Z","shell.execute_reply.started":"2022-07-29T09:07:08.466324Z","shell.execute_reply":"2022-07-29T09:07:08.853099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:07:14.371696Z","iopub.execute_input":"2022-07-29T09:07:14.372069Z","iopub.status.idle":"2022-07-29T09:07:14.401130Z","shell.execute_reply.started":"2022-07-29T09:07:14.372019Z","shell.execute_reply":"2022-07-29T09:07:14.400226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"Predicted\"] = predictions\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:07:16.602767Z","iopub.execute_input":"2022-07-29T09:07:16.603144Z","iopub.status.idle":"2022-07-29T09:07:16.742516Z","shell.execute_reply.started":"2022-07-29T09:07:16.603111Z","shell.execute_reply":"2022-07-29T09:07:16.741564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:07:18.774349Z","iopub.execute_input":"2022-07-29T09:07:18.774749Z","iopub.status.idle":"2022-07-29T09:07:18.788457Z","shell.execute_reply.started":"2022-07-29T09:07:18.774713Z","shell.execute_reply":"2022-07-29T09:07:18.787192Z"},"trusted":true},"execution_count":null,"outputs":[]}]}