{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install scikit-lego","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:47:43.775781Z","iopub.execute_input":"2022-07-31T23:47:43.776149Z","iopub.status.idle":"2022-07-31T23:47:54.354158Z","shell.execute_reply.started":"2022-07-31T23:47:43.776118Z","shell.execute_reply":"2022-07-31T23:47:54.352827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom scipy.stats import shapiro\nfrom termcolor import colored\nfrom scipy import stats\nfrom sklearn.metrics import silhouette_score\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom sklearn.preprocessing import PowerTransformer, MaxAbsScaler\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom tqdm import tqdm\nfrom sklego.mixture import BayesianGMMClassifier\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T23:47:54.356547Z","iopub.execute_input":"2022-07-31T23:47:54.357234Z","iopub.status.idle":"2022-07-31T23:47:54.366098Z","shell.execute_reply.started":"2022-07-31T23:47:54.357191Z","shell.execute_reply":"2022-07-31T23:47:54.364766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nSAMPLE = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:47:54.367529Z","iopub.execute_input":"2022-07-31T23:47:54.368130Z","iopub.status.idle":"2022-07-31T23:47:54.953805Z","shell.execute_reply.started":"2022-07-31T23:47:54.368093Z","shell.execute_reply":"2022-07-31T23:47:54.952657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(25,25)})\nfor i, column in enumerate(list(DATA.columns), 1):\n    plt.subplot(5,6,i)\n    p=sns.histplot(x=column,data=DATA.sample(1000),stat='count',kde=True,color='green')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:47:54.956851Z","iopub.execute_input":"2022-07-31T23:47:54.957232Z","iopub.status.idle":"2022-07-31T23:48:01.310799Z","shell.execute_reply.started":"2022-07-31T23:47:54.957194Z","shell.execute_reply":"2022-07-31T23:48:01.309968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40, 25))\nsns.set_style('white')\nmask = np.triu(np.ones_like(DATA.corr(), dtype=np.bool))\nheatmap = sns.heatmap(DATA.corr(), mask=mask,annot=True, cmap='BrBG', linewidths = 2)\nheatmap.set_title('Triangle Correlation Heatmap', fontdict={'fontsize':30}, pad=16);","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:01.312233Z","iopub.execute_input":"2022-07-31T23:48:01.312757Z","iopub.status.idle":"2022-07-31T23:48:04.486816Z","shell.execute_reply.started":"2022-07-31T23:48:01.312721Z","shell.execute_reply":"2022-07-31T23:48:04.485964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA = DATA.drop(['id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:04.488334Z","iopub.execute_input":"2022-07-31T23:48:04.488915Z","iopub.status.idle":"2022-07-31T23:48:04.500511Z","shell.execute_reply.started":"2022-07-31T23:48:04.488868Z","shell.execute_reply":"2022-07-31T23:48:04.499318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Univariate normality test\nfor col in DATA.columns:\n    features = []\n    stat, p_value = shapiro(DATA[col])\n    alpha = 0.05    # significance level\n    if p_value > alpha: \n        result = colored('Accepted', 'green')\n    else:\n        result = colored('Rejected','red')\n#         features = features + \"colname\"\n    print('Feature: {}\\t Hypothesis: {}'.format(col, result))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:04.501988Z","iopub.execute_input":"2022-07-31T23:48:04.502436Z","iopub.status.idle":"2022-07-31T23:48:04.770865Z","shell.execute_reply.started":"2022-07-31T23:48:04.502390Z","shell.execute_reply":"2022-07-31T23:48:04.769766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Outlier processing\ntmp_DATA = DATA\nplt.figure(figsize=(20,10)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmp_DATA)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:04.775181Z","iopub.execute_input":"2022-07-31T23:48:04.776175Z","iopub.status.idle":"2022-07-31T23:48:07.831378Z","shell.execute_reply.started":"2022-07-31T23:48:04.776135Z","shell.execute_reply":"2022-07-31T23:48:07.830408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:07.832873Z","iopub.execute_input":"2022-07-31T23:48:07.833870Z","iopub.status.idle":"2022-07-31T23:48:08.019654Z","shell.execute_reply.started":"2022-07-31T23:48:07.833830Z","shell.execute_reply":"2022-07-31T23:48:08.018312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normalize data\n","metadata":{}},{"cell_type":"code","source":"df = DATA.copy()\ncols = list(df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:08.024771Z","iopub.execute_input":"2022-07-31T23:48:08.027041Z","iopub.status.idle":"2022-07-31T23:48:08.036613Z","shell.execute_reply.started":"2022-07-31T23:48:08.027007Z","shell.execute_reply":"2022-07-31T23:48:08.035507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#and here are what we got\n# DATA=DATA.drop('id',axis=1)\ncols_select =['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\ndff = DATA[cols_select]\n\ndffs = dff.copy()\ndffs = PowerTransformer().fit_transform(dffs)\ndffs = pd.DataFrame(dffs, columns=cols_select)\ndffs\n\n# scaled_data.columns = DATA.columns\n# scaled_data.head()\n# drop_feats = [f'f_0{i}' for i in range(7)]\n# drop_feats = drop_feats + [f'f_{i}' for i in range(14,22)]\n# scaled_data = scaled_data.drop(drop_feats, axis=1)\n# scaled_data.head()\n# DATA_rest = DATA.drop(best_features, axis=1)\n# DATA_rest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:08.038575Z","iopub.execute_input":"2022-07-31T23:48:08.039166Z","iopub.status.idle":"2022-07-31T23:48:09.591633Z","shell.execute_reply.started":"2022-07-31T23:48:08.039127Z","shell.execute_reply":"2022-07-31T23:48:09.589450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# power = PowerTransformer().fit(data_ma)\n# data_p = power.transform(data_ma)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:09.593294Z","iopub.execute_input":"2022-07-31T23:48:09.594167Z","iopub.status.idle":"2022-07-31T23:48:09.598762Z","shell.execute_reply.started":"2022-07-31T23:48:09.594126Z","shell.execute_reply":"2022-07-31T23:48:09.597720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dffs = data_p","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:09.600168Z","iopub.execute_input":"2022-07-31T23:48:09.601086Z","iopub.status.idle":"2022-07-31T23:48:09.608079Z","shell.execute_reply.started":"2022-07-31T23:48:09.601050Z","shell.execute_reply":"2022-07-31T23:48:09.606989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_prime = pd.read_csv('../input/the-fine-art-of-fine-tuning/submission.csv', index_col=[0])\n\nsub_prime['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:09.609555Z","iopub.execute_input":"2022-07-31T23:48:09.610560Z","iopub.status.idle":"2022-07-31T23:48:09.652516Z","shell.execute_reply.started":"2022-07-31T23:48:09.610523Z","shell.execute_reply":"2022-07-31T23:48:09.651535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_prime['Predicted'] += -1\n\nsub_prime['Predicted'].value_counts().plot(kind='bar')\nsub_prime['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:09.653737Z","iopub.execute_input":"2022-07-31T23:48:09.654073Z","iopub.status.idle":"2022-07-31T23:48:09.966409Z","shell.execute_reply.started":"2022-07-31T23:48:09.654037Z","shell.execute_reply":"2022-07-31T23:48:09.965421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"support = pd.read_csv('../input/tps-jul22-clustering-ensembling/submission.csv', index_col=[0])\n\nsupport['Predicted'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:09.967928Z","iopub.execute_input":"2022-07-31T23:48:09.968279Z","iopub.status.idle":"2022-07-31T23:48:10.010299Z","shell.execute_reply.started":"2022-07-31T23:48:09.968242Z","shell.execute_reply":"2022-07-31T23:48:10.009375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"support['Predicted'] += -1\nsupport['Predicted'].value_counts().plot(kind='bar')\nsupport['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:10.011773Z","iopub.execute_input":"2022-07-31T23:48:10.012115Z","iopub.status.idle":"2022-07-31T23:48:10.323321Z","shell.execute_reply.started":"2022-07-31T23:48:10.012080Z","shell.execute_reply":"2022-07-31T23:48:10.322379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_2 = pd.DataFrame(data_p, columns = data_1.columns)\n# data_test = data_2[cols_select]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:48:10.324871Z","iopub.execute_input":"2022-07-31T23:48:10.325233Z","iopub.status.idle":"2022-07-31T23:48:10.330552Z","shell.execute_reply.started":"2022-07-31T23:48:10.325196Z","shell.execute_reply":"2022-07-31T23:48:10.329439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(dffs)\ny = np.array(sub_prime)\ns = np.array(support)\n\nbgm = BayesianGMMClassifier(n_components=7, random_state=1, tol=0.001, max_iter=200, n_init=3, verbose=0, init_params = 'kmeans')\n\nbgm.fit(X,y)\nproba = bgm.predict_proba(X)\n\nbgm.fit(X,s)\nprobs = bgm.predict_proba(X)\n\nproba.shape, probs.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:49.566875Z","iopub.execute_input":"2022-07-31T23:50:49.567495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prob = np.concatenate((proba, probs*0.82), axis=1)\nprob.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.526915Z","iopub.status.idle":"2022-07-31T23:50:46.535158Z","shell.execute_reply.started":"2022-07-31T23:50:46.534835Z","shell.execute_reply":"2022-07-31T23:50:46.534866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(prob, axis=1)\npred, min(pred), max(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.539908Z","iopub.status.idle":"2022-07-31T23:50:46.542711Z","shell.execute_reply.started":"2022-07-31T23:50:46.542394Z","shell.execute_reply":"2022-07-31T23:50:46.542423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.zeros(shape=(7, 7), dtype=int)\nfor n1, n2 in zip(y, s):\n    clusters[n1, n2] += 1\n    \nclusters","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.544313Z","iopub.status.idle":"2022-07-31T23:50:46.545187Z","shell.execute_reply.started":"2022-07-31T23:50:46.544891Z","shell.execute_reply":"2022-07-31T23:50:46.544918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_clusters = np.argmax(clusters, axis=0)\nmax_clusters","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.551638Z","iopub.status.idle":"2022-07-31T23:50:46.556878Z","shell.execute_reply.started":"2022-07-31T23:50:46.556574Z","shell.execute_reply":"2022-07-31T23:50:46.556604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(pred)):\n    \n    if (pred[i] == 7): \n        pred[i] = max_clusters[0]\n    if (pred[i] == 8): \n        pred[i] = max_clusters[1]\n    if (pred[i] == 9): \n        pred[i] = max_clusters[2]\n    if (pred[i] == 10): \n        pred[i] = max_clusters[3]\n    if (pred[i] == 11): \n        pred[i] = max_clusters[4]\n    if (pred[i] == 12): \n        pred[i] = max_clusters[5]\n    if (pred[i] == 13): \n        pred[i] = max_clusters[6]        \n\npred, min(pred), max(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.558452Z","iopub.status.idle":"2022-07-31T23:50:46.559333Z","shell.execute_reply.started":"2022-07-31T23:50:46.559043Z","shell.execute_reply":"2022-07-31T23:50:46.559071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_bgmm = BayesianGaussianMixture(n_components=7, covariance_type='full', max_iter=100, n_init=5, init_params='random', random_state=0)\n# preds_bgmm = model_bgmm.fit_predict(data_p)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.560894Z","iopub.status.idle":"2022-07-31T23:50:46.561758Z","shell.execute_reply.started":"2022-07-31T23:50:46.561463Z","shell.execute_reply":"2022-07-31T23:50:46.561503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_scaled = pd.DataFrame(PowerTransformer().fit_transform(data_1), columns=data_1.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.574436Z","iopub.status.idle":"2022-07-31T23:50:46.575814Z","shell.execute_reply.started":"2022-07-31T23:50:46.575511Z","shell.execute_reply":"2022-07-31T23:50:46.575540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = SAMPLE.copy()\nsub['Predicted'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.577279Z","iopub.status.idle":"2022-07-31T23:50:46.578052Z","shell.execute_reply.started":"2022-07-31T23:50:46.577787Z","shell.execute_reply":"2022-07-31T23:50:46.577810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)\nprint(\"submission successful\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.579556Z","iopub.status.idle":"2022-07-31T23:50:46.580327Z","shell.execute_reply.started":"2022-07-31T23:50:46.580073Z","shell.execute_reply":"2022-07-31T23:50:46.580097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_bgmm = BayesianGaussianMixture(n_components = 7, # default = 1\n#                           covariance_type = 'full', # defalut = 'full'\n#                           max_iter = 200, # default = 100\n#                           init_params = 'kmeans', #default = kmeans,'k-means++', 'random'\n#                           random_state = 1, # default = None\n#                           verbose = 1, # default = 0\n#                           n_init = 3\n                                  \n#                          )\n# model = model_bgmm\n# model.fit(data_p)\n# prediction_1 = model.predict(data_p)\n# prediction_1","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.581756Z","iopub.status.idle":"2022-07-31T23:50:46.582540Z","shell.execute_reply.started":"2022-07-31T23:50:46.582270Z","shell.execute_reply":"2022-07-31T23:50:46.582293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = np.array(data_scaled)\n# y = prediction_1\n\n# X,y","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.583911Z","iopub.status.idle":"2022-07-31T23:50:46.585492Z","shell.execute_reply.started":"2022-07-31T23:50:46.585217Z","shell.execute_reply":"2022-07-31T23:50:46.585241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ax = sns.countplot(x = prediction_1)\n# for p in ax.patches:\n#     ax.annotate('{:}'.format(p.get_height()), (p.get_x()+0.25, p.get_height()+0.01 ))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.587425Z","iopub.status.idle":"2022-07-31T23:50:46.590126Z","shell.execute_reply.started":"2022-07-31T23:50:46.589858Z","shell.execute_reply":"2022-07-31T23:50:46.589882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_bgmc = BayesianGMMClassifier(n_components = 7, # default = 1\n#                           covariance_type = 'full', # defalut = 'full'\n#                           max_iter = 700, # default = 100,\n#                           tol =1e-3,\n#                           init_params = 'kmeans', #default = kmeans,'k-means++', 'random'\n#                           random_state = 0, # default = None\n#                           n_init = 3  \n#                          )","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.591527Z","iopub.status.idle":"2022-07-31T23:50:46.592747Z","shell.execute_reply.started":"2022-07-31T23:50:46.592492Z","shell.execute_reply":"2022-07-31T23:50:46.592517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fold = 5\n\n# for i in range(fold):\n#     print(f'-----------{i}-----------')\n#     model = model_bgmc\n#     model.fit(X,y)\n#     print(f'-----------{i}_fitting----------')\n#     prediction_2 = model.predict(X)\n#     X = np.array(data_scaled)\n#     y =  prediction_2 \n#     print(prediction_2)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.600551Z","iopub.status.idle":"2022-07-31T23:50:46.601315Z","shell.execute_reply.started":"2022-07-31T23:50:46.601059Z","shell.execute_reply":"2022-07-31T23:50:46.601083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pl = sns.countplot(x=prediction_2)\n# pl.set_title(\"Distribution of clusters\")\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.602784Z","iopub.status.idle":"2022-07-31T23:50:46.603563Z","shell.execute_reply.started":"2022-07-31T23:50:46.603296Z","shell.execute_reply":"2022-07-31T23:50:46.603319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.604950Z","iopub.status.idle":"2022-07-31T23:50:46.605716Z","shell.execute_reply.started":"2022-07-31T23:50:46.605448Z","shell.execute_reply":"2022-07-31T23:50:46.605485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission[\"Predicted\"] =  prediction_2\n# submission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.607100Z","iopub.status.idle":"2022-07-31T23:50:46.607869Z","shell.execute_reply.started":"2022-07-31T23:50:46.607613Z","shell.execute_reply":"2022-07-31T23:50:46.607636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T23:50:46.615968Z","iopub.status.idle":"2022-07-31T23:50:46.616774Z","shell.execute_reply.started":"2022-07-31T23:50:46.616515Z","shell.execute_reply":"2022-07-31T23:50:46.616541Z"},"trusted":true},"execution_count":null,"outputs":[]}]}