{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TPS Jul 22 - Cluster_Ensembles with BayesianGaussianMixture","metadata":{}},{"cell_type":"code","source":"!sudo apt -y install libc6-dev-i386","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:45:00.351386Z","iopub.execute_input":"2022-07-11T14:45:00.351835Z","iopub.status.idle":"2022-07-11T14:45:17.503106Z","shell.execute_reply.started":"2022-07-11T14:45:00.351804Z","shell.execute_reply":"2022-07-11T14:45:17.502063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sudo apt-get -y install metis","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:45:17.505241Z","iopub.execute_input":"2022-07-11T14:45:17.505554Z","iopub.status.idle":"2022-07-11T14:45:24.357243Z","shell.execute_reply.started":"2022-07-11T14:45:17.505526Z","shell.execute_reply":"2022-07-11T14:45:24.355431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install Cluster_Ensembles","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T14:45:24.359366Z","iopub.execute_input":"2022-07-11T14:45:24.359804Z","iopub.status.idle":"2022-07-11T14:46:08.436329Z","shell.execute_reply.started":"2022-07-11T14:45:24.359767Z","shell.execute_reply":"2022-07-11T14:46:08.435309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Resolve scikit-learn incompatibility\n\nWe need to replace a reference to `jaccard_similarity_score` with `jaccard_score` to be compatible with the most recent scikit-learn.","metadata":{}},{"cell_type":"code","source":"!curl https://raw.githubusercontent.com/GGiecold-zz/Cluster_Ensembles/master/src/Cluster_Ensembles/Cluster_Ensembles.py > /tmp/Cluster_Ensembles.py\n!sed -i s/jaccard_similarity_score/jaccard_score/g /tmp/Cluster_Ensembles.py\n!cp /tmp/Cluster_Ensembles.py /opt/conda/lib/python3.7/site-packages/Cluster_Ensembles/Cluster_Ensembles.py\n\n!grep jaccard /opt/conda/lib/python3.7/site-packages/Cluster_Ensembles/Cluster_Ensembles.py","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:48:21.470141Z","iopub.execute_input":"2022-07-11T14:48:21.470737Z","iopub.status.idle":"2022-07-11T14:48:25.342073Z","shell.execute_reply.started":"2022-07-11T14:48:21.470691Z","shell.execute_reply":"2022-07-11T14:48:25.340306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nfrom pathlib import Path\n\nimport numpy as np \nimport pandas as pd\nimport random\n\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.preprocessing import PowerTransformer, RobustScaler\n\nimport warnings\nwarnings.simplefilter('ignore')\n\npd.set_option('display.max_columns', None)\npd.set_option('display.float_format', '{:.3f}'.format)\n\nINPUT = Path('../input/tabular-playground-series-jul-2022')\nSEED = 420","metadata":{"execution":{"iopub.status.busy":"2022-07-11T15:56:10.710443Z","iopub.execute_input":"2022-07-11T15:56:10.710880Z","iopub.status.idle":"2022-07-11T15:56:12.069549Z","shell.execute_reply.started":"2022-07-11T15:56:10.710787Z","shell.execute_reply":"2022-07-11T15:56:12.068483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(INPUT / 'data.csv',index_col='id')\ndata.info()\n\ndf = data[['f_07','f_08','f_09','f_10','f_11','f_12','f_13',\n           'f_22','f_23','f_24','f_25','f_26','f_27','f_28']]\n\nscaler = RobustScaler()\ntransformer = PowerTransformer()\n\nX = df.to_numpy()\nif scaler is not None:\n    X = scaler.fit_transform(X)\nif transformer is not None:\n    X[:, 0:7] = transformer.fit_transform(X[:, 0:7])\nX = pd.DataFrame(X, columns = df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:48:27.212326Z","iopub.execute_input":"2022-07-11T14:48:27.213215Z","iopub.status.idle":"2022-07-11T14:48:29.408381Z","shell.execute_reply.started":"2022-07-11T14:48:27.213152Z","shell.execute_reply":"2022-07-11T14:48:29.406982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generate multiple estimators\n- use different seeds\n- use only 1 init; try to introduce some variance.\n","metadata":{}},{"cell_type":"code","source":"n_clusters = 7\nn_estimators = 32\nestimator = BayesianGaussianMixture\nclusterings = np.empty((n_estimators, X.shape[0]))\n\nprint(f'fitting {n_estimators} estimators: ',end='')\nfor i in range(n_estimators):\n    seed = 100 * i\n    print(seed, end='...')\n    model = estimator(n_components=n_clusters, \n                      covariance_type='full',\n                      max_iter=200,\n                      n_init=1,\n                      random_state=seed)\n    clusterings[i,:] = model.fit(X).predict(X)   ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:51:01.290569Z","iopub.execute_input":"2022-07-11T14:51:01.290970Z","iopub.status.idle":"2022-07-11T14:54:11.568785Z","shell.execute_reply.started":"2022-07-11T14:51:01.290937Z","shell.execute_reply":"2022-07-11T14:54:11.567179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sanity check we got a clustering then build consensus","metadata":{}},{"cell_type":"code","source":"clusterings","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:54:52.967679Z","iopub.execute_input":"2022-07-11T14:54:52.969023Z","iopub.status.idle":"2022-07-11T14:54:52.979581Z","shell.execute_reply.started":"2022-07-11T14:54:52.968958Z","shell.execute_reply":"2022-07-11T14:54:52.978633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import Cluster_Ensembles as CE\n\nconsensus_labels = CE.cluster_ensembles(clusterings, verbose = True, N_clusters_max = n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:55:06.142940Z","iopub.execute_input":"2022-07-11T14:55:06.143426Z","iopub.status.idle":"2022-07-11T14:55:10.880002Z","shell.execute_reply.started":"2022-07-11T14:55:06.143387Z","shell.execute_reply":"2022-07-11T14:55:10.878053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Sanity check that we did indeed get some kind of labeling","metadata":{}},{"cell_type":"code","source":"consensus_labels","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:55:14.546247Z","iopub.execute_input":"2022-07-11T14:55:14.547645Z","iopub.status.idle":"2022-07-11T14:55:14.556644Z","shell.execute_reply.started":"2022-07-11T14:55:14.547591Z","shell.execute_reply":"2022-07-11T14:55:14.555250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame(consensus_labels, index=data.index, columns=['Predicted'])\nsubmission.to_csv('submission.csv')","metadata":{},"execution_count":null,"outputs":[]}]}