{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T09:13:03.152669Z","iopub.execute_input":"2022-08-03T09:13:03.153091Z","iopub.status.idle":"2022-08-03T09:13:03.161646Z","shell.execute_reply.started":"2022-08-03T09:13:03.153044Z","shell.execute_reply":"2022-08-03T09:13:03.160763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:22:00.626307Z","iopub.execute_input":"2022-08-03T09:22:00.626757Z","iopub.status.idle":"2022-08-03T09:22:00.631585Z","shell.execute_reply.started":"2022-08-03T09:22:00.626718Z","shell.execute_reply":"2022-08-03T09:22:00.630502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Do share your feedback","metadata":{}},{"cell_type":"markdown","source":"## Idea is to:-\n1. EDA\n2. Apply Preprocessing as required (scaling, transformation etc..)\n3. Fit ML algorithms. Will try to find right cluster size.Start with Kmeans and apply others\n4. Visualize the clusters","metadata":{}},{"cell_type":"code","source":"#sklearn imports\nimport pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import PCA  #Principal Component Analysis\nfrom sklearn.manifold import TSNE      #T-Distributed Stochastic Neighbor Embedding\nfrom sklearn.cluster import KMeans    \nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.preprocessing import StandardScaler \nimport sklearn.metrics as metrics\nfrom sklearn.metrics import silhouette_score\n\npd.set_option('display.max_columns', 31)\n\n\n#plotly imports\nimport plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:58:10.193349Z","iopub.execute_input":"2022-08-03T07:58:10.193803Z","iopub.status.idle":"2022-08-03T07:58:10.212132Z","shell.execute_reply.started":"2022-08-03T07:58:10.193769Z","shell.execute_reply":"2022-08-03T07:58:10.211124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_org = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/data.csv\")\ndf_org.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:34:01.278507Z","iopub.execute_input":"2022-08-03T06:34:01.278891Z","iopub.status.idle":"2022-08-03T06:34:02.518192Z","shell.execute_reply.started":"2022-08-03T06:34:01.278859Z","shell.execute_reply":"2022-08-03T06:34:02.517026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_org.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:34:03.049593Z","iopub.execute_input":"2022-08-03T06:34:03.050727Z","iopub.status.idle":"2022-08-03T06:34:03.086204Z","shell.execute_reply.started":"2022-08-03T06:34:03.050680Z","shell.execute_reply":"2022-08-03T06:34:03.084967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_org.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:34:06.057713Z","iopub.execute_input":"2022-08-03T06:34:06.058584Z","iopub.status.idle":"2022-08-03T06:34:06.282888Z","shell.execute_reply.started":"2022-08-03T06:34:06.058538Z","shell.execute_reply":"2022-08-03T06:34:06.281998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##correlation matrix\ncorr = df_org.corr()\ncorr.style.background_gradient(cmap='coolwarm')\n\n#no correlations among features","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:35:02.148719Z","iopub.execute_input":"2022-08-03T06:35:02.149175Z","iopub.status.idle":"2022-08-03T06:35:02.578308Z","shell.execute_reply.started":"2022-08-03T06:35:02.149137Z","shell.execute_reply":"2022-08-03T06:35:02.577108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create training set\ndf_train = df_org.drop(['id'], axis=1)\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:35:35.969588Z","iopub.execute_input":"2022-08-03T06:35:35.970023Z","iopub.status.idle":"2022-08-03T06:35:35.996832Z","shell.execute_reply.started":"2022-08-03T06:35:35.969988Z","shell.execute_reply":"2022-08-03T06:35:35.995628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot histogram to visualize data distribution\ndf_train.sample(5000).hist(figsize=(14,12));\n#f_07 to f_13 is not Gausian distribution. These are integar columns\n# for float columns- mean is almost at 0, and Gausian distributed","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:40:51.049926Z","iopub.execute_input":"2022-08-03T06:40:51.050843Z","iopub.status.idle":"2022-08-03T06:40:54.070254Z","shell.execute_reply.started":"2022-08-03T06:40:51.050799Z","shell.execute_reply":"2022-08-03T06:40:54.069121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing - standardize the data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:08.841228Z","iopub.execute_input":"2022-08-03T06:41:08.841611Z","iopub.status.idle":"2022-08-03T06:41:08.847363Z","shell.execute_reply.started":"2022-08-03T06:41:08.841578Z","shell.execute_reply":"2022-08-03T06:41:08.846135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_scaled_np = StandardScaler().fit_transform(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:09.449507Z","iopub.execute_input":"2022-08-03T06:41:09.449885Z","iopub.status.idle":"2022-08-03T06:41:09.511618Z","shell.execute_reply.started":"2022-08-03T06:41:09.449854Z","shell.execute_reply":"2022-08-03T06:41:09.510376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get a pandas df with same index and column\ndf_train_scaled = pd.DataFrame(train_scaled_np, columns=df_train.columns, index=df_train.index)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:13.590184Z","iopub.execute_input":"2022-08-03T06:41:13.590590Z","iopub.status.idle":"2022-08-03T06:41:13.596384Z","shell.execute_reply.started":"2022-08-03T06:41:13.590556Z","shell.execute_reply":"2022-08-03T06:41:13.595159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_scaled.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:16.369478Z","iopub.execute_input":"2022-08-03T06:41:16.370207Z","iopub.status.idle":"2022-08-03T06:41:16.589274Z","shell.execute_reply.started":"2022-08-03T06:41:16.370158Z","shell.execute_reply":"2022-08-03T06:41:16.588136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_scaled.sample(5000).hist(figsize=(14,12));\n#scaling has brought mean to zero for all columns\n#float columns are distributed between -2.5 to 2.5","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:31.687118Z","iopub.execute_input":"2022-08-03T06:41:31.687487Z","iopub.status.idle":"2022-08-03T06:41:34.619495Z","shell.execute_reply.started":"2022-08-03T06:41:31.687458Z","shell.execute_reply":"2022-08-03T06:41:34.618737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:44:19.930730Z","iopub.execute_input":"2022-08-03T06:44:19.931156Z","iopub.status.idle":"2022-08-03T06:44:19.936913Z","shell.execute_reply.started":"2022-08-03T06:44:19.931122Z","shell.execute_reply":"2022-08-03T06:44:19.935582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#understanding outliers\n#df_train.sample(5000).boxplot(column = categorical_cols)\ndf_train.sample(50000).boxplot(figsize=(20,8))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:44:32.805205Z","iopub.execute_input":"2022-08-03T06:44:32.805627Z","iopub.status.idle":"2022-08-03T06:44:34.443960Z","shell.execute_reply.started":"2022-08-03T06:44:32.805594Z","shell.execute_reply":"2022-08-03T06:44:34.442752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\nfrom sklearn.preprocessing import PowerTransformer\npt = PowerTransformer(method='yeo-johnson', standardize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:52:38.170328Z","iopub.execute_input":"2022-08-03T06:52:38.170758Z","iopub.status.idle":"2022-08-03T06:52:38.176139Z","shell.execute_reply.started":"2022-08-03T06:52:38.170724Z","shell.execute_reply":"2022-08-03T06:52:38.175146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_scaled[categorical_cols] = pd.DataFrame(pt.fit_transform(df_train[categorical_cols]), columns = categorical_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:53:50.675319Z","iopub.execute_input":"2022-08-03T06:53:50.675703Z","iopub.status.idle":"2022-08-03T06:53:51.458033Z","shell.execute_reply.started":"2022-08-03T06:53:50.675672Z","shell.execute_reply":"2022-08-03T06:53:51.456819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_scaled.boxplot(figsize=(10,5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:53:58.308552Z","iopub.execute_input":"2022-08-03T06:53:58.308942Z","iopub.status.idle":"2022-08-03T06:54:01.353031Z","shell.execute_reply.started":"2022-08-03T06:53:58.308908Z","shell.execute_reply":"2022-08-03T06:54:01.352002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA\nGet the sense of how many reduced dimensions can explain  whole data","metadata":{}},{"cell_type":"code","source":"%%time\n#PCA with three principal components\npca_model = PCA(n_components=29)\ndf_pca = pca_model.fit_transform(df_train_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:57:10.189358Z","iopub.execute_input":"2022-08-03T06:57:10.189975Z","iopub.status.idle":"2022-08-03T06:57:10.343567Z","shell.execute_reply.started":"2022-08-03T06:57:10.189931Z","shell.execute_reply":"2022-08-03T06:57:10.341933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_model.explained_variance_ratio_, pca_model.explained_variance_ratio_.cumsum()\n#none of the component is playing any major role","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:57:11.575600Z","iopub.execute_input":"2022-08-03T06:57:11.575983Z","iopub.status.idle":"2022-08-03T06:57:11.583934Z","shell.execute_reply.started":"2022-08-03T06:57:11.575952Z","shell.execute_reply":"2022-08-03T06:57:11.582937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  k-mean clustering ","metadata":{}},{"cell_type":"code","source":"# model = KMeans(n_clusters=4, init=\"k-means++\")\n# # fit the model\n# model.fit(df_train_scaled)\n# #Add the cluster vector to our DataFrame, X\n# # df_train_scaled[\"Cluster\"] = model.labels_\n# clusters = pd.unique(model.labels_)\n# clusters","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:58:23.705787Z","iopub.execute_input":"2022-08-03T06:58:23.706177Z","iopub.status.idle":"2022-08-03T06:58:23.710401Z","shell.execute_reply.started":"2022-08-03T06:58:23.706145Z","shell.execute_reply":"2022-08-03T06:58:23.709545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Will try with a range of clusters and will do some plotting to arrive at cluster size\nstart with small dataset and subsequently try with whole\n","metadata":{}},{"cell_type":"code","source":"%%time\n\nK=range(2,12)\nwss = []\nsilhoute_score = []\nfor k in K:\n    print(\"k: \", k)\n    kmeans= KMeans(n_clusters=k,init=\"k-means++\", random_state=211)\n    kmeans= kmeans.fit(df_train_scaled)\n    labels = kmeans.labels_\n    wss_iter = kmeans.inertia_\n    wss.append(wss_iter)\n    sil_score = metrics.silhouette_score(df_train_scaled, labels, metric=\"euclidean\", sample_size=30000)\n    silhoute_score.append(sil_score)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:25:00.048465Z","iopub.execute_input":"2022-08-03T07:25:00.049813Z","iopub.status.idle":"2022-08-03T07:28:13.310667Z","shell.execute_reply.started":"2022-08-03T07:25:00.049769Z","shell.execute_reply":"2022-08-03T07:28:13.309310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the metrics into df\ncluster_metrics = pd.DataFrame({'Clusters' : K, 'WSS' : wss, 'silhoute_score':silhoute_score})\ncluster_metrics","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:28:13.312604Z","iopub.execute_input":"2022-08-03T07:28:13.312988Z","iopub.status.idle":"2022-08-03T07:28:13.328251Z","shell.execute_reply.started":"2022-08-03T07:28:13.312954Z","shell.execute_reply":"2022-08-03T07:28:13.326830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x = 'Clusters', y = 'WSS', data = cluster_metrics, marker=\"o\")\n#knee jerk point","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:28:13.330183Z","iopub.execute_input":"2022-08-03T07:28:13.331209Z","iopub.status.idle":"2022-08-03T07:28:13.547630Z","shell.execute_reply.started":"2022-08-03T07:28:13.331169Z","shell.execute_reply":"2022-08-03T07:28:13.546795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x = 'Clusters', y = 'silhoute_score', data = cluster_metrics, marker=\"o\")\n#higher better","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:28:13.550026Z","iopub.execute_input":"2022-08-03T07:28:13.551321Z","iopub.status.idle":"2022-08-03T07:28:14.070449Z","shell.execute_reply.started":"2022-08-03T07:28:13.551261Z","shell.execute_reply":"2022-08-03T07:28:14.069157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ---> k-mean clustering do suggests 4 clusters","metadata":{}},{"cell_type":"code","source":"model = KMeans(n_clusters=4, init=\"k-means++\")\n# fit the model\nmodel.fit(df_train_scaled)\n#Add the cluster vector to our DataFrame, X\n# df_train_scaled[\"Cluster\"] = model.labels_\nclusters = pd.unique(model.labels_)\nclusters","metadata":{"execution":{"iopub.status.busy":"2022-07-30T11:40:43.466400Z","iopub.execute_input":"2022-07-30T11:40:43.466886Z","iopub.status.idle":"2022-07-30T11:40:50.132861Z","shell.execute_reply.started":"2022-07-30T11:40:43.466856Z","shell.execute_reply":"2022-07-30T11:40:50.131612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(model.labels_, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T11:40:50.134227Z","iopub.execute_input":"2022-07-30T11:40:50.134551Z","iopub.status.idle":"2022-07-30T11:40:50.145868Z","shell.execute_reply.started":"2022-07-30T11:40:50.134521Z","shell.execute_reply":"2022-07-30T11:40:50.144652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### submission","metadata":{}},{"cell_type":"code","source":"# submission = pd.DataFrame({'Id':df_org['id'].values,\n#                            'Predicted':model.labels_})\n# submission.to_csv(\"submission.csv\", index = False)\n# submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T11:40:50.147268Z","iopub.execute_input":"2022-07-30T11:40:50.147721Z","iopub.status.idle":"2022-07-30T11:40:50.152297Z","shell.execute_reply.started":"2022-07-30T11:40:50.147663Z","shell.execute_reply":"2022-07-30T11:40:50.151009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### kmean score on public score is quite low, so trying other models\n##### kmean score with 4 cluster. - 0.22381\n##### Kmean score with 7 cluster - 0.26360","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GaussianMixture","metadata":{}},{"cell_type":"markdown","source":"#### trying both GaussianMixture , BayesianGaussianMixture","metadata":{}},{"cell_type":"code","source":"%%time\n\nn_components_range=range(3,8)\nbic_values = []\naic_values = []\n\nfor n_components in n_components_range:\n    print(\"n_components: \", n_components)\n    # define the model\n    gm_model = GaussianMixture(n_components=n_components,\n                                n_init=1,\n                                init_params='kmeans',\n                              random_state=None)\n    # fit the model\n    gm_model.fit(df_train_scaled)\n    bic_values.append(gm_model.bic(df_train_scaled))\n    aic_values.append(gm_model.bic(df_train_scaled))\n\nbic_values, aic_values","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:27:23.410491Z","iopub.execute_input":"2022-07-31T17:27:23.411626Z","iopub.status.idle":"2022-07-31T17:28:26.374661Z","shell.execute_reply.started":"2022-07-31T17:27:23.411565Z","shell.execute_reply":"2022-07-31T17:28:26.373194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x = n_components_range, y = bic_values, marker=\"o\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:24:17.451409Z","iopub.execute_input":"2022-08-03T07:24:17.451840Z","iopub.status.idle":"2022-08-03T07:24:17.505046Z","shell.execute_reply.started":"2022-08-03T07:24:17.451807Z","shell.execute_reply":"2022-08-03T07:24:17.503469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x = n_components_range, y = aic_values, marker=\"o\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:56:53.772370Z","iopub.execute_input":"2022-07-31T15:56:53.773216Z","iopub.status.idle":"2022-07-31T15:56:53.986621Z","shell.execute_reply.started":"2022-07-31T15:56:53.773171Z","shell.execute_reply":"2022-07-31T15:56:53.985013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nn_components = 7\n# define the model\ngm_model = GaussianMixture(n_components=n_components)\n# fit the model\ngm_model.fit(df_train_scaled)\n# assign a cluster to each example\nyhat_gm = gm_model.predict(df_train_scaled)\n# retrieve unique clusters\nclusters = pd.unique(yhat_gm)\n\n\n\n#gm_model.bic(df_train_scaled), gm_model.aic(df_train_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:28:50.459910Z","iopub.execute_input":"2022-07-31T17:28:50.460343Z","iopub.status.idle":"2022-07-31T17:29:05.957535Z","shell.execute_reply.started":"2022-07-31T17:28:50.460307Z","shell.execute_reply":"2022-07-31T17:29:05.956265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gm_model.bic(df_train_scaled), gm_model.aic(df_train_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:29:21.337832Z","iopub.execute_input":"2022-07-31T17:29:21.338306Z","iopub.status.idle":"2022-07-31T17:29:22.030465Z","shell.execute_reply.started":"2022-07-31T17:29:21.338268Z","shell.execute_reply":"2022-07-31T17:29:22.029031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yhat_gm","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:29:29.789139Z","iopub.execute_input":"2022-07-31T17:29:29.790140Z","iopub.status.idle":"2022-07-31T17:29:29.798358Z","shell.execute_reply.started":"2022-07-31T17:29:29.790086Z","shell.execute_reply":"2022-07-31T17:29:29.796943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_scaled.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:29:41.117659Z","iopub.execute_input":"2022-07-31T17:29:41.118118Z","iopub.status.idle":"2022-07-31T17:29:41.157580Z","shell.execute_reply.started":"2022-07-31T17:29:41.118082Z","shell.execute_reply":"2022-07-31T17:29:41.156087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"# submission = pd.DataFrame({'Id':df_org['id'].values,\n#                            'Predicted':yhat_gm})\n# submission.to_csv(\"submission.csv\", index = False)\n# submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T16:23:47.703437Z","iopub.execute_input":"2022-07-31T16:23:47.703889Z","iopub.status.idle":"2022-07-31T16:23:47.884648Z","shell.execute_reply.started":"2022-07-31T16:23:47.703851Z","shell.execute_reply":"2022-07-31T16:23:47.883446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T16:23:50.053189Z","iopub.execute_input":"2022-07-31T16:23:50.053991Z","iopub.status.idle":"2022-07-31T16:23:50.066116Z","shell.execute_reply.started":"2022-07-31T16:23:50.053924Z","shell.execute_reply":"2022-07-31T16:23:50.064744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### public score - 0.52266\n##### good improvement over kmeans clearly","metadata":{}},{"cell_type":"markdown","source":"### BayesianGaussianMixture","metadata":{}},{"cell_type":"code","source":"%%time\n\nn_components = 7\n# define the model\nbgm_model = BayesianGaussianMixture(n_components=n_components)\n# fit the model\nbgm_model.fit(df_train_scaled)\n# assign a cluster to each example\nyhat_bgm = bgm_model.predict(df_train_scaled)\n# retrieve unique clusters\nclusters = pd.unique(yhat_bgm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:24:52.767434Z","iopub.execute_input":"2022-08-03T09:24:52.767864Z","iopub.status.idle":"2022-08-03T09:26:01.661007Z","shell.execute_reply.started":"2022-08-03T09:24:52.767833Z","shell.execute_reply":"2022-08-03T09:26:01.659495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_predict_proba = bgm_model.predict_proba(df_train_scaled)\nbgm_predict_proba","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:30:16.443249Z","iopub.execute_input":"2022-08-03T09:30:16.443675Z","iopub.status.idle":"2022-08-03T09:30:16.841364Z","shell.execute_reply.started":"2022-08-03T09:30:16.443644Z","shell.execute_reply":"2022-08-03T09:30:16.839808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#max probability is corresponding to max probability amount 7 clusters\nbgm_predict_proba_max = bgm_predict_proba.max(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:30:19.013136Z","iopub.execute_input":"2022-08-03T09:30:19.013527Z","iopub.status.idle":"2022-08-03T09:30:19.026433Z","shell.execute_reply.started":"2022-08-03T09:30:19.013496Z","shell.execute_reply":"2022-08-03T09:30:19.025034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(y=bgm_predict_proba_max)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:30:20.138559Z","iopub.execute_input":"2022-08-03T09:30:20.139613Z","iopub.status.idle":"2022-08-03T09:30:20.327675Z","shell.execute_reply.started":"2022-08-03T09:30:20.139572Z","shell.execute_reply":"2022-08-03T09:30:20.326523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(bgm_predict_proba_max < .5)/len(bgm_predict_proba_max)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:30:23.063016Z","iopub.execute_input":"2022-08-03T09:30:23.063789Z","iopub.status.idle":"2022-08-03T09:30:23.285124Z","shell.execute_reply.started":"2022-08-03T09:30:23.063752Z","shell.execute_reply":"2022-08-03T09:30:23.283898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'Id':df_org['id'].values,\n                           'Predicted':yhat_bgm})\nsubmission.to_csv(\"submission.csv\", index = False)\nsubmission.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:30:45.286877Z","iopub.execute_input":"2022-08-03T09:30:45.287320Z","iopub.status.idle":"2022-08-03T09:30:45.415640Z","shell.execute_reply.started":"2022-08-03T09:30:45.287284Z","shell.execute_reply":"2022-08-03T09:30:45.414404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### public score - 0.59899\n##### good improvement over GMM clearly","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Choose GMM cluster if BGMM predict_proba is low ?","metadata":{}},{"cell_type":"code","source":"yhat_gm","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:18.129310Z","iopub.execute_input":"2022-07-31T17:31:18.130441Z","iopub.status.idle":"2022-07-31T17:31:18.137920Z","shell.execute_reply.started":"2022-07-31T17:31:18.130395Z","shell.execute_reply":"2022-07-31T17:31:18.136609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_concat = pd.DataFrame({'yhat_gm':yhat_gm, 'yhat_bgm':yhat_bgm, 'bgm_predict_proba_max':bgm_predict_proba_max})\ny_concat.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:21.441121Z","iopub.execute_input":"2022-07-31T17:31:21.441548Z","iopub.status.idle":"2022-07-31T17:31:21.457023Z","shell.execute_reply.started":"2022-07-31T17:31:21.441512Z","shell.execute_reply":"2022-07-31T17:31:21.455745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_concat['bgm_predict_proba_max']<.5","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:24.292425Z","iopub.execute_input":"2022-07-31T17:31:24.293189Z","iopub.status.idle":"2022-07-31T17:31:24.303397Z","shell.execute_reply.started":"2022-07-31T17:31:24.293140Z","shell.execute_reply":"2022-07-31T17:31:24.302458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### just selecting for all proba less than 60 percent, we will choose GMM","metadata":{}},{"cell_type":"code","source":"yhat_selected = []\nthreshold_proba = .5\nfor y_gm, y_bgm, bgm_proba in zip(yhat_gm, yhat_bgm, bgm_predict_proba_max):\n    if bgm_proba < threshold_proba:\n        yhat_selected.append(y_gm)\n    else:\n        yhat_selected.append(y_bgm)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:29.972668Z","iopub.execute_input":"2022-07-31T17:31:29.973545Z","iopub.status.idle":"2022-07-31T17:31:30.062288Z","shell.execute_reply.started":"2022-07-31T17:31:29.973498Z","shell.execute_reply":"2022-07-31T17:31:30.061073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(yhat_selected), yhat_selected[:6]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:33.604965Z","iopub.execute_input":"2022-07-31T17:31:33.606282Z","iopub.status.idle":"2022-07-31T17:31:33.614666Z","shell.execute_reply.started":"2022-07-31T17:31:33.606226Z","shell.execute_reply":"2022-07-31T17:31:33.613753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yhat_selected[97995:97997], y_concat[97995:97997]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:36.044280Z","iopub.execute_input":"2022-07-31T17:31:36.045518Z","iopub.status.idle":"2022-07-31T17:31:36.056636Z","shell.execute_reply.started":"2022-07-31T17:31:36.045461Z","shell.execute_reply":"2022-07-31T17:31:36.055320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### submission","metadata":{}},{"cell_type":"code","source":"# submission = pd.DataFrame({'Id':df_org['id'].values,\n#                            'Predicted':yhat_selected})\n# submission.to_csv(\"submission.csv\", index = False)\n# submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:45.242364Z","iopub.execute_input":"2022-07-31T17:31:45.242872Z","iopub.status.idle":"2022-07-31T17:31:45.495627Z","shell.execute_reply.started":"2022-07-31T17:31:45.242830Z","shell.execute_reply":"2022-07-31T17:31:45.494225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.tail(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T17:31:48.403139Z","iopub.execute_input":"2022-07-31T17:31:48.404229Z","iopub.status.idle":"2022-07-31T17:31:48.416625Z","shell.execute_reply.started":"2022-07-31T17:31:48.404184Z","shell.execute_reply":"2022-07-31T17:31:48.415343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### public score - 0.55041\n#### score has reduced, have to try different technique","metadata":{}},{"cell_type":"markdown","source":"# Visualizing cluster","metadata":{}},{"cell_type":"markdown","source":"PCA could be used to reduce the dimenstion to 1,2 & 3. And same could be plotted to visiualize. Provided PCA components are able to explain variance.\nWhich is not the case here. PCA explained_variance_ratio_ is low for first 3 PCs","metadata":{}},{"cell_type":"markdown","source":"Links refered:\n\nhttps://www.kaggle.com/code/minc33/visualizing-high-dimensional-clusters/notebook <br>\nhttps://stats.stackexchange.com/questions/324843/visualization-problem-of-15-attributes-which-is-clustered-into-2-cluster","metadata":{}},{"cell_type":"code","source":"cluster_size = 7 \nplot_x_df = df_train_scaled.copy()\nplot_x_df['Cluster'] = yhat_bgm","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:13:42.627263Z","iopub.execute_input":"2022-08-03T08:13:42.627652Z","iopub.status.idle":"2022-08-03T08:13:42.644345Z","shell.execute_reply.started":"2022-08-03T08:13:42.627621Z","shell.execute_reply":"2022-08-03T08:13:42.643037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_x_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:09:00.687337Z","iopub.execute_input":"2022-08-03T08:09:00.688194Z","iopub.status.idle":"2022-08-03T08:09:00.721929Z","shell.execute_reply.started":"2022-08-03T08:09:00.688149Z","shell.execute_reply":"2022-08-03T08:09:00.720920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotX is a DataFrame containing 5000 values sampled randomly from X\nplotX = pd.DataFrame(np.array(plot_x_df.sample(20000)))\n\n#Rename plotX's columns since it was briefly converted to an np.array above\nplotX.columns = plot_x_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:07:46.348181Z","iopub.execute_input":"2022-08-03T08:07:46.348558Z","iopub.status.idle":"2022-08-03T08:07:46.368050Z","shell.execute_reply.started":"2022-08-03T08:07:46.348529Z","shell.execute_reply":"2022-08-03T08:07:46.367210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PCA with one principal component\npca_1d = PCA(n_components=1)\n\n#PCA with two principal components\npca_2d = PCA(n_components=2)\n\n#PCA with three principal components\npca_3d = PCA(n_components=3)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:07:50.026784Z","iopub.execute_input":"2022-08-03T08:07:50.027235Z","iopub.status.idle":"2022-08-03T08:07:50.033800Z","shell.execute_reply.started":"2022-08-03T08:07:50.027201Z","shell.execute_reply":"2022-08-03T08:07:50.032498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This DataFrame holds that single principal component mentioned above\nPCs_1d = pd.DataFrame(pca_1d.fit_transform(plotX.drop([\"Cluster\"], axis=1)))\n\n#This DataFrame contains the two principal components that will be used\n#for the 2-D visualization mentioned above\nPCs_2d = pd.DataFrame(pca_2d.fit_transform(plotX.drop([\"Cluster\"], axis=1)))\n\n#And this DataFrame contains three principal components that will aid us\n#in visualizing our clusters in 3-D\nPCs_3d = pd.DataFrame(pca_3d.fit_transform(plotX.drop([\"Cluster\"], axis=1)))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:07:53.633382Z","iopub.execute_input":"2022-08-03T08:07:53.633823Z","iopub.status.idle":"2022-08-03T08:07:54.137253Z","shell.execute_reply.started":"2022-08-03T08:07:53.633786Z","shell.execute_reply":"2022-08-03T08:07:54.135482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_1d.explained_variance_ratio_, pca_2d.explained_variance_ratio_, pca_3d.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:07:59.017066Z","iopub.execute_input":"2022-08-03T08:07:59.017557Z","iopub.status.idle":"2022-08-03T08:07:59.027492Z","shell.execute_reply.started":"2022-08-03T08:07:59.017510Z","shell.execute_reply":"2022-08-03T08:07:59.025990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PCs_1d.columns = [\"PC1_1d\"]\n\n#\"PC1_2d\" means: 'The first principal component of the components created for 2-D visualization, by PCA.'\n#And \"PC2_2d\" means: 'The second principal component of the components created for 2-D visualization, by PCA.'\nPCs_2d.columns = [\"PC1_2d\", \"PC2_2d\"]\n\nPCs_3d.columns = [\"PC1_3d\", \"PC2_3d\", \"PC3_3d\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:08:01.327754Z","iopub.execute_input":"2022-08-03T08:08:01.328438Z","iopub.status.idle":"2022-08-03T08:08:01.334295Z","shell.execute_reply.started":"2022-08-03T08:08:01.328403Z","shell.execute_reply":"2022-08-03T08:08:01.333342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plotX = pd.concat([plotX,PCs_1d,PCs_2d,PCs_3d], axis=1, join='inner')\nplotX[\"dummy\"] = 0","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:08:03.787910Z","iopub.execute_input":"2022-08-03T08:08:03.788340Z","iopub.status.idle":"2022-08-03T08:08:03.795660Z","shell.execute_reply.started":"2022-08-03T08:08:03.788310Z","shell.execute_reply":"2022-08-03T08:08:03.794800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This is needed so we can display plotly plots properly\ninit_notebook_mode(connected=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T08:45:31.270085Z","iopub.execute_input":"2022-07-30T08:45:31.270870Z","iopub.status.idle":"2022-07-30T08:45:31.342786Z","shell.execute_reply.started":"2022-07-30T08:45:31.270820Z","shell.execute_reply":"2022-07-30T08:45:31.341844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instructions for building the 1-D plot\ndata = []\nfor cluster in range(cluster_size):\n    #trace is for 'Cluster x'\n    trace = go.Scatter(\n                        x = plotX[plotX[\"Cluster\"] == cluster][\"PC1_1d\"],\n                        y = plotX[plotX[\"Cluster\"] == cluster][\"dummy\"],\n                        mode = \"markers\",\n                        name = f\"Cluster {cluster}\",\n                        text = None)\n    data.append(trace)\n\ntitle = \"Visualizing Clusters in One Dimension Using PCA\"\n\nlayout = dict(title = title,\n              xaxis= dict(title= 'PC1',ticklen= 5,zeroline= False),\n              yaxis= dict(title= '',ticklen= 5,zeroline= False)\n             )\n\nfig = dict(data = data, layout = layout)\n\n# iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:21:04.667292Z","iopub.execute_input":"2022-08-03T08:21:04.667671Z","iopub.status.idle":"2022-08-03T08:21:05.910254Z","shell.execute_reply.started":"2022-08-03T08:21:04.667634Z","shell.execute_reply":"2022-08-03T08:21:05.909144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instructions for building the 2-D plot\ndata = []\nfor cluster in range(cluster_size):\n    trace = go.Scatter(\n                        x = plotX[plotX[\"Cluster\"] == cluster][\"PC1_2d\"],     #PC1_2d\n                        y = plotX[plotX[\"Cluster\"] == cluster][\"PC2_2d\"],     #PC2_2d\n                        mode = \"markers\",\n                        name = f\"Cluster {cluster}\",\n                        text = None)\n    data.append(trace)\n\n\n\ntitle = \"Visualizing Clusters in Two Dimensions Using PCA\"\n\nlayout = dict(title = title,\n              xaxis= dict(title= 'PC1',ticklen= 5,zeroline= False),\n              yaxis= dict(title= 'PC2',ticklen= 5,zeroline= False)\n             )\n\nfig = dict(data = data, layout = layout)\n\n# iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:14:29.433563Z","iopub.execute_input":"2022-08-03T09:14:29.434226Z","iopub.status.idle":"2022-08-03T09:14:29.501156Z","shell.execute_reply.started":"2022-08-03T09:14:29.434190Z","shell.execute_reply":"2022-08-03T09:14:29.499938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instructions for building the 3-D plot\ndata = []\nfor cluster in range(cluster_size): \n    #trace is for 'Cluster 0'\n    trace = go.Scatter3d(\n                        x = plotX[plotX[\"Cluster\"] == cluster][\"PC1_3d\"],\n                        y = plotX[plotX[\"Cluster\"] == cluster][\"PC2_3d\"],\n                        z = plotX[plotX[\"Cluster\"] == cluster][\"PC3_3d\"],\n                        mode = \"markers\",\n                        name = f\"Cluster {cluster}\",\n                        text = None)\n    data.append(trace)\n\n\ntitle = \"Visualizing Clusters in Three Dimensions Using PCA\"\n\nlayout = dict(title = title,\n              xaxis= dict(title= 'PC1',ticklen= 5,zeroline= False),\n              yaxis= dict(title= 'PC2',ticklen= 5,zeroline= False)\n             )\n\nfig = dict(data = data, layout = layout)\n\n# iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:31:13.024724Z","iopub.execute_input":"2022-08-03T08:31:13.025224Z","iopub.status.idle":"2022-08-03T08:31:13.061280Z","shell.execute_reply.started":"2022-08-03T08:31:13.025188Z","shell.execute_reply":"2022-08-03T08:31:13.060317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}