{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Original notebook from \"https://www.kaggle.com/code/imnaho/choose-colums-bayesiangaussianmixture/notebook?scriptVersionId=101704407\" and annotated by me ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nimport itertools\n#from sklearn.mixture import GaussianMixture\nfrom sklearn.mixture import BayesianGaussianMixture\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T08:45:18.000078Z","iopub.execute_input":"2022-07-25T08:45:18.00067Z","iopub.status.idle":"2022-07-25T08:45:19.729948Z","shell.execute_reply.started":"2022-07-25T08:45:18.000547Z","shell.execute_reply":"2022-07-25T08:45:19.728915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# read data","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\nsubmission=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nprint('train data shape is :',train.shape)\nprint('submission data shape is :',submission.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:19.731868Z","iopub.execute_input":"2022-07-25T08:45:19.732774Z","iopub.status.idle":"2022-07-25T08:45:21.265987Z","shell.execute_reply.started":"2022-07-25T08:45:19.732728Z","shell.execute_reply":"2022-07-25T08:45:21.264856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check data simply","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.267231Z","iopub.execute_input":"2022-07-25T08:45:21.267576Z","iopub.status.idle":"2022-07-25T08:45:21.309822Z","shell.execute_reply.started":"2022-07-25T08:45:21.267545Z","shell.execute_reply":"2022-07-25T08:45:21.308535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.31307Z","iopub.execute_input":"2022-07-25T08:45:21.313816Z","iopub.status.idle":"2022-07-25T08:45:21.323988Z","shell.execute_reply.started":"2022-07-25T08:45:21.313761Z","shell.execute_reply":"2022-07-25T08:45:21.32255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include='all').T","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.327148Z","iopub.execute_input":"2022-07-25T08:45:21.328899Z","iopub.status.idle":"2022-07-25T08:45:21.556345Z","shell.execute_reply.started":"2022-07-25T08:45:21.328846Z","shell.execute_reply":"2022-07-25T08:45:21.555131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a=[col for col in train.columns if train[col].isnull().any()]\nprint(a)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.558041Z","iopub.execute_input":"2022-07-25T08:45:21.558703Z","iopub.status.idle":"2022-07-25T08:45:21.57603Z","shell.execute_reply.started":"2022-07-25T08:45:21.558657Z","shell.execute_reply":"2022-07-25T08:45:21.574924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"there is no missing value","metadata":{}},{"cell_type":"code","source":"train[train.iloc[:,1:].duplicated()]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.577685Z","iopub.execute_input":"2022-07-25T08:45:21.57799Z","iopub.status.idle":"2022-07-25T08:45:21.82231Z","shell.execute_reply.started":"2022-07-25T08:45:21.577961Z","shell.execute_reply":"2022-07-25T08:45:21.821493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"there is no dumplicate rows","metadata":{}},{"cell_type":"markdown","source":"# simple EDA","metadata":{}},{"cell_type":"markdown","source":"**check data distribution**","metadata":{}},{"cell_type":"code","source":"col_list=train.columns\nncols = 5\nnrows = 6\n\nfig, axes = plt.subplots(nrows, ncols, figsize=(30,30), facecolor='#EAEAF2')\n\nfor r in range(nrows):\n    for c in range(ncols):\n        col = col_list[r*ncols+c]\n        sns.histplot(x=train[col], ax=axes[r, c])\n        axes[r, c].set_ylabel('')\n        axes[r, c].set_xlabel(col, fontsize=15, fontweight='bold')\n        axes[r, c].tick_params(labelsize=10, width=0.5)\n        axes[r, c].xaxis.offsetText.set_fontsize(20)\n        axes[r, c].yaxis.offsetText.set_fontsize(20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:21.823786Z","iopub.execute_input":"2022-07-25T08:45:21.824576Z","iopub.status.idle":"2022-07-25T08:45:34.388659Z","shell.execute_reply.started":"2022-07-25T08:45:21.824528Z","shell.execute_reply":"2022-07-25T08:45:34.387453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**check single correlation heatmap**","metadata":{}},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(13,8)})\ndf_corr=train.iloc[:,1:].corr()\nsns.heatmap(df_corr, vmax=1, vmin=-1, center=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:34.39028Z","iopub.execute_input":"2022-07-25T08:45:34.390722Z","iopub.status.idle":"2022-07-25T08:45:35.321318Z","shell.execute_reply.started":"2022-07-25T08:45:34.390683Z","shell.execute_reply":"2022-07-25T08:45:35.319723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# prepare training data","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import FastICA\n\nica = FastICA(n_components=3)\nS_1 = ica.fit_transform(train.iloc[:,8:15])\nS_2 = ica.fit_transform(train.iloc[:,23:])\nS_3 = ica.fit_transform(train.iloc[:,1:8])\nS_4 = ica.fit_transform(train.iloc[:,15:23])\n\ncomp_df_1=pd.DataFrame(S_1,columns=['01','02','03'])\ncomp_df_2=pd.DataFrame(S_2,columns=['04','05','06'])\ncomp_df_3=pd.DataFrame(S_3,columns=['07','08','09'])\ncomp_df_4=pd.DataFrame(S_4,columns=['10','11','12'])\ncomp_df_1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:35.324443Z","iopub.execute_input":"2022-07-25T08:45:35.325194Z","iopub.status.idle":"2022-07-25T08:45:45.127076Z","shell.execute_reply.started":"2022-07-25T08:45:35.325149Z","shell.execute_reply":"2022-07-25T08:45:45.125702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data=pd.concat([train,comp_df_1,comp_df_2,comp_df_3,comp_df_4],axis=1)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:45.134513Z","iopub.execute_input":"2022-07-25T08:45:45.138581Z","iopub.status.idle":"2022-07-25T08:45:45.233447Z","shell.execute_reply.started":"2022-07-25T08:45:45.138501Z","shell.execute_reply":"2022-07-25T08:45:45.232522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**standardization**","metadata":{}},{"cell_type":"markdown","source":"# RobustScaler\nBriefly, Robustscaler is strong on outliers which often bother us deeply in real world. This scaler use IQR ( Range between 1st quartile and the 3rd quartile) for scaling. I love this scaler because he always help me greatly !!!","metadata":{}},{"cell_type":"code","source":"train_data=train_data.iloc[:,1:]\n\nscaler = StandardScaler()\nscaler.fit(train_data)\nX=pd.DataFrame(scaler.transform(train_data),columns=train_data.columns)\nX.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:45.23496Z","iopub.execute_input":"2022-07-25T08:45:45.235738Z","iopub.status.idle":"2022-07-25T08:45:45.698091Z","shell.execute_reply.started":"2022-07-25T08:45:45.235692Z","shell.execute_reply":"2022-07-25T08:45:45.696778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**make column list**","metadata":{}},{"cell_type":"code","source":"order_list_=train_data.iloc[:,29:].columns\norder_list_=order_list_.append(train_data.iloc[:,7:14].columns)\norder_list_=order_list_.append(train_data.iloc[:,22:29].columns)\norder_list_=order_list_.append(train_data.iloc[:,0:7].columns)\norder_list_=order_list_.append(train_data.iloc[:,14:22].columns)\nprint(len(order_list_))\norder_list_","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:45:45.699939Z","iopub.execute_input":"2022-07-25T08:45:45.700655Z","iopub.status.idle":"2022-07-25T08:45:45.723249Z","shell.execute_reply.started":"2022-07-25T08:45:45.700603Z","shell.execute_reply":"2022-07-25T08:45:45.72184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**choose columns**","metadata":{}},{"cell_type":"markdown","source":"# What Is a Normal Distribution/ Gaussian Distribution?\nNormal distribution, also known as the Gaussian distribution, is a probability distribution that is symmetric about the mean, showing that data near the mean are more frequent in occurrence than data far from the mean","metadata":{}},{"cell_type":"markdown","source":"# Gaussian mixture model \nA Gaussian mixture model is a probabilistic model that assumes all the data points are generated from a mixture of a finite number of Gaussian distributions with unknown parameters. One can think of mixture models as generalizing k-means clustering to incorporate information about the **covariance** structure of the data as well as the centers of the **latent** Gaussians.\n\nScikit-learn implements different classes to estimate Gaussian mixture models, that correspond to different estimation strategies, detailed below.\n\n","metadata":{}},{"cell_type":"markdown","source":"**Covariance** measures the directional relationship between the returns on two assets. A positive covariance means that asset returns move together while a negative covariance means they move inversely.\n\nCovariance is calculated by analyzing at-return surprises (standard deviations from the expected return) or by multiplying the correlation between the two random variables by the standard deviation of each variable.","metadata":{}},{"cell_type":"markdown","source":"**latent** (of a quality or state) existing but not yet developed or manifest; hidden or concealed.\n","metadata":{}},{"cell_type":"markdown","source":"# GaussianMixture\nThe GaussianMixture object implements the **expectation-maximization** (EM) algorithm for fitting mixture-of-Gaussian models. It can also draw confidence ellipsoids for multivariate models, and compute the Bayesian Information Criterion to assess the number of clusters in the data. A GaussianMixture.fit method is provided that learns a Gaussian Mixture Model from train data. Given test data, it can assign to each sample the Gaussian it mostly probably belong to using the GaussianMixture.predict method.\n\nThe GaussianMixture comes with different options to constrain the covariance of the difference classes estimated: spherical, diagonal, tied or full covariance.","metadata":{}},{"cell_type":"markdown","source":" **Expectation-maximization** is a well-founded statistical algorithm to get around this problem by an iterative process. First one assumes random components (randomly centered on data points, learned from k-means, or even just normally distributed around the origin) and computes for each point a probability of being generated by each component of the model. Then, one tweaks the parameters to maximize the likelihood of the data given those assignments. Repeating this process is guaranteed to always converge to a local optimum.\n \n With the help of the Expectation-maximization algorithm, you can figure which points came from which latent component","metadata":{}},{"cell_type":"markdown","source":"# Variational Bayesian Gaussian Mixture\nVariational inference is an extension of expectation-maximization that maximizes a lower bound on model evidence (including priors) instead of data likelihood. The principle behind variational methods is the same as expectation-maximization (that is both are iterative algorithms that alternate between finding the probabilities for each point to be generated by each mixture and fitting the mixture to these assigned points), but variational methods add regularization by integrating information from prior distributions. This avoids the singularities often found in expectation-maximization solutions but introduces some subtle biases to the model. Inference is often notably slower, but not usually as much so as to render usage unpractical.\n\nDue to its Bayesian nature, the variational algorithm needs more hyper- parameters than expectation-maximization, the most important of these being the concentration parameter weight_concentration_prior. Specifying a low value for the concentration prior will make the model put most of the weight on few components set the remaining components weights very close to zero. High values of the concentration prior will allow a larger number of components to be active in the mixture.\n\nThe parameters implementation of the BayesianGaussianMixture class proposes two types of prior for the weights distribution: a finite mixture model with Dirichlet distribution and an infinite mixture model with the Dirichlet Process. In practice Dirichlet Process inference algorithm is approximated and uses a truncated distribution with a fixed maximum number of components (called the Stick-breaking representation). The number of components actually used almost always depends on the data.\n\n\n**In other words, a Dirichlet process is a probability distribution whose range is itself a set of probability distributions.**","metadata":{}},{"cell_type":"code","source":"test_list=[]\nbest_score=0\nfor col in order_list_.tolist():  \n    test_list.append(col)\n    X_test=X[test_list]\n    cls_gaussian = BayesianGaussianMixture(n_components=7,tol=0.0001,\n                             max_iter=1000,init_params='kmeans')\n    result = cls_gaussian.fit(X_test)\n    result_score=np.abs(cls_gaussian.score(X_test))\n    diff_sum=0\n    print('score is : ',result_score)\n    if best_score<result_score:\n        best_score=result_score\n    else:\n        test_list.pop()\n#         The pop() method removes the element at the specified position.","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:14.216118Z","iopub.execute_input":"2022-07-25T08:51:14.217411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test_list))\nX_test=X[test_list]\ncls_gaussian= BayesianGaussianMixture(n_components=7,tol=0.0001,\n                             max_iter=1000,init_params='kmeans')\nsubmission['Predicted']=cls_gaussian.fit_predict(X_test)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:11.303044Z","iopub.status.idle":"2022-07-25T08:51:11.30418Z","shell.execute_reply.started":"2022-07-25T08:51:11.303853Z","shell.execute_reply":"2022-07-25T08:51:11.303886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# create submission file","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:11.305966Z","iopub.status.idle":"2022-07-25T08:51:11.306559Z","shell.execute_reply.started":"2022-07-25T08:51:11.306254Z","shell.execute_reply":"2022-07-25T08:51:11.306281Z"},"trusted":true},"execution_count":null,"outputs":[]}]}