{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T13:36:07.106328Z","iopub.execute_input":"2022-07-31T13:36:07.106817Z","iopub.status.idle":"2022-07-31T13:36:07.140315Z","shell.execute_reply.started":"2022-07-31T13:36:07.106710Z","shell.execute_reply":"2022-07-31T13:36:07.139442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install scikit-lego","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:07.141962Z","iopub.execute_input":"2022-07-31T13:36:07.142472Z","iopub.status.idle":"2022-07-31T13:36:21.195442Z","shell.execute_reply.started":"2022-07-31T13:36:07.142440Z","shell.execute_reply":"2022-07-31T13:36:21.194003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom scipy.stats import shapiro\nfrom termcolor import colored\nfrom scipy import stats\nfrom sklearn.metrics import silhouette_score\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom sklearn.preprocessing import PowerTransformer, MinMaxScaler, RobustScaler, MaxAbsScaler\nfrom sklego.mixture import BayesianGMMClassifier, GMMClassifier\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:21.197329Z","iopub.execute_input":"2022-07-31T13:36:21.197696Z","iopub.status.idle":"2022-07-31T13:36:22.417196Z","shell.execute_reply.started":"2022-07-31T13:36:21.197661Z","shell.execute_reply":"2022-07-31T13:36:22.415999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:22.419955Z","iopub.execute_input":"2022-07-31T13:36:22.420315Z","iopub.status.idle":"2022-07-31T13:36:24.189534Z","shell.execute_reply.started":"2022-07-31T13:36:22.420275Z","shell.execute_reply":"2022-07-31T13:36:24.188380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df. describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:24.191274Z","iopub.execute_input":"2022-07-31T13:36:24.192024Z","iopub.status.idle":"2022-07-31T13:36:24.427639Z","shell.execute_reply.started":"2022-07-31T13:36:24.191980Z","shell.execute_reply":"2022-07-31T13:36:24.426420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:24.428999Z","iopub.execute_input":"2022-07-31T13:36:24.429351Z","iopub.status.idle":"2022-07-31T13:36:24.444909Z","shell.execute_reply.started":"2022-07-31T13:36:24.429318Z","shell.execute_reply":"2022-07-31T13:36:24.443706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(25,25)})\nfor i, column in enumerate(list(df.columns), 1):\n    plt.subplot(5,6,i)\n    p=sns.histplot(x=column,data=df.sample(1000),stat='count',kde=True,color='green')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:24.446488Z","iopub.execute_input":"2022-07-31T13:36:24.447113Z","iopub.status.idle":"2022-07-31T13:36:31.009820Z","shell.execute_reply.started":"2022-07-31T13:36:24.447078Z","shell.execute_reply":"2022-07-31T13:36:31.008911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:31.011105Z","iopub.execute_input":"2022-07-31T13:36:31.011612Z","iopub.status.idle":"2022-07-31T13:36:31.039505Z","shell.execute_reply.started":"2022-07-31T13:36:31.011579Z","shell.execute_reply":"2022-07-31T13:36:31.038289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df.drop('id',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:31.040916Z","iopub.execute_input":"2022-07-31T13:36:31.041519Z","iopub.status.idle":"2022-07-31T13:36:31.054096Z","shell.execute_reply.started":"2022-07-31T13:36:31.041481Z","shell.execute_reply":"2022-07-31T13:36:31.052781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dff = df.copy()\ndffs = MaxAbsScaler().fit_transform(dff)\ndffs_ = PowerTransformer().fit_transform(dffs)\n\npower_features =[ 'f_07','f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24','f_25','f_26','f_27','f_28']\n\ndffs_ = pd.DataFrame(dffs_, columns=dff.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:31.057697Z","iopub.execute_input":"2022-07-31T13:36:31.058080Z","iopub.status.idle":"2022-07-31T13:36:34.745159Z","shell.execute_reply.started":"2022-07-31T13:36:31.058046Z","shell.execute_reply":"2022-07-31T13:36:34.743914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dffs_","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:34.746686Z","iopub.execute_input":"2022-07-31T13:36:34.747179Z","iopub.status.idle":"2022-07-31T13:36:34.784570Z","shell.execute_reply.started":"2022-07-31T13:36:34.747132Z","shell.execute_reply":"2022-07-31T13:36:34.783593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from yellowbrick.cluster import KElbowVisualizer\n\nmodel = KMeans()\nvisualizer = KElbowVisualizer(model, k=(4,12))\nvisualizer.fit(df)     \nvisualizer.show()  ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:36:34.785926Z","iopub.execute_input":"2022-07-31T13:36:34.786304Z","iopub.status.idle":"2022-07-31T13:37:27.148869Z","shell.execute_reply.started":"2022-07-31T13:36:34.786272Z","shell.execute_reply":"2022-07-31T13:37:27.147660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dffs_\n# Test Data for predictions later\ntest_data = dffs_[power_features].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:37:27.150425Z","iopub.execute_input":"2022-07-31T13:37:27.150771Z","iopub.status.idle":"2022-07-31T13:37:27.164595Z","shell.execute_reply.started":"2022-07-31T13:37:27.150730Z","shell.execute_reply":"2022-07-31T13:37:27.163345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Fit Bayesian Gaussian Mixture\nprint(\"Fitting Bayesian Gaussian Mixture..\")\nbgm = BayesianGaussianMixture(\n    n_components=7,\n    max_iter=300,\n    n_init=10,\n    random_state=2,\n    verbose_interval=100,\n)\n\nbgm_labels = bgm.fit_predict(dffs_[power_features])\nbgm_proba = bgm.predict_proba(dffs_[power_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:37:27.166122Z","iopub.execute_input":"2022-07-31T13:37:27.166733Z","iopub.status.idle":"2022-07-31T13:43:45.048310Z","shell.execute_reply.started":"2022-07-31T13:37:27.166697Z","shell.execute_reply":"2022-07-31T13:43:45.046913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Using idea from: https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting\n\n# Creating Best data based on predicted probability of BGM model\nn_components = 7\ndffs_[\"predict\"] = bgm_labels\ndffs_[\"predict_proba\"] = 0\n\nfor n in range(n_components):\n    dffs_[f\"bgm_proba_{n}\"] = bgm_proba[:, n]\n    dffs_.loc[dffs_.predict == n, \"bgm_proba\"] = dffs_[\n        f\"bgm_proba_{n}\"\n    ]\n\ntrain_index = np.array([])\nfor n in range(n_components):\n    median = dffs_[dffs_.predict == n][\"bgm_proba\"].median()\n\n    # Experiment with different thresholds\n    # Higher thereshold might overfit\n    n_inx = dffs_[\n        (dffs_.predict == n) & (dffs_.bgm_proba > 0.675)\n    ].index\n\n    train_index = np.concatenate((train_index, n_inx))\n    print(\n        f\"class:{n}\",\n        f\"median: {round(median,4)}\",\n        \"Training data:\"\n        + str(round(len(n_inx) / len(dffs_[(dffs_.predict == n)]), 2) * 100)\n        + \"%\",\n    )\n\n\nprint(f\"\\nSize of Training data : {len(train_index)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:43:45.054489Z","iopub.execute_input":"2022-07-31T13:43:45.055669Z","iopub.status.idle":"2022-07-31T13:43:45.214909Z","shell.execute_reply.started":"2022-07-31T13:43:45.055595Z","shell.execute_reply":"2022-07-31T13:43:45.213667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = dffs_.loc[train_index][power_features]\ny = dffs_.loc[train_index][\"predict\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:43:45.216825Z","iopub.execute_input":"2022-07-31T13:43:45.217644Z","iopub.status.idle":"2022-07-31T13:43:45.279026Z","shell.execute_reply.started":"2022-07-31T13:43:45.217598Z","shell.execute_reply":"2022-07-31T13:43:45.277711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# https://www.kaggle.com/code/karlcini/bayesiangmmclassifier\n\nbgm = BayesianGMMClassifier(\n    n_components=7,\n    random_state=42,\n    # tol =1e-3,\n    covariance_type=\"full\",\n    max_iter=500,\n    n_init=7,\n    init_params=\"kmeans\", # you can use k-means++\n)\nbgm.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:43:45.280600Z","iopub.execute_input":"2022-07-31T13:43:45.281604Z","iopub.status.idle":"2022-07-31T13:49:42.699115Z","shell.execute_reply.started":"2022-07-31T13:43:45.281557Z","shell.execute_reply":"2022-07-31T13:49:42.697857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = bgm.predict(X)\naccuracy_score(y, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:56:32.832176Z","iopub.execute_input":"2022-07-31T13:56:32.832731Z","iopub.status.idle":"2022-07-31T13:56:34.229775Z","shell.execute_reply.started":"2022-07-31T13:56:32.832681Z","shell.execute_reply":"2022-07-31T13:56:34.228470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:56:40.141076Z","iopub.execute_input":"2022-07-31T13:56:40.141503Z","iopub.status.idle":"2022-07-31T13:56:40.167928Z","shell.execute_reply.started":"2022-07-31T13:56:40.141463Z","shell.execute_reply":"2022-07-31T13:56:40.166985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = bgm.predict(test_data)\nss[\"Predicted\"] = predictions\nss.to_csv(\n    \"submission.csv\",\n    index=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:56:41.621972Z","iopub.execute_input":"2022-07-31T13:56:41.622416Z","iopub.status.idle":"2022-07-31T13:56:43.177228Z","shell.execute_reply.started":"2022-07-31T13:56:41.622375Z","shell.execute_reply":"2022-07-31T13:56:43.176017Z"},"trusted":true},"execution_count":null,"outputs":[]}]}