{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T15:24:19.009863Z","iopub.execute_input":"2022-07-28T15:24:19.010359Z","iopub.status.idle":"2022-07-28T15:24:19.034721Z","shell.execute_reply.started":"2022-07-28T15:24:19.010319Z","shell.execute_reply":"2022-07-28T15:24:19.031493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libaries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import RobustScaler, PowerTransformer, StandardScaler\nfrom sklearn.mixture import BayesianGaussianMixture, GaussianMixture\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import StratifiedKFold\n\nimport lightgbm as lgb\n\nimport seaborn as sns\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom matplotlib import pyplot as plt\n%matplotlib inline\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:19.039141Z","iopub.execute_input":"2022-07-28T15:24:19.040426Z","iopub.status.idle":"2022-07-28T15:24:22.403064Z","shell.execute_reply.started":"2022-07-28T15:24:19.040383Z","shell.execute_reply":"2022-07-28T15:24:22.401833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv', index_col='id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:22.404889Z","iopub.execute_input":"2022-07-28T15:24:22.405306Z","iopub.status.idle":"2022-07-28T15:24:23.641421Z","shell.execute_reply.started":"2022-07-28T15:24:22.405263Z","shell.execute_reply":"2022-07-28T15:24:23.640153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:23.645125Z","iopub.execute_input":"2022-07-28T15:24:23.645929Z","iopub.status.idle":"2022-07-28T15:24:23.654111Z","shell.execute_reply.started":"2022-07-28T15:24:23.645882Z","shell.execute_reply":"2022-07-28T15:24:23.652794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:23.656098Z","iopub.execute_input":"2022-07-28T15:24:23.657019Z","iopub.status.idle":"2022-07-28T15:24:23.690091Z","shell.execute_reply.started":"2022-07-28T15:24:23.656978Z","shell.execute_reply":"2022-07-28T15:24:23.688835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- There are 29 features\n- There are 98,000 data points\n- There are 7 discrete features (f_07 to f_13) and other are continuous.","metadata":{}},{"cell_type":"markdown","source":"## Mising values","metadata":{}},{"cell_type":"code","source":"print('missing values:', df.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:23.691887Z","iopub.execute_input":"2022-07-28T15:24:23.692639Z","iopub.status.idle":"2022-07-28T15:24:23.707302Z","shell.execute_reply.started":"2022-07-28T15:24:23.692579Z","shell.execute_reply":"2022-07-28T15:24:23.705626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no missing values","metadata":{}},{"cell_type":"markdown","source":"## Duplicate values","metadata":{}},{"cell_type":"code","source":"print('duplicat values:', df.duplicated().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:23.709434Z","iopub.execute_input":"2022-07-28T15:24:23.710260Z","iopub.status.idle":"2022-07-28T15:24:23.926662Z","shell.execute_reply.started":"2022-07-28T15:24:23.710214Z","shell.execute_reply":"2022-07-28T15:24:23.925003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no duplicate values","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,14))\nfor i, f in enumerate(df.columns):\n    plt.subplot(6, 5, i+1)\n    sns.histplot(x=df[f])\n    plt.title(f'feature: {f}')\n\nfig.suptitle('Feature distributions', size=20)\nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:23.928881Z","iopub.execute_input":"2022-07-28T15:24:23.929602Z","iopub.status.idle":"2022-07-28T15:24:40.557006Z","shell.execute_reply.started":"2022-07-28T15:24:23.929554Z","shell.execute_reply":"2022-07-28T15:24:40.555661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Discrete features are positive skew and have similar distributions\n- Continous features are normally distributed\n- It seems no outliers","metadata":{}},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"markdown","source":"## Scalling\n> It is always important to scale the data for clustering problems so that it is easier to compare the distance between data points. [***Samuel Cortinhas***](https://www.kaggle.com/code/samuelcortinhas/tps-july-22-unsupervised-clustering)","metadata":{}},{"cell_type":"code","source":"# from https://www.kaggle.com/code/samuelcortinhas/tps-july-22-unsupervised-clustering\n\n# scaled_data = pd.DataFrame(RobustScaler().fit_transform(df))\n# scaled_data = pd.DataFrame(StandardScaler().fit_transform(df))\nscaled_data = pd.DataFrame(PowerTransformer().fit_transform(df))\n\nscaled_data.columns = df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:40.562286Z","iopub.execute_input":"2022-07-28T15:24:40.562670Z","iopub.status.idle":"2022-07-28T15:24:44.043830Z","shell.execute_reply.started":"2022-07-28T15:24:40.562625Z","shell.execute_reply":"2022-07-28T15:24:44.042577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Elbow method","metadata":{}},{"cell_type":"code","source":"elbow_m = KElbowVisualizer(KMeans(random_state=23), k=(4, 12))\nelbow_m.fit(scaled_data)\nelbow_m.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:24:44.045729Z","iopub.execute_input":"2022-07-28T15:24:44.046188Z","iopub.status.idle":"2022-07-28T15:26:14.628788Z","shell.execute_reply.started":"2022-07-28T15:24:44.046117Z","shell.execute_reply":"2022-07-28T15:26:14.627456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BEST_CLUSTER_NUM = 6","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:26:14.631022Z","iopub.execute_input":"2022-07-28T15:26:14.631946Z","iopub.status.idle":"2022-07-28T15:26:14.638318Z","shell.execute_reply.started":"2022-07-28T15:26:14.631899Z","shell.execute_reply":"2022-07-28T15:26:14.636569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_bgm = BayesianGaussianMixture(n_components=BEST_CLUSTER_NUM)\npreds = model_bgm.fit_predict(scaled_data)\nsilhouette_score(scaled_data, preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:26:14.640338Z","iopub.execute_input":"2022-07-28T15:26:14.641081Z","iopub.status.idle":"2022-07-28T15:27:10.051979Z","shell.execute_reply.started":"2022-07-28T15:26:14.641039Z","shell.execute_reply":"2022-07-28T15:27:10.050367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We got a very lowe score","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nfor i in range(model_bgm.means_.shape[0]):\n    plt.scatter(np.arange(scaled_data.shape[1]), model_bgm.means_[i])\nplt.xticks(ticks=np.arange(scaled_data.shape[1]), label=scaled_data.columns)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:10.061706Z","iopub.execute_input":"2022-07-28T15:27:10.063025Z","iopub.status.idle":"2022-07-28T15:27:10.494517Z","shell.execute_reply.started":"2022-07-28T15:27:10.062986Z","shell.execute_reply":"2022-07-28T15:27:10.493164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above figure, we can see that the f_00 to f07 and f_14 to f_21 are not much useful. They don't separate the cluster at all. So, we can drop these.","metadata":{}},{"cell_type":"code","source":"features = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\nscaled_data_crop = scaled_data[features]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:10.496225Z","iopub.execute_input":"2022-07-28T15:27:10.497169Z","iopub.status.idle":"2022-07-28T15:27:10.507221Z","shell.execute_reply.started":"2022-07-28T15:27:10.497106Z","shell.execute_reply":"2022-07-28T15:27:10.505889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_km = KMeans(n_clusters=BEST_CLUSTER_NUM)\nkm_preds = model_km.fit_predict(scaled_data_crop)\nsilhouette_score(scaled_data_crop, km_preds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_gm = GaussianMixture(n_components=BEST_CLUSTER_NUM)\ngm_preds = model_gm.fit_predict(scaled_data_crop)\nsilhouette_score(scaled_data_crop, gm_preds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_preds = model_bgm.fit_predict(scaled_data_crop)\nsilhouette_score(scaled_data_crop, bgm_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:10.509073Z","iopub.execute_input":"2022-07-28T15:27:10.509870Z","iopub.status.idle":"2022-07-28T15:27:37.104266Z","shell.execute_reply.started":"2022-07-28T15:27:10.509823Z","shell.execute_reply":"2022-07-28T15:27:37.102575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems better than previous score","metadata":{}},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ashaykatrojwar/eda-pca-bayesian-gaussian-mixture \n    \npp = model_bgm.predict_proba(scaled_data_crop)\ndf_new = pd.DataFrame(scaled_data_crop.copy(), columns=scaled_data_crop.columns)\ndf_new[[f'predict_proba_{i}' for i in range(BEST_CLUSTER_NUM)]] = pp\ndf_new['preds'] = bgm_preds\ndf_new['predict_proba'] = np.max(pp, axis=1)\ndf_new['predict_index'] = np.argmax(pp, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:37.165426Z","iopub.execute_input":"2022-07-28T15:27:37.166320Z","iopub.status.idle":"2022-07-28T15:27:37.383203Z","shell.execute_reply.started":"2022-07-28T15:27:37.166275Z","shell.execute_reply":"2022-07-28T15:27:37.381517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:37.389785Z","iopub.execute_input":"2022-07-28T15:27:37.391083Z","iopub.status.idle":"2022-07-28T15:27:37.430015Z","shell.execute_reply.started":"2022-07-28T15:27:37.391022Z","shell.execute_reply":"2022-07-28T15:27:37.428713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_index = np.array([])\nfor n in range(BEST_CLUSTER_NUM):\n    index = df_new[(df_new.preds == n) & (df_new.predict_proba > 0.68)].index\n    train_index = np.concatenate((train_index, index))\n\n    \ntrain_index","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:27:37.431673Z","iopub.execute_input":"2022-07-28T15:27:37.432485Z","iopub.status.idle":"2022-07-28T15:27:37.469925Z","shell.execute_reply.started":"2022-07-28T15:27:37.432440Z","shell.execute_reply":"2022-07-28T15:27:37.468446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_new=df_new.loc[train_index][features]\ny=df_new.loc[train_index]['preds']\n\nparams_lgb = {'learning_rate': 0.06,\n              'objective': 'multiclass',\n              'boosting': 'gbdt',\n              'n_jobs': -1,\n              'verbosity': -1, \n              'num_classes':BEST_CLUSTER_NUM} \n\nmodel_list=[]\n\ngkf = StratifiedKFold(10, shuffle=True, random_state=42)\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X_new,y)):   \n\n    x_train, y_train = X_new.iloc[train_idx] ,y.iloc[train_idx]\n    x_valid, y_valid = X_new.iloc[valid_idx], y.iloc[valid_idx]\n    train_dataset = lgb.Dataset(x_train, y_train)\n    valid_dataset = lgb.Dataset(x_valid, y_valid)\n    \n    model = lgb.train(params = params_lgb, \n                train_set = train_dataset, \n                valid_sets =  valid_dataset, \n                num_boost_round = 5000, \n                callbacks=[lgb.early_stopping(stopping_rounds=300, verbose=True), \n                           lgb.log_evaluation(period=200)])  \n    \n    model_list.append(model) ","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:28:45.410876Z","iopub.execute_input":"2022-07-28T15:28:45.411263Z","iopub.status.idle":"2022-07-28T15:39:58.099530Z","shell.execute_reply.started":"2022-07-28T15:28:45.411233Z","shell.execute_reply":"2022-07-28T15:39:58.097925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_preds=0\nfor model in model_list:\n    lgb_preds+=model.predict(df_new[features])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:46:32.277875Z","iopub.execute_input":"2022-07-28T15:46:32.278349Z","iopub.status.idle":"2022-07-28T15:52:51.424546Z","shell.execute_reply.started":"2022-07-28T15:46:32.278319Z","shell.execute_reply":"2022-07-28T15:52:51.423271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(lgb_preds,axis=1)\nsilhouette_score(lgb_preds, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:53:21.429238Z","iopub.execute_input":"2022-07-28T15:53:21.429588Z","iopub.status.idle":"2022-07-28T15:55:32.241936Z","shell.execute_reply.started":"2022-07-28T15:53:21.429560Z","shell.execute_reply":"2022-07-28T15:55:32.240594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It give us the highest score","metadata":{}},{"cell_type":"code","source":"final_preds = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:57:36.576496Z","iopub.execute_input":"2022-07-28T15:57:36.577200Z","iopub.status.idle":"2022-07-28T15:57:36.583350Z","shell.execute_reply.started":"2022-07-28T15:57:36.577151Z","shell.execute_reply":"2022-07-28T15:57:36.581700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,4))\nsns.countplot(x=final_preds)\nplt.title('predicted clusters')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:57:39.444836Z","iopub.execute_input":"2022-07-28T15:57:39.445275Z","iopub.status.idle":"2022-07-28T15:57:39.676688Z","shell.execute_reply.started":"2022-07-28T15:57:39.445231Z","shell.execute_reply":"2022-07-28T15:57:39.675389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submit = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nsubmit['Predicted'] = final_preds\nsubmit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:57:43.332378Z","iopub.execute_input":"2022-07-28T15:57:43.332884Z","iopub.status.idle":"2022-07-28T15:57:43.528798Z","shell.execute_reply.started":"2022-07-28T15:57:43.332852Z","shell.execute_reply":"2022-07-28T15:57:43.527311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References:\nhttps://www.kaggle.com/code/ashaykatrojwar/feature-selection-eda\n\nhttps://www.kaggle.com/code/ashaykatrojwar/eda-pca-bayesian-gaussian-mixture\n\nhttps://www.kaggle.com/code/samuelcortinhas/tps-july-22-unsupervised-clustering\n\nhttps://www.kaggle.com/competitions/tabular-playground-series-jul-2022/discussion/334808\n\nhttps://www.kaggle.com/code/ashaykatrojwar/eda-pca-bayesian-gaussian-mixture\n\nhttps://www.kaggle.com/code/ashaykatrojwar/feature-selection-iqr-outliers-eda","metadata":{}}]}