{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import standard libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nimport seaborn as sns\n\n# Yellowbrick is a handy library to find the exact elbow-point when searching for number of clusters\nfrom yellowbrick.cluster import KElbowVisualizer\n\n# Import required sklearn libraries\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.model_selection import GridSearchCV, train_test_split\nfrom sklearn.metrics import classification_report\n\n# LightGBM Classifier\nimport lightgbm as lgb\n\n# Ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Set seed for all cells\nimport os\n\nRANDOM_SEED = 48\n\ndef seed_everything(seed=RANDOM_SEED):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:21.916214Z","iopub.execute_input":"2022-07-26T00:15:21.917562Z","iopub.status.idle":"2022-07-26T00:15:23.109238Z","shell.execute_reply.started":"2022-07-26T00:15:21.917494Z","shell.execute_reply":"2022-07-26T00:15:23.108025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load dataset\ndf = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:23.702765Z","iopub.execute_input":"2022-07-26T00:15:23.703179Z","iopub.status.idle":"2022-07-26T00:15:24.578639Z","shell.execute_reply.started":"2022-07-26T00:15:23.703145Z","shell.execute_reply":"2022-07-26T00:15:24.577467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing data\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:26.098490Z","iopub.execute_input":"2022-07-26T00:15:26.099708Z","iopub.status.idle":"2022-07-26T00:15:26.121502Z","shell.execute_reply.started":"2022-07-26T00:15:26.099666Z","shell.execute_reply":"2022-07-26T00:15:26.120651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for correlations within the dataset\ncorr = df.corr().round(2)\n\nplt.figure(figsize=(20,10))\nsns.heatmap(corr,annot=True,square=False,cmap=\"coolwarm\",vmin=-1,vmax=1,center=0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:27.686878Z","iopub.execute_input":"2022-07-26T00:15:27.687253Z","iopub.status.idle":"2022-07-26T00:15:31.396220Z","shell.execute_reply.started":"2022-07-26T00:15:27.687224Z","shell.execute_reply":"2022-07-26T00:15:31.395370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As can be seen in above heatmap, only few features really contribute. These features will be extracted from the dataset and used to create a model.","metadata":{}},{"cell_type":"code","source":"# Extract columns that have some correlation\ncols_corr = ['f_07','f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13','f_22', 'f_23', 'f_24', 'f_25','f_26', 'f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:40.119734Z","iopub.execute_input":"2022-07-26T00:15:40.120140Z","iopub.status.idle":"2022-07-26T00:15:40.126395Z","shell.execute_reply.started":"2022-07-26T00:15:40.120109Z","shell.execute_reply":"2022-07-26T00:15:40.125259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create DataFrame \ndf_corr = df[cols_corr]\ndf_corr.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:41.142414Z","iopub.execute_input":"2022-07-26T00:15:41.143893Z","iopub.status.idle":"2022-07-26T00:15:41.163802Z","shell.execute_reply.started":"2022-07-26T00:15:41.143847Z","shell.execute_reply":"2022-07-26T00:15:41.162900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scale the data using PowerTransformer in order to make data more Gaussian-like\nscaler = PowerTransformer()\nX_scaled = pd.DataFrame(scaler.fit_transform(df_corr),columns=df_corr.columns)\nX_scaled.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:15:42.980547Z","iopub.execute_input":"2022-07-26T00:15:42.980964Z","iopub.status.idle":"2022-07-26T00:15:44.760537Z","shell.execute_reply.started":"2022-07-26T00:15:42.980929Z","shell.execute_reply":"2022-07-26T00:15:44.759331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find no. of clusters with yellowbrick library\nelbow = KElbowVisualizer(KMeans(),k=(2,20))\nelbow.fit(X_scaled)\nelbow.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:36.377415Z","iopub.execute_input":"2022-07-19T10:52:36.378137Z","iopub.status.idle":"2022-07-19T10:54:48.876443Z","shell.execute_reply.started":"2022-07-19T10:52:36.378099Z","shell.execute_reply":"2022-07-19T10:54:48.875103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bayesian Gaussian Mixture with 7 clusters. Gridsearch {'init_params': 'kmeans', 'tol': 0.01, 'warm_start': True}\nBGMmodel = BayesianGaussianMixture(n_components=7,init_params=\"kmeans\",tol=0.01,warm_start=True,n_init=100,random_state=48)\n\n# params = {\n#     \"tol\": [0.01,0.1,1],\n#     \"init_params\": [\"kmeans\", \"random\"],\n#     \"warm_start\": [True,False]\n# }\n\n# BGM_grid = GridSearchCV(estimator=BGMmodel,param_grid=params,verbose=2,cv=3,scoring=\"adjusted_rand_score\")\n# BGM_grid.fit(X_scaled)\n# BGM_grid.best_params_\n\nBGMmodel.fit(X_scaled.values) ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:16:49.910750Z","iopub.execute_input":"2022-07-26T00:16:49.911190Z","iopub.status.idle":"2022-07-26T00:17:16.051604Z","shell.execute_reply.started":"2022-07-26T00:16:49.911157Z","shell.execute_reply":"2022-07-26T00:17:16.050190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the results for probability and cluster\nprob = BGMmodel.predict_proba(X_scaled)\ncluster = BGMmodel.predict(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:17:35.600130Z","iopub.execute_input":"2022-07-26T00:17:35.600533Z","iopub.status.idle":"2022-07-26T00:17:35.982533Z","shell.execute_reply.started":"2022-07-26T00:17:35.600498Z","shell.execute_reply":"2022-07-26T00:17:35.981108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new dataframe\nX_new = pd.DataFrame({\"Probability\":np.max(prob,axis=1),\"Cluster\":cluster})\nX_new = X_scaled.join(X_new)\nX_new.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:17:39.182292Z","iopub.execute_input":"2022-07-26T00:17:39.182708Z","iopub.status.idle":"2022-07-26T00:17:39.216493Z","shell.execute_reply.started":"2022-07-26T00:17:39.182676Z","shell.execute_reply":"2022-07-26T00:17:39.215300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a more reliable dataframe for training purposes. Only rows with a probability > 80% are selected\nX_reliable = X_new[X_new[\"Probability\"] > 0.80]\nX_reliable.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:18:03.014987Z","iopub.execute_input":"2022-07-26T00:18:03.015425Z","iopub.status.idle":"2022-07-26T00:18:03.049342Z","shell.execute_reply.started":"2022-07-26T00:18:03.015386Z","shell.execute_reply":"2022-07-26T00:18:03.048142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define X and y\nX = X_reliable.drop([\"Probability\",\"Cluster\"],axis=1)\ny = X_reliable[\"Cluster\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:18:07.342690Z","iopub.execute_input":"2022-07-26T00:18:07.343129Z","iopub.status.idle":"2022-07-26T00:18:07.351014Z","shell.execute_reply.started":"2022-07-26T00:18:07.343093Z","shell.execute_reply":"2022-07-26T00:18:07.350039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train/test/split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.10, random_state=48)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:18:08.454544Z","iopub.execute_input":"2022-07-26T00:18:08.455439Z","iopub.status.idle":"2022-07-26T00:18:08.474486Z","shell.execute_reply.started":"2022-07-26T00:18:08.455386Z","shell.execute_reply":"2022-07-26T00:18:08.473652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGBM Classifier to train a model. Gridsearch {{'boosting_type': 'goss', 'learning_rate': 0.3, 'n_estimators': 450}\nlgbm = lgb.LGBMClassifier(boosting_type=\"goss\",learning_rate=0.3,n_estimators=450,random_state=48,class_weight=\"balanced\")\n\n# params = {\n#     \"boosting_type\": ['goss'],\n#     \"learning_rate\": [0.2,0.3,0.4],\n#     \"n_estimators\": [400,450,500],\n# }\n\n# lgbm_grid = GridSearchCV(estimator=lgbm,param_grid=params,cv=3,verbose=2)\n\nlgbm.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:04:58.001888Z","iopub.execute_input":"2022-07-26T01:04:58.002297Z","iopub.status.idle":"2022-07-26T01:05:11.407244Z","shell.execute_reply.started":"2022-07-26T01:04:58.002266Z","shell.execute_reply":"2022-07-26T01:05:11.405954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Classification report with test set\ntest_results_lgbm = lgbm.predict(X_test)\nprint(classification_report(y_test,test_results_lgbm))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:05:18.350978Z","iopub.execute_input":"2022-07-26T01:05:18.351381Z","iopub.status.idle":"2022-07-26T01:05:19.272857Z","shell.execute_reply.started":"2022-07-26T01:05:18.351350Z","shell.execute_reply":"2022-07-26T01:05:19.271455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run the full (original X_scaled) set through the XGBoost Classifier\nresults_lgbm = lgbm.predict(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:05:32.614815Z","iopub.execute_input":"2022-07-26T01:05:32.615240Z","iopub.status.idle":"2022-07-26T01:05:45.632144Z","shell.execute_reply.started":"2022-07-26T01:05:32.615206Z","shell.execute_reply":"2022-07-26T01:05:45.630674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization of results in a reduced 3 dimensional plane\n\n# Reduce features to 3 dimensions\npca = PCA(n_components=3)\nreduced3d = pca.fit_transform(X_scaled)\n\n# Create a dataframe\ndf_3d = pd.DataFrame({\"x\":reduced3d[:,0],\"y\":reduced3d[:,1],\"z\":reduced3d[:,2],\"Clusters\":results_lgbm})\ndf_3d.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:05:53.722419Z","iopub.execute_input":"2022-07-26T01:05:53.723447Z","iopub.status.idle":"2022-07-26T01:05:54.216850Z","shell.execute_reply.started":"2022-07-26T01:05:53.723391Z","shell.execute_reply":"2022-07-26T01:05:54.215770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(10,10))\nax = fig.add_subplot(111, projection='3d')\ncolors = df_3d['Clusters']\nax.scatter(df_3d['x'],df_3d['y'],df_3d['z'],c=colors,cmap=\"Set2\",alpha=0.3)\n\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:05:56.762304Z","iopub.execute_input":"2022-07-26T01:05:56.762720Z","iopub.status.idle":"2022-07-26T01:05:58.691201Z","shell.execute_reply.started":"2022-07-26T01:05:56.762684Z","shell.execute_reply":"2022-07-26T01:05:58.689991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization of results in a reduced 2 dimensional plane\n\n# Reduce features to 2 dimensions\npca = PCA(n_components=2)\nreduced2d = pca.fit_transform(X_scaled)\n\n# Create a dataframe\ndf_2d = pd.DataFrame({\"x\":reduced2d[:,0],\"y\":reduced2d[:,1],\"Clusters\":results_lgbm})\n\n# Plot\nplt.figure(figsize=(10,10))\nsns.scatterplot(x=df_2d[\"x\"],y=df_2d[\"y\"],data=df_2d,hue=df_2d[\"Clusters\"],alpha=0.5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:06:07.763178Z","iopub.execute_input":"2022-07-26T01:06:07.763567Z","iopub.status.idle":"2022-07-26T01:06:11.335895Z","shell.execute_reply.started":"2022-07-26T01:06:07.763537Z","shell.execute_reply":"2022-07-26T01:06:11.334649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.countplot(x=df_2d[\"Clusters\"])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:06:11.714127Z","iopub.execute_input":"2022-07-26T01:06:11.715288Z","iopub.status.idle":"2022-07-26T01:06:11.908368Z","shell.execute_reply.started":"2022-07-26T01:06:11.715238Z","shell.execute_reply":"2022-07-26T01:06:11.907147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DateFrame for submission\nsubmission = pd.DataFrame({\"Id\":df[\"id\"],\"Predicted\":results_lgbm + 1})\n\nsubmission.to_csv(\"submissionBGM.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T01:06:20.951069Z","iopub.execute_input":"2022-07-26T01:06:20.951529Z","iopub.status.idle":"2022-07-26T01:06:21.121814Z","shell.execute_reply.started":"2022-07-26T01:06:20.951494Z","shell.execute_reply":"2022-07-26T01:06:21.120506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}