{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport plotly as py\nimport plotly.graph_objs as go\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import RobustScaler,PowerTransformer, StandardScaler, MaxAbsScaler\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom collections import Counter\nfrom tqdm import tqdm\n\npy.offline.init_notebook_mode(connected=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:15.364743Z","iopub.execute_input":"2022-07-13T06:48:15.365174Z","iopub.status.idle":"2022-07-13T06:48:16.122916Z","shell.execute_reply.started":"2022-07-13T06:48:15.365071Z","shell.execute_reply":"2022-07-13T06:48:16.121522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import Dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:19.426292Z","iopub.execute_input":"2022-07-13T06:48:19.426687Z","iopub.status.idle":"2022-07-13T06:48:20.154394Z","shell.execute_reply.started":"2022-07-13T06:48:19.426655Z","shell.execute_reply":"2022-07-13T06:48:20.153258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:20.482924Z","iopub.execute_input":"2022-07-13T06:48:20.483342Z","iopub.status.idle":"2022-07-13T06:48:20.511237Z","shell.execute_reply.started":"2022-07-13T06:48:20.483308Z","shell.execute_reply":"2022-07-13T06:48:20.510151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Preprocessing","metadata":{}},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:22.338678Z","iopub.execute_input":"2022-07-13T06:48:22.339652Z","iopub.status.idle":"2022-07-13T06:48:22.346902Z","shell.execute_reply.started":"2022-07-13T06:48:22.339581Z","shell.execute_reply":"2022-07-13T06:48:22.345629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:23.604294Z","iopub.execute_input":"2022-07-13T06:48:23.605021Z","iopub.status.idle":"2022-07-13T06:48:23.816072Z","shell.execute_reply.started":"2022-07-13T06:48:23.604983Z","shell.execute_reply":"2022-07-13T06:48:23.814920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:24.464757Z","iopub.execute_input":"2022-07-13T06:48:24.465401Z","iopub.status.idle":"2022-07-13T06:48:24.488687Z","shell.execute_reply.started":"2022-07-13T06:48:24.465346Z","shell.execute_reply":"2022-07-13T06:48:24.487610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:25.731149Z","iopub.execute_input":"2022-07-13T06:48:25.731803Z","iopub.status.idle":"2022-07-13T06:48:25.749185Z","shell.execute_reply.started":"2022-07-13T06:48:25.731764Z","shell.execute_reply":"2022-07-13T06:48:25.747979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Great! we don't have any null value","metadata":{}},{"cell_type":"markdown","source":"EDA and Visualization","metadata":{}},{"cell_type":"code","source":"# sns.pairplot(df)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:28.892331Z","iopub.execute_input":"2022-07-13T06:48:28.892751Z","iopub.status.idle":"2022-07-13T06:48:28.898691Z","shell.execute_reply.started":"2022-07-13T06:48:28.892716Z","shell.execute_reply":"2022-07-13T06:48:28.897665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Apply corelation technique","metadata":{}},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:30.470337Z","iopub.execute_input":"2022-07-13T06:48:30.470967Z","iopub.status.idle":"2022-07-13T06:48:30.503183Z","shell.execute_reply.started":"2022-07-13T06:48:30.470901Z","shell.execute_reply":"2022-07-13T06:48:30.501922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now, plot the data\n\nplt.figure(figsize=(22,18))\nax = sns.heatmap(df.corr(), annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:31.190149Z","iopub.execute_input":"2022-07-13T06:48:31.190923Z","iopub.status.idle":"2022-07-13T06:48:35.716615Z","shell.execute_reply.started":"2022-07-13T06:48:31.190874Z","shell.execute_reply":"2022-07-13T06:48:35.715523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation(dataset, threshold):\n    col_corr = set()  # Set of all the names of correlated columns\n    corr_matrix = dataset.corr()\n    for i in range(len(corr_matrix.columns)):\n        for j in range(i):\n            if abs(corr_matrix.iloc[i, j]) > threshold: # we are interested in absolute coeff value\n                colname = corr_matrix.columns[i]  # getting the name of column\n                col_corr.add(colname)\n    return col_corr","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:35.718857Z","iopub.execute_input":"2022-07-13T06:48:35.720261Z","iopub.status.idle":"2022-07-13T06:48:35.729871Z","shell.execute_reply.started":"2022-07-13T06:48:35.720215Z","shell.execute_reply":"2022-07-13T06:48:35.728343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_features = correlation(df, 0.1)\nlen(set(corr_features))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:35.731746Z","iopub.execute_input":"2022-07-13T06:48:35.733317Z","iopub.status.idle":"2022-07-13T06:48:36.022444Z","shell.execute_reply.started":"2022-07-13T06:48:35.733274Z","shell.execute_reply":"2022-07-13T06:48:36.021283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_features\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:36.025358Z","iopub.execute_input":"2022-07-13T06:48:36.025900Z","iopub.status.idle":"2022-07-13T06:48:36.033365Z","shell.execute_reply.started":"2022-07-13T06:48:36.025859Z","shell.execute_reply":"2022-07-13T06:48:36.032039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_corr = df.drop(corr_features,axis=1)\nX_corr","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:36.035617Z","iopub.execute_input":"2022-07-13T06:48:36.036086Z","iopub.status.idle":"2022-07-13T06:48:36.081772Z","shell.execute_reply.started":"2022-07-13T06:48:36.036046Z","shell.execute_reply":"2022-07-13T06:48:36.080633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Standardization","metadata":{"execution":{"iopub.status.busy":"2022-07-12T06:42:44.959454Z","iopub.execute_input":"2022-07-12T06:42:44.959808Z","iopub.status.idle":"2022-07-12T06:42:44.977715Z","shell.execute_reply.started":"2022-07-12T06:42:44.959775Z","shell.execute_reply":"2022-07-12T06:42:44.976163Z"}}},{"cell_type":"code","source":"cols = df.drop(columns=['id']).columns\ncols","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:39.889181Z","iopub.execute_input":"2022-07-13T06:48:39.889562Z","iopub.status.idle":"2022-07-13T06:48:39.904642Z","shell.execute_reply.started":"2022-07-13T06:48:39.889531Z","shell.execute_reply":"2022-07-13T06:48:39.903271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = df[cols]\ndata","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:43.860193Z","iopub.execute_input":"2022-07-13T06:48:43.860615Z","iopub.status.idle":"2022-07-13T06:48:43.920641Z","shell.execute_reply.started":"2022-07-13T06:48:43.860561Z","shell.execute_reply":"2022-07-13T06:48:43.919496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select scaler and transformer\n\n#scaler = StandardScaler()\n#df = scaler.fit_transform(df)\n\nAbs_scaler = MaxAbsScaler().fit(data)\ndata = Abs_scaler.fit_transform(data)\n\n#rob_scaler = RobustScaler().fit(data)\n#data = rob_scaler.fit_transform(data)\n\npower_transformer = PowerTransformer().fit(data)\ndata = power_transformer.transform(data)\n\n#quantile_transformer = QuantileTransformer(output_distribution='normal').fit(data)\n#data = quantile_transformer.transform(data)\n\nX_scaled = pd.DataFrame(data, columns=cols)\nX_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:45.628529Z","iopub.execute_input":"2022-07-13T06:48:45.629576Z","iopub.status.idle":"2022-07-13T06:48:49.585890Z","shell.execute_reply.started":"2022-07-13T06:48:45.629536Z","shell.execute_reply":"2022-07-13T06:48:49.584517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rb_scaler=RobustScaler()\nX=rb_scaler.fit_transform(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:49.588225Z","iopub.execute_input":"2022-07-13T06:48:49.589004Z","iopub.status.idle":"2022-07-13T06:48:49.722608Z","shell.execute_reply.started":"2022-07-13T06:48:49.588957Z","shell.execute_reply":"2022-07-13T06:48:49.721434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=3,random_state=1)\npca.fit(X_scaled)\nPCA_ds = pd.DataFrame(pca.transform(X_scaled), columns=([\"col1\",\"col2\",\"col3\"]))\nPCA_ds.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:49.724897Z","iopub.execute_input":"2022-07-13T06:48:49.725667Z","iopub.status.idle":"2022-07-13T06:48:50.318961Z","shell.execute_reply.started":"2022-07-13T06:48:49.725619Z","shell.execute_reply":"2022-07-13T06:48:50.317665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PCA_ds","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:50.326329Z","iopub.execute_input":"2022-07-13T06:48:50.329976Z","iopub.status.idle":"2022-07-13T06:48:50.358117Z","shell.execute_reply.started":"2022-07-13T06:48:50.329925Z","shell.execute_reply":"2022-07-13T06:48:50.356829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x =PCA_ds[\"col1\"]\ny =PCA_ds[\"col2\"]\nz =PCA_ds[\"col3\"]\n#To plot\nfig = plt.figure(figsize=(10,8))\nax = fig.add_subplot(111, projection=\"3d\")\nax.scatter(x,y,z, c=\"maroon\", marker=\"o\" )\nax.set_title(\"A 3D Projection Of Data In The Reduced Dimension\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:52.336552Z","iopub.execute_input":"2022-07-13T06:48:52.337511Z","iopub.status.idle":"2022-07-13T06:48:54.165899Z","shell.execute_reply.started":"2022-07-13T06:48:52.337470Z","shell.execute_reply":"2022-07-13T06:48:54.164645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\nprint('Elbow Method to determine the number of clusters to be formed:')\nElbow_M = KElbowVisualizer(KMeans(random_state=23), k=(4,12))\nElbow_M.fit(X)\nElbow_M.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:48:54.733733Z","iopub.execute_input":"2022-07-13T06:48:54.734438Z","iopub.status.idle":"2022-07-13T06:50:12.309830Z","shell.execute_reply.started":"2022-07-13T06:48:54.734402Z","shell.execute_reply":"2022-07-13T06:50:12.308797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer = PowerTransformer()\nX=transformer.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:50:12.312262Z","iopub.execute_input":"2022-07-13T06:50:12.313032Z","iopub.status.idle":"2022-07-13T06:50:15.989921Z","shell.execute_reply.started":"2022-07-13T06:50:12.312991Z","shell.execute_reply":"2022-07-13T06:50:15.988727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:50:15.991498Z","iopub.execute_input":"2022-07-13T06:50:15.992584Z","iopub.status.idle":"2022-07-13T06:50:16.001769Z","shell.execute_reply.started":"2022-07-13T06:50:15.992531Z","shell.execute_reply":"2022-07-13T06:50:16.000458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BGM = BayesianGaussianMixture(n_components=7,covariance_type='full',random_state=10)\n# fit model and predict clusters\npreds = BGM.fit_predict(X)\nPCA_ds[\"Clusters\"] = preds\n#Adding the Clusters feature to the orignal dataframe.\ndf[\"Clusters\"]= preds","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:50:55.107498Z","iopub.execute_input":"2022-07-13T06:50:55.108674Z","iopub.status.idle":"2022-07-13T06:51:55.038692Z","shell.execute_reply.started":"2022-07-13T06:50:55.108616Z","shell.execute_reply":"2022-07-13T06:51:55.037422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(10,8))\nax = plt.subplot(111, projection='3d', label=\"bla\")\nax.scatter(x, y, z, s=40, c=PCA_ds[\"Clusters\"], marker='o' )\nax.set_title(\"The Plot Of The Clusters\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:52:15.905070Z","iopub.execute_input":"2022-07-13T06:52:15.905503Z","iopub.status.idle":"2022-07-13T06:52:18.808485Z","shell.execute_reply.started":"2022-07-13T06:52:15.905475Z","shell.execute_reply":"2022-07-13T06:52:18.807458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.Predicted=pd.DataFrame(preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:52:26.418190Z","iopub.execute_input":"2022-07-13T06:52:26.418574Z","iopub.status.idle":"2022-07-13T06:52:26.426222Z","shell.execute_reply.started":"2022-07-13T06:52:26.418542Z","shell.execute_reply":"2022-07-13T06:52:26.424908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pl = sns.countplot(x=df[\"Clusters\"])\npl.set_title(\"Distribution Of The Clusters\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:52:26.966276Z","iopub.execute_input":"2022-07-13T06:52:26.967025Z","iopub.status.idle":"2022-07-13T06:52:27.210104Z","shell.execute_reply.started":"2022-07-13T06:52:26.966985Z","shell.execute_reply":"2022-07-13T06:52:27.209060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission File**","metadata":{}},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T06:52:29.137135Z","iopub.execute_input":"2022-07-13T06:52:29.138145Z","iopub.status.idle":"2022-07-13T06:52:29.291427Z","shell.execute_reply.started":"2022-07-13T06:52:29.138094Z","shell.execute_reply":"2022-07-13T06:52:29.290257Z"},"trusted":true},"execution_count":null,"outputs":[]}]}