{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T05:08:27.865559Z","iopub.execute_input":"2022-07-05T05:08:27.866068Z","iopub.status.idle":"2022-07-05T05:08:27.875319Z","shell.execute_reply.started":"2022-07-05T05:08:27.866029Z","shell.execute_reply":"2022-07-05T05:08:27.874265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading and Understanding Data","metadata":{}},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport datetime as dt\n\nimport sklearn\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\n\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:27.985016Z","iopub.execute_input":"2022-07-05T05:08:27.985771Z","iopub.status.idle":"2022-07-05T05:08:27.991092Z","shell.execute_reply.started":"2022-07-05T05:08:27.985736Z","shell.execute_reply":"2022-07-05T05:08:27.99015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:27.993081Z","iopub.execute_input":"2022-07-05T05:08:27.993713Z","iopub.status.idle":"2022-07-05T05:08:28.828245Z","shell.execute_reply.started":"2022-07-05T05:08:27.993681Z","shell.execute_reply":"2022-07-05T05:08:28.826911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Column Types","metadata":{}},{"cell_type":"code","source":"df_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:28.829794Z","iopub.execute_input":"2022-07-05T05:08:28.830134Z","iopub.status.idle":"2022-07-05T05:08:28.853718Z","shell.execute_reply.started":"2022-07-05T05:08:28.830105Z","shell.execute_reply":"2022-07-05T05:08:28.852345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Shape","metadata":{}},{"cell_type":"code","source":"df_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:28.85518Z","iopub.execute_input":"2022-07-05T05:08:28.855571Z","iopub.status.idle":"2022-07-05T05:08:28.862894Z","shell.execute_reply.started":"2022-07-05T05:08:28.85554Z","shell.execute_reply":"2022-07-05T05:08:28.861909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Missing values","metadata":{}},{"cell_type":"code","source":"df_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:28.865788Z","iopub.execute_input":"2022-07-05T05:08:28.866355Z","iopub.status.idle":"2022-07-05T05:08:28.882901Z","shell.execute_reply.started":"2022-07-05T05:08:28.866307Z","shell.execute_reply":"2022-07-05T05:08:28.881887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Unique Values","metadata":{}},{"cell_type":"code","source":"for col in df_data.columns[1:]:\n    n = len(pd.unique(df_data[col]))\n    print(\"Column - \" , col , \"Unique Values\" , n,end='\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:28.884111Z","iopub.execute_input":"2022-07-05T05:08:28.884643Z","iopub.status.idle":"2022-07-05T05:08:28.990854Z","shell.execute_reply.started":"2022-07-05T05:08:28.884613Z","shell.execute_reply":"2022-07-05T05:08:28.989645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f_07 to f_13 have less than 45 unique values. Hence considering them as categorical","metadata":{}},{"cell_type":"code","source":"features = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13']\nfor feature in features:\n            df_data[feature] = df_data[feature].astype('object')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:28.992577Z","iopub.execute_input":"2022-07-05T05:08:28.993905Z","iopub.status.idle":"2022-07-05T05:08:29.027085Z","shell.execute_reply.started":"2022-07-05T05:08:28.993856Z","shell.execute_reply":"2022-07-05T05:08:29.026037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Removing id and storing it in df_id","metadata":{}},{"cell_type":"code","source":"df_id = df_data.pop('id')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Outlier Treatment and Scaling numeric variables","metadata":{}},{"cell_type":"code","source":"num_cols = df_data.select_dtypes(include = ['int64','float64']).columns.tolist()\nfor col in num_cols:\n    Q1 = df_data[col].quantile(0.05)\n    Q3 = df_data[col].quantile(0.95)\n    IQR = Q3 - Q1\n    df_data = df_data[(df_data[col] >= Q1 - 1.5*IQR) & (df_data[col] <= Q3 + 1.5*IQR)]\n\nscaler = StandardScaler()\n\ndf_data[num_cols] = scaler.fit_transform(df_data[num_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:29.028629Z","iopub.execute_input":"2022-07-05T05:08:29.029793Z","iopub.status.idle":"2022-07-05T05:08:29.493335Z","shell.execute_reply.started":"2022-07-05T05:08:29.029752Z","shell.execute_reply":"2022-07-05T05:08:29.492204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One Hot Encoding - categorical variables","metadata":{}},{"cell_type":"code","source":"df_data = pd.get_dummies(df_data, columns = features)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:29.494759Z","iopub.execute_input":"2022-07-05T05:08:29.495192Z","iopub.status.idle":"2022-07-05T05:08:29.726713Z","shell.execute_reply.started":"2022-07-05T05:08:29.495159Z","shell.execute_reply":"2022-07-05T05:08:29.725665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:29.728112Z","iopub.execute_input":"2022-07-05T05:08:29.728454Z","iopub.status.idle":"2022-07-05T05:08:29.735714Z","shell.execute_reply.started":"2022-07-05T05:08:29.728423Z","shell.execute_reply":"2022-07-05T05:08:29.734441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PCA","metadata":{}},{"cell_type":"code","source":"pca = PCA(random_state=42)\ndf = pca.fit_transform(df_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:29.749986Z","iopub.execute_input":"2022-07-05T05:08:29.750446Z","iopub.status.idle":"2022-07-05T05:08:32.490716Z","shell.execute_reply.started":"2022-07-05T05:08:29.750378Z","shell.execute_reply":"2022-07-05T05:08:32.489308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:32.495104Z","iopub.execute_input":"2022-07-05T05:08:32.495471Z","iopub.status.idle":"2022-07-05T05:08:32.503429Z","shell.execute_reply.started":"2022-07-05T05:08:32.49543Z","shell.execute_reply":"2022-07-05T05:08:32.502198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(range(1,len(pca.explained_variance_ratio_)+1), pca.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:32.504624Z","iopub.execute_input":"2022-07-05T05:08:32.504977Z","iopub.status.idle":"2022-07-05T05:08:33.12763Z","shell.execute_reply.started":"2022-07-05T05:08:32.504948Z","shell.execute_reply":"2022-07-05T05:08:33.126205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_cumu = np.cumsum(pca.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:33.128988Z","iopub.execute_input":"2022-07-05T05:08:33.129334Z","iopub.status.idle":"2022-07-05T05:08:33.134191Z","shell.execute_reply.started":"2022-07-05T05:08:33.129295Z","shell.execute_reply":"2022-07-05T05:08:33.132969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(range(1,len(var_cumu)+1), var_cumu)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:33.135977Z","iopub.execute_input":"2022-07-05T05:08:33.136447Z","iopub.status.idle":"2022-07-05T05:08:33.317531Z","shell.execute_reply.started":"2022-07-05T05:08:33.136376Z","shell.execute_reply":"2022-07-05T05:08:33.316425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=25,random_state=42)\ndf = pca.fit_transform(df_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:33.319013Z","iopub.execute_input":"2022-07-05T05:08:33.319324Z","iopub.status.idle":"2022-07-05T05:08:36.828648Z","shell.execute_reply.started":"2022-07-05T05:08:33.319297Z","shell.execute_reply":"2022-07-05T05:08:36.827411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"# instantiate\n#scaler = StandardScaler()\n\n#df_id = df_data.pop('id')\n# fit_transform\n#df_scaled = scaler.fit_transform(df_data)\n#df_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:36.830563Z","iopub.execute_input":"2022-07-05T05:08:36.830912Z","iopub.status.idle":"2022-07-05T05:08:36.835813Z","shell.execute_reply.started":"2022-07-05T05:08:36.830882Z","shell.execute_reply":"2022-07-05T05:08:36.834478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# k-means with some arbitrary k\nkmeans = KMeans(n_clusters=4, max_iter=100)\nkmeans.fit(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:36.837323Z","iopub.execute_input":"2022-07-05T05:08:36.838422Z","iopub.status.idle":"2022-07-05T05:08:41.409342Z","shell.execute_reply.started":"2022-07-05T05:08:36.838362Z","shell.execute_reply":"2022-07-05T05:08:41.40793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans.labels_","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:41.41123Z","iopub.execute_input":"2022-07-05T05:08:41.411623Z","iopub.status.idle":"2022-07-05T05:08:41.419975Z","shell.execute_reply.started":"2022-07-05T05:08:41.411592Z","shell.execute_reply":"2022-07-05T05:08:41.418621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Finding Optimal Number of Clusters","metadata":{}},{"cell_type":"markdown","source":"To determine the optimal number of clusters, we select the value of k at the “elbow” ie the point after which the inertia start decreasing in a linear fashion\n\nInertia - Sum of squared distances of samples to their closest cluster center.","metadata":{}},{"cell_type":"code","source":"# elbow-curve/SSD\nssd = []\nrange_n_clusters = [4, 5, 6, 7, 8,9,10,11,12]\nfor num_clusters in range_n_clusters:\n    kmeans = KMeans(n_clusters=num_clusters, max_iter=100)\n    kmeans.fit(df)\n    \n    ssd.append(kmeans.inertia_)\n    \n# plot the SSDs for each n_clusters\n# ssd\nplt.plot(ssd)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:08:41.421699Z","iopub.execute_input":"2022-07-05T05:08:41.422222Z","iopub.status.idle":"2022-07-05T05:09:25.019052Z","shell.execute_reply.started":"2022-07-05T05:08:41.422174Z","shell.execute_reply":"2022-07-05T05:09:25.017891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Silhouette Analysis\n\n\n\n* The value of the silhouette score range lies between -1 to 1. It is calculated using the mean intra-cluster distance (a) and the mean nearest-cluster distance (b) for each sample. Score =  (b - a) / max(a, b)\n\n* A score closer to 1 indicates that the data point is very similar to other data points in the cluster, \n\n* A score closer to -1 indicates that the data point is not similar to the data points in its cluster.\n\n* The best value is 1 and the worst value is -1. Values near 0 indicate overlapping clusters","metadata":{}},{"cell_type":"code","source":"# silhouette analysis\nrange_n_clusters = [4, 5, 6, 7, 8,9,10,11,12]\n\nfor num_clusters in range_n_clusters:\n    \n    # intialise kmeans\n    kmeans = KMeans(n_clusters=num_clusters, max_iter=100)\n    kmeans.fit(df)\n    \n    cluster_labels = kmeans.labels_\n    \n    # silhouette score\n    silhouette_avg = silhouette_score(df, cluster_labels)\n    print(\"For n_clusters={0}, the silhouette score is {1}\".format(num_clusters, silhouette_avg))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:09:25.020615Z","iopub.execute_input":"2022-07-05T05:09:25.020972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final model with k=7\nkmeans = KMeans(n_clusters=7, max_iter=50)\nkmeans.fit(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label=  kmeans.fit_predict(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data['label']=label","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame()\noutput['predicted']= df_data['label']\noutput['id']= df_id\n\noutput.set_index('id',inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\noutput.to_csv(\"submission.csv\", index=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}