{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Import Libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt \nimport seaborn as sns \nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_samples, silhouette_score\nimport matplotlib.cm as cm\nimport numpy as np\nimport matplotlib.style as style\nimport os ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:51:58.126769Z","iopub.execute_input":"2022-07-07T08:51:58.127162Z","iopub.status.idle":"2022-07-07T08:51:58.972927Z","shell.execute_reply.started":"2022-07-07T08:51:58.127129Z","shell.execute_reply":"2022-07-07T08:51:58.971945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check Data\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        file_size = round(os.path.getsize(os.path.join(dirname, filename)) / (1e9), 4)\n        print('*' * 70)\n        print(f\"Filename : {filename} \\t File Size : {file_size} GB\")\n        print('*' * 70)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:51:58.974947Z","iopub.execute_input":"2022-07-07T08:51:58.975322Z","iopub.status.idle":"2022-07-07T08:51:58.988879Z","shell.execute_reply.started":"2022-07-07T08:51:58.975288Z","shell.execute_reply":"2022-07-07T08:51:58.987829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = \"../input/tabular-playground-series-jul-2022/data.csv\"\ndf = pd.read_csv(DATA_DIR)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:51:58.991732Z","iopub.execute_input":"2022-07-07T08:51:58.992596Z","iopub.status.idle":"2022-07-07T08:52:00.104083Z","shell.execute_reply.started":"2022-07-07T08:51:58.992561Z","shell.execute_reply":"2022-07-07T08:52:00.103018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check Statistics among features\ndf.describe(include='all') ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:00.107509Z","iopub.execute_input":"2022-07-07T08:52:00.108332Z","iopub.status.idle":"2022-07-07T08:52:00.296820Z","shell.execute_reply.started":"2022-07-07T08:52:00.108293Z","shell.execute_reply":"2022-07-07T08:52:00.295817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check datatype\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:00.298534Z","iopub.execute_input":"2022-07-07T08:52:00.299294Z","iopub.status.idle":"2022-07-07T08:52:00.326089Z","shell.execute_reply.started":"2022-07-07T08:52:00.299246Z","shell.execute_reply":"2022-07-07T08:52:00.325134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see there are 2 datatypes of features for this data namely int64 and float64 that has no missing values. Each feature has 98000 rows.","metadata":{}},{"cell_type":"code","source":"SUBMISSION_DIR = '../input/tabular-playground-series-jul-2022/sample_submission.csv'\nsubmission =pd.read_csv(SUBMISSION_DIR)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:00.327325Z","iopub.execute_input":"2022-07-07T08:52:00.327677Z","iopub.status.idle":"2022-07-07T08:52:00.361113Z","shell.execute_reply.started":"2022-07-07T08:52:00.327642Z","shell.execute_reply":"2022-07-07T08:52:00.360131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our task is to predict the clusters given the id on the test data stated `For this challenge, you are given (simulated) manufacturing control data that can be clustered into different control states. Your task is to cluster the data into these control states. You are not given any training data, and you are not told how many possible control states there are. This is a completely unsupervised problem, one you might encounter in a real-world setting.`","metadata":{}},{"cell_type":"code","source":"# Check Distribution of the features\nfeatures = [feature for feature in df.columns if feature not in \"id\"]\nsns.set(rc={'figure.figsize':(15,15)})\nfor i, column in enumerate(features, 1):\n    plt.subplot(5,6,i)\n    p=sns.histplot(x=column,data=df.sample(1000),stat='count',kde=True,color='blue')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:00.362847Z","iopub.execute_input":"2022-07-07T08:52:00.363286Z","iopub.status.idle":"2022-07-07T08:52:06.421807Z","shell.execute_reply.started":"2022-07-07T08:52:00.363250Z","shell.execute_reply":"2022-07-07T08:52:06.413029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler =MinMaxScaler()\ndf_scaled = scaler.fit_transform(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:06.423002Z","iopub.execute_input":"2022-07-07T08:52:06.423379Z","iopub.status.idle":"2022-07-07T08:52:06.491423Z","shell.execute_reply.started":"2022-07-07T08:52:06.423343Z","shell.execute_reply":"2022-07-07T08:52:06.490358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check sclaing features\ndf_scale = pd.DataFrame(df_scaled)\ndf_scale.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:06.496368Z","iopub.execute_input":"2022-07-07T08:52:06.498707Z","iopub.status.idle":"2022-07-07T08:52:06.538657Z","shell.execute_reply.started":"2022-07-07T08:52:06.498666Z","shell.execute_reply":"2022-07-07T08:52:06.537752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Building the Model\n#KMeans Algorithm to decide the optimum cluster number , KMeans++ using Elbow Mmethod\n#to figure out K for KMeans, I will use ELBOW Method on KMEANS++ Calculation\nfrom sklearn.cluster import KMeans\nwcss=[]\n\nfor i in range(4,12):\n    kmeans = KMeans(n_clusters= i, init='k-means++', random_state=0)\n    kmeans.fit(df_scale)\n    wcss.append(kmeans.inertia_)\n\n    #inertia_ is the formula used to segregate the data points into clusters","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:52:06.545350Z","iopub.execute_input":"2022-07-07T08:52:06.547873Z","iopub.status.idle":"2022-07-07T08:53:17.006027Z","shell.execute_reply.started":"2022-07-07T08:52:06.547829Z","shell.execute_reply":"2022-07-07T08:53:17.005005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing the ELBOW method to get the optimal value of K \nfig = plt.figure(figsize=(10,8))\nplt.plot(range(4,12), wcss)\nplt.title('The Elbow Method')\nplt.xlabel('no of clusters')\nplt.ylabel('wcss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:53:17.007466Z","iopub.execute_input":"2022-07-07T08:53:17.008065Z","iopub.status.idle":"2022-07-07T08:53:17.267138Z","shell.execute_reply.started":"2022-07-07T08:53:17.008027Z","shell.execute_reply":"2022-07-07T08:53:17.266045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validate the clusters","metadata":{}},{"cell_type":"code","source":"range_n_clusters = [i for i in range(4,12)]\nX =df_scaled\nfor n_clusters in range_n_clusters:\n    # Create a subplot with 1 row and 2 columns\n    fig, (ax1, ax2) = plt.subplots(1, 2)\n    fig.set_size_inches(18, 7)\n\n    # The 1st subplot is the silhouette plot\n    # The silhouette coefficient can range from -1, 1 but in this example all\n    # lie within [-0.1, 1]\n    ax1.set_xlim([-0.1, 1])\n    # The (n_clusters+1)*10 is for inserting blank space between silhouette\n    # plots of individual clusters, to demarcate them clearly.\n    ax1.set_ylim([0, len(X) + (n_clusters + 1) * 10])\n\n    # Initialize the clusterer with n_clusters value and a random generator\n    # seed of 10 for reproducibility.\n    clusterer = KMeans(n_clusters=n_clusters, random_state=10)\n    cluster_labels = clusterer.fit_predict(X)\n\n    # The silhouette_score gives the average value for all the samples.\n    # This gives a perspective into the density and separation of the formed\n    # clusters\n    silhouette_avg = silhouette_score(X, cluster_labels)\n    print(\n        \"For n_clusters =\",\n        n_clusters,\n        \"The average silhouette_score is :\",\n        silhouette_avg,\n    )\n\n    # Compute the silhouette scores for each sample\n    sample_silhouette_values = silhouette_samples(X, cluster_labels)\n\n    y_lower = 10\n    for i in range(n_clusters):\n        # Aggregate the silhouette scores for samples belonging to\n        # cluster i, and sort them\n        ith_cluster_silhouette_values = sample_silhouette_values[cluster_labels == i]\n\n        ith_cluster_silhouette_values.sort()\n\n        size_cluster_i = ith_cluster_silhouette_values.shape[0]\n        y_upper = y_lower + size_cluster_i\n\n        color = cm.nipy_spectral(float(i) / n_clusters)\n        ax1.fill_betweenx(\n            np.arange(y_lower, y_upper),\n            0,\n            ith_cluster_silhouette_values,\n            facecolor=color,\n            edgecolor=color,\n            alpha=0.7,\n        )\n\n        # Label the silhouette plots with their cluster numbers at the middle\n        ax1.text(-0.05, y_lower + 0.5 * size_cluster_i, str(i))\n\n        # Compute the new y_lower for next plot\n        y_lower = y_upper + 10  # 10 for the 0 samples\n\n    ax1.set_title(\"The silhouette plot for the various clusters.\")\n    ax1.set_xlabel(\"The silhouette coefficient values\")\n    ax1.set_ylabel(\"Cluster label\")\n\n    # The vertical line for average silhouette score of all the values\n    ax1.axvline(x=silhouette_avg, color=\"red\", linestyle=\"--\")\n\n    ax1.set_yticks([])  # Clear the yaxis labels / ticks\n    ax1.set_xticks([-0.1, 0, 0.2, 0.4, 0.6, 0.8, 1])\n\n    # 2nd Plot showing the actual clusters formed\n    colors = cm.nipy_spectral(cluster_labels.astype(float) / n_clusters)\n    ax2.scatter(\n        X[:, 0], X[:, 1], marker=\".\", s=30, lw=0, alpha=0.7, c=colors, edgecolor=\"k\"\n    )\n\n    # Labeling the clusters\n    centers = clusterer.cluster_centers_\n    # Draw white circles at cluster centers\n    ax2.scatter(\n        centers[:, 0],\n        centers[:, 1],\n        marker=\"o\",\n        c=\"white\",\n        alpha=1,\n        s=200,\n        edgecolor=\"k\",\n    )\n\n    for i, c in enumerate(centers):\n        ax2.scatter(c[0], c[1], marker=\"$%d$\" % i, alpha=1, s=50, edgecolor=\"k\")\n\n    ax2.set_title(\"The visualization of the clustered data.\")\n    ax2.set_xlabel(\"Feature space for the 1st feature\")\n    ax2.set_ylabel(\"Feature space for the 2nd feature\")\n\n    plt.suptitle(\n        \"Silhouette analysis for KMeans clustering on sample data with n_clusters = %d\"\n        % n_clusters,\n        fontsize=14,\n        fontweight=\"bold\",\n    )\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T08:53:17.319658Z","iopub.execute_input":"2022-07-07T08:53:17.320280Z","iopub.status.idle":"2022-07-07T09:24:13.088299Z","shell.execute_reply.started":"2022-07-07T08:53:17.320238Z","shell.execute_reply":"2022-07-07T09:24:13.087402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Important Points:\n- The Silhouette coefficient of +1 indicates that the sample is far away from the neighboring clusters.\n- The Silhouette coefficient of 0 indicates that the sample is on or very close to the decision boundary between two neighboring clusters.\n- Silhouette coefficient <0 indicates that those samples might have been assigned to the wrong cluster or are outliers.","metadata":{}},{"cell_type":"markdown","source":"## Observations from Silhouette Plots:\n- For n_clusters= 5, 6, 9, 10, 11 are bad pick due to some outliers(negative values) regardless its high score of Silhouette score.\n- For n_clusters=7, all the plots are more or less of similar thickness and hence are of similar sizes, as can be considered as best ‘k’ compared to n_clusters=4.","metadata":{}},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:04:44.916419Z","iopub.execute_input":"2022-07-07T10:04:44.917174Z","iopub.status.idle":"2022-07-07T10:04:44.924192Z","shell.execute_reply.started":"2022-07-07T10:04:44.917138Z","shell.execute_reply":"2022-07-07T10:04:44.922580Z"}}}]}