{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T19:10:48.227036Z","iopub.execute_input":"2022-07-30T19:10:48.227510Z","iopub.status.idle":"2022-07-30T19:10:48.259406Z","shell.execute_reply.started":"2022-07-30T19:10:48.227396Z","shell.execute_reply":"2022-07-30T19:10:48.258488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A thank you note \n\n***the techniques used in this notebook are inspired with the helpful guide of @-medali1992 , thank you !***\n****\n\n***this article was my beacon to understanding diffrent dimensionality reduction methods and how to implement them [https://towardsdatascience.com/the-similarity-between-t-sne-umap-pca-and-other-mappings-c6453b80f303](http://)***\n","metadata":{}},{"cell_type":"markdown","source":"# Importing the data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:10:54.819895Z","iopub.execute_input":"2022-07-30T19:10:54.820899Z","iopub.status.idle":"2022-07-30T19:10:56.166514Z","shell.execute_reply.started":"2022-07-30T19:10:54.820851Z","shell.execute_reply":"2022-07-30T19:10:56.165152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dropping id ","metadata":{}},{"cell_type":"code","source":"df.drop(\"id\", axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:11:01.667405Z","iopub.execute_input":"2022-07-30T19:11:01.667985Z","iopub.status.idle":"2022-07-30T19:11:01.767084Z","shell.execute_reply.started":"2022-07-30T19:11:01.667935Z","shell.execute_reply":"2022-07-30T19:11:01.765102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T18:35:13.842167Z","iopub.execute_input":"2022-07-29T18:35:13.842648Z","iopub.status.idle":"2022-07-29T18:35:13.873851Z","shell.execute_reply.started":"2022-07-29T18:35:13.842609Z","shell.execute_reply":"2022-07-29T18:35:13.872544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:11:08.917590Z","iopub.execute_input":"2022-07-30T19:11:08.918005Z","iopub.status.idle":"2022-07-30T19:11:08.926743Z","shell.execute_reply.started":"2022-07-30T19:11:08.917971Z","shell.execute_reply":"2022-07-30T19:11:08.924532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking to see if there's any missing values\nour data is null value free :D !\n","metadata":{}},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T18:18:46.468788Z","iopub.execute_input":"2022-07-29T18:18:46.469610Z","iopub.status.idle":"2022-07-29T18:18:46.485242Z","shell.execute_reply.started":"2022-07-29T18:18:46.469572Z","shell.execute_reply":"2022-07-29T18:18:46.484415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Taking a look at our data","metadata":{}},{"cell_type":"code","source":"X = df.iloc[:,0]\ny = df.iloc[:,1]\nheatmap, xedges, yedges = np.histogram2d(X, y, bins=50)\nextent = [xedges[0], xedges[-1], yedges[0], yedges[-1]]\n\nplt.clf()\nplt.figure(figsize= (6,6))\nplt.imshow(heatmap.T, extent=extent, origin='lower')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:12:45.182557Z","iopub.execute_input":"2022-07-30T19:12:45.183723Z","iopub.status.idle":"2022-07-30T19:12:45.459179Z","shell.execute_reply.started":"2022-07-30T19:12:45.183669Z","shell.execute_reply":"2022-07-30T19:12:45.458231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dimensionality Reduction  \n**In order to get a better visualsiation for our data we use dimensionality reduction. \nhere well be implementing two diffrent methods, one linear and the other non_linear : PCA, and T_SNE**","metadata":{}},{"cell_type":"markdown","source":"# Applying PCA \n** we'll start with the linear method  **","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\npca = PCA(n_components = 2)\nndf = pca.fit_transform(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:12:51.823909Z","iopub.execute_input":"2022-07-30T19:12:51.824328Z","iopub.status.idle":"2022-07-30T19:12:52.526704Z","shell.execute_reply.started":"2022-07-30T19:12:51.824294Z","shell.execute_reply":"2022-07-30T19:12:52.525064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K_maen with PCA","metadata":{}},{"cell_type":"markdown","source":"**first let's take a look at the optimal k number using elbow method**","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom yellowbrick.cluster import KElbowVisualizer\n\nElbow = KElbowVisualizer(KMeans(random_state=42,init='k-means++'), k=(4,12))\nElbow.fit(ndf)\nElbow.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:16:11.318684Z","iopub.execute_input":"2022-07-30T19:16:11.319107Z","iopub.status.idle":"2022-07-30T19:16:36.360644Z","shell.execute_reply.started":"2022-07-30T19:16:11.319073Z","shell.execute_reply":"2022-07-30T19:16:36.359198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"** Now let's apply that number in our K_mean **","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(init = \"k-means++\", n_clusters = 7)\nkmeans.fit(ndf)\ny_kmeans = kmeans.predict(ndf)\nplt.figure(figsize= (12,12))\nX = ndf[:,0]\ny = ndf[:,1]\nplt.scatter(X, y , c=y_kmeans,  cmap='viridis')\n\ncenters = kmeans.cluster_centers_\nplt.scatter(centers[:, 0], centers[:, 1], c='black', s=200, alpha=0.5)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:44:10.388314Z","iopub.execute_input":"2022-07-30T19:44:10.388943Z","iopub.status.idle":"2022-07-30T19:44:15.486504Z","shell.execute_reply.started":"2022-07-30T19:44:10.388891Z","shell.execute_reply":"2022-07-30T19:44:15.484812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K_means with T-SNE \n**Now let's try the non linear method of dimensionality reduction**","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE\n\n\ntsne = TSNE(n_components = 3, verbose = 1,n_jobs = -1) \ntsne_results = tsne.fit_transform(df)\n\nkmeans = KMeans(n_clusters = 7,init = \"k-means++\", random_state = 42)\nclusters = kmeans.fit_predict(tsne_results)\nplt.figure(figsize= (12,12))\nplt.scatter(tsne_results[:,0], tsne_results[:,1], c=y_kmeans,  cmap='viridis')\n\ncenters = kmeans.cluster_centers_\nplt.scatter(centers[:, 0], centers[:, 1], c='black', s=200, alpha=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:56:41.967198Z","iopub.execute_input":"2022-07-30T20:56:41.967664Z","iopub.status.idle":"2022-07-30T22:01:49.521740Z","shell.execute_reply.started":"2022-07-30T20:56:41.967629Z","shell.execute_reply":"2022-07-30T22:01:49.519547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# submission","metadata":{}},{"cell_type":"code","source":"Predicted = pd.Series(kmeans.predict(tsne_results), name = 'Predicted')\nPredicted.to_csv('submissiontnse.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:18:21.747325Z","iopub.execute_input":"2022-07-30T22:18:21.748605Z","iopub.status.idle":"2022-07-30T22:18:21.877929Z","shell.execute_reply.started":"2022-07-30T22:18:21.748568Z","shell.execute_reply":"2022-07-30T22:18:21.876754Z"},"trusted":true},"execution_count":null,"outputs":[]}]}