{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Using a  Mapper algorithm for visualization of high-dimensional data****","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"[Info on Keppler](http://https://github.com/scikit-tda/kepler-mapper) along with tutorials\n\n[Kernel](https://www.kaggle.com/noelano/topological-analysis-of-premier-league-players/notebook) provides good information and uses on premier league dataset","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"!pip install kmapper","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport cv2\nfrom glob import glob\ntraining_data_path = \"../input/siic-isic-224x224-images/train/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nimages_path = glob(\"../input/siic-isic-224x224-images/train/*\")\ndf = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"benign=df[df['benign_malignant']=='benign'].sample(600)\nbenign.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"malignant=df[df['benign_malignant']=='malignant']\nmalignant.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Testdf=pd.concat([benign,malignant])\nTestdf.shape\n# df.merge(images_df, left_on='image_name')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_list=Testdf['image_name']\nimage_list[0:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# labels_2 = []\nimages_df=pd.DataFrame()\n\nfor imageName in image_list:\n#column_name=imagePath.split('/')[-1].split('.')[0]\n  imagePath=training_data_path+imageName+'.png'\n#   print(imagePath)\n  image=cv2.imread(imagePath)\n  image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n  image=cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n  # image=cv2.resize(image, (300, 300),interpolation=cv2.INTER_AREA)\n#   print(imageName)\n  images_df[imageName]=image.flatten()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_df=images_df.transpose()\nimages_df.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_name=images_df.index\nimages_df['image_name']=image_name","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_df['image_name'].head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Testdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Testdf=Testdf.merge(images_df,right_on='image_name',left_on='image_name')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Testdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Using TSNE for dimensionality Reduction","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import kmapper as km\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport sklearn\nfrom sklearn import datasets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=images_df.drop(['image_name'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Initialize to use t-SNE with 2 components (reduces data to 2 dimensions). Also note high overlap_percentage.\nmapper_full = km.KeplerMapper(verbose=2)\n\n# Fit and transform data\nprojected_data_full = mapper_full.fit_transform(X,\n                                      projection=sklearn.manifold.TSNE())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create the graph (we cluster on the projected data and suffer projection loss)\ngraph_full = mapper_full.map(projected_data_full,\n#                    clusterer=sklearn.cluster.DBSCAN(eps=0.3, min_samples=15),\n                   clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                    random_state=1618033),          \n                   cover=km.Cover(15, 0.7))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y=Testdf['benign_malignant']\n##Tooltips with the target y-labels for every cluster member\nmapper_full.visualize(graph_full,\n                 title=\"Skin Cancer Mapper with  Y Labels \",\n                 path_html=\"/kaggle/working/skin_cancer_tsne_ylabels_only.html\",\n                 custom_tooltips=Y)\n\n# ##Tooltips with the target y-labels for every cluster member\n# mapper_full.visualize(graph_full,\n#                  title=\"Skin Cancer Mapper with  Image NAmes \",\n#                  path_html=\"/kaggle/working/skin_cancer_image_name_tsne_image_names.html\",\n#                  custom_tooltips=Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Matplotlib examples\nkm.draw_matplotlib(graph_full)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y=Testdf['image_name']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"##Tooltips with the target y-labels for every cluster member\nmapper_full.visualize(graph_full,\n                 title=\"Skin Cancer Mapper with  Image NAmes \",\n                 path_html=\"/kaggle/working/skin_cancer_image_name_tsne_image_names.html\",\n                 custom_tooltips=Y)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Viewing Images from one cluster","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"Testdf[Testdf['image_name'].isin (['ISIC_6187353', 'ISIC_3593913', 'ISIC_2837876', 'ISIC_4301050', 'ISIC_4378851' ,\n                                   'ISIC_0645454', 'ISIC_0961235', 'ISIC_1116483', 'ISIC_1975042', 'ISIC_3244067', \n                                   'ISIC_3253484', 'ISIC_3408231', 'ISIC_3993924', 'ISIC_4730066' ,'ISIC_7075474', \n                                   'ISIC_7181296', 'ISIC_8417873', 'ISIC_8872158' ,'ISIC_8882374', 'ISIC_9509757' ,\n                                   'ISIC_9910791'])]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Looking at  this cluster- It contains only malignant images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# img=images_df[images_df['image_name']=='ISIC_6187353']\n# plt.imshow(np.array(img.iloc[:,0:50176]).reshape(224,224),cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # plt.imshow(training_data_path+'ISIC_3593913.jpg')\n# img=images_df[images_df['image_name']=='ISIC_3593913']\n# plt.imshow(np.array(img.iloc[:,0:50176]).reshape(224,224),cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # plt.imshow(training_data_path+'ISIC_6187353.png')\n# img=images_df[images_df['image_name']=='ISIC_6187353']\n# plt.imshow(np.array(img.iloc[:,0:50176]).reshape(224,224),cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# #ISIC_7075474\n# img=images_df[images_df['image_name']=='ISIC_7075474']\n# plt.imshow(np.array(img.iloc[:,0:50176]).reshape(224,224),cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Using MDS(Multi-dimensional Scaling)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"X=images_df.drop(['image_name'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.manifold import MDS\n\nmapper = km.KeplerMapper(verbose=0)\nlens = mapper.fit_transform(X, projection=MDS())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# Create the simplicial complex\ngraph = mapper.map(lens,\n                   X,\n                   cover=km.Cover(n_cubes=15, perc_overlap=0.7),\n                   clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                    random_state=1618033))\n\ny=Testdf['benign_malignant']\n# Visualization\nmapper.visualize(graph,\n                 path_html=\"Skin-cancer_MDS.html\",\n                 title=\"Melanoma skin Cancer Dataset\",\n                 custom_tooltips=y)\n\n\n# import matplotlib.pyplot as plt\nkm.draw_matplotlib(graph)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=images_df.drop(['image_name'],axis=1)\nfrom sklearn.metrics import pairwise_distances\nfrom sklearn.manifold import MDS\nX1_dist = pairwise_distances(X, metric= 'l2')\n\nmapper = km.KeplerMapper(verbose=0)\nlens2 = mapper.fit_transform(X1_dist, projection=MDS(dissimilarity='precomputed'))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create the simplicial complex\ngraph = mapper.map(lens2,\n                   X,\n                   cover=km.Cover(n_cubes=15, perc_overlap=0.7),\n                   clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                    random_state=1618033))\n\ny=Testdf['benign_malignant']\n# Visualization\nmapper.visualize(graph,\n                 path_html=\"Skin-cancer_MDS_l2_20.html\",\n                 title=\"Melanoma skin Cancer Dataset\",\n                 custom_tooltips=y)\n\n\n# import matplotlib.pyplot as plt\nkm.draw_matplotlib(graph)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Increasing the resolution of the graph","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create the simplicial complex\ngraph = mapper.map(lens2,\n                   X,\n                   cover=km.Cover(n_cubes=30, perc_overlap=0.7),\n                   clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                    random_state=1618033))\n\ny=Testdf['benign_malignant']\n# Visualization\nmapper.visualize(graph,\n                 path_html=\"Skin-cancer_MDS_l2_30.html\",\n                 title=\"Melanoma skin Cancer Dataset\",\n                 custom_tooltips=y)\n\n\n# import matplotlib.pyplot as plt\nkm.draw_matplotlib(graph)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Using PCA","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"X=images_df.drop(['image_name'],axis=1)\n\n\nfrom sklearn.decomposition import PCA\nmapper = km.KeplerMapper(verbose=0)\nlens = mapper.fit_transform(X, projection=PCA(0.8))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create the simplicial complex\ngraph = mapper.map(lens,\n                   X,\n                   cover=km.Cover(n_cubes=15, perc_overlap=0.7),\n                   clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                    random_state=1618033))\n\ny=Testdf['benign_malignant']\n# Visualization\nmapper.visualize(graph,\n                 path_html=\"Skin-cancer_PCA.html\",\n                 title=\"Melanoma skin Cancer Dataset\",\n                 custom_tooltips=y)\n\n\n# import matplotlib.pyplot as plt\nkm.draw_matplotlib(graph)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#  Using Anamoly Detection ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import sklearn\nfrom sklearn import ensemble\n\n# Create a custom 1-D lens with Isolation Forest\nmodel = ensemble.IsolationForest(random_state=1729)\nmodel.fit(X)\nlens1 = model.decision_function(X).reshape((X.shape[0], 1))\n\n# Create another 1-D lens with L2-norm\nmapper = km.KeplerMapper(verbose=0)\nlens2 = mapper.fit_transform(X, projection=\"l2norm\")\n\n# Combine both lenses to get a 2-D [Isolation Forest, L^2-Norm] lens\nlens = np.c_[lens1, lens2]\n\n# Define the simplicial complex\nscomplex = mapper.map(lens,\n                      X,\n                      nr_cubes=15,\n                      overlap_perc=0.7,\n                      clusterer=sklearn.cluster.KMeans(n_clusters=2,\n                                                       random_state=3471))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y=Testdf['benign_malignant']\n# Visualization\nmapper.visualize(graph,\n                 path_html=\"Skin-cancer_IsolationForest.html\",\n                 title=\"Melanoma skin Cancer Dataset\",\n                 custom_tooltips=y)\n\n\n# import matplotlib.pyplot as plt\nkm.draw_matplotlib(graph)\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}