{"cells":[{"metadata":{"_uuid":"1f29091d498b010e56f96be5fc7f491931c1e6db"},"cell_type":"markdown","source":"Here, I have tried to create a 2D representation of the data to understand the patterns and how they are distributed. I have used parts of code from the kernel by Henrique Mello linked here- https://www.kaggle.com/hrmello/base-cnn-classification-from-scratch"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nfrom skimage.io import imread\nfrom glob import glob\n\nimport itertools\nimport shutil\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelBinarizer\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"base_tile_dir = '../input/train/'\ndf = pd.DataFrame({'path': glob(os.path.join(base_tile_dir,'*.tif'))})\ndf['id'] = df.path.map(lambda x: x.split('/')[3].split(\".\")[0])\nlabels = pd.read_csv(\"../input/train_labels.csv\")\ndf = df.merge(labels, on = \"id\")\ndf.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"639f66aa7ef469d80880cb9fb74843b0f9b4a1e3"},"cell_type":"code","source":"df0 = df[df.label == 0].sample(10000, random_state = 42)\ndf1 = df[df.label == 1].sample(10000, random_state = 42)\ndf = pd.concat([df0, df1], ignore_index=True).reset_index()\ndf = df[[\"path\", \"id\", \"label\"]]\ndf.sample(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65aea497f30cd96ebadfba77cf4c427a053a44c0"},"cell_type":"code","source":"df['image'] = df['path'].map(imread)\ndf.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1df55ea4848473e151c9396ddc8ec75e8212d28c"},"cell_type":"code","source":"input_images = np.stack(list(df.image), axis = 0)\ninput_images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c02a0f359381197a5e6798b80214a4fc6807a951"},"cell_type":"code","source":"encoder = LabelBinarizer()\ny = encoder.fit_transform(df.label)\nnsamples, nx, ny, nz = input_images.shape\nx = input_images.reshape((nsamples,nx*ny*nz))\nx.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5be09a7449a1a536969046495a30043c74b9b932"},"cell_type":"code","source":"import umap","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9518b5208ad070fbfebae83e693c823440664dd0"},"cell_type":"code","source":"embedding = umap.UMAP(n_components=2, min_dist=0.3, metric='correlation',random_state=42, verbose=True).fit_transform(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f39a171e86bf0684a2ad19dcd37dff48eccbf6b"},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 10))\nplt.scatter(\n    embedding[:, 0], embedding[:, 1], cmap=\"Spectral\", s=0.1\n)\nplt.title(\"Cancer detection data embedded into two dimensions by UMAP\", fontsize=18)\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9656a682a3b79f3fad2dfda599677bf2cf0b816c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}