{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport pandas as pd\nimport numpy as np\nimport matplotlib.image as mpimg\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Activation, BatchNormalization\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\n\nimport cv2\nimport os\nimport pickle\nfrom IPython.lib.display import Audio\nimport zipfile \n\nfrom sklearn.utils import shuffle   \nfrom sklearn.model_selection import train_test_split \nimport shutil   \nimport matplotlib.pyplot as plt\n\nimport plotly.graph_objects as go\n\nimport plotly.figure_factory as ff\n%matplotlib inline\n\ntf.random.set_seed(101)","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:29.838547Z","iopub.execute_input":"2021-12-05T15:31:29.838999Z","iopub.status.idle":"2021-12-05T15:31:39.494403Z","shell.execute_reply.started":"2021-12-05T15:31:29.838857Z","shell.execute_reply":"2021-12-05T15:31:39.493415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"../input/histopathologic-cancer-detection/\"\n\n# we load the training set and test set\ntrain = pd.read_csv(path + 'train_labels.csv')\ntest = pd.read_csv(path + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:39.496289Z","iopub.execute_input":"2021-12-05T15:31:39.496523Z","iopub.status.idle":"2021-12-05T15:31:40.299192Z","shell.execute_reply.started":"2021-12-05T15:31:39.496478Z","shell.execute_reply":"2021-12-05T15:31:40.298340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train))\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.301150Z","iopub.execute_input":"2021-12-05T15:31:40.301713Z","iopub.status.idle":"2021-12-05T15:31:40.307041Z","shell.execute_reply.started":"2021-12-05T15:31:40.301665Z","shell.execute_reply":"2021-12-05T15:31:40.306077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.308330Z","iopub.execute_input":"2021-12-05T15:31:40.308549Z","iopub.status.idle":"2021-12-05T15:31:40.334850Z","shell.execute_reply.started":"2021-12-05T15:31:40.308522Z","shell.execute_reply":"2021-12-05T15:31:40.334008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data = train\nprint(df_data.shape)\ntrain['id']=df_data['id'].apply(lambda x: x+'.tif')\ndf_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.336679Z","iopub.execute_input":"2021-12-05T15:31:40.336898Z","iopub.status.idle":"2021-12-05T15:31:40.413886Z","shell.execute_reply.started":"2021-12-05T15:31:40.336872Z","shell.execute_reply":"2021-12-05T15:31:40.413033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset contains 220,025 training images, which are labeled 0 or 1.\n0 is negative, i.e. no cancer, and 1 is positive, i.e. the image contains (cancer) metastases.\n\nThe dataset also includes 57,458 test images. The test samples are unmarked so will not be used to build and evaluate our model.","metadata":{}},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"\ntrain.label.value_counts() ","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.415391Z","iopub.execute_input":"2021-12-05T15:31:40.416223Z","iopub.status.idle":"2021-12-05T15:31:40.427108Z","shell.execute_reply.started":"2021-12-05T15:31:40.416183Z","shell.execute_reply":"2021-12-05T15:31:40.426283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round((train.label.value_counts() / len(train)).to_frame()*100,2)","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.429280Z","iopub.execute_input":"2021-12-05T15:31:40.429523Z","iopub.status.idle":"2021-12-05T15:31:40.444608Z","shell.execute_reply.started":"2021-12-05T15:31:40.429477Z","shell.execute_reply":"2021-12-05T15:31:40.443691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The training dataset consists of 130,908 negatives and 89,117 positives, which is approximately 59.5% and 40.5%, respectively, as shown in the pie chart below.","metadata":{}},{"cell_type":"markdown","source":"# Visualization of Images","metadata":{}},{"cell_type":"code","source":"import cv2\nfig, axs = plt.subplots(3,3,figsize=(8, 5), dpi=150)\n\n\nimages = []\nfor i in range(3):\n    for j in range(3):\n        \n        tran = np.random.randint(0,1000)\n                \n        image = cv2.imread(path + \"train/\" + df_data.iloc[tran]['id'])\n        images.append(axs[i, j].imshow(image))\n        \n        if df_data.iloc[tran]['label'] == 1:\n            axs[i,j].set_title('Tumor')\n        else:\n            axs[i,j].set_title('No Cancer')\n            \n        axs[i,j].set_xticks([])\n        axs[i,j].set_yticks([])\n        \n\n    \nplt.show()\ndel images","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:40.446278Z","iopub.execute_input":"2021-12-05T15:31:40.446719Z","iopub.status.idle":"2021-12-05T15:31:41.086177Z","shell.execute_reply.started":"2021-12-05T15:31:40.446676Z","shell.execute_reply":"2021-12-05T15:31:41.085108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is difficult to determine the features that distinguish cancer cells from normal cells from the presented pictures. We can see that there are images of cancerous and non-cancerous cells with similar colors and with a large and a small number of round nodes. Let's look at how the frequency of the color channels of a randomly selected image is plotted for two possible categories.","metadata":{}},{"cell_type":"code","source":"# with mpimg\ncancer_data = df_data[(df_data.label==1)]\ncancer_image = cancer_data.iloc[900]['id']\nimg = mpimg.imread(path + \"train/\" + cancer_image)\nplt.imshow(img)\nplt.title(\"Cancer Cell\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:41.087531Z","iopub.execute_input":"2021-12-05T15:31:41.087812Z","iopub.status.idle":"2021-12-05T15:31:41.310535Z","shell.execute_reply.started":"2021-12-05T15:31:41.087779Z","shell.execute_reply":"2021-12-05T15:31:41.309867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With cv2\ncancer_data = df_data[(df_data.label==1)]\ncancer_image = cancer_data.iloc[900]['id']\nimg = cv2.imread(path + \"train/\" + cancer_image)\nplt.imshow(img)\nplt.title(\"Cancer Cell\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:41.311456Z","iopub.execute_input":"2021-12-05T15:31:41.312048Z","iopub.status.idle":"2021-12-05T15:31:41.521928Z","shell.execute_reply.started":"2021-12-05T15:31:41.312014Z","shell.execute_reply":"2021-12-05T15:31:41.521074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"plt.hist(img[:, :, 0].ravel(), bins = 256, color = 'red')\nplt.hist(img[:, :, 1].ravel(), bins = 256, color = 'Green')\nplt.hist(img[:, :, 2].ravel(), bins = 256, color = 'Blue')\nplt.xlabel('Intensity')\nplt.ylabel('Quantity')\nplt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])\nplt.title(\"The frequency of the color channels of the cancer cells\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:41.523054Z","iopub.execute_input":"2021-12-05T15:31:41.523263Z","iopub.status.idle":"2021-12-05T15:31:43.384948Z","shell.execute_reply.started":"2021-12-05T15:31:41.523237Z","shell.execute_reply":"2021-12-05T15:31:43.384073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_cancer_data = df_data[(df_data.label==0)]\nnon_cancer_image = non_cancer_data.iloc[500]['id']\n\nimg = cv2.imread(path + \"train/\" + non_cancer_image)\nplt.imshow(img)\nplt.title(\"No Cancer Cell\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:43.386389Z","iopub.execute_input":"2021-12-05T15:31:43.386774Z","iopub.status.idle":"2021-12-05T15:31:43.604160Z","shell.execute_reply.started":"2021-12-05T15:31:43.386738Z","shell.execute_reply":"2021-12-05T15:31:43.603316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img[:, :, 0].ravel(), bins = 256, color = 'red')\nplt.hist(img[:, :, 1].ravel(), bins = 256, color = 'Green')\nplt.hist(img[:, :, 2].ravel(), bins = 256, color = 'Blue')\nplt.xlabel('Intensity')\nplt.ylabel('Quantity')\nplt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])\nplt.title(\"The frequency of the color channels in the absence of cancer cells\")\nplt.show()\n\ndel img, non_cancer_data, cancer_data","metadata":{"execution":{"iopub.status.busy":"2021-12-05T15:31:43.605285Z","iopub.execute_input":"2021-12-05T15:31:43.605507Z","iopub.status.idle":"2021-12-05T15:31:45.570623Z","shell.execute_reply.started":"2021-12-05T15:31:43.605472Z","shell.execute_reply":"2021-12-05T15:31:45.570002Z"},"trusted":true},"execution_count":null,"outputs":[]}]}