{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom glob import glob \nimport time\nimport numpy as np  # for math operations, one of the most commonly used libraries\nimport pandas as pd  # for handling data and data frames, another essential library\nimport cv2  # OpenCV library, which we will use to \"read\" images and transform them\nimport matplotlib.pyplot as plt  # to visualize data\nfrom tqdm.notebook import tqdm # to see the progress\nimport gc #garbage collection to save RAM\n\n# the output of plotting commands is displayed inline within Jupyter notebook:\n%matplotlib inline  \n\nfrom sklearn.model_selection import train_test_split as tts # to split train and validation data\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix\n\n# PyTorch libraries to build a Machine Learning Model:\nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-28T05:20:45.381740Z","iopub.execute_input":"2023-10-28T05:20:45.382887Z","iopub.status.idle":"2023-10-28T05:20:52.346066Z","shell.execute_reply.started":"2023-10-28T05:20:45.382833Z","shell.execute_reply":"2023-10-28T05:20:52.344763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"data_train_path = \"../input/histopathologic-cancer-detection/train\"\ndata_test_path = \"../input/histopathologic-cancer-detection/test\"\ndata_train_labels = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:20:52.348177Z","iopub.execute_input":"2023-10-28T05:20:52.348826Z","iopub.status.idle":"2023-10-28T05:20:52.759465Z","shell.execute_reply.started":"2023-10-28T05:20:52.348781Z","shell.execute_reply":"2023-10-28T05:20:52.758173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Making Dataframe","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame({'path': glob(os.path.join(data_train_path,'*.tif'))}) # load the filenames\n# df['path'][0]\ndf['id'] = df.path.map(lambda x: x.split('/')[4].split(\".\")[0]) # keep only the file names in 'id'\n# df['id'][0]\nprint(df.head(5))\ndf = df.merge(data_train_labels, on = \"id\") # merge labels and filepaths\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:20:52.761046Z","iopub.execute_input":"2023-10-28T05:20:52.762302Z","iopub.status.idle":"2023-10-28T05:21:04.973257Z","shell.execute_reply.started":"2023-10-28T05:20:52.762258Z","shell.execute_reply":"2023-10-28T05:21:04.972136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" function for loading image data and their corresponding labels","metadata":{}},{"cell_type":"code","source":"def load_data(N, df):\n    \"\"\" This function loads N images using the data df \"\"\"\n    # Allocate a numpy array for the images (N, 96x96px, 3 channels, values 0 - 255)\n    X = np.zeros([N, 96, 96, 3], dtype=np.uint8)\n\n    # Convert the labels to a numpy array too\n    y = np.squeeze(df['label'].values)[0:N]  # Corrected the line\n\n    # Read images one by one, tqdm notebook displays a progress bar\n    for i, row in tqdm(df.iterrows(), total=N):\n        if i == N:\n            break\n        X[i] = cv2.imread(row['path'])\n\n    return X, y\n","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:04.976337Z","iopub.execute_input":"2023-10-28T05:21:04.976959Z","iopub.status.idle":"2023-10-28T05:21:04.984616Z","shell.execute_reply.started":"2023-10-28T05:21:04.976918Z","shell.execute_reply":"2023-10-28T05:21:04.983432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# say n=10000\nN = 1000\nx,y = load_data(N,df)\nprint(len(x))\nprint(len(y))","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:04.986179Z","iopub.execute_input":"2023-10-28T05:21:04.986615Z","iopub.status.idle":"2023-10-28T05:21:15.424931Z","shell.execute_reply.started":"2023-10-28T05:21:04.986573Z","shell.execute_reply":"2023-10-28T05:21:15.423561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"x = x / 255.0  # Normalize pixel values","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:15.426439Z","iopub.execute_input":"2023-10-28T05:21:15.426881Z","iopub.status.idle":"2023-10-28T05:21:15.541708Z","shell.execute_reply.started":"2023-10-28T05:21:15.426829Z","shell.execute_reply":"2023-10-28T05:21:15.540310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(8, 9), dpi=150)\nnp.random.seed(100) #we can use the seed to get a different set of random images\nimage_count = 8  # number of images we want to have\n\n# iterate over a list of random numbers, then plot images with labels\nfor plot_idx, image_idx in enumerate(np.random.randint(0, N, image_count)):\n    ax = fig.add_subplot(4, 4, plot_idx+1)\n    plt.imshow(x[image_idx])\n    ax.set_title(\"Label: \" + str(y[image_idx]))","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:15.543287Z","iopub.execute_input":"2023-10-28T05:21:15.543839Z","iopub.status.idle":"2023-10-28T05:21:17.326104Z","shell.execute_reply.started":"2023-10-28T05:21:15.543798Z","shell.execute_reply":"2023-10-28T05:21:17.325018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(4, 2),dpi=100)\n#plot bars of numbers of positive and negative samples\nnumber_of_negatives = (y==0).sum()\nnumber_of_positives = (y==1).sum()\nnegative_samples = x[y==0]\npositive_samples = x[y==1]\nplt.bar([1, 0], [number_of_negatives, number_of_positives])\nplt.xticks([1, 0], [\"Negative N = \" + str(number_of_negatives), \n                    \"Positive N = \" + str(number_of_positives)])\nplt.ylabel(\"# of samples\")\n...","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:17.327297Z","iopub.execute_input":"2023-10-28T05:21:17.328220Z","iopub.status.idle":"2023-10-28T05:21:17.616983Z","shell.execute_reply.started":"2023-10-28T05:21:17.328185Z","shell.execute_reply":"2023-10-28T05:21:17.615813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nr_of_bins = 256 #each possible pixel value will get a bin in the following histograms\nfig,axs = plt.subplots(4,2,sharey=True,figsize=(8,8),dpi=120)  # 4 by 2 plot\nrgb_list = [\"Red\", \"Green\", \"Blue\", \"RGB\"]\n\n#append a histogram to each of the 4*2=8 positions of the axs\nfor row_idx in range(0, 4):\n    for col_idx in range(0, 2):\n        if row_idx < 3:\n            axs[row_idx, 0].set_ylabel(\"Relative Frequency\")\n            axs[row_idx, 1].set_ylabel(rgb_list[row_idx], rotation=\"horizontal\",\n                                       labelpad=35, fontsize=12)\n            # show the color channels\n            if col_idx == 0: #show positive samples\n                axs[row_idx, 0].hist(positive_samples[:, :, :, row_idx].flatten(),\n                                     bins=nr_of_bins, density = True)\n            elif col_idx == 1: #show negative sampels\n                axs[row_idx, 1].hist(negative_samples[:, :, :, row_idx].flatten(),\n                                     bins=nr_of_bins, density = True)\n                \n        else:\n            # show the rgb as cumulative\n            if col_idx == 0: #show positive samples\n                axs[row_idx, 0].hist(positive_samples.flatten(),\n                                     bins=nr_of_bins, density = True)\n            elif col_idx == 1: #show negative sampels\n                axs[row_idx, 1].hist(negative_samples.flatten(),\n                                     bins=nr_of_bins, density = True)\n                \naxs[0, 0].set_title(\"Positive Samples\")\naxs[0, 1].set_title(\"Negative Samples\")\n\n...","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:17.618533Z","iopub.execute_input":"2023-10-28T05:21:17.618887Z","iopub.status.idle":"2023-10-28T05:21:24.239229Z","shell.execute_reply.started":"2023-10-28T05:21:17.618853Z","shell.execute_reply":"2023-10-28T05:21:24.237956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_temp,y_train,y_temp = tts(x,y,test_size = 0.3 ,shuffle =True ,stratify=y)\nx_train.shape\nx_val,x_test,y_val,y_test = tts(x_temp,y_temp,test_size = 0.33 ,shuffle =True ,stratify=y_temp)\nx_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:24.243161Z","iopub.execute_input":"2023-10-28T05:21:24.243534Z","iopub.status.idle":"2023-10-28T05:21:24.357756Z","shell.execute_reply.started":"2023-10-28T05:21:24.243500Z","shell.execute_reply":"2023-10-28T05:21:24.356459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 1: Convolutional Neural Network (CNN)","metadata":{}},{"cell_type":"code","source":"class CNNModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv1 = nn.Conv2d(3, 16, 3, padding=1)\n        self.conv2 = nn.Conv2d(16, 32, 3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.fc1 = nn.Linear(32 * 24 * 24, 128)\n        self.fc2 = nn.Linear(128, 1)\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = x.reshape(-1, 32 * 24 * 24)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:24.359223Z","iopub.execute_input":"2023-10-28T05:21:24.359615Z","iopub.status.idle":"2023-10-28T05:21:24.369050Z","shell.execute_reply.started":"2023-10-28T05:21:24.359582Z","shell.execute_reply":"2023-10-28T05:21:24.368002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_cnn = CNNModel()\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model_cnn.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:24.370509Z","iopub.execute_input":"2023-10-28T05:21:24.370817Z","iopub.status.idle":"2023-10-28T05:21:24.445086Z","shell.execute_reply.started":"2023-10-28T05:21:24.370789Z","shell.execute_reply":"2023-10-28T05:21:24.443666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training loop for CNN model","metadata":{}},{"cell_type":"code","source":"x_train_tensor = torch.Tensor(x_train).permute(0, 3, 1, 2)\ny_train_tensor = torch.Tensor(y_train.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:24.446726Z","iopub.execute_input":"2023-10-28T05:21:24.447068Z","iopub.status.idle":"2023-10-28T05:21:24.541941Z","shell.execute_reply.started":"2023-10-28T05:21:24.447024Z","shell.execute_reply":"2023-10-28T05:21:24.541089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(10):  # Adjust the number of epochs\n    model_cnn.train()\n    optimizer.zero_grad()\n    outputs = model_cnn(x_train_tensor)\n    loss = criterion(outputs, y_train_tensor)\n    loss.backward()\n    optimizer.step()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:24.543529Z","iopub.execute_input":"2023-10-28T05:21:24.544190Z","iopub.status.idle":"2023-10-28T05:21:54.466750Z","shell.execute_reply.started":"2023-10-28T05:21:24.544148Z","shell.execute_reply":"2023-10-28T05:21:54.465455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate the CNN model on the test data","metadata":{}},{"cell_type":"code","source":"model_cnn.eval()\nx_test_tensor = torch.Tensor(x_test).permute(0, 3, 1, 2)\ny_test_tensor = torch.Tensor(y_test.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:54.468420Z","iopub.execute_input":"2023-10-28T05:21:54.468787Z","iopub.status.idle":"2023-10-28T05:21:54.476861Z","shell.execute_reply.started":"2023-10-28T05:21:54.468753Z","shell.execute_reply":"2023-10-28T05:21:54.475532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with torch.no_grad():\n    y_pred_cnn = model_cnn(x_test_tensor)\n    y_pred_cnn = torch.sigmoid(y_pred_cnn).numpy()\n\naccuracy_cnn = accuracy_score(y_test, y_pred_cnn > 0.5)\nprecision_cnn = precision_score(y_test, y_pred_cnn > 0.5)\nrecall_cnn = recall_score(y_test, y_pred_cnn > 0.5)\nf1_score_cnn = f1_score(y_test, y_pred_cnn > 0.5)\nconfusion_matrix_cnn = confusion_matrix(y_test, y_pred_cnn > 0.5)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:54.478107Z","iopub.execute_input":"2023-10-28T05:21:54.478447Z","iopub.status.idle":"2023-10-28T05:21:54.648364Z","shell.execute_reply.started":"2023-10-28T05:21:54.478416Z","shell.execute_reply":"2023-10-28T05:21:54.647116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNN Model performance Matrix","metadata":{}},{"cell_type":"code","source":"print(\"CNN Model Metrics:\")\nprint(\"Accuracy:\", accuracy_cnn)\nprint(\"Precision:\", precision_cnn)\nprint(\"Recall:\", recall_cnn)\nprint(\"F1 Score:\", f1_score_cnn)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_cnn)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:54.651242Z","iopub.execute_input":"2023-10-28T05:21:54.652191Z","iopub.status.idle":"2023-10-28T05:21:54.659858Z","shell.execute_reply.started":"2023-10-28T05:21:54.652140Z","shell.execute_reply":"2023-10-28T05:21:54.658698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 2: Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"x_train_rf = x_train.reshape(x_train.shape[0], -1)\nx_test_rf = x_test.reshape(x_test.shape[0], -1)\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=64)\nrf_classifier.fit(x_train_rf, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:21:54.661583Z","iopub.execute_input":"2023-10-28T05:21:54.662830Z","iopub.status.idle":"2023-10-28T05:22:01.275871Z","shell.execute_reply.started":"2023-10-28T05:21:54.662791Z","shell.execute_reply":"2023-10-28T05:22:01.274712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_rf = rf_classifier.predict(x_test_rf)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:22:01.277147Z","iopub.execute_input":"2023-10-28T05:22:01.277572Z","iopub.status.idle":"2023-10-28T05:22:01.298380Z","shell.execute_reply.started":"2023-10-28T05:22:01.277537Z","shell.execute_reply":"2023-10-28T05:22:01.297160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_rf = accuracy_score(y_test, y_pred_rf)\nprecision_rf = precision_score(y_test, y_pred_rf)\nrecall_rf = recall_score(y_test, y_pred_rf)\nf1_score_rf = f1_score(y_test, y_pred_rf)\nconfusion_matrix_rf = confusion_matrix(y_test, y_pred_rf)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:22:01.299849Z","iopub.execute_input":"2023-10-28T05:22:01.300213Z","iopub.status.idle":"2023-10-28T05:22:01.316300Z","shell.execute_reply.started":"2023-10-28T05:22:01.300181Z","shell.execute_reply":"2023-10-28T05:22:01.314917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Random Forest Model Metrics:\")\nprint(\"Accuracy:\", accuracy_rf)\nprint(\"Precision:\", precision_rf)\nprint(\"Recall:\", recall_rf)\nprint(\"F1 Score:\", f1_score_rf)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_rf)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:22:01.318146Z","iopub.execute_input":"2023-10-28T05:22:01.318596Z","iopub.status.idle":"2023-10-28T05:22:01.328458Z","shell.execute_reply.started":"2023-10-28T05:22:01.318552Z","shell.execute_reply":"2023-10-28T05:22:01.327360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data for plotting\nmodels = ['Random Forest', 'CNN']\naccuracies = [accuracy_rf, accuracy_cnn]\nprecisions = [precision_rf, precision_cnn]","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:22:01.329546Z","iopub.execute_input":"2023-10-28T05:22:01.330451Z","iopub.status.idle":"2023-10-28T05:22:01.341718Z","shell.execute_reply.started":"2023-10-28T05:22:01.330395Z","shell.execute_reply":"2023-10-28T05:22:01.340258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a grouped bar plot\nfig, ax = plt.subplots()\n\nbar_width = 0.35\nindex = range(len(models))\n\n# Plot accuracy\nbar1 = ax.bar(index, accuracies, bar_width, label='Accuracy', color='b')\n\n# Plot precision\nbar2 = ax.bar([i + bar_width for i in index], precisions, bar_width, label='Precision', color='g')\n\n# Set labels and title\nax.set_xlabel('Models')\nax.set_ylabel('Scores')\nax.set_title('Comparison of Accuracy and Precision')\nax.set_xticks([i + bar_width / 2 for i in index])\nax.set_xticklabels(models)\nax.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-28T05:22:01.343260Z","iopub.execute_input":"2023-10-28T05:22:01.343666Z","iopub.status.idle":"2023-10-28T05:22:01.654373Z","shell.execute_reply.started":"2023-10-28T05:22:01.343632Z","shell.execute_reply":"2023-10-28T05:22:01.653139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 3: DenseNet-121","metadata":{}},{"cell_type":"markdown","source":"# Resources\n","metadata":{}},{"cell_type":"markdown","source":"## Convolutional Neural Networks (CNNs):\n\n* **Stanford University's CS231n course:** CS231n Convolutional Neural Networks for Visual Recognition\n* **Fast.ai's Practical Deep Learning for Coders course:** Fast.ai Deep Learning for Coders\n\n## ResNet (Residual Neural Network):\n\n* **Original ResNet paper by Kaiming He et al.:** Deep Residual Learning for Image Recognition\n* **ResNet explained on the Towards Data Science blog:** Understanding and Implementing Architectures of ResNet and ResNext for state-of-the-art Image Classification\n* \n## DenseNet (Densely Connected Convolutional Networks):\n\n* **Original DenseNet paper by Gao Huang et al.:** Densely Connected Convolutional Networks\n* **DenseNet explained on the Towards Data Science blog:** An introduction to DenseNet\n\n## VGG (Visual Geometry Group) Network:\n\n* **Original VGG paper by Karen Simonyan and Andrew Zisserman:** Very Deep Convolutional Networks for Large-Scale Image Recognition\n* **VGG architecture explained on the UFLDL Tutorial:** Convolutional Neural Networks (UFLDL)\n\n## Decision Trees and Random Forests:\n\n* Scikit-learn's documentation on Decision Trees: **Decision Trees**\n* Scikit-learn's documentation on Random Forests: **Random Forests**","metadata":{}}]}