{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"!pip install memory-profiler\n%load_ext memory_profiler\n# %%memit\n\nimport numpy as np\nfrom glob import glob\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn import svm\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.svm import SVC\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.model_selection import RepeatedStratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix, roc_auc_score\nimport tensorflow as tf\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.keras.metrics import AUC\n\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\n\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, Activation, MaxPool2D, ReLU\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom PIL import Image, ImageDraw\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA\n\nimport gc\n\n# %%memit","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-01T11:47:26.584914Z","iopub.execute_input":"2023-12-01T11:47:26.585529Z","iopub.status.idle":"2023-12-01T11:47:59.504227Z","shell.execute_reply.started":"2023-12-01T11:47:26.585476Z","shell.execute_reply":"2023-12-01T11:47:59.502805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install --pre --upgrade pythainlp\n\n# from pythainlp.chat.core import ChatBotModel\n# import torch\n\n# chatbot = ChatBotModel()\n# chatbot.load_model(device=\"cpu\" ,torch_dtype=torch.bfloat16, load_in_8bit=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:47:59.508144Z","iopub.execute_input":"2023-12-01T11:47:59.508662Z","iopub.status.idle":"2023-12-01T11:47:59.515705Z","shell.execute_reply.started":"2023-12-01T11:47:59.508621Z","shell.execute_reply":"2023-12-01T11:47:59.514518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load","metadata":{}},{"cell_type":"code","source":"data_train_path = \"../input/histopathologic-cancer-detection/train\"\ndata_test_path = \"../input/histopathologic-cancer-detection/test\"\ndata_train_labels = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:47:59.519436Z","iopub.execute_input":"2023-12-01T11:47:59.520127Z","iopub.status.idle":"2023-12-01T11:47:59.922991Z","shell.execute_reply.started":"2023-12-01T11:47:59.520068Z","shell.execute_reply":"2023-12-01T11:47:59.919092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\ndf = pd.DataFrame({'path': glob(os.path.join(data_train_path,'*.tif'))}) # load the filenames\n# df['path'][0]\ndf['id'] = df.path.map(lambda x: x.split('/')[4].split(\".\")[0]) # keep only the file names in 'id'\n# df['id'][0]\ndf = df.merge(data_train_labels, on = \"id\") # merge labels and filepaths\ndf.head(10)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:47:59.927659Z","iopub.execute_input":"2023-12-01T11:47:59.928124Z","iopub.status.idle":"2023-12-01T11:48:02.139296Z","shell.execute_reply.started":"2023-12-01T11:47:59.928084Z","shell.execute_reply":"2023-12-01T11:48:02.137838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load images and labels\ndef load_data(N, df):\n    X = np.zeros([N, 96, 96, 3], dtype=np.uint8)\n    y = np.squeeze(df['label'].values)[0:N]\n\n    for i, (_, row) in tqdm(enumerate(df.iterrows()), total=N):\n        if i == N:\n            break\n        X[i] = cv2.imread(row['path'])\n\n    return X, y\ndf['path'][1]","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:48:02.141526Z","iopub.execute_input":"2023-12-01T11:48:02.141895Z","iopub.status.idle":"2023-12-01T11:48:02.154764Z","shell.execute_reply.started":"2023-12-01T11:48:02.141862Z","shell.execute_reply":"2023-12-01T11:48:02.152963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Data preprocessing\nN = 20000\nx, y = load_data(N, df)","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:48:02.156640Z","iopub.execute_input":"2023-12-01T11:48:02.157056Z","iopub.status.idle":"2023-12-01T11:48:43.810604Z","shell.execute_reply.started":"2023-12-01T11:48:02.157022Z","shell.execute_reply":"2023-12-01T11:48:43.807962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some Eda","metadata":{}},{"cell_type":"code","source":"%%memit\n\nmalignant = data_train_labels.loc[data_train_labels['label']==1]['id'].values    # get the ids of malignant cases\nnormal = data_train_labels.loc[data_train_labels['label']==0]['id'].values       # get the ids of the normal cases\n\ndef plot_fig(ids,title,nrows=5,ncols=15):\n\n    fig,ax = plt.subplots(nrows,ncols,figsize=(18,6))\n    plt.subplots_adjust(wspace=0, hspace=0)\n    for i,j in enumerate(ids[:nrows*ncols]):\n        fname = os.path.join(data_train_path ,j +'.tif')\n        img = Image.open(fname)\n        idcol = ImageDraw.Draw(img)\n        idcol.rectangle(((0,0),(95,95)),outline='white')\n        plt.subplot(nrows, ncols, i+1)\n        plt.imshow(np.array(img))\n        plt.axis('off')\n\n    plt.suptitle(title, y=0.94)\nplot_fig(malignant,'Malignant Cases')","metadata":{"execution":{"iopub.status.busy":"2023-12-01T11:48:43.813659Z","iopub.execute_input":"2023-12-01T11:48:43.815336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\nplot_fig(normal,'Non-Malignant Cases')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Assuming X is a list of images and y is a list of corresponding labels\n\n# Set the intensity thresholds\ndark_threshold = 50\nbright_threshold = 240\n\n# Create a single figure for visualization\nfig, axs = plt.subplots(1, 2, figsize=(12, 6))\n\nfor image, y1 in zip(x, y):\n    # Calculate average intensity of the image\n    avg_intensity = np.mean(image)\n\n    # Determine if the image is too dark or too bright\n    is_dark = avg_intensity <= dark_threshold\n    is_bright = avg_intensity >= bright_threshold\n\n    # Display the image in the appropriate subplot\n    if is_dark or is_bright:\n        ax = axs[0] if is_dark else axs[1]\n        ax.imshow(image, cmap='gray')\n        ax.set_title(f\"{'Dark' if is_dark else 'Bright'} Image - Avg Intensity: {avg_intensity:.2f} - Label: {y1}\")\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Calculate the number of positive and negative samples\nnumber_of_negatives = (y == 0).sum()\nnumber_of_positives = (y == 1).sum()\n\n# Filter samples based on labels\nnegative_samples = x[y == 0]\npositive_samples = x[y == 1]\n\n# Create a bar chart\nfig = plt.figure(figsize=(4, 2), dpi=100)\nplt.bar([1, 0], [number_of_negatives, number_of_positives])\nplt.xticks([1, 0], [\"Negative N = \" + str(number_of_negatives),\n                        \"Positive N = \" + str(number_of_positives)])\nplt.ylabel(\"# of samples\")\n\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\ndef plot_histograms_optimized(positive_samples, negative_samples, nr_of_bins=256):\n    fig, axs = plt.subplots(4, 2, sharey=True, figsize=(8, 8), dpi=120)  # 4 by 2 plot\n    rgb_list = [\"Red\", \"Green\", \"Blue\", \"RGB\"]\n\n    # Set labels and titles outside of the loop\n    axs[0, 0].set_title(\"Positive Samples\")\n    axs[0, 1].set_title(\"Negative Samples\")\n\n    for row_idx in range(3):\n        axs[row_idx, 0].set_ylabel(\"Relative Frequency\")\n        axs[row_idx, 1].set_ylabel(rgb_list[row_idx], rotation=\"horizontal\", labelpad=35, fontsize=12)\n\n        # show the color channels\n        axs[row_idx, 0].hist(\n            positive_samples[:, :, :, row_idx].ravel(),\n            bins=nr_of_bins, density=True\n        )\n        axs[row_idx, 1].hist(\n            negative_samples[:, :, :, row_idx].ravel(),\n            bins=nr_of_bins, density=True\n        )\n\n    # show the rgb as cumulative\n    axs[3, 0].hist(\n        positive_samples.ravel(),\n        bins=nr_of_bins, density=True\n    )\n    axs[3, 1].hist(\n        negative_samples.ravel(),\n        bins=nr_of_bins, density=True\n    )\n\n    return fig, axs\n\nfig, axs = plot_histograms_optimized(positive_samples, negative_samples)\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Test Split","metadata":{}},{"cell_type":"code","source":"%%memit\n\n# normalizing data\nx = x / 255.0\nx = x.reshape(x.shape[0], -1)\n\n# Train-test split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.3, shuffle=True, stratify=y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\ndel malignant\ndel normal\ndel number_of_negatives\ndel number_of_positives\ndel negative_samples\ndel positive_samples\ndel fig\ndel axs\ndel x\ndel y","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import classification_report, confusion_matrix\n# from tensorflow.keras.models import load_model\n\n# Load the trained model\n# model = load_model(\"trained_model.h5\")\n\n# Predictions on the validation set\n# val_predictions = model.predict(val_generator)\n\n# Convert probabilities to binary predictions\n# val_predictions_binary = (val_predictions > 0.5).astype(int)\n\n\n# Save the trained model\n# model.save(\"trained_model.h5\")\n\n# Predictions on the validation set using the trained model\n# y_pred_cnn = model.predict(val_generator)\n\n\n# True labels\n# val_true_labels = val_generator.classes\n\n# # Classification Report\n# classification_report_str = classification_report(val_true_labels, y_pred_cnn, target_names=val_generator.class_indices)\n# print(\"Classification Report:\\n\", classification_report_str)\n\n# # Confusion Matrix\n# conf_matrix = confusion_matrix(val_true_labels, val_predictions_binary)\n# plt.figure(figsize=(8, 6))\n# plt.imshow(conf_matrix, interpolation='nearest', cmap=plt.cm.Blues)\n# plt.title('Confusion Matrix')\n# plt.colorbar()\n# plt.xlabel('Predicted')\n# plt.ylabel('True')\n# plt.show()\n\n# # Plot training history\n# plt.figure(figsize=(12, 5))\n\n\n# # # Plot training & validation accuracy values\n# # plt.subplot(1, 2, 1)\n# # plt.plot(history.history['acc']) \n# # plt.plot(history.history['val_acc']) \n# # plt.title('Model accuracy')\n# # plt.xlabel('Epoch')\n# # plt.ylabel('Accuracy')\n# # plt.legend(['Train', 'Validation'], loc='upper left')\n\n\n# # Plot training & validation loss values\n# plt.subplot(1, 2, 2)\n# plt.plot(history.history['loss'])\n# plt.plot(history.history['val_loss'])\n# plt.title('Model loss')\n# plt.xlabel('Epoch')\n# plt.ylabel('Loss')\n# plt.legend(['Train', 'Validation'], loc='upper left')\n\n# plt.tight_layout()\n# # plt.show()\n# # print(history.history.keys())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reshape images for LDA\n# x_train_flatten = x_train.reshape(x_train.shape[0], -1)\n# x_test_flatten = x_test.reshape(x_test.shape[0], -1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"%%memit\n\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=64)\nrf_classifier.fit(x_train, y_train)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\ny_pred_rf = rf_classifier.predict(x_test)\n\n# Evaluation metrics for Random Forest\naccuracy_rf = accuracy_score(y_test, y_pred_rf)\nprecision_rf = precision_score(y_test, y_pred_rf)\nrecall_rf = recall_score(y_test, y_pred_rf)\nf1_score_rf = f1_score(y_test, y_pred_rf),\nconfusion_matrix_rf = confusion_matrix(y_test, y_pred_rf)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Print RF metrics\nprint(\"RF Model Metrics:\")\nprint(\"Accuracy:\", accuracy_rf)\nprint(\"Precision:\", precision_rf)\nprint(\"Recall:\", recall_rf)\nprint(\"F1 Score:\", f1_score_rf)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_rf)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LDA","metadata":{}},{"cell_type":"code","source":"%%memit\n\n# Initialize and train LDA model\nlda_classifier = LDA()\nx_lda = lda_classifier.fit(x_train, y_train).transform(x_train)\nx_lda_test = lda_classifier.transform(x_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Make predictions on the test set\ny_pred_lda = lda_classifier.predict(x_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Evaluate LDA performance\naccuracy_lda = accuracy_score(y_test, y_pred_lda)\nprecision_lda = precision_score(y_test, y_pred_lda)\nrecall_lda = recall_score(y_test, y_pred_lda)\nf1_score_lda = f1_score(y_test, y_pred_lda)\nconfusion_matrix_lda = confusion_matrix(y_test, y_pred_lda)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%memit\n\n# Print LDA metrics\nprint(\"LDA Model Metrics:\")\nprint(\"Accuracy:\", accuracy_lda)\nprint(\"Precision:\", precision_lda)\nprint(\"Recall:\", recall_lda)\nprint(\"F1 Score:\", f1_score_lda)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_lda)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying kNN","metadata":{}},{"cell_type":"code","source":"%%memit\n\nmodel_KNN = KNeighborsClassifier(n_neighbors=3, n_jobs=-1, metric='euclidean')\nmodel_KNN.fit(x_train, y_train)\n\ny_pred_knn = model_KNN.predict(x_test)\naccuracy_knn = accuracy_score(y_test, y_pred_knn)\nprint(\"KNN accuracy: \" + str(accuracy_knn))\n\naccuracy_knn= accuracy_score(y_test, y_pred_knn)\nprecision_knn = precision_score(y_test, y_pred_knn)\nrecall_knn = recall_score(y_test, y_pred_knn)\nf1_score_knn = f1_score(y_test, y_pred_knn)\nconfusion_matrix_knn= confusion_matrix(y_test, y_pred_knn)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print KNN metrics\nprint(\"KNN Model Metrics:\")\nprint(\"Accuracy:\", accuracy_knn)\nprint(\"Precision:\", precision_knn)\nprint(\"Recall:\", recall_knn)\nprint(\"F1 Score:\", f1_score_knn)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_knn)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SVM","metadata":{}},{"cell_type":"code","source":"%%memit\n\nclf = svm.SVC(kernel='rbf', class_weight = 'balanced')\nclf.fit(x_train, y_train)\n\ny_pred_svm = clf.predict(x_test)\naccuracy_svm = accuracy_score(np.array(y_test), y_pred_svm)\nprint(\"SVM Accuracy : \"+str(accuracy_svm))\n\naccuracy_lda = accuracy_score(y_test, y_pred_svm)\nprecision_lda = precision_score(y_test, y_pred_svm)\nrecall_lda = recall_score(y_test, y_pred_svm)\nf1_score_lda = f1_score(y_test, y_pred_svm)\nconfusion_matrix_lda = confusion_matrix(y_test, y_pred_svm)\n\n# Print LDA metrics\nprint(\"LDA Model Metrics:\")\nprint(\"Accuracy:\", accuracy_lda)\nprint(\"Precision:\", precision_lda)\nprint(\"Recall:\", recall_lda)\nprint(\"F1 Score:\", f1_score_lda)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix_lda)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying SVM and KNN on LDA transformed data","metadata":{}},{"cell_type":"code","source":"%%memit\n\nmodel_ldaknn = KNeighborsClassifier(n_neighbors=10, n_jobs=-1)\nmodel_ldaknn.fit(x_lda, y_train)\ny_pred = model_ldaknn.predict(x_lda_test)\naccuracy_lda_knn = accuracy_score(np.array(y_test), y_pred)\nprint(\"KNN accuracy with LDA data : \" + str(accuracy_lda_knn))\n\n\nclf_lda = svm.SVC(kernel='rbf', class_weight = 'balanced')\nclf_lda.fit(x_lda, y_train)\ny_pred = clf_lda.predict(x_lda_test)\naccuracy_svm_lda = accuracy_score(np.array(y_test), y_pred)\nprint(\"SVM Accuracy with LDA : \"+str(accuracy_svm_lda))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Comparing SVM and KNN on LDA transformed data","metadata":{}},{"cell_type":"code","source":"# Data for plotting\nmodels = ['LDA', 'KNN', 'SVM', 'KNN_LDA', 'SVM_LDA']\naccuracies = [accuracy_lda, accuracy_knn, accuracy_lda_knn, accuracy_svm_lda, accuracy_svm]\n\n# Create a grouped bar plot\nfig, ax = plt.subplots()\n\nbar_width = 0.4\nindex = range(len(models))\n\n# Plot accuracy\nbar1 = ax.bar(index, accuracies, bar_width, label='Accuracy', color='g')\n\n# Adjust the position of bars for the next set of data\nindex = [i + bar_width for i in index]\n\n# Set labels and title\nax.set_xlabel('Models')\nax.set_ylabel('Scores')\nax.set_title('Comparison of Metrics')\nax.set_xticks([i + 0.5 * bar_width for i in index])  # Adjust the position of x-ticks\nax.set_xticklabels(models)\nax.legend()\n\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble models","metadata":{}},{"cell_type":"markdown","source":"## Using Adaboost with DTC","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\nbase_model = DecisionTreeClassifier(max_depth=10)\nadaboost_model = AdaBoostClassifier(base_model, n_estimators=10, random_state=3445)\nadaboost_model.fit(x_train, y_train)\npredictions = adaboost_model.predict(x_test)\naccuracy_DTC_ada = accuracy_score(y_test, predictions)\nprint(f'AdaBoost Model Accuracy: {accuracy_DTC_ada}')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Using Adaboost on SVM","metadata":{}},{"cell_type":"code","source":"base_model = SVC(kernel='rbf', probability=True)\nadaboost_model = AdaBoostClassifier(base_model, n_estimators=3, random_state=2323)\nadaboost_model.fit(x_train, y_train)\npredictions = adaboost_model.predict(x_test)\n# Evaluate the performance\naccuracy_svm_ada = accuracy_score(y_test, predictions)\nprint(f'AdaBoost with SVM Accuracy: {accuracy_svm_ada}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Using Bagging on CNN","metadata":{}},{"cell_type":"code","source":"# Define a simple CNN model\ncnn_model_BG = Sequential()\ncnn_model_BG.add(Conv2D(32, kernel_size=(3, 3), activation='relu', input_shape=(96, 96, 3)))\ncnn_model_BG.add(MaxPooling2D(pool_size=(2, 2)))\ncnn_model_BG.add(Flatten())\ncnn_model_BG.add(Dense(128, activation='relu'))\ncnn_model_BG.add(Dense(1, activation='sigmoid'))\n\ncnn_model_BG.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nbagging_model = BaggingClassifier(base_estimator=cnn_model_BG, n_estimators=3, random_state=3453)\nbagging_model.fit(X_train, y_train)\npredictions = bagging_model.predict(X_test)\naccuracy_cnn_bag = accuracy_score(y_test, predictions)\nprint(f'Bagging with CNN Accuracy: {accuracy_cnn_bag}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### RAM optimization","metadata":{}},{"cell_type":"code","source":"del rf_classifier\ndel lda_classifier\ndel x_lda\ndel x_lda_test\ndel model_KNN\ndel clf\ndel model_ldaknn\ndel base_model_1\ndel base_model_2\ndel base_model_3\ndel ensemble_model\ndel cnn_model_BG\ndel bagging_model\ndel x_train\ndel x_test\ndel y_train\ndel y_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN","metadata":{}},{"cell_type":"markdown","source":"## Data Augmentation","metadata":{}},{"cell_type":"code","source":"df1 = df\ndf1['label'] = df1['label'].astype(str)\n\n\n# Define the number of images you want to load\nnum_images = 20000\n\n# Create a subset of your DataFrame\nsubset_df = df1.head(num_images)\n\n# Split the data into training and validation sets\ntrain_df, temp_df = train_test_split(subset_df, test_size=0.33, random_state=42)\ntest_df, val_df = train_test_split(temp_df, test_size=0.5, random_state=42)\n\n# Define image dimensions and other parameters\nimg_height, img_width = 96, 96\nbatch_size = 32\n\n# Create ImageDataGenerators\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest')\n\nval_datagen = ImageDataGenerator(\n    rescale=1./255,)\n\ntest_datagen = ImageDataGenerator(\n    rescale=1./255,)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    x_col='path',  # Replace with your actual column name\n    y_col='label',        # Replace with your actual column name\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='binary',     # Use 'binary' for binary classification\n    shuffle=True\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    val_df,\n    x_col='path',\n    y_col='label',\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='binary',\n    shuffle=False  # No need to shuffle validation data\n)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    test_df,\n    x_col='path',\n    y_col='label',\n    target_size=(img_height, img_width),\n    batch_size=batch_size,\n    class_mode='binary',\n    shuffle=False  # No need to shuffle validation data\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNN MODEL","metadata":{}},{"cell_type":"code","source":"\n# Check if GPU is available\nphysical_devices = tf.config.list_physical_devices('GPU')\nif physical_devices:\n    # Set memory growth for all GPUs to be the same\n    for device in physical_devices:\n        tf.config.experimental.set_memory_growth(device, True)\n    print(\"GPU is available and configured.\")\nelse:\n    print(\"GPU is not available. Training on CPU.\")\n\n# Define the model\nmodel = Sequential([   \n    Conv2D(32, (3, 3), strides=(1, 1), padding='valid', input_shape=(96, 96, 3)),\n    BatchNormalization(),\n    ReLU(),\n    MaxPooling2D(pool_size=(2, 2), strides=(2, 2)),\n    \n    Conv2D(64, (2, 2), strides=(1, 1), padding='valid'),\n    BatchNormalization(),\n    ReLU(),\n    MaxPooling2D(pool_size=(2, 2), strides=(2, 2)),\n    \n    Conv2D(128, (3, 3), strides=(1, 1), padding='valid'),\n    BatchNormalization(),\n    ReLU(),\n    MaxPooling2D(pool_size=(2, 2), strides=(2, 2)),\n    \n    Conv2D(256, (3, 3), strides=(1, 1), padding='valid'),\n    BatchNormalization(),\n    ReLU(),\n    MaxPooling2D(pool_size=(2, 2), strides=(2, 2)),\n    \n    Conv2D(512, (3, 3), strides=(1, 1), padding='valid'),\n    BatchNormalization(),\n    ReLU(),\n    MaxPooling2D(pool_size=(2, 2), strides=(2, 2)),\n    \n    Dropout(0.5),\n    Flatten(),\n    \n    Dense(1024, activation='relu'),\n    Dropout(0.4),\n    \n    Dense(512, activation='relu'),\n    Dropout(0.4),\n    \n    Dense(1, activation='sigmoid')\n])\nmodel.compile(optimizer=Adam(learning_rate=0.00015), loss=BinaryCrossentropy(), metrics=[AUC()])\n\n# Display the model summary\nmodel.summary()\n\n# Set up Early Stopping\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\n# Train the model\nhistory = model.fit(\n    train_generator,\n    epochs=20,\n    validation_data=val_generator,\n    callbacks=[early_stopping]\n)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate the model\n","metadata":{}},{"cell_type":"code","source":"test_loss_cnn, test_auc_cnn = model.evaluate(val_generator)\nprint(f\"Test Loss: {test_loss}, test AUC: {test_auc}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\ntest_loss, test_auc = model.evaluate(test_generator)\nprint(f\"Validation Loss: {test_loss}, Validation AUC: {test_auc}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparing Models","metadata":{}},{"cell_type":"code","source":"# Data for plotting\nmodels = ['RF', 'Model_CNN', 'LDA', 'KNN', 'SVM', 'KNN_LDA', 'SVM_LDA','Adaboost with DTC', 'Adaboost with SVM','Bagging with CNN' ]\naccuracies = [accuracy_rf, test_auc_cnn, accuracy_lda, accuracy_knn, accuracy_lda_knn, accuracy_svm_lda, accuracy_svm, accuracy_DTC_ada, accuracy_svm_ada, accuracy_cnn_bag]\n\n# Create a grouped bar plot\nfig, ax = plt.subplots()\n\nbar_width = 0.4\nindex = range(len(models))\n\n# Plot accuracy\nbar1 = ax.bar(index, accuracies, bar_width, label='Accuracy', color='g')\n\n# Adjust the position of bars for the next set of data\nindex = [i + bar_width for i in index]\n\n# Set labels and title\nax.set_xlabel('Models')\nax.set_ylabel('Scores')\nax.set_title('Comparison of Metrics')\nax.set_xticks([i + 0.5 * bar_width for i in index])  # Adjust the position of x-ticks\nax.set_xticklabels(models)\nax.legend()\n\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}