{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":1866454,"sourceType":"datasetVersion","datasetId":1110966},{"sourceId":2027624,"sourceType":"datasetVersion","datasetId":1213331},{"sourceId":2032939,"sourceType":"datasetVersion","datasetId":1216164}],"dockerImageVersionId":30042,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom time import time\nimport shutil\nimport multiprocessing\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nimport keras\nimport cv2\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import TensorBoard\nfrom keras.models import Sequential\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D, MaxPooling2D\nfrom keras.layers.core import Activation, Flatten, Dropout, Dense\nfrom keras.callbacks import TensorBoard\nfrom keras import backend as K\nfrom sklearn.metrics import confusion_matrix,roc_curve,auc\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:06.926812Z","iopub.execute_input":"2025-09-10T02:36:06.927184Z","iopub.status.idle":"2025-09-10T02:36:16.229182Z","shell.execute_reply.started":"2025-09-10T02:36:06.927154Z","shell.execute_reply":"2025-09-10T02:36:16.228055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -V","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:16.231657Z","iopub.execute_input":"2025-09-10T02:36:16.232027Z","iopub.status.idle":"2025-09-10T02:36:17.433595Z","shell.execute_reply.started":"2025-09-10T02:36:16.231992Z","shell.execute_reply":"2025-09-10T02:36:17.432513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preliminary data analysis","metadata":{}},{"cell_type":"code","source":"df1 = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ndf2a = pd.read_csv(\"../input/cancerdata2/df_train.csv\")\ndf2b = pd.read_csv(\"../input/cancerdata2/df_val.csv\")\ndf2 = result = df2a.append(df2b)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:17.436012Z","iopub.execute_input":"2025-09-10T02:36:17.436548Z","iopub.status.idle":"2025-09-10T02:36:18.114230Z","shell.execute_reply.started":"2025-09-10T02:36:17.436497Z","shell.execute_reply":"2025-09-10T02:36:18.112883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:18.115515Z","iopub.execute_input":"2025-09-10T02:36:18.115879Z","iopub.status.idle":"2025-09-10T02:36:18.135363Z","shell.execute_reply.started":"2025-09-10T02:36:18.115845Z","shell.execute_reply":"2025-09-10T02:36:18.134230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def draw_category_images(col_name,figure_cols, df, IMAGE_PATH):\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    f, ax = plt.subplots(nrows=len(categories),ncols=figure_cols, \n                         figsize=(4*figure_cols,4*len(categories))) # adjust size here\n    # draw a number of images for each location\n    for i, cat in enumerate(categories):\n        sample = df[df[col_name]==cat].sample(figure_cols) # figure_cols is also the sample size\n        for j in range(0,figure_cols):\n            file=IMAGE_PATH + sample.iloc[j]['id']\n            print(file)\n            im=cv2.imread(file)\n            ax[i, j].imshow(im, resample=True, cmap='gray')\n            ax[i, j].set_title(cat, fontsize=16)  \n    plt.tight_layout()\n    f.savefig(\"sample_data\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:18.138840Z","iopub.execute_input":"2025-09-10T02:36:18.139203Z","iopub.status.idle":"2025-09-10T02:36:18.151647Z","shell.execute_reply.started":"2025-09-10T02:36:18.139149Z","shell.execute_reply":"2025-09-10T02:36:18.150273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"draw_category_images('label',4, df2a, '../input/histopathologic-cancer-detection/train/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:18.153487Z","iopub.execute_input":"2025-09-10T02:36:18.153837Z","iopub.status.idle":"2025-09-10T02:36:20.318892Z","shell.execute_reply.started":"2025-09-10T02:36:18.153804Z","shell.execute_reply":"2025-09-10T02:36:20.317115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot = df1.label.value_counts().plot(kind='bar', title=\"Distribution of Data\")\nplot.set_xlabel(\"Image Label\")\nplot.set_ylabel(\"Number of Images\")\nplt.tight_layout()\nplt.savefig(\"bar_chart\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:20.320666Z","iopub.execute_input":"2025-09-10T02:36:20.321143Z","iopub.status.idle":"2025-09-10T02:36:20.539325Z","shell.execute_reply.started":"2025-09-10T02:36:20.321092Z","shell.execute_reply":"2025-09-10T02:36:20.538231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df1.label.value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:20.540795Z","iopub.execute_input":"2025-09-10T02:36:20.541189Z","iopub.status.idle":"2025-09-10T02:36:20.551146Z","shell.execute_reply.started":"2025-09-10T02:36:20.541157Z","shell.execute_reply":"2025-09-10T02:36:20.549829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot = df2.label.value_counts().plot(kind='bar', title=\"Distribution of Data\")\nplot.set_xlabel(\"Image Label\")\nplot.set_ylabel(\"Number of Images\")\nplot.invert_xaxis()\nplt.tight_layout()\nplt.savefig(\"bar_chart_2\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:36:20.552534Z","iopub.execute_input":"2025-09-10T02:36:20.552905Z","iopub.status.idle":"2025-09-10T02:36:20.743714Z","shell.execute_reply.started":"2025-09-10T02:36:20.552873Z","shell.execute_reply":"2025-09-10T02:36:20.742624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"--------------------------------\n# Evaluate Model","metadata":{}},{"cell_type":"code","source":"model = keras.models.load_model(\"../input/modpredying/vgg19_model\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:47:49.730335Z","iopub.execute_input":"2025-09-10T02:47:49.730778Z","iopub.status.idle":"2025-09-10T02:47:53.687490Z","shell.execute_reply.started":"2025-09-10T02:47:49.730736Z","shell.execute_reply":"2025-09-10T02:47:53.686294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df3 = pd.read_csv(\"../input/cancerdata2/df_test.csv\", dtype=object)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:47:53.689837Z","iopub.execute_input":"2025-09-10T02:47:53.690308Z","iopub.status.idle":"2025-09-10T02:47:53.711657Z","shell.execute_reply.started":"2025-09-10T02:47:53.690262Z","shell.execute_reply":"2025-09-10T02:47:53.710806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_path = \"../input/histopathologic-cancer-detection/train\"\n\ntest_data_generator = ImageDataGenerator(rescale=1./255).flow_from_dataframe(dataframe = df3,\n                                                                             directory=test_path,\n                                                                             x_col = \"id\",\n                                                                             y_col = \"label\",\n                                                                             target_size=(96,96),\n                                                                             batch_size=16,\n                                                                             class_mode = 'binary',\n                                                                             shuffle=False,\n                                                                             validate_filenames = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:47:53.713142Z","iopub.execute_input":"2025-09-10T02:47:53.713421Z","iopub.status.idle":"2025-09-10T02:47:53.769745Z","shell.execute_reply.started":"2025-09-10T02:47:53.713395Z","shell.execute_reply":"2025-09-10T02:47:53.768717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.rcParams.update({'font.size': 16})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:47:53.771124Z","iopub.execute_input":"2025-09-10T02:47:53.771597Z","iopub.status.idle":"2025-09-10T02:47:53.776543Z","shell.execute_reply.started":"2025-09-10T02:47:53.771563Z","shell.execute_reply":"2025-09-10T02:47:53.775237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict_generator(test_data_generator, steps=len(test_data_generator), verbose = 1)\nfalse_positive_rate, true_positive_rate, threshold = roc_curve(test_data_generator.classes, np.round(predictions))\narea_under_curve = auc(false_positive_rate, true_positive_rate)\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(false_positive_rate, true_positive_rate, label='AUC = {:.3f}'.format(area_under_curve))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.tight_layout()\nplt.savefig(\"roc_curve\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T02:47:53.779635Z","iopub.execute_input":"2025-09-10T02:47:53.780037Z","iopub.status.idle":"2025-09-10T03:00:37.539733Z","shell.execute_reply.started":"2025-09-10T02:47:53.780000Z","shell.execute_reply":"2025-09-10T03:00:37.538220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val = model.evaluate(test_data_generator, verbose = 1)\nprint(val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:00:37.542079Z","iopub.execute_input":"2025-09-10T03:00:37.542417Z","iopub.status.idle":"2025-09-10T03:13:18.198424Z","shell.execute_reply.started":"2025-09-10T03:00:37.542387Z","shell.execute_reply":"2025-09-10T03:13:18.197400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"figure = plt.figure(figsize=(8, 8))\n\nconfusion_m = confusion_matrix(test_data_generator.classes, np.round(predictions))\n\ngroup_names = ['True Neg','False Pos','False Neg','True Pos']\ngroup_counts = [\"{0:0.0f}\".format(value) for value in confusion_m.flatten()]\ngroup_percentages = [\"{0:.2%}\".format(value) for value in confusion_m.flatten()/np.sum(confusion_m)]\nlabels = [f\"{v1}\\n{v2}\\n{v3}\" for v1, v2, v3 in zip(group_names,group_counts,group_percentages)]\nlabels = np.asarray(labels).reshape(2,2)\n\nax = sns.heatmap(confusion_m/np.sum(confusion_m), annot=labels, fmt='',cmap=plt.cm.Blues, cbar = False, annot_kws={\"fontsize\":24})\nax.invert_xaxis()\nax.invert_yaxis()\nplt.ylabel('Actual label')\nplt.xlabel('Predicted label')\nplt.tight_layout()\nplt.savefig(\"confusion_matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:13:18.200460Z","iopub.execute_input":"2025-09-10T03:13:18.200947Z","iopub.status.idle":"2025-09-10T03:13:18.413383Z","shell.execute_reply.started":"2025-09-10T03:13:18.200795Z","shell.execute_reply":"2025-09-10T03:13:18.411656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}