{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":167782,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":142741,"modelId":165319}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:19:10.235003Z","iopub.execute_input":"2024-11-15T13:19:10.235480Z","iopub.status.idle":"2024-11-15T13:19:13.647372Z","shell.execute_reply.started":"2024-11-15T13:19:10.235417Z","shell.execute_reply":"2024-11-15T13:19:13.646562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"test_images = '/kaggle/input/histopathologic-cancer-detection/test/'\n\ntest_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\ntest_df['id'] = test_df['id'] + '.tif'\ntest_df['label'] = test_df['label'].astype(str)\n\nprint('Test Set Size:', test_df.shape)\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:19:13.649103Z","iopub.execute_input":"2024-11-15T13:19:13.649622Z","iopub.status.idle":"2024-11-15T13:19:13.748077Z","shell.execute_reply.started":"2024-11-15T13:19:13.649584Z","shell.execute_reply":"2024-11-15T13:19:13.747192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# View Image Sample","metadata":{}},{"cell_type":"code","source":"# Sample 16 images and labels from the training set\nsample_images = test_df.sample(16)\n\n# Set up the figure and axes\nfig, axes = plt.subplots(4, 4, figsize=(6, 6))\nfig.tight_layout(pad=1.0)\n\n# Loop through the images and display each one with its label\nfor i, ax in enumerate(axes.flat):\n    # Get the filename and label for each sample\n    id = sample_images.iloc[i]['id']  \n    label = sample_images.iloc[i]['label']  \n\n    # Load the image from file\n    img = mpimg.imread(os.path.join(test_images, id))\n\n    # Display the image\n    ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Label: {label}\")\n    ax.axis('off')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:19:13.749231Z","iopub.execute_input":"2024-11-15T13:19:13.749532Z","iopub.status.idle":"2024-11-15T13:19:15.118641Z","shell.execute_reply.started":"2024-11-15T13:19:13.749498Z","shell.execute_reply":"2024-11-15T13:19:15.117592Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create data generator and loader","metadata":{}},{"cell_type":"code","source":"# Data generator\ntest_datagen = ImageDataGenerator(rescale=1/255) # Normalize pixel values\n\n# Data loader\ntest_loader = test_datagen.flow_from_dataframe(\n    dataframe = test_df,\n    directory = test_images,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = 64,\n    seed = 1,\n    shuffle = False,\n    class_mode = 'categorical',\n    target_size = (32,32)\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:19:15.119967Z","iopub.execute_input":"2024-11-15T13:19:15.120425Z","iopub.status.idle":"2024-11-15T13:20:32.312544Z","shell.execute_reply.started":"2024-11-15T13:19:15.120360Z","shell.execute_reply":"2024-11-15T13:20:32.311503Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"cnn = keras.models.load_model('/kaggle/input/week4/keras/default/1/cnn_wk4.keras')\ncnn.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:20:32.314858Z","iopub.execute_input":"2024-11-15T13:20:32.315166Z","iopub.status.idle":"2024-11-15T13:20:33.835777Z","shell.execute_reply.started":"2024-11-15T13:20:32.315133Z","shell.execute_reply":"2024-11-15T13:20:33.834800Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict test images","metadata":{}},{"cell_type":"code","source":"test_preds = cnn.predict(test_loader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:20:33.836994Z","iopub.execute_input":"2024-11-15T13:20:33.837340Z","iopub.status.idle":"2024-11-15T13:25:19.010848Z","shell.execute_reply.started":"2024-11-15T13:20:33.837296Z","shell.execute_reply":"2024-11-15T13:25:19.009947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save predictions to csv","metadata":{}},{"cell_type":"code","source":"# Convert probabilities to predicted class labels\npredicted_labels = np.argmax(test_preds, axis=1)\n\nsubmission_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\n# Create the submission DataFrame using filenames from test_df\nsubmission = pd.DataFrame({'id': submission_df['id'], 'label': predicted_labels})\n\n# Save the DataFrame to a CSV file for submission\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:25:19.012227Z","iopub.execute_input":"2024-11-15T13:25:19.012544Z","iopub.status.idle":"2024-11-15T13:25:19.219408Z","shell.execute_reply.started":"2024-11-15T13:25:19.012511Z","shell.execute_reply":"2024-11-15T13:25:19.218393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:25:19.220537Z","iopub.execute_input":"2024-11-15T13:25:19.220836Z","iopub.status.idle":"2024-11-15T13:25:19.230735Z","shell.execute_reply.started":"2024-11-15T13:25:19.220802Z","shell.execute_reply":"2024-11-15T13:25:19.229729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate frequency distribution\nfrequency_distribution = (submission.label.value_counts() / len(submission)).to_frame()\n\n# Plotting the frequency distribution as a bar chart\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']  # light green for benign, light red for malignant\n\n# Plotting bar chart with specified colors\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-15T13:25:19.232016Z","iopub.execute_input":"2024-11-15T13:25:19.232341Z","iopub.status.idle":"2024-11-15T13:25:19.477831Z","shell.execute_reply.started":"2024-11-15T13:25:19.232288Z","shell.execute_reply":"2024-11-15T13:25:19.476919Z"}},"outputs":[],"execution_count":null}]}