{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)\nimport keras\nprint(keras.__version__)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:02.011151Z","iopub.execute_input":"2023-11-11T03:48:02.011912Z","iopub.status.idle":"2023-11-11T03:48:13.244481Z","shell.execute_reply.started":"2023-11-11T03:48:02.011880Z","shell.execute_reply":"2023-11-11T03:48:13.243535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Activation, Flatten, Dropout, Dense\nfrom tensorflow.keras.optimizers import Adam\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:13.246544Z","iopub.execute_input":"2023-11-11T03:48:13.247382Z","iopub.status.idle":"2023-11-11T03:48:13.705458Z","shell.execute_reply.started":"2023-11-11T03:48:13.247345Z","shell.execute_reply":"2023-11-11T03:48:13.704514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the training data into a DataFrame. \n# Print the shape of the resulting DataFrame.\n\nhcd = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\nprint(hcd.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:13.706820Z","iopub.execute_input":"2023-11-11T03:48:13.707209Z","iopub.status.idle":"2023-11-11T03:48:14.071465Z","shell.execute_reply.started":"2023-11-11T03:48:13.707176Z","shell.execute_reply":"2023-11-11T03:48:14.070530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the head of the train DataFrame. \nhcd.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:14.073548Z","iopub.execute_input":"2023-11-11T03:48:14.073854Z","iopub.status.idle":"2023-11-11T03:48:14.089924Z","shell.execute_reply.started":"2023-11-11T03:48:14.073828Z","shell.execute_reply":"2023-11-11T03:48:14.088820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label distrobution\n(hcd.label.value_counts() / len(hcd)).to_frame()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:14.091236Z","iopub.execute_input":"2023-11-11T03:48:14.091601Z","iopub.status.idle":"2023-11-11T03:48:14.111306Z","shell.execute_reply.started":"2023-11-11T03:48:14.091569Z","shell.execute_reply":"2023-11-11T03:48:14.110423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Adding a variable for the image directory\nimg_dir = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:14.112481Z","iopub.execute_input":"2023-11-11T03:48:14.113177Z","iopub.status.idle":"2023-11-11T03:48:14.118810Z","shell.execute_reply.started":"2023-11-11T03:48:14.113127Z","shell.execute_reply":"2023-11-11T03:48:14.117903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = hcd.sample(n=9).reset_index()\n\nplt.figure(figsize=(3,3))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'{img_dir}/{row.id}.tif')    \n    label = row.label\n\n    plt.subplot(3,3,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:14.120105Z","iopub.execute_input":"2023-11-11T03:48:14.120844Z","iopub.status.idle":"2023-11-11T03:48:14.956641Z","shell.execute_reply.started":"2023-11-11T03:48:14.120817Z","shell.execute_reply":"2023-11-11T03:48:14.955205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using data generators \ntrain_df, valid_df = train_test_split(hcd, test_size=0.2, random_state=39, stratify=hcd.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:14.958240Z","iopub.execute_input":"2023-11-11T03:48:14.958705Z","iopub.status.idle":"2023-11-11T03:48:15.074290Z","shell.execute_reply.started":"2023-11-11T03:48:14.958664Z","shell.execute_reply":"2023-11-11T03:48:15.073331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scaling images \ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:15.075355Z","iopub.execute_input":"2023-11-11T03:48:15.075637Z","iopub.status.idle":"2023-11-11T03:48:15.080616Z","shell.execute_reply.started":"2023-11-11T03:48:15.075614Z","shell.execute_reply":"2023-11-11T03:48:15.079474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['id'] = train_df['id'] + '.tif'\nvalid_df['id'] = valid_df['id'] + '.tif'","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:15.083480Z","iopub.execute_input":"2023-11-11T03:48:15.083824Z","iopub.status.idle":"2023-11-11T03:48:15.146131Z","shell.execute_reply.started":"2023-11-11T03:48:15.083786Z","shell.execute_reply":"2023-11-11T03:48:15.145137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1/255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Only rescaling for the validation data\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n\n# Convert the 'label' column to string\ntrain_df['label'] = train_df['label'].astype(str)\nvalid_df['label'] = valid_df['label'].astype(str)\n\n# Now create the generators again\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)\n\nvalidation_generator = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T03:48:15.147238Z","iopub.execute_input":"2023-11-11T03:48:15.147542Z","iopub.status.idle":"2023-11-11T04:01:04.936942Z","shell.execute_reply.started":"2023-11-11T03:48:15.147517Z","shell.execute_reply":"2023-11-11T04:01:04.936128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\nmodel = Sequential()\n\n# Convolutional layer\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(96, 96, 3)))\nmodel.add(MaxPooling2D((2, 2)))\n\n# Second layer\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\n\n# Third layer\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\n\nmodel.add(Flatten())\n\nmodel.add(Dense(128, activation='relu'))\n\n# Dropout layer\nmodel.add(Dropout(0.5))\n\n# Output layer\nmodel.add(Dense(1, activation='sigmoid'))\n\n# Optimizer\noptimizer = Adam(learning_rate=0.001)\n\n# Compiling the model\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:01:04.938054Z","iopub.execute_input":"2023-11-11T04:01:04.938330Z","iopub.status.idle":"2023-11-11T04:01:07.649577Z","shell.execute_reply.started":"2023-11-11T04:01:04.938305Z","shell.execute_reply":"2023-11-11T04:01:07.648670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nhistory = model.fit(\n    train_generator, \n    validation_data=validation_generator, \n    epochs=1, \n    verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the accuracy\naccuracy = model.evaluate(validation_generator)\nprint(f\"Accuracy on the validation set: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-11T04:03:24.056192Z","iopub.execute_input":"2023-11-11T04:03:24.056926Z","iopub.status.idle":"2023-11-11T04:08:42.846750Z","shell.execute_reply.started":"2023-11-11T04:03:24.056889Z","shell.execute_reply":"2023-11-11T04:08:42.845775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save Model and History","metadata":{}},{"cell_type":"code","source":"import pickle\nmodel_filename = \"model.h5\"\nmodel.save(model_filename)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T01:16:00.613017Z","iopub.execute_input":"2023-11-13T01:16:00.613465Z","iopub.status.idle":"2023-11-13T01:16:01.049707Z","shell.execute_reply.started":"2023-11-13T01:16:00.613430Z","shell.execute_reply":"2023-11-13T01:16:01.047453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/history.pkl', 'wb') as f:\n    pickle.dump(history.history, f)","metadata":{},"execution_count":null,"outputs":[]}]}