{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25563,"databundleVersionId":2094376,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-03T16:53:53.370225Z","iopub.execute_input":"2024-11-03T16:53:53.370806Z","iopub.status.idle":"2024-11-03T16:53:53.377241Z","shell.execute_reply.started":"2024-11-03T16:53:53.370761Z","shell.execute_reply":"2024-11-03T16:53:53.375968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#IMPORT REQUIRED LIBRARIES:\n\nfrom tensorflow.keras.layers import Input, Lambda, Dense, Flatten\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img\nfrom tensorflow.keras.models import Sequential\nimport tensorflow as tf\nimport numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\ngpus = tf.config.list_physical_devices('GPU')\nprint(\"GPUs:\", gpus)\n\n# Check if GPUs are available\nif gpus:\n    print(\"GPU is available.\")\nelse:\n    print(\"GPU is not available.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:55:15.675471Z","iopub.execute_input":"2024-11-03T16:55:15.675769Z","iopub.status.idle":"2024-11-03T16:55:15.900281Z","shell.execute_reply.started":"2024-11-03T16:55:15.675736Z","shell.execute_reply":"2024-11-03T16:55:15.899346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_path = \"/kaggle/input/plant-pathology-2021-fgvc8/train_images\"\ntrain_images_labels_path = \"/kaggle/input/plant-pathology-2021-fgvc8/train.csv\"\ndf = pd.read_csv(train_images_labels_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:56:49.272656Z","iopub.execute_input":"2024-11-03T16:56:49.273053Z","iopub.status.idle":"2024-11-03T16:56:49.294415Z","shell.execute_reply.started":"2024-11-03T16:56:49.273017Z","shell.execute_reply":"2024-11-03T16:56:49.293651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_df, test_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df['labels'])\n\n# Further split train_val_df into training and validation sets\n# train_df, val_df = train_test_split(train_val_df, test_size=0.115, random_state=42, stratify=train_val_df['labels'])\n# Summary of the split\nprint(f\"Default set: {len(df)} samples\")\nprint(f\"Training set: {len(train_df)} samples\")\n# print(f\"Validation set: {len(val_df)} samples\")\nprint(f\"Test set: {len(test_df)} samples\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:09.817826Z","iopub.execute_input":"2024-11-03T18:46:09.818206Z","iopub.status.idle":"2024-11-03T18:46:09.851051Z","shell.execute_reply.started":"2024-11-03T18:46:09.818173Z","shell.execute_reply":"2024-11-03T18:46:09.850229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_df.labels.value_counts().index\nlabels_count = list(train_df.labels.value_counts().values)\n\nprint(f\"Number of labels: {len(labels)}\")\nprint(labels_count)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:12.809748Z","iopub.execute_input":"2024-11-03T18:46:12.810673Z","iopub.status.idle":"2024-11-03T18:46:12.820643Z","shell.execute_reply.started":"2024-11-03T18:46:12.810625Z","shell.execute_reply":"2024-11-03T18:46:12.819593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:15.630966Z","iopub.execute_input":"2024-11-03T18:46:15.631855Z","iopub.status.idle":"2024-11-03T18:46:15.641509Z","shell.execute_reply.started":"2024-11-03T18:46:15.631815Z","shell.execute_reply":"2024-11-03T18:46:15.640517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35,15))\nplt.bar(labels, labels_count)\nplt.title(\"Number of instances per class\",fontweight=\"bold\",fontsize=40)\nplt.xlabel(\"Classes\",fontsize = 30)\nplt.xticks(rotation=20,fontsize = 20,fontweight = \"bold\")\nplt.yticks(fontsize = 20,fontweight = \"bold\")\nplt.ylabel(\"Count\",fontsize=30)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:17.329586Z","iopub.execute_input":"2024-11-03T18:46:17.330536Z","iopub.status.idle":"2024-11-03T18:46:18.015800Z","shell.execute_reply.started":"2024-11-03T18:46:17.330495Z","shell.execute_reply":"2024-11-03T18:46:18.014858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augmentacja\nim_size = 224\nSEED = 42\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nBATCH_SIZE = 32\ntrain_datagen = ImageDataGenerator(rescale = 1/255.,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    horizontal_flip=True,\n    validation_split = 0.2,\n    zoom_range = 0.2,\n    shear_range = 0.2,\n    vertical_flip = False)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    directory = train_images_path,\n    x_col = \"image\",\n    y_col = \"labels\",\n    target_size = (im_size,im_size),\n    class_mode='categorical',\n    batch_size = BATCH_SIZE,\n    subset = \"training\",\n    shuffle = True,\n    seed = SEED,\n    validate_filenames = False\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:19.645913Z","iopub.execute_input":"2024-11-03T18:46:19.646302Z","iopub.status.idle":"2024-11-03T18:46:19.708735Z","shell.execute_reply.started":"2024-11-03T18:46:19.646265Z","shell.execute_reply":"2024-11-03T18:46:19.707978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    directory = train_images_path,\n    x_col = \"image\",\n    y_col = \"labels\",\n    target_size = (im_size,im_size),\n    class_mode='categorical',\n    batch_size = BATCH_SIZE,\n    subset = \"validation\",\n    shuffle = True,\n    seed = SEED,\n    validate_filenames = False\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:46:22.558235Z","iopub.execute_input":"2024-11-03T18:46:22.559126Z","iopub.status.idle":"2024-11-03T18:46:22.597456Z","shell.execute_reply.started":"2024-11-03T18:46:22.559078Z","shell.execute_reply":"2024-11-03T18:46:22.596655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input,\n    rescale=1/255.0\n)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=train_images_path,\n    x_col=\"image\",\n    y_col=None,\n    batch_size=BATCH_SIZE,\n    seed=42,\n    shuffle=False,\n    class_mode=None,\n    target_size=(im_size, im_size)\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T19:56:09.417125Z","iopub.execute_input":"2024-11-03T19:56:09.417815Z","iopub.status.idle":"2024-11-03T19:56:16.158098Z","shell.execute_reply.started":"2024-11-03T19:56:09.417773Z","shell.execute_reply":"2024-11-03T19:56:16.157118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D, Dense, Dropout, BatchNormalization\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', padding='same', input_shape=(224, 224, 3)),\n    BatchNormalization(),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    Conv2D(64, (3, 3), activation='relu', padding='same'),\n    BatchNormalization(),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    Conv2D(128, (3, 3), activation='relu', padding='same'),\n    BatchNormalization(),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    Conv2D(256, (3, 3), activation='relu', padding='same'),\n    BatchNormalization(),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    # Use GlobalAveragePooling2D instead of Flatten to reduce parameters\n    GlobalAveragePooling2D(),\n    \n    Dense(256, activation='relu'),\n    Dropout(0.5),\n    \n    Dense(12, activation='softmax')  # Adjust the output for your number of classes\n])\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:00:40.771868Z","iopub.execute_input":"2024-11-03T17:00:40.772775Z","iopub.status.idle":"2024-11-03T17:00:40.899573Z","shell.execute_reply.started":"2024-11-03T17:00:40.772735Z","shell.execute_reply":"2024-11-03T17:00:40.898662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate train and validation steps\ntrain_steps = (len(train_df) + BATCH_SIZE - 1) // BATCH_SIZE  # Round up division\nval_steps = (len(val_df) + BATCH_SIZE - 1) // BATCH_SIZE  # Round up division    ","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:48:25.555519Z","iopub.execute_input":"2024-11-03T18:48:25.556222Z","iopub.status.idle":"2024-11-03T18:48:25.560525Z","shell.execute_reply.started":"2024-11-03T18:48:25.556183Z","shell.execute_reply":"2024-11-03T18:48:25.559773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n        train_generator,\n        epochs=5,\n        validation_data=val_generator,\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T20:07:09.599361Z","iopub.execute_input":"2024-11-03T20:07:09.599803Z","iopub.status.idle":"2024-11-03T21:20:19.323343Z","shell.execute_reply.started":"2024-11-03T20:07:09.599763Z","shell.execute_reply":"2024-11-03T21:20:19.322289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\n# Generate predictions on the test set\npredictions = model.predict(test_generator)\npredicted_classes = predictions.argmax(axis=-1)  # Get the class with the highest probability\n\n# Map indices back to class labels\nclass_indices = train_generator.class_indices  # Assuming same class indices as train generator\nlabels = {v: k for k, v in class_indices.items()}\npredicted_labels = [labels[idx] for idx in predicted_classes]\n\n# Extract the true labels from test_df\ntrue_labels = test_df[\"labels\"].tolist()\n\n# Calculate accuracy\naccuracy = accuracy_score(true_labels, predicted_labels)\n\nprint(f\"Test Accuracy: {accuracy * 100:.2f}%\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T21:54:35.199804Z","iopub.execute_input":"2024-11-03T21:54:35.200791Z","iopub.status.idle":"2024-11-03T21:57:41.073007Z","shell.execute_reply.started":"2024-11-03T21:54:35.200749Z","shell.execute_reply":"2024-11-03T21:57:41.072023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"first_cnn.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T19:05:23.100980Z","iopub.execute_input":"2024-11-03T19:05:23.101801Z","iopub.status.idle":"2024-11-03T19:05:23.175454Z","shell.execute_reply.started":"2024-11-03T19:05:23.101759Z","shell.execute_reply":"2024-11-03T19:05:23.174670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check a single batch of training data\nx_train, y_train = next(train_generator)\nprint(x_train.shape, y_train.shape)\n\n# Check a single batch of validation data\nx_val, y_val = next(val_generator)\nprint(x_val.shape, y_val.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:40:03.976481Z","iopub.execute_input":"2024-11-03T17:40:03.977296Z","iopub.status.idle":"2024-11-03T17:40:07.608047Z","shell.execute_reply.started":"2024-11-03T17:40:03.977252Z","shell.execute_reply":"2024-11-03T17:40:07.606958Z"},"trusted":true},"execution_count":null,"outputs":[]}]}