{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-16T07:43:23.193113Z","iopub.execute_input":"2025-10-16T07:43:23.193540Z","iopub.status.idle":"2025-10-16T07:43:43.594324Z","shell.execute_reply.started":"2025-10-16T07:43:23.193516Z","shell.execute_reply":"2025-10-16T07:43:43.593138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\nDATASET_PATH = \"/kaggle/input/cassava-leaf-disease-classification\"\ntrain_df = pd.read_csv(os.path.join(DATASET_PATH, \"train.csv\"))\ntrain_df['label'] = train_df['label'].astype(str)\n\n# Chia train/val\ntrain_df, val_df = train_test_split(train_df, test_size=0.15, stratify=train_df['label'], random_state=42)\nprint(train_df.shape, val_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T07:44:49.963835Z","iopub.execute_input":"2025-10-16T07:44:49.965171Z","iopub.status.idle":"2025-10-16T07:44:50.202938Z","shell.execute_reply.started":"2025-10-16T07:44:49.965135Z","shell.execute_reply":"2025-10-16T07:44:50.201926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nIMAGE_SIZE = (128, 128)\nBATCH_SIZE = 32\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=25,\n    zoom_range=0.2,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    train_df,\n    directory=os.path.join(DATASET_PATH, \"train_images\"),\n    x_col=\"image_id\",\n    y_col=\"label\",\n    target_size=IMAGE_SIZE,\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE\n)\n\nval_gen = val_datagen.flow_from_dataframe(\n    val_df,\n    directory=os.path.join(DATASET_PATH, \"train_images\"),\n    x_col=\"image_id\",\n    y_col=\"label\",\n    target_size=IMAGE_SIZE,\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T07:44:55.254744Z","iopub.execute_input":"2025-10-16T07:44:55.255045Z","iopub.status.idle":"2025-10-16T07:45:12.202683Z","shell.execute_reply.started":"2025-10-16T07:44:55.255025Z","shell.execute_reply":"2025-10-16T07:45:12.201874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models\n\nmodel = models.Sequential([\n    layers.Conv2D(32, (3,3), activation='relu', input_shape=(128,128,3)),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(2,2),\n    \n    layers.Conv2D(64, (3,3), activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(2,2),\n\n    layers.Conv2D(128, (3,3), activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(2,2),\n    \n    layers.Conv2D(256, (3,3), activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(2,2),\n    \n    layers.Flatten(),\n    layers.Dense(256, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(5, activation='softmax')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T07:45:15.411576Z","iopub.execute_input":"2025-10-16T07:45:15.412639Z","iopub.status.idle":"2025-10-16T07:45:15.598263Z","shell.execute_reply.started":"2025-10-16T07:45:15.412576Z","shell.execute_reply":"2025-10-16T07:45:15.597189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\n\nmodel.compile(\n    optimizer=Adam(learning_rate=1e-2),\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:00:29.920835Z","iopub.execute_input":"2025-10-16T08:00:29.921443Z","iopub.status.idle":"2025-10-16T08:00:29.956504Z","shell.execute_reply.started":"2025-10-16T08:00:29.921411Z","shell.execute_reply":"2025-10-16T08:00:29.955808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping\n\ncallbacks = [\n    ReduceLROnPlateau(monitor='val_loss', factor=0.3, patience=3),\n    EarlyStopping(monitor='val_loss', patience=6, restore_best_weights=True)\n]\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=10,\n    callbacks=callbacks,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:00:41.329685Z","iopub.execute_input":"2025-10-16T08:00:41.330404Z","iopub.status.idle":"2025-10-16T09:43:22.090847Z","shell.execute_reply.started":"2025-10-16T08:00:41.330371Z","shell.execute_reply":"2025-10-16T09:43:22.089722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntest_dir = os.path.join(DATASET_PATH, \"test_images\")\ntest_images = sorted(os.listdir(test_dir))\n\ntest_gen = ImageDataGenerator(rescale=1./255).flow_from_dataframe(\n    pd.DataFrame({\"image_id\": test_images}),\n    directory=test_dir,\n    x_col=\"image_id\",\n    y_col=None,\n    target_size=IMAGE_SIZE,\n    class_mode=None,\n    shuffle=False,\n    batch_size=BATCH_SIZE\n)\n\npreds = model.predict(test_gen)\npred_labels = np.argmax(preds, axis=1)\n\nsubmission = pd.DataFrame({\n    \"image_id\": test_images,\n    \"label\": pred_labels\n})\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(\"submission.csv saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T09:43:27.717234Z","iopub.execute_input":"2025-10-16T09:43:27.717559Z","iopub.status.idle":"2025-10-16T09:43:28.052262Z","shell.execute_reply.started":"2025-10-16T09:43:27.717539Z","shell.execute_reply":"2025-10-16T09:43:28.051415Z"}},"outputs":[],"execution_count":null}]}