{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mimage\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom tensorflow.keras.utils import img_to_array\nfrom IPython.display import FileLink\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPool2D, Flatten, Dense, BatchNormalization, Dropout\nfrom keras.utils import to_categorical\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:44.629776Z","iopub.execute_input":"2023-06-02T15:11:44.630318Z","iopub.status.idle":"2023-06-02T15:11:52.62082Z","shell.execute_reply.started":"2023-06-02T15:11:44.630289Z","shell.execute_reply":"2023-06-02T15:11:52.62003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = mimage.imread(\"/kaggle/input/noaa-right-whale-recognition/w_7489.jpg\")\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:33:29.240258Z","iopub.execute_input":"2023-06-02T14:33:29.241067Z","iopub.status.idle":"2023-06-02T14:33:31.642299Z","shell.execute_reply.started":"2023-06-02T14:33:29.241032Z","shell.execute_reply":"2023-06-02T14:33:31.641465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/noaa-right-whale-recognition/train.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/noaa-right-whale-recognition/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:52.622403Z","iopub.execute_input":"2023-06-02T15:11:52.623173Z","iopub.status.idle":"2023-06-02T15:11:52.948439Z","shell.execute_reply.started":"2023-06-02T15:11:52.623143Z","shell.execute_reply":"2023-06-02T15:11:52.947503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:52.949733Z","iopub.execute_input":"2023-06-02T15:11:52.950134Z","iopub.status.idle":"2023-06-02T15:11:52.990263Z","shell.execute_reply.started":"2023-06-02T15:11:52.9501Z","shell.execute_reply":"2023-06-02T15:11:52.989289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission[\"whale_00195\"] = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:52.992526Z","iopub.execute_input":"2023-06-02T15:11:52.992814Z","iopub.status.idle":"2023-06-02T15:11:53.009415Z","shell.execute_reply.started":"2023-06-02T15:11:52.99279Z","shell.execute_reply":"2023-06-02T15:11:53.008517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = sample_submission.columns[1:]\nclass_names_labels = {i:class_name for i, class_name in enumerate(class_names)}\n\nNUM_CLASSES = len(class_names)\nIMAGE_SIZE = (256, 256)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:53.010638Z","iopub.execute_input":"2023-06-02T15:11:53.010917Z","iopub.status.idle":"2023-06-02T15:11:53.015369Z","shell.execute_reply.started":"2023-06-02T15:11:53.010893Z","shell.execute_reply":"2023-06-02T15:11:53.014508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[~train_df[\"Image\"].isin(['w_7489.jpg'])]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(mode):\n    if mode == \"train\":\n        images = []\n        for img_name in tqdm(train_df[\"Image\"].values):\n            with Image.open(f\"/kaggle/input/cropped-imgs/{img_name}\") as image:\n                images.append(img_to_array(image.resize(IMAGE_SIZE, resample=Image.NEAREST)))\n        return np.array(images, dtype=\"float32\")\n    if mode == \"test\":\n        images = []\n        for img_name in tqdm(sample_submission[\"Image\"].values):\n            with Image.open(f\"/kaggle/input/cropped-imgs/{img_name}\") as image:\n                images.append(img_to_array(image.resize(IMAGE_SIZE, resample=Image.NEAREST)))\n        return np.array(images, dtype=\"float32\")           ","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:53.016702Z","iopub.execute_input":"2023-06-02T15:11:53.016962Z","iopub.status.idle":"2023-06-02T15:11:53.02691Z","shell.execute_reply.started":"2023-06-02T15:11:53.016932Z","shell.execute_reply":"2023-06-02T15:11:53.02608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = load_data(mode=\"train\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = ImageDataGenerator(rotation_range=8, width_shift_range=0.1, \n                                     height_shift_range=0.1,\n                                     zoom_range=0.1,rescale=1/255.0,validation_split=0.1)\ntest_generator = ImageDataGenerator(rescale=1/255.0)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:11:53.028178Z","iopub.execute_input":"2023-06-02T15:11:53.028459Z","iopub.status.idle":"2023-06-02T15:11:53.040599Z","shell.execute_reply.started":"2023-06-02T15:11:53.028436Z","shell.execute_reply":"2023-06-02T15:11:53.039699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator.fit(X_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = [class_names_labels[label] for label in train_df[\"whaleID\"]]\ny_train = to_categorical(y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    Conv2D(128, kernel_size=5, input_shape = (IMAGE_SIZE[0], IMAGE_SIZE[0], 3),activation=\"relu\"),\n    MaxPool2D(2),\n    BatchNormalization(),\n    Conv2D(64, kernel_size=3, activation=\"relu\"),\n    MaxPool2D(2),\n    BatchNormalization(),\n    Conv2D(32, kernel_size=3, activation=\"relu\"),\n    MaxPool2D(2),\n    BatchNormalization(),\n    Conv2D(16, kernel_size=3, activation=\"relu\"),\n    MaxPool2D(2),\n    BatchNormalization(),\n    Flatten(),\n    Dense(4096, activation=\"relu\"),\n    Dropout(0.25),\n    Dense(NUM_CLASSES, activation=\"softmax\")\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=\"adam\", loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MCP = ModelCheckpoint('model.h5',verbose=1,save_best_only=True,monitor='val_accuracy',mode='max')\nES = EarlyStopping(monitor='val_accuracy',min_delta=0,verbose=0,restore_best_weights = True,patience=4,mode='max')\nRLP = ReduceLROnPlateau(monitor='val_loss',patience=3,factor=0.2,min_lr=0.0001)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=1000\nmodel.fit(train_generator.flow(X_train, y_train, batch_size=32, subset=\"training\"), validation_data=train_generator.flow(X_train, y_train, batch_size=16, subset=\"validation\"), epochs=epochs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"/kaggle/working/model.h5\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink('model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = load_data(mode=\"test\")","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:12:06.732366Z","iopub.execute_input":"2023-06-02T15:12:06.732755Z","iopub.status.idle":"2023-06-02T15:12:50.429786Z","shell.execute_reply.started":"2023-06-02T15:12:06.732725Z","shell.execute_reply":"2023-06-02T15:12:50.427505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gnr = test_generator.flow(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:12:50.431076Z","iopub.status.idle":"2023-06-02T15:12:50.431478Z","shell.execute_reply.started":"2023-06-02T15:12:50.431299Z","shell.execute_reply":"2023-06-02T15:12:50.431317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"saved_model = load_model(\"/kaggle/input/model/model.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:12:50.432678Z","iopub.status.idle":"2023-06-02T15:12:50.433034Z","shell.execute_reply.started":"2023-06-02T15:12:50.43286Z","shell.execute_reply":"2023-06-02T15:12:50.432876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = np.argmax(saved_model.predict(test_gnr), axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:12:50.434284Z","iopub.status.idle":"2023-06-02T15:12:50.435137Z","shell.execute_reply.started":"2023-06-02T15:12:50.434932Z","shell.execute_reply":"2023-06-02T15:12:50.434961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(sample_submission.shape[0]):\n    sample_submission.at[i,class_names_labels[predictions[i]]] = 1","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:50:17.923483Z","iopub.execute_input":"2023-06-02T14:50:17.924437Z","iopub.status.idle":"2023-06-02T14:50:18.114302Z","shell.execute_reply.started":"2023-06-02T14:50:17.9244Z","shell.execute_reply":"2023-06-02T14:50:18.113344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T14:52:59.283926Z","iopub.execute_input":"2023-06-02T14:52:59.284358Z","iopub.status.idle":"2023-06-02T14:52:59.908616Z","shell.execute_reply.started":"2023-06-02T14:52:59.284327Z","shell.execute_reply":"2023-06-02T14:52:59.907625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}