{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\nfrom tensorflow.keras.utils import to_categorical             ## encode numbers to one-hot encode form\nfrom sklearn.preprocessing import LabelEncoder                ## encode string labels to numbers \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-17T13:57:52.492057Z","iopub.execute_input":"2023-04-17T13:57:52.492833Z","iopub.status.idle":"2023-04-17T13:57:52.499542Z","shell.execute_reply.started":"2023-04-17T13:57:52.492792Z","shell.execute_reply":"2023-04-17T13:57:52.498266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/happy-whale-and-dolphin/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:57:54.072074Z","iopub.execute_input":"2023-04-17T13:57:54.072615Z","iopub.status.idle":"2023-04-17T13:57:54.188169Z","shell.execute_reply.started":"2023-04-17T13:57:54.072566Z","shell.execute_reply":"2023-04-17T13:57:54.186961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=train_df.drop_duplicates(subset=['individual_id'],keep='last')","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:57:55.448339Z","iopub.execute_input":"2023-04-17T13:57:55.448818Z","iopub.status.idle":"2023-04-17T13:57:55.486616Z","shell.execute_reply.started":"2023-04-17T13:57:55.448772Z","shell.execute_reply":"2023-04-17T13:57:55.485490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Loading_Images(data, m, dataset): #โหลดภาพมาใส่ใน X เพื่อนำไปเทรน\n    X_train = np.zeros((m, 32, 32, 3))\n    count = 0\n    for fig in tqdm(data['image']):\n        img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+fig, target_size=(32, 32, 3))\n        x = image.img_to_array(img) #แปลงรูปภาพเป็น numpy array\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    return X_train","metadata":{"execution":{"iopub.status.busy":"2023-04-17T12:47:11.026122Z","iopub.execute_input":"2023-04-17T12:47:11.026526Z","iopub.status.idle":"2023-04-17T12:47:11.032887Z","shell.execute_reply.started":"2023-04-17T12:47:11.026485Z","shell.execute_reply":"2023-04-17T12:47:11.031663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Loading_Images(train_df, train_df.shape[0], 'train_images')\nX /= 255","metadata":{"execution":{"iopub.status.busy":"2023-04-17T12:47:11.034657Z","iopub.execute_input":"2023-04-17T12:47:11.035028Z","iopub.status.idle":"2023-04-17T13:03:28.350965Z","shell.execute_reply.started":"2023-04-17T12:47:11.034992Z","shell.execute_reply":"2023-04-17T13:03:28.349952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder  ","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:03:28.355708Z","iopub.execute_input":"2023-04-17T13:03:28.358143Z","iopub.status.idle":"2023-04-17T13:03:28.367445Z","shell.execute_reply.started":"2023-04-17T13:03:28.358090Z","shell.execute_reply":"2023-04-17T13:03:28.366235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, label_encoder = labels(train_df['individual_id'])","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:03:28.373262Z","iopub.execute_input":"2023-04-17T13:03:28.376464Z","iopub.status.idle":"2023-04-17T13:03:28.544337Z","shell.execute_reply.started":"2023-04-17T13:03:28.376422Z","shell.execute_reply":"2023-04-17T13:03:28.543092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:57:59.244557Z","iopub.execute_input":"2023-04-17T13:57:59.245339Z","iopub.status.idle":"2023-04-17T13:58:00.295139Z","shell.execute_reply.started":"2023-04-17T13:57:59.245289Z","shell.execute_reply":"2023-04-17T13:58:00.292657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.optimizers import SGD","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:58:02.497700Z","iopub.execute_input":"2023-04-17T13:58:02.498108Z","iopub.status.idle":"2023-04-17T13:58:02.505122Z","shell.execute_reply.started":"2023-04-17T13:58:02.498072Z","shell.execute_reply":"2023-04-17T13:58:02.503875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet = ResNet50(include_top=False, weights='imagenet', input_shape=(32, 32, 3))\n# ตัดชั้น Fully-connected Layer ออก\nx = resnet.output\n\nx = Flatten()(x)\n\n# เพิ่มชั้น Dense สำหรับการจำแนกภาพ\npredictions = Dense(15587, activation='softmax')(x)\n\n# สร้างโมเดลใหม่จาก ResNet 50 ที่ถูกตัดชั้น Fully-connected Layer ออกแล้ว\nmodel = Model(inputs=resnet.input, outputs=predictions)\n# model.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:58:04.436048Z","iopub.execute_input":"2023-04-17T13:58:04.436736Z","iopub.status.idle":"2023-04-17T13:58:06.525709Z","shell.execute_reply.started":"2023-04-17T13:58:04.436697Z","shell.execute_reply":"2023-04-17T13:58:06.524605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import SGD\nSGD = SGD(learning_rate=0.04)\nbatch_size = 200\nepochs = 30\n\n# กำหนด optimizer และ loss function\nloss = 'categorical_crossentropy'\n\n# คอมไพล์โมเดล\nmodel.compile(optimizer=SGD, loss=loss, metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:58:08.113841Z","iopub.execute_input":"2023-04-17T13:58:08.114667Z","iopub.status.idle":"2023-04-17T13:58:08.132218Z","shell.execute_reply.started":"2023-04-17T13:58:08.114622Z","shell.execute_reply":"2023-04-17T13:58:08.131044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train,y_train, batch_size=batch_size, epochs=epochs)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:58:10.345560Z","iopub.execute_input":"2023-04-17T13:58:10.346136Z","iopub.status.idle":"2023-04-17T14:01:24.791026Z","shell.execute_reply.started":"2023-04-17T13:58:10.346098Z","shell.execute_reply":"2023-04-17T14:01:24.790008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.models import Model\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras.preprocessing import image\n# from tensorflow.keras.layers import AveragePooling2D, MaxPooling2D, Dropout\n# from tensorflow.keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\n# from keras.optimizers import SGD\n# model = Sequential()\n# model.add(Conv2D(16, (3,3), strides = (1, 1), input_shape = (32, 32, 3)))\n# model.add(Activation('relu'))\n# model.add(MaxPooling2D((2, 2)))\n# model.add(Conv2D(32, (2, 2), strides = (1,1)))\n# model.add(Activation('relu'))\n# model.add(MaxPooling2D())\n# model.add(Conv2D(64, (2, 2), strides = (1,1)))\n# model.add(Activation('relu'))\n# model.add(MaxPooling2D())\n\n# model.add(Flatten())\n# model.add(Dense(1600, activation=\"relu\"))\n# model.add(Dense(15587, activation='softmax'))\n# SGD = SGD(learning_rate=0.04)\n# model.compile(loss='categorical_crossentropy', optimizer=\"SGD\", metrics=['accuracy'])\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:07:03.363721Z","iopub.execute_input":"2023-04-17T13:07:03.364081Z","iopub.status.idle":"2023-04-17T13:07:03.369336Z","shell.execute_reply.started":"2023-04-17T13:07:03.364048Z","shell.execute_reply":"2023-04-17T13:07:03.368244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = model.fit(X_train,y_train, epochs=200, batch_size=200,validation_split = 0.2, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T13:07:03.370888Z","iopub.execute_input":"2023-04-17T13:07:03.371539Z","iopub.status.idle":"2023-04-17T13:07:03.386341Z","shell.execute_reply.started":"2023-04-17T13:07:03.371500Z","shell.execute_reply":"2023-04-17T13:07:03.385293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt \nplt.plot(history.history['loss'])                                               ## plot loss curve during training \nplt.legend(['train loss'])\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.show()\n\nplt.plot(history.history['accuracy'])                                           ## plot accuracy curve during training \nplt.legend(['train accuracy'])\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T14:47:11.846981Z","iopub.execute_input":"2023-04-17T14:47:11.847928Z","iopub.status.idle":"2023-04-17T14:47:12.288231Z","shell.execute_reply.started":"2023-04-17T14:47:11.847878Z","shell.execute_reply":"2023-04-17T14:47:12.287114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\ntest = os.listdir(\"/kaggle/input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2023-04-17T14:47:52.351067Z","iopub.execute_input":"2023-04-17T14:47:52.352020Z","iopub.status.idle":"2023-04-17T14:47:52.369457Z","shell.execute_reply.started":"2023-04-17T14:47:52.351984Z","shell.execute_reply":"2023-04-17T14:47:52.368129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(n=250)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T14:47:54.136348Z","iopub.execute_input":"2023-04-17T14:47:54.136782Z","iopub.status.idle":"2023-04-17T14:47:54.153138Z","shell.execute_reply.started":"2023-04-17T14:47:54.136743Z","shell.execute_reply":"2023-04-17T14:47:54.151495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=5000\nbatch_start = 0\nbatch_end = batch_size\nL = len(test_df)\n\nwhile batch_start < L:\n    limit = min(batch_end, L)\n    test_df_batch = test_df.iloc[batch_start:limit]\n    print(type(test_df_batch))\n    X = Loading_Images(test_df_batch, test_df_batch.shape[0], \"test_images\")\n    X /= 255\n    predictions = model.predict(np.array(X), verbose=1)\n    for i, pred in enumerate(predictions):\n        p=pred.argsort()[-5:][::-1]\n        idx=-1\n        s=''\n        s1=''\n        s2=''\n        for x in p:\n            idx=idx+1\n            if pred[x]>0.6:\n                s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n            else:\n                s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n        s= s1 + ' new_individual' + s2\n        s = s.strip(' ')\n        test_df.loc[ batch_start + i, 'predictions'] = s\n    batch_start += batch_size   \n    batch_end += batch_size","metadata":{"execution":{"iopub.status.busy":"2023-04-17T14:47:56.111712Z","iopub.execute_input":"2023-04-17T14:47:56.112315Z","iopub.status.idle":"2023-04-17T15:26:06.700362Z","shell.execute_reply.started":"2023-04-17T14:47:56.112275Z","shell.execute_reply":"2023-04-17T15:26:06.699293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv',index=False)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T15:38:23.475752Z","iopub.execute_input":"2023-04-17T15:38:23.476158Z","iopub.status.idle":"2023-04-17T15:38:23.546360Z","shell.execute_reply.started":"2023-04-17T15:38:23.476124Z","shell.execute_reply":"2023-04-17T15:38:23.545242Z"},"trusted":true},"execution_count":null,"outputs":[]}]}