{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport gc\nimport sys\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nfrom tqdm.autonotebook import tqdm\nimport cv2\nfrom cv2 import getRotationMatrix2D, warpAffine, BORDER_REPLICATE\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport keras.backend as K\nfrom keras.models import Sequential\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom keras.models import Model\n\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T12:59:40.590344Z","iopub.execute_input":"2022-06-05T12:59:40.593478Z","iopub.status.idle":"2022-06-05T12:59:40.631421Z","shell.execute_reply.started":"2022-06-05T12:59:40.59337Z","shell.execute_reply":"2022-06-05T12:59:40.630282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Loading_Images(data, m, dataset):\n    print(\"Loading images\")\n    X_train = np.zeros((m, 32, 32, 3))\n    count = 0\n    for fig in tqdm(data['image']):\n        img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+fig, target_size=(32, 32, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    return X_train\n\ndef prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder","metadata":{"execution":{"iopub.status.busy":"2022-06-05T12:59:40.656171Z","iopub.execute_input":"2022-06-05T12:59:40.656691Z","iopub.status.idle":"2022-06-05T12:59:40.662272Z","shell.execute_reply.started":"2022-06-05T12:59:40.656661Z","shell.execute_reply":"2022-06-05T12:59:40.661214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#============匯入資料============#\n#------匯入訓練檔案------#\ntrain_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\") #這是圖片名稱、鯨豚種類、個體編號的檔案\ntrain_img = Loading_Images(train_df, train_df.shape[0], \"train_images\")\ntrain_img /= 255\n\n#------把兩個合成DataFrame：X_train，這會跑三分鐘左右------#\nX_train = pd.DataFrame(columns=['image', 'species', 'individual_id', 'data'])\nfor i in tqdm(range(len(train_df))):\n    X_train = X_train.append(train_df.iloc[i], ignore_index=True)\n    X_train.at[len(X_train)-1, 'data'] = train_img[i]\n\nprint(X_train.head())\n\n#------清理不會用到的資料------#\ndel train_df, train_img\n","metadata":{"execution":{"iopub.status.busy":"2022-06-05T12:59:40.725187Z","iopub.execute_input":"2022-06-05T12:59:40.725683Z","iopub.status.idle":"2022-06-05T13:00:50.466345Z","shell.execute_reply.started":"2022-06-05T12:59:40.725653Z","shell.execute_reply":"2022-06-05T13:00:50.464133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#============image augmentation============#\n#------計算每個鯨豚種類的個數------#\nspecies_value_counts = X_train['species'].value_counts()\n\n#------如果該鯨豚種類的數量少於數量最多的鯨豚種類的1/5（下面簡稱「稀有種」），就把他的鯨豚種類名稱加到 minor_labels------#\nminor_labels = np.array([])\nfor i in range(len(species_value_counts)):\n    if(species_value_counts[i] < (species_value_counts[0] / 5)):\n        minor_labels = np.append(minor_labels, species_value_counts.index[i])\n        \nprint(minor_labels)\n\n#------清理不會用到的資料------#\ndel species_value_counts\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.467607Z","iopub.status.idle":"2022-06-05T13:00:50.468079Z","shell.execute_reply.started":"2022-06-05T13:00:50.46783Z","shell.execute_reply":"2022-06-05T13:00:50.467857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#------把所有屬於稀有種的鯨豚資料複製出來------#\nminor_species = pd.DataFrame(columns=['image', 'species', 'individual_id', 'data'])\nfor i in tqdm(range(len(X_train))):\n    for ml in minor_labels:\n        if(X_train.at[i, 'species'] == ml):\n            minor_species = minor_species.append(X_train.iloc[i], ignore_index=True)\n            break\n\nprint(minor_species.head())\n\n#------清理不會用到的資料------#\ndel minor_labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.469976Z","iopub.status.idle":"2022-06-05T13:00:50.470361Z","shell.execute_reply.started":"2022-06-05T13:00:50.470186Z","shell.execute_reply":"2022-06-05T13:00:50.470204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#------圖片旋轉的前置設定，要餵到圖片旋轉模板 getRotationMatrix2D 裡------#\nh = w = 32 #------------#這是圖片長和寬\ncenter = (w / 2, h / 2) #這是圖片中心點座標\nangle1 = 30 #-----------#這是旋轉角度一\nangle2 = -30 #----------#這是旋轉角度二（旋轉的照片我設定了兩種，一種是順時針30度，一種是逆時針30度）\nscale = 1 #-------------#這個我不知道是什麼，反正就是這樣設\n\n#------這是圖片旋轉的模板，這會餵到旋轉的 function 裡，我不知道他為什麼要這麼麻煩------#\nM1 = getRotationMatrix2D(center, angle1, scale) #這是順轉30度的模板\nM2 = getRotationMatrix2D(center, angle2, scale) #這是逆轉30度的模板\n\n#------把每一筆複製出來的稀有種資料全部做 augmentation，水平反轉*1、旋轉*2，一共三種（我把原資料覆蓋掉，不然拼回X_train原始資料會重複）------#\naug_minor_species = pd.DataFrame(columns=['image', 'species', 'individual_id', 'data'])\n\nfor i in tqdm(range(len(minor_species))):\n    aug_minor_species = aug_minor_species.append(minor_species.iloc[i], ignore_index=True)\n    id = len(aug_minor_species)-1 #這是第一份資料的 index，因為我會複製成三筆資料，如果不記會不方便\n    for t in range(2):\n        aug_minor_species = aug_minor_species.append(aug_minor_species.iloc[-1], ignore_index=True)\n    \n    #這裡是 augmentation 的部份，順序是翻轉、順轉、逆轉，旋轉的 borderMode 我設定圖片旋轉後的空白處會用原本圖片邊邊的顏色補滿\n    aug_minor_species.at[id, 'data']  = cv2.flip(aug_minor_species.at[id, 'data'], 1)\n    aug_minor_species.at[id+1, 'data']  = cv2.warpAffine(aug_minor_species.at[id, 'data'], M1, (w, h), borderMode=BORDER_REPLICATE)\n    aug_minor_species.at[id+2, 'data']  = cv2.warpAffine(aug_minor_species.at[id, 'data'], M2, (w, h), borderMode=BORDER_REPLICATE)\n    \nprint(aug_minor_species[35:41])\n\n#------清理不會用到的資料------#\ndel h, w, center, angle1, angle2, scale, M1, M2\ndel minor_species\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.471466Z","iopub.status.idle":"2022-06-05T13:00:50.472233Z","shell.execute_reply.started":"2022-06-05T13:00:50.472025Z","shell.execute_reply":"2022-06-05T13:00:50.472048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#------把 X_train 和 augmenttion 拼起來------#\nminor_aug_X_train = pd.concat([X_train, aug_minor_species], ignore_index=True)\n\n#------洗牌------#\nminor_aug_X_train = minor_aug_X_train.sample(frac=1).reset_index(drop=True)\n\n#------這裡是觀察 augmentation 前後的所有鯨豚數量------#\n# print(minor_aug_X_train.value_counts(subset=['species']))\n# print(X_train.value_counts(subset=['species']))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.473522Z","iopub.status.idle":"2022-06-05T13:00:50.473911Z","shell.execute_reply.started":"2022-06-05T13:00:50.473734Z","shell.execute_reply":"2022-06-05T13:00:50.473753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#------把拼起來後的資料的訓練資料（圖片 array 部份）和 label（個體編號） 拆開來------#\nX_train = np.stack(minor_aug_X_train['data'].values) #----------------------#我讓訓練資料直接蓋過原本的 X_train\ny_train, label_encoder = prepare_labels(minor_aug_X_train['individual_id']) #y_train 是 one-hot encoder，我不確定 label encoder是什麼\n\n#------清理不會用到的資料------#\ndel minor_aug_X_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.474872Z","iopub.status.idle":"2022-06-05T13:00:50.475266Z","shell.execute_reply.started":"2022-06-05T13:00:50.4751Z","shell.execute_reply":"2022-06-05T13:00:50.475119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#============訓練模型============# 這裡開始都是我直接貼過來的\nmodel = Sequential()\n\nmodel.add(Conv2D(32, (6, 6), strides = (1, 1), input_shape = (32, 32, 3)))\nmodel.add(BatchNormalization(axis = 3))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1)))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3)))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation=\"relu\"))\nmodel.add(Dropout(0.85))\n\nmodel.add(Dense(y_train.shape[1], activation='softmax'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['Accuracy', 'Precision', 'Recall'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.476491Z","iopub.status.idle":"2022-06-05T13:00:50.476831Z","shell.execute_reply.started":"2022-06-05T13:00:50.476664Z","shell.execute_reply":"2022-06-05T13:00:50.47668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=200, batch_size=200, validation_split=0.2, verbose=1)\nmodel.save('./last.h5')","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.478227Z","iopub.status.idle":"2022-06-05T13:00:50.47888Z","shell.execute_reply.started":"2022-06-05T13:00:50.478679Z","shell.execute_reply":"2022-06-05T13:00:50.478701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef show_train_history(train_history):\n  plt.plot(train_history.history['Accuracy'])\n  plt.plot(train_history.history['val_Accuracy'])\n  plt.xticks([i for i in range(0, len(train_history.history['Accuracy']))])\n  plt.title('Train History')\n  plt.ylabel('Accuracy')\n  plt.xlabel('epoch')\n  plt.legend(['train', 'validation'], loc = 'upper left')\n  plt.show()\n\nshow_train_history(history)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.480305Z","iopub.status.idle":"2022-06-05T13:00:50.480669Z","shell.execute_reply.started":"2022-06-05T13:00:50.480508Z","shell.execute_reply":"2022-06-05T13:00:50.480526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.482004Z","iopub.status.idle":"2022-06-05T13:00:50.482325Z","shell.execute_reply.started":"2022-06-05T13:00:50.482177Z","shell.execute_reply":"2022-06-05T13:00:50.482193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#這個好像是用來設定 submission.csv 的檔案排版，因為要跟 sample_sbmission 一模一樣\ncol = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(n=250)\n\ndel col\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.483089Z","iopub.status.idle":"2022-06-05T13:00:50.483439Z","shell.execute_reply.started":"2022-06-05T13:00:50.483272Z","shell.execute_reply":"2022-06-05T13:00:50.483289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=5000\nbatch_start = 0\nbatch_end = batch_size\nL = len(test_df)\n\nwhile batch_start < L:\n    limit = min(batch_end, L)\n    test_df_batch = test_df.iloc[batch_start:limit]\n    print(type(test_df_batch))\n    X = Loading_Images(test_df_batch, test_df_batch.shape[0], \"test_images\")\n    X /= 255\n    predictions = model.predict(np.array(X), verbose=1)\n    for i, pred in enumerate(predictions):\n        p=pred.argsort()[-5:][::-1]\n        idx=-1\n        s=''\n        s1=''\n        s2=''\n        for x in p:\n            idx=idx+1\n            if pred[x]>0.7:\n                s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n            else:\n                s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n        s= s1 + ' new_individual' + s2\n        s = s.strip(' ')\n        test_df.loc[ batch_start + i, 'predictions'] = s\n    batch_start += batch_size   \n    batch_end += batch_size\n    del X\n    del test_df_batch\n    del predictions\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T13:00:50.484834Z","iopub.status.idle":"2022-06-05T13:00:50.485211Z","shell.execute_reply.started":"2022-06-05T13:00:50.485047Z","shell.execute_reply":"2022-06-05T13:00:50.485064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv',index=False)\ntest_df.head()","metadata":{},"execution_count":null,"outputs":[]}]}