{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\nfolder = '/kaggle/input/happy-mammals-with-128x128-image-size/train_images_128'\nimages = []\n# for dirname, _, filenames in os.walk(folder):\n#     for filename in filenames:\n# #         print(os.path.join(dirname, filename))\n\n# print(\"load train data sucsess\")\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-23T06:46:09.797271Z","iopub.execute_input":"2022-05-23T06:46:09.797606Z","iopub.status.idle":"2022-05-23T06:46:10.021016Z","shell.execute_reply.started":"2022-05-23T06:46:09.797569Z","shell.execute_reply":"2022-05-23T06:46:10.020069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize data","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport os\nimport cv2\npath = '/kaggle/input/happy-whale-and-dolphin/'\ndir = os.listdir(path)\nprint(dir)\ntrain_data = pd.read_csv(path+'train.csv')\nprint('Number in train.csv:', len(train_data))","metadata":{"execution":{"iopub.status.busy":"2022-05-25T07:26:09.000599Z","iopub.execute_input":"2022-05-25T07:26:09.0011Z","iopub.status.idle":"2022-05-25T07:26:11.924289Z","shell.execute_reply.started":"2022-05-25T07:26:09.001048Z","shell.execute_reply":"2022-05-25T07:26:11.923278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print sample of train.csv\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:11.864275Z","iopub.execute_input":"2022-05-23T06:46:11.864492Z","iopub.status.idle":"2022-05-23T06:46:11.883726Z","shell.execute_reply.started":"2022-05-23T06:46:11.864466Z","shell.execute_reply":"2022-05-23T06:46:11.882772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.loc[train_data['image'] == '80b5373b87942b.jpg']","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:11.886529Z","iopub.execute_input":"2022-05-23T06:46:11.887159Z","iopub.status.idle":"2022-05-23T06:46:11.913278Z","shell.execute_reply.started":"2022-05-23T06:46:11.887112Z","shell.execute_reply":"2022-05-23T06:46:11.912426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number train images:', len(os.listdir(path+'train_images/')))\nprint('Number test images:', len(os.listdir(path+'test_images/')))","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:11.915116Z","iopub.execute_input":"2022-05-23T06:46:11.915714Z","iopub.status.idle":"2022-05-23T06:46:14.798862Z","shell.execute_reply.started":"2022-05-23T06:46:11.915668Z","shell.execute_reply":"2022-05-23T06:46:14.7977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_species=train_data['species'].value_counts()\nprint(count_species)\nprint(f\"\\nTotal number of species: {len(count_species)}\")","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:14.800266Z","iopub.execute_input":"2022-05-23T06:46:14.800581Z","iopub.status.idle":"2022-05-23T06:46:14.818683Z","shell.execute_reply.started":"2022-05-23T06:46:14.80054Z","shell.execute_reply":"2022-05-23T06:46:14.817836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 15))\nplt.rcParams[\"font.size\"] = 18\nplt.barh(train_data[\"species\"].value_counts().sort_values(ascending=True).index,\n         train_data[\"species\"].value_counts().sort_values(ascending=True),\n         tick_label = train_data[\"species\"].value_counts().sort_values(ascending=True).index)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:14.819958Z","iopub.execute_input":"2022-05-23T06:46:14.820587Z","iopub.status.idle":"2022-05-23T06:46:15.258713Z","shell.execute_reply.started":"2022-05-23T06:46:14.820546Z","shell.execute_reply":"2022-05-23T06:46:15.257642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['label'] = train_data.species.map(lambda x: 'whale' if 'whale' in x else 'dolphin')\ndata = train_data['label'].value_counts().reset_index()\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:15.26001Z","iopub.execute_input":"2022-05-23T06:46:15.26025Z","iopub.status.idle":"2022-05-23T06:46:15.292606Z","shell.execute_reply.started":"2022-05-23T06:46:15.260221Z","shell.execute_reply":"2022-05-23T06:46:15.291749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(data.iloc[0:, 0])\nplt.figure(figsize=(15, 10))\nplt.rcParams[\"font.size\"] = 12\nplt.title(\"Dolphin and whale\")\nplt.bar(data.iloc[0:, 0], data.iloc[0:, 1], color=['r','b'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:15.293603Z","iopub.execute_input":"2022-05-23T06:46:15.294415Z","iopub.status.idle":"2022-05-23T06:46:15.477331Z","shell.execute_reply.started":"2022-05-23T06:46:15.294362Z","shell.execute_reply":"2022-05-23T06:46:15.476162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_ = \"../input/happy-whale-and-dolphin/train_images/\"\nimage = plt.imread(PATH_+train_data.iloc[0, 0])\nplt.figure(figsize=(15, 6))\nplt.imshow(image)\n\nprint(train_data.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:15.48036Z","iopub.execute_input":"2022-05-23T06:46:15.480641Z","iopub.status.idle":"2022-05-23T06:46:15.889668Z","shell.execute_reply.started":"2022-05-23T06:46:15.480596Z","shell.execute_reply":"2022-05-23T06:46:15.886507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = plt.imread(PATH_+train_data.iloc[150, 0])\nplt.figure(figsize=(15, 6))\nplt.imshow(image)\n\nprint(train_data.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:15.891336Z","iopub.execute_input":"2022-05-23T06:46:15.892364Z","iopub.status.idle":"2022-05-23T06:46:16.492055Z","shell.execute_reply.started":"2022-05-23T06:46:15.892312Z","shell.execute_reply":"2022-05-23T06:46:16.491153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_ = \"../input/happy-whale-and-dolphin/train_images/\"\nimage = plt.imread(PATH_+train_data.iloc[25492, 0])\nplt.figure(figsize=(15, 6))\nplt.imshow(image)\n\nprint(train_data.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:16.493399Z","iopub.execute_input":"2022-05-23T06:46:16.49362Z","iopub.status.idle":"2022-05-23T06:46:18.0197Z","shell.execute_reply.started":"2022-05-23T06:46:16.493594Z","shell.execute_reply":"2022-05-23T06:46:18.018889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clean data and visualize label","metadata":{}},{"cell_type":"code","source":"# Clean data\ntrain_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.021182Z","iopub.execute_input":"2022-05-23T06:46:18.0217Z","iopub.status.idle":"2022-05-23T06:46:18.094517Z","shell.execute_reply.started":"2022-05-23T06:46:18.021644Z","shell.execute_reply":"2022-05-23T06:46:18.093654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.096021Z","iopub.execute_input":"2022-05-23T06:46:18.096278Z","iopub.status.idle":"2022-05-23T06:46:18.107037Z","shell.execute_reply.started":"2022-05-23T06:46:18.096241Z","shell.execute_reply":"2022-05-23T06:46:18.106052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.108351Z","iopub.execute_input":"2022-05-23T06:46:18.108648Z","iopub.status.idle":"2022-05-23T06:46:18.152034Z","shell.execute_reply.started":"2022-05-23T06:46:18.10861Z","shell.execute_reply":"2022-05-23T06:46:18.151303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.154354Z","iopub.execute_input":"2022-05-23T06:46:18.154649Z","iopub.status.idle":"2022-05-23T06:46:18.190628Z","shell.execute_reply.started":"2022-05-23T06:46:18.154607Z","shell.execute_reply":"2022-05-23T06:46:18.18986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['individual_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.193452Z","iopub.execute_input":"2022-05-23T06:46:18.194092Z","iopub.status.idle":"2022-05-23T06:46:18.219803Z","shell.execute_reply.started":"2022-05-23T06:46:18.194054Z","shell.execute_reply":"2022-05-23T06:46:18.21877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(20, 15))\n# plt.rcParams[\"font.size\"] = 18\n# plt.barh(train_df[\"individual_id\"].value_counts().sort_values(ascending=True).index,\n#          train_df[\"individual_id\"].value_counts().sort_values(ascending=True))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:18.221574Z","iopub.execute_input":"2022-05-23T06:46:18.222134Z","iopub.status.idle":"2022-05-23T06:46:18.227729Z","shell.execute_reply.started":"2022-05-23T06:46:18.222089Z","shell.execute_reply":"2022-05-23T06:46:18.226106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare data for model","metadata":{}},{"cell_type":"code","source":"#Function\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\ndef Loading_Images(data, m, dataset):\n    print(\"Loading images\")\n    X_train = np.zeros((m, 32, 32, 3))\n    count = 0\n    for fig in tqdm(data['image']):\n        img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+fig, target_size=(32, 32, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    return X_train","metadata":{"execution":{"iopub.status.busy":"2022-05-25T07:26:24.392596Z","iopub.execute_input":"2022-05-25T07:26:24.392899Z","iopub.status.idle":"2022-05-25T07:26:32.21704Z","shell.execute_reply.started":"2022-05-25T07:26:24.392872Z","shell.execute_reply":"2022-05-25T07:26:32.216059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Loading_Images(train_df, train_df.shape[0], \"train_images\")","metadata":{"execution":{"iopub.status.busy":"2022-05-23T06:46:25.226355Z","iopub.execute_input":"2022-05-23T06:46:25.22663Z","iopub.status.idle":"2022-05-23T08:10:18.33045Z","shell.execute_reply.started":"2022-05-23T06:46:25.226584Z","shell.execute_reply":"2022-05-23T08:10:18.32808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Nomarlize data\nX /= 255","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:18.334932Z","iopub.execute_input":"2022-05-23T08:10:18.335351Z","iopub.status.idle":"2022-05-23T08:10:20.470194Z","shell.execute_reply.started":"2022-05-23T08:10:18.335278Z","shell.execute_reply":"2022-05-23T08:10:20.469011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:20.473274Z","iopub.execute_input":"2022-05-23T08:10:20.474312Z","iopub.status.idle":"2022-05-23T08:10:20.484691Z","shell.execute_reply.started":"2022-05-23T08:10:20.47426Z","shell.execute_reply":"2022-05-23T08:10:20.48367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\ndef prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder","metadata":{"execution":{"iopub.status.busy":"2022-05-25T07:26:35.300508Z","iopub.execute_input":"2022-05-25T07:26:35.300908Z","iopub.status.idle":"2022-05-25T07:26:36.273093Z","shell.execute_reply.started":"2022-05-25T07:26:35.300865Z","shell.execute_reply":"2022-05-25T07:26:36.271707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, label_encoder = prepare_labels(train_df['individual_id'])","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:22.266038Z","iopub.execute_input":"2022-05-23T08:10:22.26677Z","iopub.status.idle":"2022-05-23T08:10:22.658229Z","shell.execute_reply.started":"2022-05-23T08:10:22.266733Z","shell.execute_reply":"2022-05-23T08:10:22.657282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:22.659634Z","iopub.execute_input":"2022-05-23T08:10:22.660841Z","iopub.status.idle":"2022-05-23T08:10:22.668213Z","shell.execute_reply.started":"2022-05-23T08:10:22.660788Z","shell.execute_reply":"2022-05-23T08:10:22.667204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Model","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom tensorflow.keras import layers\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\nfrom tensorflow.keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\nmodel = Sequential()\n\nmodel.add(Conv2D(32, (6, 6), strides = (1, 1), input_shape = (32, 32, 3)))\nmodel.add(BatchNormalization())\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1)))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3)))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation=\"relu\"))\n\nmodel.add(Dense(y.shape[1], activation='softmax'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:22.670012Z","iopub.execute_input":"2022-05-23T08:10:22.67071Z","iopub.status.idle":"2022-05-23T08:10:23.035228Z","shell.execute_reply.started":"2022-05-23T08:10:22.670662Z","shell.execute_reply":"2022-05-23T08:10:23.034509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X, y, epochs=150, batch_size=128, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:10:23.036456Z","iopub.execute_input":"2022-05-23T08:10:23.036794Z","iopub.status.idle":"2022-05-23T10:23:15.30604Z","shell.execute_reply.started":"2022-05-23T08:10:23.036763Z","shell.execute_reply":"2022-05-23T10:23:15.305255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:32:07.436562Z","iopub.execute_input":"2022-05-23T10:32:07.436884Z","iopub.status.idle":"2022-05-23T10:32:09.768021Z","shell.execute_reply.started":"2022-05-23T10:32:07.436846Z","shell.execute_reply":"2022-05-23T10:32:09.76647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แสดงค่าความแม่นยำในรูปแบบกราฟ\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:23:15.69674Z","iopub.execute_input":"2022-05-23T10:23:15.697151Z","iopub.status.idle":"2022-05-23T10:23:15.930122Z","shell.execute_reply.started":"2022-05-23T10:23:15.697095Z","shell.execute_reply":"2022-05-23T10:23:15.929014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แสดงค่า Model loss\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['loss'])\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:23:15.931776Z","iopub.execute_input":"2022-05-23T10:23:15.932162Z","iopub.status.idle":"2022-05-23T10:23:16.142327Z","shell.execute_reply.started":"2022-05-23T10:23:15.932112Z","shell.execute_reply":"2022-05-23T10:23:16.141378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# เตรียมใช้โมเดลกับ test_images","metadata":{}},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:23:16.143961Z","iopub.execute_input":"2022-05-23T10:23:16.144323Z","iopub.status.idle":"2022-05-23T10:23:16.824878Z","shell.execute_reply.started":"2022-05-23T10:23:16.144278Z","shell.execute_reply":"2022-05-23T10:23:16.823811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# เตรียม Variable เพื่อใช้โมเดล predict ใน format ที่การแข่งขันต้องการ\ncol = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:23:16.826607Z","iopub.execute_input":"2022-05-23T10:23:16.826893Z","iopub.status.idle":"2022-05-23T10:23:16.844044Z","shell.execute_reply.started":"2022-05-23T10:23:16.82686Z","shell.execute_reply":"2022-05-23T10:23:16.843035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=5000\nbatch_start = 0\nbatch_end = batch_size\nL = len(test_df)\n\nwhile batch_start < L:\n    limit = min(batch_end, L)\n    test_df_batch = test_df.iloc[batch_start:limit]\n    print(type(test_df_batch))\n    X = Loading_Images(test_df_batch, test_df_batch.shape[0], \"test_images\")\n    X /= 255\n    predictions = model.predict(np.array(X), verbose=1)\n    for i, pred in enumerate(predictions):\n        p=pred.argsort()[-5:][::-1]\n        idx=-1\n        s=''\n        s1=''\n        s2=''\n        for x in p:\n            idx=idx+1\n            if pred[x]>0.6:\n                s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n            else:\n                s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n        s= s1 + ' new_individual' + s2\n        s = s.strip(' ')\n        test_df.loc[ batch_start + i, 'predictions'] = s\n    batch_start += batch_size   \n    batch_end += batch_size","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:23:16.845414Z","iopub.execute_input":"2022-05-23T10:23:16.845694Z","iopub.status.idle":"2022-05-23T10:31:16.404038Z","shell.execute_reply.started":"2022-05-23T10:23:16.84566Z","shell.execute_reply":"2022-05-23T10:31:16.402571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# สร้างไฟล์ CSV เพื่อเตรียม submit\n\ntest_df.to_csv('submission.csv',index=False)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:31:16.405106Z","iopub.status.idle":"2022-05-23T10:31:16.405499Z","shell.execute_reply.started":"2022-05-23T10:31:16.405328Z","shell.execute_reply":"2022-05-23T10:31:16.405346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"evaluate_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\", nrows=10000)\neva = Loading_Images(evaluate_df, evaluate_df.shape[0], \"train_images\")","metadata":{"execution":{"iopub.status.busy":"2022-05-25T08:27:37.532494Z","iopub.execute_input":"2022-05-25T08:27:37.532913Z","iopub.status.idle":"2022-05-25T08:41:11.24283Z","shell.execute_reply.started":"2022-05-25T08:27:37.53287Z","shell.execute_reply":"2022-05-25T08:41:11.241083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\neva_y, label_encoder = prepare_labels(evaluate_df['individual_id'])\nprint(len(eva_y))","metadata":{"execution":{"iopub.status.busy":"2022-05-25T08:41:45.042695Z","iopub.execute_input":"2022-05-25T08:41:45.043401Z","iopub.status.idle":"2022-05-25T08:41:45.376644Z","shell.execute_reply.started":"2022-05-25T08:41:45.043367Z","shell.execute_reply":"2022-05-25T08:41:45.374947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nreconstructed_model = keras.models.load_model(\"../input/model-for-evaluation-happywhale/model.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-05-25T07:44:12.235679Z","iopub.execute_input":"2022-05-25T07:44:12.236842Z","iopub.status.idle":"2022-05-25T07:44:13.853326Z","shell.execute_reply.started":"2022-05-25T07:44:12.236766Z","shell.execute_reply":"2022-05-25T07:44:13.851587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eva/=255\nx = np.array(eva)\npredict = reconstructed_model.predict(x, verbose=1)\nans = label_encoder.inverse_transform([0])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T07:44:17.408611Z","iopub.execute_input":"2022-05-25T07:44:17.40906Z","iopub.status.idle":"2022-05-25T07:44:25.815948Z","shell.execute_reply.started":"2022-05-25T07:44:17.409022Z","shell.execute_reply":"2022-05-25T07:44:25.813976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p=predict[0].argsort()[-1:]      #sort index and pick the most values but as it index ofcouse\nprint(len(predict[0]))\nprint(len(eva_y[0]))\nans = label_encoder.inverse_transform(p)\nprint(\"Predict = \",ans)\n\nprint(p)\ny=eva_y[0].argsort()[-1:]\nprint(y)\ny_ans = label_encoder.inverse_transform(y)\nprint(\"actual = \",y_ans)\n\ncount=0;\nfor i, pred in enumerate(predict):\n    p=pred.argsort()[-1:] \n    y=eva_y[i].argsort()[-1:] \n    if(p[0] == y[0]):\n        count+=1\n        print(\"equal_index = \",i)\nprint(\"total = \",count)\nprint(\"total test = \", len(predict))\nprint(\"Acculancy percentage = \",count/len(predict))","metadata":{"execution":{"iopub.status.busy":"2022-05-25T09:19:39.005896Z","iopub.execute_input":"2022-05-25T09:19:39.006246Z","iopub.status.idle":"2022-05-25T09:19:54.716069Z","shell.execute_reply.started":"2022-05-25T09:19:39.006211Z","shell.execute_reply":"2022-05-25T09:19:54.714319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check if equal_index true\np=predict[1].argsort()[-1:]      #sort index and pick the most values but as it index ofcouse\nprint(len(predict[0]))\nprint(len(eva_y[0]))\nans = label_encoder.inverse_transform(p)\nprint(\"Predict = \",ans)\n\nprint(p)\ny=eva_y[1].argsort()[-1:]\nprint(y)\ny_ans = label_encoder.inverse_transform(y)\nprint(\"actual = \",y_ans)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T08:14:03.499556Z","iopub.execute_input":"2022-05-25T08:14:03.499916Z","iopub.status.idle":"2022-05-25T08:14:03.511308Z","shell.execute_reply.started":"2022-05-25T08:14:03.499884Z","shell.execute_reply":"2022-05-25T08:14:03.510504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#validate y label\ny_i=eva_y[0].argsort()[-1:]\nprint(\"sum of y at index 0 = \",eva_y[0].sum())\nprint(\"max value(as an index) at index 0 = \",y_i)\nprint(\"y 0 at max value(as an index) = \", eva_y[0][y_i])","metadata":{"execution":{"iopub.status.busy":"2022-05-25T09:11:59.404271Z","iopub.execute_input":"2022-05-25T09:11:59.404627Z","iopub.status.idle":"2022-05-25T09:11:59.413573Z","shell.execute_reply.started":"2022-05-25T09:11:59.404599Z","shell.execute_reply":"2022-05-25T09:11:59.412518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluation with 5 inputs\ncount_s=0;\nfor i, pred in enumerate(predict):\n    p=pred.argsort()[-5:][::-1] \n    y=eva_y[i].argsort()[-1:]\n    for j in range(5):\n        if(p[j] == y[0]):\n            count_s+=1\n            print(\"equal_index = \"+str(i)+\" and number of predict that correct = \"+str(j))\n            continue\nprint(\"total in 5 ans = \",count_s)\nprint(\"total test = \", len(predict))\nprint(\"Accuracy = \", count_s/len(predict))","metadata":{"execution":{"iopub.status.busy":"2022-05-25T09:28:14.134439Z","iopub.execute_input":"2022-05-25T09:28:14.134852Z","iopub.status.idle":"2022-05-25T09:28:29.805529Z","shell.execute_reply.started":"2022-05-25T09:28:14.134807Z","shell.execute_reply":"2022-05-25T09:28:29.803909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluate with MAP@5\ncount_m5=0;\nsumofPrecis=[];\nfor i, pred in enumerate(predict):\n    p=pred.argsort()[-5:][::-1] \n    y=eva_y[i].argsort()[-1:]\n    for j in range(5):\n        if(p[j] == y[0]):\n            count_m5+=1\n            sumofPrecis.append(pred[p[j]])\n            print(\"equal_index = \"+str(i)+\" and number of predict that correct = \"+str(j)+\" precission = \", pred[p[j]])\n            continue\nprint(\"total in 5 ans = \",count_m5)\nprint(\"total test = \", len(predict))\nprint(\"Accuracy = \", count_m5/len(predict))\nprint(\"Sum of best 5 correct precission(MAP@5) = \", sum(sumofPrecis))\nprint(\"Accuracy(MAP@5) = \", sum(sumofPrecis)/len(predict))","metadata":{"execution":{"iopub.status.busy":"2022-05-25T09:37:34.412335Z","iopub.execute_input":"2022-05-25T09:37:34.412768Z","iopub.status.idle":"2022-05-25T09:37:50.386703Z","shell.execute_reply.started":"2022-05-25T09:37:34.412719Z","shell.execute_reply":"2022-05-25T09:37:50.384665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TEST","metadata":{}},{"cell_type":"code","source":"# เตรียม Variable เพื่อใช้โมเดล predict ใน format ที่การแข่งขันต้องการ\n# col = ['image']\n# test_dff = pd.DataFrame(test, columns=col)\n# test_dff['predictions'] = ''\n# test_dff_batch = test_dff.iloc[0:1]\n# test_dff_batch\n\n# X = Loading_Images(test_dff_batch, test_dff_batch.shape[0], \"test_images\")\n# X /= 255\n\n# predictions = model.predict(np.array(X), verbose=1)\n# np.set_printoptions(suppress=True)\n# print('last value = ',predictions[0][len(predictions[0])-1])\n# print('max = ',np.max(predictions[0]))\n# print('min = ',np.min(predictions[0]))\n\n# for i, pred in enumerate(predictions):\n#         p=pred.argsort()[-5:][::-1] #argsort basicly return index of sorting increasing array\n#         print(p)\n#         print(pred.argsort()[-5:])\n#         print(pred.argsort())\n#         print('test min? = ', pred[9483])","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:31:16.406841Z","iopub.status.idle":"2022-05-23T10:31:16.407218Z","shell.execute_reply.started":"2022-05-23T10:31:16.407015Z","shell.execute_reply":"2022-05-23T10:31:16.407039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ss = np.array([1.48,1.41,0.0,0.1])\n# print(ss.argsort())","metadata":{"execution":{"iopub.status.busy":"2022-05-23T10:31:16.41222Z","iopub.status.idle":"2022-05-23T10:31:16.413012Z","shell.execute_reply.started":"2022-05-23T10:31:16.412593Z","shell.execute_reply":"2022-05-23T10:31:16.412665Z"},"trusted":true},"execution_count":null,"outputs":[]}]}