{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Yelp Restuarant Photo Classification**","metadata":{}},{"cell_type":"markdown","source":"## **Introduction**\n**Yelp Restuarant Classification** \n\n* เป็นการระบุคุณลักษณะของธุรกิจ จากรูปภาพที่ผู้ใช้ได้อัปโหลดลงมา โดยมีการจำแนกภาพที่ผู้ใช้ถ่ายว่าเป็นคุณลักษณะใดบ้าง\n\n**จุดมุ่งหมาย**\n* เพื่อเรียนรู้ในการทำ Image Classification ของชุดข้อมูล Yelp Restaurant Photo ClassificationYelp Restuarant Classification มีเป้าหมายในการระบุคุณลักษณะของธุรกิจจากรูปภาพจำนวนมากที่ผู้ใช้ได้อัปโหลดลงมาใน Yelp โดยการจำแนกภาพที่ผู้ใช้ถ่ายว่าเป็นคุณลักษณะไหนได้บ้าง (ในหนึ่งภาพอาจเป็นได้หลายคุณลักษณะ) \n\n**คุณลักษณะที่แตกต่างกันมี 9 ประเภท ได้แก่**\n\n0: good_for_lunch\n\n1: good_for_dinner\n\n2: takes_reservations\n\n3: outdoor_seating\n\n4: restaurant_is_expensive\n\n5: has_alcohol\n\n6: has_table_service\n\n7: ambience_is_classy\n\n8: good_for_kids","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:32:54.459455Z","iopub.execute_input":"2023-11-27T15:32:54.460501Z","iopub.status.idle":"2023-11-27T15:32:54.469224Z","shell.execute_reply.started":"2023-11-27T15:32:54.460447Z","shell.execute_reply":"2023-11-27T15:32:54.467509Z"}}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-28T03:54:25.192653Z","iopub.execute_input":"2023-11-28T03:54:25.193096Z","iopub.status.idle":"2023-11-28T03:54:25.656004Z","shell.execute_reply.started":"2023-11-28T03:54:25.193064Z","shell.execute_reply":"2023-11-28T03:54:25.655148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Import Libraries**","metadata":{}},{"cell_type":"code","source":"# libraries for simple anaysis & visualizaition\nimport os\nimport sys\nimport cv2\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport random\n\n# libraies for classification\nimport time # measuring time used.\nfrom PIL import Image\nfrom tqdm import tqdm # for progress bar\nfrom torchvision import transforms # for preprocessing\nfrom torchvision import models\nimport torch # general importing\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data.dataset import TensorDataset\nfrom torch.utils.data.dataloader import DataLoader\nfrom torch.utils.tensorboard import SummaryWriter # dashboard\nfrom sklearn.metrics import precision_score, recall_score, f1_score # for measuring scores\nfrom torch import nn # Neural Network\nfrom torch.utils.data.dataloader import DataLoader\nfrom sklearn import metrics, model_selection, preprocessing\nfrom PIL import Image\nimport tensorflow as tf\nfrom tensorflow import keras\nimport tarfile\nfrom keras.preprocessing.image import img_to_array, load_img, ImageDataGenerator #Converts a PIL Image instance to a Numpy array. #Loads an image into PIL format. \nfrom keras.utils import to_categorical #Converts a class vector (integers) to binary class matrix.\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\nfrom keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom keras.layers import Activation, Flatten, Dense, Dropout, Lambda\nfrom keras.optimizers import Adam #Adam that implements the Optimizer\nfrom keras.applications.vgg16 import VGG16 #Instantiates the VGG16 model.\nfrom keras.layers import GlobalAveragePooling2D, Input, Conv2D, multiply, LocallyConnected2D\nimport keras.backend as K\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping\nimport random\nfrom keras.preprocessing import image\nfrom keras import models\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom concurrent.futures import ThreadPoolExecutor\nfrom glob import glob\nfrom shutil import copyfile","metadata":{"execution":{"iopub.status.busy":"2023-11-28T03:54:25.657755Z","iopub.execute_input":"2023-11-28T03:54:25.658587Z","iopub.status.idle":"2023-11-28T03:54:46.000667Z","shell.execute_reply.started":"2023-11-28T03:54:25.658553Z","shell.execute_reply":"2023-11-28T03:54:45.999419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Extract Files**\nเนื่องจากไฟล์ที่ได้มาเป็นนามสกุล .tgz หมายความว่าเป็นที่ zip เอาไว้จึงต้องทำการแตกไฟล์ก่อน","metadata":{}},{"cell_type":"code","source":"#กำหนดฟังก์ชันสำหรับแตกไฟล์ข้อมูลจากไฟล์ tar archive และรวมการจัดการกับ archive ที่ซ้อนกัน\nimport os, sys, tarfile\nfrom tqdm import tqdm\ndef extract(tar_url, extract_path='.'):\n    print(tar_url)\n    tar = tarfile.open(tar_url, 'r')\n    for item in tqdm(tar):\n        tar.extract(item, extract_path)\n        if item.name.find(\".tgz\") != -1 or item.name.find(\".tar\") != -1:\n            extract(item.name, \"./\" + item.name[:item.name.rfind('/')])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T03:54:46.002280Z","iopub.execute_input":"2023-11-28T03:54:46.003237Z","iopub.status.idle":"2023-11-28T03:54:46.012254Z","shell.execute_reply.started":"2023-11-28T03:54:46.003187Z","shell.execute_reply":"2023-11-28T03:54:46.011321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#แตกไฟล์จากไฟล์ tar archive ที่ถูกระบุ (train_photos)\ntry:\n    extract('../input/yelp-restaurant-photo-classification/train_photos' + '.tgz')\n    print ('Done.')\nexcept:\n    print('error')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T03:54:46.014721Z","iopub.execute_input":"2023-11-28T03:54:46.015109Z","iopub.status.idle":"2023-11-28T03:59:06.930237Z","shell.execute_reply.started":"2023-11-28T03:54:46.015075Z","shell.execute_reply":"2023-11-28T03:59:06.928796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#แตกไฟล์จากไฟล์ tar archive ที่ถูกระบุ (test_photos)\ntry:\n    extract('../input/yelp-restaurant-photo-classification/test_photos.tgz')\n    print ('Done.')\nexcept:\n    print('error')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T03:59:06.932700Z","iopub.execute_input":"2023-11-28T03:59:06.933108Z","iopub.status.idle":"2023-11-28T04:03:30.546444Z","shell.execute_reply.started":"2023-11-28T03:59:06.933073Z","shell.execute_reply":"2023-11-28T04:03:30.544997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#แตกไฟล์จากไฟล์ tar archive ที่ถูกระบุ (train.csv, train_photo_to_biz_ids.csv, test_photo_to_biz.csv, sample_submission.csv)\ntry:\n    extract('../input/yelp-restaurant-photo-classification/train.csv.tgz')\n    extract('../input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz')\n    extract('../input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz')\n    extract('../input/yelp-restaurant-photo-classification/sample_submission.csv.tgz')\n    print ('Done.')\nexcept:\n    print('error')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.548361Z","iopub.execute_input":"2023-11-28T04:03:30.548723Z","iopub.status.idle":"2023-11-28T04:03:30.822425Z","shell.execute_reply.started":"2023-11-28T04:03:30.548692Z","shell.execute_reply":"2023-11-28T04:03:30.821684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"train.csv\")\ndf.head(4)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.823871Z","iopub.execute_input":"2023-11-28T04:03:30.824522Z","iopub.status.idle":"2023-11-28T04:03:30.870422Z","shell.execute_reply.started":"2023-11-28T04:03:30.824480Z","shell.execute_reply":"2023-11-28T04:03:30.869559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('data shape : ', df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.872178Z","iopub.execute_input":"2023-11-28T04:03:30.872627Z","iopub.status.idle":"2023-11-28T04:03:30.880231Z","shell.execute_reply.started":"2023-11-28T04:03:30.872584Z","shell.execute_reply":"2023-11-28T04:03:30.879073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ในเครือของ Yelp จะมีร้านอาหารย่อยมากมายหลายสาขา ซึ่งข้อมูลต่อไปนี้หมายถึง\n### รูปที่ถ่ายแต่ละรูปถูกถ่ายขึ้นบริเวณร้านอาหารอะไร","metadata":{}},{"cell_type":"markdown","source":"# Loading data","metadata":{}},{"cell_type":"code","source":"# ย่อมาจาก photo to restuarant id data-frame\np2r_df = pd.read_csv(\"train_photo_to_biz_ids.csv\")\np2r_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.882211Z","iopub.execute_input":"2023-11-28T04:03:30.882585Z","iopub.status.idle":"2023-11-28T04:03:30.971246Z","shell.execute_reply.started":"2023-11-28T04:03:30.882552Z","shell.execute_reply":"2023-11-28T04:03:30.969779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of mapping photo_id to restaurant_id : \", p2r_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.976249Z","iopub.execute_input":"2023-11-28T04:03:30.977507Z","iopub.status.idle":"2023-11-28T04:03:30.982797Z","shell.execute_reply.started":"2023-11-28T04:03:30.977436Z","shell.execute_reply":"2023-11-28T04:03:30.981734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_dir = 'train_photos'\nimg_paths = os.listdir(imgs_dir)\n\nprint('Number of images:', len(img_paths))","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:30.984361Z","iopub.execute_input":"2023-11-28T04:03:30.984747Z","iopub.status.idle":"2023-11-28T04:03:31.303593Z","shell.execute_reply.started":"2023-11-28T04:03:30.984717Z","shell.execute_reply":"2023-11-28T04:03:31.302321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Clean & Manage Data**\nเพื่อความง่ายต่อการนำไปเป็นข้อมูลที่มีประโยชน์ในตอนทำ model","metadata":{}},{"cell_type":"markdown","source":"## remove hidden files","metadata":{}},{"cell_type":"code","source":"train_dir = 'train_photos'\ndef remove_hidden_files(train_dir):\n    for filename in os.listdir(train_dir):\n        if filename.startswith('.'):\n            file_path = os.path.join(train_dir, filename)\n            os.remove(file_path)\nremove_hidden_files(train_dir)\n\ntest_dir = 'test_photos'\ndef remove_hidden_files(test_dir):\n    for filename in os.listdir(test_dir):\n        if filename.startswith('.'):\n            file_path = os.path.join(test_dir, filename)\n            os.remove(file_path)\nremove_hidden_files(test_dir)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:03:31.305711Z","iopub.execute_input":"2023-11-28T04:03:31.306101Z","iopub.status.idle":"2023-11-28T04:04:03.860262Z","shell.execute_reply.started":"2023-11-28T04:03:31.306068Z","shell.execute_reply":"2023-11-28T04:04:03.858829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = 'train_photos'\ntrain_imgs = os.listdir(train_dir)\n\ntest_dir = 'test_photos'\ntest_imgs = os.listdir(test_dir)\n\nprint('Number of training images:', len(train_imgs))\nprint('Number of testing images:', len(test_imgs))","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:03.861800Z","iopub.execute_input":"2023-11-28T04:04:03.862214Z","iopub.status.idle":"2023-11-28T04:04:04.185493Z","shell.execute_reply.started":"2023-11-28T04:04:03.862179Z","shell.execute_reply":"2023-11-28T04:04:04.184233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge Data","metadata":{}},{"cell_type":"code","source":"# ควมรวมข้อมูลจากสอง dataframes เป็น 1 เพื่อข้อมูลที่มีประโยชน์ยิ่งขึ้น\ndata = pd.merge(df, p2r_df,\n     on='business_id', how='left')\ndata.dropna()\ndata.sample(10)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:04.186994Z","iopub.execute_input":"2023-11-28T04:04:04.187446Z","iopub.status.idle":"2023-11-28T04:04:04.326693Z","shell.execute_reply.started":"2023-11-28T04:04:04.187411Z","shell.execute_reply":"2023-11-28T04:04:04.325440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.dropna(subset=['labels'])\n# สร้าง column ใหม่ที่เป็น list of labels\ndata['labs_arr'] = data['labels'].apply(lambda lb: str(lb).split(' '))\ndata.sample(20)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:04.328375Z","iopub.execute_input":"2023-11-28T04:04:04.328868Z","iopub.status.idle":"2023-11-28T04:04:04.563801Z","shell.execute_reply.started":"2023-11-28T04:04:04.328824Z","shell.execute_reply":"2023-11-28T04:04:04.562701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"แสดงจำนวน counts ของแต่ละ labels","metadata":{}},{"cell_type":"code","source":"# Count all labels in training set\nall_labels = ' '.join(list(data['labels'].fillna('nan').values)).split()\nfrom collections import Counter\nlabel_counts = Counter(all_labels)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:04.565085Z","iopub.execute_input":"2023-11-28T04:04:04.565427Z","iopub.status.idle":"2023-11-28T04:04:04.708509Z","shell.execute_reply.started":"2023-11-28T04:04:04.565398Z","shell.execute_reply":"2023-11-28T04:04:04.707123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in label_counts:\n    print('Label {0} appears {1} times in training dataset'.format(key, label_counts[key]))","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:04.709845Z","iopub.execute_input":"2023-11-28T04:04:04.710231Z","iopub.status.idle":"2023-11-28T04:04:04.716525Z","shell.execute_reply.started":"2023-11-28T04:04:04.710197Z","shell.execute_reply":"2023-11-28T04:04:04.715327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" สร้าง new column data ขึ้นมาเพื่อใช้เก็บ list of interger tags โดยเฉพาะ","metadata":{}},{"cell_type":"code","source":"labs_int = data['labs_arr'].tolist()\nfor r in tqdm(labs_int):\n    for c in range(len(r)):\n        r[c] = int(r[c]) # parse str to int\ndata['labs_int'] = labs_int\ndata.sample(5)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:04.717927Z","iopub.execute_input":"2023-11-28T04:04:04.718333Z","iopub.status.idle":"2023-11-28T04:04:05.476417Z","shell.execute_reply.started":"2023-11-28T04:04:04.718300Z","shell.execute_reply":"2023-11-28T04:04:05.475106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_notation = {\n    0: 'good_for_lunch',\n    1: 'good_for_dinner',\n    2: 'takes_reservations',\n    3: 'outdoor_seating',\n    4: 'restaurant_is_expensive',\n    5: 'has_alcohol',\n    6: 'has_table_service',\n    7: 'ambience_is_classy',\n    8: 'good_for_kids'\n}","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:05.478107Z","iopub.execute_input":"2023-11-28T04:04:05.479287Z","iopub.status.idle":"2023-11-28T04:04:05.485878Z","shell.execute_reply.started":"2023-11-28T04:04:05.479237Z","shell.execute_reply":"2023-11-28T04:04:05.484652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# สร้าง column ใหม่ 'labels_names' เพื่อเก็บชื่อ labels จาก 'label_notation'\ndata['labels_names'] = data['labs_int'].apply(lambda labels: [label_notation[label] for label in labels])\n\n# ดูตัวอย่างข้อมูลหลังจากแปลง\nprint(data[['labs_int', 'labels_names']].head())","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:05.487406Z","iopub.execute_input":"2023-11-28T04:04:05.487822Z","iopub.status.idle":"2023-11-28T04:04:06.150483Z","shell.execute_reply.started":"2023-11-28T04:04:05.487789Z","shell.execute_reply":"2023-11-28T04:04:06.149087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ทำการ Plot กราฟจำนวน counts ของแต่ละ labels","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Count occurrences of each labels_names\nlabels_names_counts = data['labels_names'].explode().value_counts()\n\n\n# Create a bar plot\nplt.figure(figsize=(12, 6))\nsns.barplot(x=labels_names_counts.index, y=labels_names_counts.values, palette='YlOrRd')\n\n# Add labels and title\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('How many images per labels are given in the data?')\n\n# Rotate x-axis labels for better readability (optional)\nplt.xticks(rotation=45, ha='right')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:06.151945Z","iopub.execute_input":"2023-11-28T04:04:06.152352Z","iopub.status.idle":"2023-11-28T04:04:06.837034Z","shell.execute_reply.started":"2023-11-28T04:04:06.152315Z","shell.execute_reply":"2023-11-28T04:04:06.835689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Visualize some Images**\nแสดงรูปภาพบางส่วน","metadata":{}},{"cell_type":"code","source":"torch.manual_seed(2020)\ntorch.cuda.manual_seed(2020)\nnp.random.seed(2020)\nrandom.seed(2020)\ntorch.backends.cudnn.deterministic = True","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:06.838947Z","iopub.execute_input":"2023-11-28T04:04:06.839863Z","iopub.status.idle":"2023-11-28T04:04:06.854294Z","shell.execute_reply.started":"2023-11-28T04:04:06.839805Z","shell.execute_reply":"2023-11-28T04:04:06.853242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have already loaded train.csv and train_photos DataFrames\n# Split the 'labels' column into a list of labels\ndata['labels_list'] = data['labels'].str.split(' ')\n\n# Find the business IDs of all businesses\nbusiness_id_list = data['business_id'].tolist()\n\n# Assuming train_photos is a DataFrame containing photo information\n# Extract the photo IDs of all photos\nall_photos = p2r_df[p2r_df.business_id.isin(business_id_list)].photo_id.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:06.856357Z","iopub.execute_input":"2023-11-28T04:04:06.856868Z","iopub.status.idle":"2023-11-28T04:04:07.466911Z","shell.execute_reply.started":"2023-11-28T04:04:06.856821Z","shell.execute_reply":"2023-11-28T04:04:07.465510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images_for_show = 20\n\nphotos_to_show = np.random.choice(all_photos,num_images_for_show**2)\n\n\nfor x in range(num_images_for_show ** 2):\n        plt.rcParams['figure.figsize'] = (7.0, 7.0)\n        plt.subplot(num_images_for_show, num_images_for_show, x+1)\n        im = Image.open(os.path.join('train_photos',''.join([str(photos_to_show[x]),'.jpg'])))\n        plt.imshow(im.resize((224,224)))\n        plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:07.468447Z","iopub.execute_input":"2023-11-28T04:04:07.468969Z","iopub.status.idle":"2023-11-28T04:04:29.632068Z","shell.execute_reply.started":"2023-11-28T04:04:07.468912Z","shell.execute_reply":"2023-11-28T04:04:29.630418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# เลือก 8 รูปอย่างสุ่มมาแสดง\nimgs_samples = random.sample(train_imgs, 8)\n\nplt.figure(figsize=(15, 10))\nfor i in range(len(imgs_samples)):\n    # readed as BGR format\n    img = cv2.imread(os.path.join(train_dir, imgs_samples[i]))\n    # แปลงภาพเป็น rgb เพื่อให้ matplotlib อ่าน\n    img = img[...,::-1]\n    # Grab image's business ID and labels\n    business = p2r_df.loc[p2r_df['photo_id'] == int(imgs_samples[i][:-4]), 'business_id']\n    labels = df.loc[df['business_id'] == business.values[0], 'labels']\n    title = \"Image ID: \" + imgs_samples[i] + ' Business: ' + str(business.values[0]) + '\\nLabels: ' + ''.join(labels.values)\n    \n    plt.subplot(2, 4, i+1)\n    plt.tight_layout(pad=0.4, w_pad=0.5, h_pad=1.0)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(title)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:29.633806Z","iopub.execute_input":"2023-11-28T04:04:29.634230Z","iopub.status.idle":"2023-11-28T04:04:32.095074Z","shell.execute_reply.started":"2023-11-28T04:04:29.634194Z","shell.execute_reply":"2023-11-28T04:04:32.093740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show some images from Train & Test","metadata":{}},{"cell_type":"markdown","source":"สร้าง Custom Dataset ที่มีข้อมูลทั้งรูปภาพและบอกว่ารูปนั้นเป็น tags ไหนอยู่พร้อม ๆ กัน\n(เก็บทั้ง features X, z อยู่ในก้อนเดียวกันเพื่อความง่ายในการจัดการ)","metadata":{}},{"cell_type":"code","source":"class NusDataset(Dataset):\n    def __init__(self, data_path, data, transforms):\n        self.transforms = transforms\n        data=data\n        samples = data['photo_id'].tolist()\n        labs=data['labs_int'].tolist()\n        self.classes = [0,1,2,3,4,5,6,7,8]\n\n        self.imgs = []\n        self.annos = []\n        self.data_path = data_path\n        #print('loading', anno_path)\n        for sample in samples:\n            self.imgs.append(sample)\n        for lab in labs:\n            self.annos.append(lab)\n            \n        for item_id in range(len(self.annos)):\n            item = self.annos[item_id]\n            vector = [cls in item for cls in self.classes]\n            self.annos[item_id] = np.array(vector, dtype=float)\n            \n    def __getitem__(self, item):\n        anno = self.annos[item]\n        img_path = os.path.join(self.data_path, str(self.imgs[item])+'.jpg')\n        img = Image.open(img_path)\n        if self.transforms is not None:\n            img = self.transforms(img)\n        return img, anno\n    \n    def __len__(self):\n        return len(self.imgs)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:32.096875Z","iopub.execute_input":"2023-11-28T04:04:32.098337Z","iopub.status.idle":"2023-11-28T04:04:32.113708Z","shell.execute_reply.started":"2023-11-28T04:04:32.098280Z","shell.execute_reply":"2023-11-28T04:04:32.112481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, df_val = model_selection.train_test_split(data, test_size=0.2, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:32.115304Z","iopub.execute_input":"2023-11-28T04:04:32.116706Z","iopub.status.idle":"2023-11-28T04:04:32.285468Z","shell.execute_reply.started":"2023-11-28T04:04:32.116659Z","shell.execute_reply":"2023-11-28T04:04:32.283353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train = NusDataset(\"train_photos\", df_train, None)\ndataset_val = NusDataset(\"train_photos\", df_val, None)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:32.294931Z","iopub.execute_input":"2023-11-28T04:04:32.296258Z","iopub.status.idle":"2023-11-28T04:04:33.509056Z","shell.execute_reply.started":"2023-11-28T04:04:32.296193Z","shell.execute_reply":"2023-11-28T04:04:33.507752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train.annos[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:33.510624Z","iopub.execute_input":"2023-11-28T04:04:33.511112Z","iopub.status.idle":"2023-11-28T04:04:33.520293Z","shell.execute_reply.started":"2023-11-28T04:04:33.511078Z","shell.execute_reply":"2023-11-28T04:04:33.519063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset_train.imgs[0]\ndataset_train.imgs[:5]","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:33.521873Z","iopub.execute_input":"2023-11-28T04:04:33.522346Z","iopub.status.idle":"2023-11-28T04:04:33.531430Z","shell.execute_reply.started":"2023-11-28T04:04:33.522312Z","shell.execute_reply":"2023-11-28T04:04:33.530246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(dataset_train[0])\na,b = dataset_train[17]\nprint(b)\n\nplt.imshow(a)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:33.532780Z","iopub.execute_input":"2023-11-28T04:04:33.533152Z","iopub.status.idle":"2023-11-28T04:04:34.073024Z","shell.execute_reply.started":"2023-11-28T04:04:33.533121Z","shell.execute_reply":"2023-11-28T04:04:34.071810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure()\n\nfor i, sample in enumerate(dataset_train):\n    print(i, sample[1])\n\n    ax = plt.subplot(1, 4, i + 1)\n    plt.tight_layout()\n    ax.set_title('Sample #{}'.format(i))\n    ax.axis('off')\n\n    plt.imshow(sample[0])\n\n    if i == 3:\n        plt.show()\n        break","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:34.074769Z","iopub.execute_input":"2023-11-28T04:04:34.075997Z","iopub.status.idle":"2023-11-28T04:04:34.839427Z","shell.execute_reply.started":"2023-11-28T04:04:34.075943Z","shell.execute_reply":"2023-11-28T04:04:34.837884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout, BatchNormalization, Conv2D, MaxPool2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing import image\n\nimg_width = 256\nimg_height = 256\nnum_images = 5000\n\nX = []\n\nfor i in tqdm(range(num_images)):\n    photo_id = data['photo_id'].iloc[i]\n    path = 'train_photos/' + str(photo_id) + '.jpg'\n    \n    img = image.load_img(path, target_size = (img_width, img_height))\n    img = image.img_to_array(img)\n    img = img/255.0\n    X.append(img)\n\nX = np.array(X)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:04:34.841120Z","iopub.execute_input":"2023-11-28T04:04:34.841631Z","iopub.status.idle":"2023-11-28T04:05:06.348477Z","shell.execute_reply.started":"2023-11-28T04:04:34.841594Z","shell.execute_reply":"2023-11-28T04:05:06.347162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:06.350545Z","iopub.execute_input":"2023-11-28T04:05:06.351199Z","iopub.status.idle":"2023-11-28T04:05:06.357944Z","shell.execute_reply.started":"2023-11-28T04:05:06.351159Z","shell.execute_reply":"2023-11-28T04:05:06.356623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = data.drop(['photo_id', 'labels','labels_names'], axis = 1)\ny = y.head(5000)\ny.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:06.359840Z","iopub.execute_input":"2023-11-28T04:05:06.360198Z","iopub.status.idle":"2023-11-28T04:05:06.389703Z","shell.execute_reply.started":"2023-11-28T04:05:06.360169Z","shell.execute_reply":"2023-11-28T04:05:06.388371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and Validation Split\nทำการแบ่งเป็น Train ทั้งหมด โดยจะแบ่งเป็น Train และ Validation ในอัตราส่วน 80:20","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0, test_size=0.2)\nX_train.shape, X_test.shape, y_train.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:06.391669Z","iopub.execute_input":"2023-11-28T04:05:06.392119Z","iopub.status.idle":"2023-11-28T04:05:08.493693Z","shell.execute_reply.started":"2023-11-28T04:05:06.392084Z","shell.execute_reply":"2023-11-28T04:05:08.492501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# ทำให้ค่า pixel อยู่ในช่วง [0, 1]\nX_train = X_train / 255.0\nX_test = X_test / 255.0\n\n# แปลงเป็น NumPy arrays\nX_train = np.array(X_train)\nX_test = np.array(X_test)\ny_train = np.array(y_train)\ny_test = np.array(y_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:08.495455Z","iopub.execute_input":"2023-11-28T04:05:08.496179Z","iopub.status.idle":"2023-11-28T04:05:19.089228Z","shell.execute_reply.started":"2023-11-28T04:05:08.496130Z","shell.execute_reply":"2023-11-28T04:05:19.087857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming y is a DataFrame with multiple label columns\ny = data.drop(['photo_id', 'labels', 'business_id', 'labs_arr'], axis=1)\ny = y.head(5000)\n\nmlb = MultiLabelBinarizer()\none_hot_labels = mlb.fit_transform(y['labs_int'])\nlabel_names = ['good_for_lunch', 'good_for_dinner', 'takes_reservations',\n               'outdoor_seating', 'restaurant_is_expensive', 'has_alcohol',\n               'has_table_service', 'ambience_is_classy', 'good_for_kids']\n\ny = pd.DataFrame(one_hot_labels, columns=label_names)\n\n# Convert DataFrame to NumPy array\ny_array = y.values","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:19.090948Z","iopub.execute_input":"2023-11-28T04:05:19.091336Z","iopub.status.idle":"2023-11-28T04:05:19.145072Z","shell.execute_reply.started":"2023-11-28T04:05:19.091303Z","shell.execute_reply":"2023-11-28T04:05:19.143816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y_array, random_state=0, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:19.149985Z","iopub.execute_input":"2023-11-28T04:05:19.150755Z","iopub.status.idle":"2023-11-28T04:05:22.419358Z","shell.execute_reply.started":"2023-11-28T04:05:19.150699Z","shell.execute_reply":"2023-11-28T04:05:22.418412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **CNN Model**\n\nเพิ่ม layer ต่าง ๆ ลงไปในโมเดลโดยใช้ Keras API ซึ่งเป็น API ที่สะดวกในการสร้างและการจัดการโมเดลประสิทธิภาพสูงในงาน Deep Learning","metadata":{}},{"cell_type":"code","source":"# Assuming X_train has the shape of your input data\n# Modify input_shape according to your data\n# Model Architecture\nmodel = Sequential()\nmodel.add(Conv2D(16, (7, 7), activation='relu', input_shape=X_train.shape[1:]))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(2, 2))\n\nmodel.add(Conv2D(32, (7, 7), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(2, 2))\n\nmodel.add(Conv2D(64, (7, 7), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(2, 2))\n\nmodel.add(Conv2D(128, (7, 7), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(2, 2))\nmodel.add(Dropout(0.5))\n\nmodel.add(Flatten())\n\nmodel.add(Dense(128, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(64, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(9, activation='softmax'))\n\nmodel.summary()\n\n\ny_train = np.array(y_train)\ny_test = np.array(y_test)\n\n# This converts one-hot encoded labels to integers\ny_train_int = np.argmax(y_train, axis=1)\ny_test_int = np.argmax(y_test, axis=1)\n\nopt = Adam(learning_rate=0.001)\n\n#Compile and Train the Model\nmodel.compile(optimizer=opt, loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n#Evaluate and Print Results\nhistory_resnet = model.fit(X_train, y_train_int, epochs=5, validation_data=(X_test, y_test_int))","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:05:22.421091Z","iopub.execute_input":"2023-11-28T04:05:22.421830Z","iopub.status.idle":"2023-11-28T04:59:46.986384Z","shell.execute_reply.started":"2023-11-28T04:05:22.421786Z","shell.execute_reply":"2023-11-28T04:59:46.982624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#คำนวณค่า loss และความแม่นยำ accuracy ของโมเดลบนชุดข้อมูลทดสอบ\nval_score, val_acc = model.evaluate(X_test, y_test_int)\ntrain_score, train_acc = model.evaluate(X_train, y_train_int)\n\nprint('Validation score:', val_score, 'Validation accuracy:', val_acc)\nprint('Train score:', train_score, '   Train accuracy:', train_acc)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:24:52.561716Z","iopub.execute_input":"2023-11-28T05:24:52.562287Z","iopub.status.idle":"2023-11-28T05:27:38.601636Z","shell.execute_reply.started":"2023-11-28T05:24:52.562245Z","shell.execute_reply":"2023-11-28T05:27:38.600590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction**","metadata":{}},{"cell_type":"code","source":"# Make predictions on the test data\npredictions = model.predict(X_test)\n\n# Assuming predictions are one-hot encoded, convert them back to class labels\npredicted_labels = np.argmax(predictions, axis=1)\n\n# Display some predictions\nfor i in range(10):  # Displaying the first 10 predictions\n    print(f\"Prediction: {predicted_labels[i]}, Actual Label: {y_test_int[i]}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:02:34.633073Z","iopub.execute_input":"2023-11-28T05:02:34.633798Z","iopub.status.idle":"2023-11-28T05:03:16.506505Z","shell.execute_reply.started":"2023-11-28T05:02:34.633757Z","shell.execute_reply":"2023-11-28T05:03:16.505232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport random\n\n# Assuming label_names is a list of label names\nlabel_names = ['good_for_lunch', 'good_for_dinner', 'takes_reservations',\n               'outdoor_seating', 'restaurant_is_expensive', 'has_alcohol',\n               'has_table_service', 'ambience_is_classy', 'good_for_kids']\n\n# Visualize random samples and their predictions\nnum_samples_to_visualize = 5\n\nfor _ in range(num_samples_to_visualize):\n    index = random.randint(0, len(X_test) - 1)\n    \n    # Display the image\n    plt.imshow(X_test[index])\n    plt.title(f\"Actual: {label_names[y_test_int[index]]}, Predicted: {label_names[predicted_labels[index]]}\")\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:03:16.508129Z","iopub.execute_input":"2023-11-28T05:03:16.509105Z","iopub.status.idle":"2023-11-28T05:03:18.726948Z","shell.execute_reply.started":"2023-11-28T05:03:16.509023Z","shell.execute_reply":"2023-11-28T05:03:18.725836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**","metadata":{}},{"cell_type":"code","source":"# Create a DataFrame for submission\nsubmission_df = pd.DataFrame({\n    'Image_ID': range(1, len(X_test) + 1),\n    'Predicted_Label': [label_names[i] for i in predicted_labels]\n})","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:10:30.904698Z","iopub.execute_input":"2023-11-28T05:10:30.905250Z","iopub.status.idle":"2023-11-28T05:10:30.922714Z","shell.execute_reply.started":"2023-11-28T05:10:30.905205Z","shell.execute_reply":"2023-11-28T05:10:30.921238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:10:36.167009Z","iopub.execute_input":"2023-11-28T05:10:36.167451Z","iopub.status.idle":"2023-11-28T05:10:36.191924Z","shell.execute_reply.started":"2023-11-28T05:10:36.167416Z","shell.execute_reply":"2023-11-28T05:10:36.190929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Project Summary**\n\n* ใช้ Conv2D layers, BatchNormalization, MaxPool2D, Dropout, และ Dense layers ซึ่งอยู่ใน CNN Model มาใช้ในการปรับค่าและทำนายผล โดยในตอนแรกพวกเราได้ลองใช้ข้อมูลทั้งหมดเพื่อทำการ extract bottle feature แต่เนื่องจากข้อมูลมีจำนวนมาก ( 234842 ไฟล์ภาพในโฟลเดอร์ train_photos และ 1190225 ไฟล์ภาพในโฟลเดอร์ test_photos ) จึงทำให้เวลาที่ประมาณว่าจะทำการสกัดข้อมูลทั้งหมดได้นั้นมากตามไปด้วย พวกเราจึงได้แก้ปัญหาด้วยการ สร้างชุดข้อมูลรูปภาพขึ้นมาใหม่ โดยให้มีจำนวนภาพในชุดข้อมูล 5000 รูป และแบ่งเป็น 4000 รูปสำหรับ train set และ 1000 รูปสำหรับ test set)\n* เราได้ทำการลองใช้ Model อื่น ๆ ไม่ว่าจะเป็น MobileNetV2, VGG16 ซึ่งได้ค่า loss ที่สูงเกินจริง และ ค่า accuray ที่ต่ำกว่า 0.001 ทำให้เราไม่เลือกใช้โมเดลเหล่านี้ \n* สุดท้ายที่เราเลือกโมเดลนี้ เพราะว่าโมเดลนี้มีการทำนายค่าได้มีประสิทธิภาพดีที่สุด จากค่า accuracy ที่สูงแสดงให้เห็นถึงการทำนายโมเดลจากรูปภาพ\n","metadata":{}},{"cell_type":"markdown","source":"# References","metadata":{}},{"cell_type":"markdown","source":"* Subbrain. (October 8, 2022). Pytorch – Neural Network. สืบค้นเมื่อวันที่ 13 พฤศจิกายน 2566. จาก https://www.sub-brain.com/datait/pytorch-convnet/\n\n* Amey Varangaonkar. (August 2018). Hands-On Convolutional Neural Networks with TensorFlow. สืบค้นเมื่อวันที่ 13 พฤศจิกายน 2566\n\n* Nuttachot Promrit. (October 6, 2020). Visualizing Kernels and Feature Maps in Deep Learning Model (CNN). สืบค้นเมื่อ 18 พฤศจิกายน 2566. จาก https://blog.pjjop.org/visualizing-filters-and-feature-maps-in-deep-learning-cnn/#:~:text=CNN%20(Convolutional%20Neural%20Network)%20เป็น,ในเชิงพื้นที่%20(Spatial%20Relationship)\n \n* Bellakhal Mohamed. ComputerVisionAtelier. สืบค้นเมื่อวันที่ 20 พฤศจิกายน 2566. จาก www.kaggle.com/code/punjuanoir/computervisionatelier\n\n* WITS CODE. (January 25, 2020). สร้างโมเดล Deep learnning: CNN. ง่ายๆ สำหรับการจำแนกรูปภาพ. สืบค้นเมื่อวันที่ 20 พฤศจิกายน 2566. จาก https://witscodes.wordpress.com/2020/01/26/สร้างโมเดล-deep-learning-cnn-convnet/","metadata":{}},{"cell_type":"markdown","source":"## สมาชิก\n* 6524650014 ณัฐธิดา แนวสุภาพ\n* 6524651111 พรนพิน เธียรฤกษ์ \n* 6524651202 ชวิศา วรรณพินทุ\n* 6524651269 ธษา โพธิ์วัฒนากุล\n* 6524651319 พิไลลักษณ์ รอดผล ","metadata":{}}]}