{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"},{"sourceId":2558,"sourceType":"modelInstanceVersion","modelInstanceId":1872}],"dockerImageVersionId":30580,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"background-color: #FFF3E0; padding: 20px; border-radius: 10px; box-shadow: 2px 2px 10px rgba(0, 0, 0, 0.1);\">\n    <h1 style=\"font-family: 'Lucida Sans', 'Lucida Sans Regular', 'Lucida Grande', 'Lucida Sans Unicode', Geneva, Verdana, sans-serif; text-align: center; color: #3A405A;\">DSI206 Project</h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"\n<div style=\"background-color: #FFE5E5; padding: 15px; border-radius: 10px; box-shadow: 2px 2px 10px rgba(0, 0, 0, 0.1);\">\n    <h1 style=\"font-family: 'Lucida Sans', 'Lucida Sans Regular', 'Lucida Grande', 'Lucida Sans Unicode', Geneva, Verdana, sans-serif; text-align: center; color: #3A405A;\">Yelp Restaurant Photo Classification</h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"\n<div class=\"alert alert-block alert-info\">\n    <h2 align=\"left\"> <font>Introduction</font></h2>\n\n</div>","metadata":{}},{"cell_type":"markdown","source":"จุดมุ่งหมาย : เพื่อศึกษาวิธีการทำโมเดล Photo Clssification โดยการแยกประเภทของรูปภาพอาหาร ในร้านอาหารต่างๆ 9 ประเภท จากรูปภาพทั้งหมด 234842 ภาพ ในชุดข้อมูล Yelp Restaurant Photo Classification ด้วยภาษา Python\n","metadata":{}},{"cell_type":"markdown","source":"มีประเภทของอาหาร 9 อย่าง ดังนี้\n\n1. good_for_lunch \n2. good_for_dinner\n3. takes_reservations\n4. outdoor_seating\n5. restaurant_is_expensive\n6. has_alcohol\n7. has_table_service\n8. ambience_is_classy\n9. good_for_kids","metadata":{}},{"cell_type":"markdown","source":"**File descriptions**\n* sample_submission.csv\n* test_photo_to_biz.csv\n* test_photos\n* train.csv\n* train_photo_to_biz_ids.csv\n* train_photos\n","metadata":{}},{"cell_type":"markdown","source":"### Members\n1. 6524651095 ปาณิสรา กุยยะรัตน์\n2. 6524651129 พิมพกานต์ คงทอง\n3. 6524651236 ณัชชา คงแก้ว\n4. 6524651301 พิชามญชุ์ พรอรุณสถาพร\n5. 6524651343 รอยมีย์ เนสะและ\n6. 6524651392 วรานิษฐ์ เตชะระพีพัฒน์","metadata":{}},{"cell_type":"markdown","source":"### ขั้นตอนการดำเนินงาน\n1. Data Exploration\n2. Data Preparation\n3. Modeling\n4. Project Summary","metadata":{}},{"cell_type":"markdown","source":"# 1.Data Exploration\n*  Imports Libraries ที่ต้องการใช้\n* นำเข้าข้อมูลจากชุดข้อมูล\n* ดูข้อมูลที่นำเข้ามาว่ามีข้อมูลอะไรบ้าง","metadata":{}},{"cell_type":"code","source":"# import Library ทั้งหมดที่ต้องใช้\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.decomposition import PCA\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import metrics\nfrom skimage import io\nimport os\nfrom tqdm import tqdm\nfrom PIL import Image\nimport random\nimport IPython.display as display_module\nfrom IPython.display import display\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout, BatchNormalization, Conv2D, MaxPool2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing import image\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T07:58:43.867397Z","iopub.execute_input":"2023-11-28T07:58:43.867857Z","iopub.status.idle":"2023-11-28T07:58:56.791381Z","shell.execute_reply.started":"2023-11-28T07:58:43.867817Z","shell.execute_reply":"2023-11-28T07:58:56.790402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"นำเข้าชุดข้อมูล Yelp Restaurant Photo Classification","metadata":{}},{"cell_type":"code","source":"%%bash\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz\ntar -xzf /kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz","metadata":{"execution":{"iopub.status.busy":"2023-11-28T07:58:56.793353Z","iopub.execute_input":"2023-11-28T07:58:56.793905Z","iopub.status.idle":"2023-11-28T08:03:06.958103Z","shell.execute_reply.started":"2023-11-28T07:58:56.793878Z","shell.execute_reply":"2023-11-28T08:03:06.957153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"อ่านไฟล์ csv แล้วนำข้อมูลไปเก็บในตัวแปรต่างๆ","metadata":{}},{"cell_type":"code","source":"# อ่านไฟล์ csv แล้วนำข้อมูลไปเก็บในตัวแปรต่างๆ\ntrain_data = pd.read_csv('train.csv')\ntrain_photo_to_biz_ids = pd.read_csv('train_photo_to_biz_ids.csv')\ntest_photo_to_biz = pd.read_csv('test_photo_to_biz.csv')\nsample_sub = pd.read_csv('sample_submission.csv')\n\n# แสดง information และตัวอย่างข้อมูล ของข้อมูล Train data\nprint(train_data.info())\nprint(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:06.959403Z","iopub.execute_input":"2023-11-28T08:03:06.959765Z","iopub.status.idle":"2023-11-28T08:03:07.503868Z","shell.execute_reply.started":"2023-11-28T08:03:06.959715Z","shell.execute_reply":"2023-11-28T08:03:07.502932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataframe แสดงตัวอย่างข้อมูลที่นำเข้ามา","metadata":{}},{"cell_type":"markdown","source":"แสดงตัวอย่างข้อมูล และจำนวนข้อมูลใน train_photo_to_biz_ids","metadata":{}},{"cell_type":"code","source":"display_module.display(train_photo_to_biz_ids.head())\nprint('Shape of train_photo_to_id:', train_photo_to_biz_ids.shape)\nprint('Number of images in training set:', train_photo_to_biz_ids.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:07.505173Z","iopub.execute_input":"2023-11-28T08:03:07.505478Z","iopub.status.idle":"2023-11-28T08:03:07.516996Z","shell.execute_reply.started":"2023-11-28T08:03:07.505451Z","shell.execute_reply":"2023-11-28T08:03:07.516116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"แสดงตัวอย่างข้อมูล และจำนวนข้อมูลใน train_data","metadata":{}},{"cell_type":"code","source":"display_module.display(train_data.head())\nprint('Shape of train data:', train_data.shape)\nprint('Number of unique businesses:', train_data.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:07.520154Z","iopub.execute_input":"2023-11-28T08:03:07.520531Z","iopub.status.idle":"2023-11-28T08:03:07.530004Z","shell.execute_reply.started":"2023-11-28T08:03:07.520501Z","shell.execute_reply":"2023-11-28T08:03:07.529111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Missing Values**\nตรวจสอบข้อมูลที่หายไปในชุดข้อมูล และแสดงรายละเอียดข้อมูลนั้น เนื่องจาก Missing Values สามารถมีผลกระทบต่อการ Train Model เราจึงทำการลบ Missing Values ออกจากชุดข้อมูล","metadata":{}},{"cell_type":"code","source":"# Business id to labels dataframe\nprint('Total number of missing labels:', train_data['labels'].isnull().sum())\ndisplay_module.display(train_data[train_data['labels'].isnull()])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:07.531038Z","iopub.execute_input":"2023-11-28T08:03:07.531302Z","iopub.status.idle":"2023-11-28T08:03:07.544975Z","shell.execute_reply.started":"2023-11-28T08:03:07.531272Z","shell.execute_reply":"2023-11-28T08:03:07.544076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ลบค่าที่หายไปจากคอลัมน์ 'labels'\ntrain_data = train_data.dropna(subset=['labels'])\n\n# ตรวจสอบข้อมูลหลังจากลบค่าที่หายไป\nprint('Total number of missing labels after removal:', train_data['labels'].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:07.546267Z","iopub.execute_input":"2023-11-28T08:03:07.546646Z","iopub.status.idle":"2023-11-28T08:03:07.558284Z","shell.execute_reply.started":"2023-11-28T08:03:07.546615Z","shell.execute_reply":"2023-11-28T08:03:07.557245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_files = [f for f in os.listdir('/kaggle/working/train_photos') if f.endswith(('.jpg', '.jpeg', '.png')) and not f.startswith(('._'))]","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:07.559479Z","iopub.execute_input":"2023-11-28T08:03:07.559772Z","iopub.status.idle":"2023-11-28T08:03:08.012557Z","shell.execute_reply.started":"2023-11-28T08:03:07.559748Z","shell.execute_reply":"2023-11-28T08:03:08.011789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merg Dataframe \nนำข้อมูลที่มีความเกี่ยวข้องกัน ของแต่ละ dataframe มาเชื่อมโยงกัน เพื่อสามารถเรียกใช้ข้อมูลได้อย่างมีประสิทธิภาพ","metadata":{}},{"cell_type":"code","source":"data=pd.merge(train_photo_to_biz_ids, train_data, on='business_id',how='left')\ndata=data.head(20000)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.013673Z","iopub.execute_input":"2023-11-28T08:03:08.013949Z","iopub.status.idle":"2023-11-28T08:03:08.066520Z","shell.execute_reply.started":"2023-11-28T08:03:08.013924Z","shell.execute_reply":"2023-11-28T08:03:08.065557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=data.dropna(subset=['labels'])\ndata = data.head(10000)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.067654Z","iopub.execute_input":"2023-11-28T08:03:08.068113Z","iopub.status.idle":"2023-11-28T08:03:08.083524Z","shell.execute_reply.started":"2023-11-28T08:03:08.068082Z","shell.execute_reply":"2023-11-28T08:03:08.082641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test=pd.merge(test_photo_to_biz ,sample_sub,on='business_id',how='left') \ndata_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.085214Z","iopub.execute_input":"2023-11-28T08:03:08.085594Z","iopub.status.idle":"2023-11-28T08:03:08.374538Z","shell.execute_reply.started":"2023-11-28T08:03:08.085559Z","shell.execute_reply":"2023-11-28T08:03:08.373651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['labs']=data['labels'].apply(lambda x:str(x).split(' '))\ndata","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.375761Z","iopub.execute_input":"2023-11-28T08:03:08.376129Z","iopub.status.idle":"2023-11-28T08:03:08.404988Z","shell.execute_reply.started":"2023-11-28T08:03:08.376096Z","shell.execute_reply":"2023-11-28T08:03:08.404103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#แปลงค่า คอลัมน์ 'labs' จากประเภท string เป็น int\nlabelsint=data.labs.tolist()\nfor i in tqdm(labelsint):\n    for j in range(len(i)):\n        i[j]=int(i[j])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.406199Z","iopub.execute_input":"2023-11-28T08:03:08.406497Z","iopub.status.idle":"2023-11-28T08:03:08.445459Z","shell.execute_reply.started":"2023-11-28T08:03:08.406473Z","shell.execute_reply":"2023-11-28T08:03:08.444537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['labsint']=labelsint\ndata\n\ndata_train=data[[\"photo_id\",\"labsint\"]]\ndata_train","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:08.448888Z","iopub.execute_input":"2023-11-28T08:03:08.449173Z","iopub.status.idle":"2023-11-28T08:03:08.466009Z","shell.execute_reply.started":"2023-11-28T08:03:08.449148Z","shell.execute_reply":"2023-11-28T08:03:08.465092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"แสดงรูปตัวอย่างใน train_photos","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\nimport random\nimport os\n\n# หาไฟล์ภาพที่ใช้ได้\nimage_files = [f for f in os.listdir('/kaggle/working/train_photos') if f.endswith(('.jpg', '.jpeg', '.png')) and not f.startswith(('._'))]\n\n# Randomly sample 8 images\nimgs_samples = random.sample(image_files, min(8, len(image_files)))\n\n# Plot random sample of 8 images\nplt.figure(figsize=(15, 10))\n\nfor i, img_file in enumerate(imgs_samples):\n    # Get the image path\n    img_path = os.path.join('/kaggle/working/train_photos', img_file)\n\n    # Try to read the image\n    img = cv2.imread(img_path)\n\n    # Check if the image is successfully loaded\n    if img is not None:\n        # Convert color channels to RGB\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        # Grab image's business ID and labels (assuming you have these variables defined)\n        business = train_photo_to_biz_ids.loc[train_photo_to_biz_ids['photo_id'] == int(img_file[:-4]), 'business_id']\n        labels = train_data.loc[train_data['business_id'] == business.values[0], 'labels']\n\n        # Annotate each image with image ID, business ID, and labels\n        title = \"Image ID: \" + img_file + ' Business: ' + str(business.values[0]) + '\\nLabels: ' + ''.join(labels.values)\n\n         # Plot the image\n        plt.subplot(2, 4, i+1)\n        plt.tight_layout(pad=0.4, w_pad=0.5, h_pad=1.0)\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(title)\n    else:\n        print(f\"Error: Unable to read image: {img_path}\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:07:13.878742Z","iopub.execute_input":"2023-11-28T08:07:13.879452Z","iopub.status.idle":"2023-11-28T08:07:16.351809Z","shell.execute_reply.started":"2023-11-28T08:07:13.879418Z","shell.execute_reply":"2023-11-28T08:07:16.350910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.Data Preparation","metadata":{}},{"cell_type":"markdown","source":"จัดเตรียมข้อมูลเพื่อนำมา train โดยโหลดรูปภาพจากไฟล์ในโฟลเดอร์ 'train_photos' และนำมาเก็บไว้ในตัวแปร X  และทำการแปลงรูปภาพให้อยู่ในรูปแบบของ NumPy array เพื่อให้สามารถนำมาใช้ในการฝึกโมเดลได้","metadata":{}},{"cell_type":"code","source":"# resize ให้มีขนาด รูปภาพให้ขนาดเท่ากัน\nimg_width = 200\nimg_height = 200\nnum_images = 10000\n\nX = []\n\nfor i in tqdm(range(num_images)):\n    photo_id = data['photo_id'].iloc[i]\n    path = 'train_photos/' + str(photo_id) + '.jpg'\n    \n    img = image.load_img(path, target_size = (img_width, img_height))\n    img = image.img_to_array(img)\n    img = img/255.0\n    X.append(img)\n\nX = np.array(X)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:10.941419Z","iopub.execute_input":"2023-11-28T08:03:10.941706Z","iopub.status.idle":"2023-11-28T08:03:37.458187Z","shell.execute_reply.started":"2023-11-28T08:03:10.941682Z","shell.execute_reply":"2023-11-28T08:03:37.457336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### One-Hot Encoding \nใช้  One-Hot Encoding เพื่อแปลงข้อมูลประเภทของรูปภาพอาหาร ที่มีอยู่ในประเภทอาหารนั้นๆ ให้กลายเป็น vector มีค่า 1,0 เพื่อให้โมเดล Machine Learning ทำงานง่ายขึ้น","metadata":{}},{"cell_type":"code","source":"# ลบคอลัมน์ที่ไม่ต้องการ\ny = data.drop(['photo_id', 'labels', 'business_id', 'labs'], axis=1)\n# ข้อมูล\n# Define label names\nlabel_names = ['good_for_lunch', 'good_for_dinner', 'takes_reservations',\n               'outdoor_seating', 'restaurant_is_expensive', 'has_alcohol',\n               'has_table_service', 'ambience_is_classy', 'good_for_kids']\n\n\n# ใช้ MultiLabelBinarizer\nmlb = MultiLabelBinarizer()\none_hot_labels = mlb.fit_transform(y['labsint'])\n\ny = pd.DataFrame(one_hot_labels, columns=label_names)\n# สร้าง DataFrame ใหม่\ny\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:37.459317Z","iopub.execute_input":"2023-11-28T08:03:37.459590Z","iopub.status.idle":"2023-11-28T08:03:37.501885Z","shell.execute_reply.started":"2023-11-28T08:03:37.459565Z","shell.execute_reply":"2023-11-28T08:03:37.500989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.Modeling CNN Model","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:37.503256Z","iopub.execute_input":"2023-11-28T08:03:37.503979Z","iopub.status.idle":"2023-11-28T08:03:40.293872Z","shell.execute_reply.started":"2023-11-28T08:03:37.503941Z","shell.execute_reply":"2023-11-28T08:03:40.292877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"กำหนด เลเยอร์ต่างๆ ให้กับ โมเดล CNN ที่เราได้นำมาใช้","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, (7,7), activation='relu', input_shape = X_train[0].shape))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(4,4))\n\n\nmodel.add(Conv2D(64, (7,7), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(4,4))\n\n\n\nmodel.add(Flatten())\n\nmodel.add(Dense(64, activation='relu'))\nmodel.add(BatchNormalization())\n\n\n\nmodel.add(Dense(9, activation='relu'))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:40.295120Z","iopub.execute_input":"2023-11-28T08:03:40.295417Z","iopub.status.idle":"2023-11-28T08:03:43.216272Z","shell.execute_reply.started":"2023-11-28T08:03:40.295390Z","shell.execute_reply":"2023-11-28T08:03:43.214684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train โมเดล","metadata":{}},{"cell_type":"code","source":"\n# 1. สร้าง Adam optimizer โดยกำหนด learning_rate=0.01\nopt = Adam(learning_rate=0.001)\n\n# 2. Compile โมเดลโดยใช้ Adam optimizer ที่สร้างขึ้น\nmodel.compile(optimizer=opt, loss='binary_crossentropy', metrics=['accuracy'])\n\n# 3. Fit โมเดลด้วยข้อมูลการฝึกและการทดสอบ\nhistory = model.fit(X_train, y_train, epochs=20, validation_data=(X_test, y_test))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:03:43.217684Z","iopub.execute_input":"2023-11-28T08:03:43.217940Z","iopub.status.idle":"2023-11-28T08:06:24.071961Z","shell.execute_reply.started":"2023-11-28T08:03:43.217918Z","shell.execute_reply":"2023-11-28T08:06:24.071121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_score, val_acc = model.evaluate(X_test, y_test, verbose=0)\ntrain_score,train_acc = model.evaluate(X_train, y_train, verbose=0)\nprint('Validation score:', val_score,'Validation accuracy:', val_acc)\nprint('Train score:', train_score,'   Train accuracy:', train_acc)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:06:24.073261Z","iopub.execute_input":"2023-11-28T08:06:24.074314Z","iopub.status.idle":"2023-11-28T08:06:43.306915Z","shell.execute_reply.started":"2023-11-28T08:06:24.074287Z","shell.execute_reply":"2023-11-28T08:06:43.305855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"กราฟแสดงค่า accuracy และ loss ของโมเดลในระหว่างการฝึก (training) และการทดสอบ (validation) ในแต่ละ epoch หลังจากการฝึกโมเดล","metadata":{}},{"cell_type":"code","source":"def plot_learningCurve(history):\n    # Plot training & validation accuracy values\n    plt.plot(history['accuracy'])\n    plt.plot(history['val_accuracy'])\n    plt.title('Model accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Val'], loc='upper left')\n    plt.show()\n\n    # Plot training & validation loss values\n    plt.plot(history['loss'])\n    plt.plot(history['val_loss'])\n    plt.title('Model loss')\n    plt.ylabel('Loss')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Val'], loc='upper left')\n    plt.show()\n\n# ใช้งานฟังก์ชัน\nplot_learningCurve(history.history)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:06:43.308229Z","iopub.execute_input":"2023-11-28T08:06:43.308532Z","iopub.status.idle":"2023-11-28T08:06:44.194578Z","shell.execute_reply.started":"2023-11-28T08:06:43.308505Z","shell.execute_reply":"2023-11-28T08:06:44.191728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)\ny_pred = np.argmax(y_pred, axis=1)\n\ny_true = np.argmax(y_test, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:06:44.195675Z","iopub.execute_input":"2023-11-28T08:06:44.195964Z","iopub.status.idle":"2023-11-28T08:06:47.253324Z","shell.execute_reply.started":"2023-11-28T08:06:44.195939Z","shell.execute_reply":"2023-11-28T08:06:47.252299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(classification_report(y_true, y_pred, target_names= label_names))","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:06:47.254649Z","iopub.execute_input":"2023-11-28T08:06:47.255023Z","iopub.status.idle":"2023-11-28T08:06:47.278928Z","shell.execute_reply.started":"2023-11-28T08:06:47.254990Z","shell.execute_reply":"2023-11-28T08:06:47.277900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"สุ่มแสดงค่าที่ได้จากการใช้โมเดลทำนาย","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing import image\nimport numpy as np\nimport os\n\nimage_directory = '/kaggle/working/test_photos'\n\n# Assuming you have already defined img_width and img_height\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# Shuffle the list of image files\nnp.random.shuffle(image_files)\n\n# Limit the number of images to predict\nnum_images_to_predict = 10\n\nfor i in range(num_images_to_predict):\n    file = image_files[i]\n    img_path = os.path.join(image_directory, file)\n    \n    # Update the target size to match the input shape of your model\n    img = image.load_img(img_path, target_size=(200, 200))\n\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array /= 255.0  # Normalize pixel values to be between 0 and 1\n\n    # ทำนาย labels\n    predictions = model.predict(img_array)\n\n    # นำ labels ที่ทำนายได้มาเรียกตามลำดับ\n    label_names = ['good_for_lunch', 'good_for_dinner', 'takes_reservations',\n                   'outdoor_seating', 'restaurant_is_expensive', 'has_alcohol',\n                   'has_table_service', 'ambience_is_classy', 'good_for_kids']\n\n    predicted_labels = [label_names[i] for i, pred in enumerate(predictions[0]) if pred > 0.5]\n\n    # แสดงรูปภาพ\n    plt.imshow(img)\n    plt.show()\n\n    # แสดง labels ที่ทำนายได้\n    print(\"Predicted labels:\", predicted_labels)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T08:06:47.280300Z","iopub.execute_input":"2023-11-28T08:06:47.280645Z","iopub.status.idle":"2023-11-28T08:06:51.338436Z","shell.execute_reply.started":"2023-11-28T08:06:47.280614Z","shell.execute_reply":"2023-11-28T08:06:51.337521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.Project Summary","metadata":{}},{"cell_type":"markdown","source":"จากการที่พวกเราทำโมเดลออกมา เราได้ค่า accuracy ประมาณ 0.1 ซึ่งเป็นค่าที่น้อยมาก แสดงให้เห็นว่า โมเดลของเรายังทำนายได้ไม่แม่นยำพอ และเราจึงทำการสร้างโมเดลเพิ่มคือ ResNet50 ผลปรากฎว่า ได้ค่า accuracy ประมาณ 0.04 ซึ่งยังได้ค่า acuuracy ที่ต่ำอยู่เช่นเดียวกัน ","metadata":{}},{"cell_type":"markdown","source":"พวกเราได้ทดลองทำอีกโมเดล คือ Model ResNet50\n\nจากการที่สร้างโมเดลมาทำนาย พบว่า Model CNN ทำนายได้ดีกว่า เนื่องจากได้ค่า accuracy มากกว่า ResNet50\n\nคลิกที่นี่เพื่อดูโมเดลที่ 2 Model [ResNet50](https://www.kaggle.com/code/nnayjya/dsi206-model-resnet50)","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color: #D9B0CD; color: white; border-radius: 20px; height:50px\">\n     <center><h1 style=\"display:block; padding:7px\"> Reference </h1></center>\n </div>\n ","metadata":{}},{"cell_type":"markdown","source":"ตัวอย่าง ModelCNN\n\nMulti-Label Image Classification in Python//7 September 2020// Models CNN//สืบค้นเมื่อ 25 พ.ย 2566//จาก [https://kgptalkie.com/multi-label-image-classification-on-movies-poster-using-cnn/](http://)\n\nสำรวจชุดข้อมูล\n\nData Exploration Yelp Classification// 2018 // Data Exploration //สืบค้นเมื่อ 25 พ.ย 2566//จาก [https://www.kaggle.com/code/enerrio/data-exploration-yelp-classification](http://)","metadata":{}}]}