{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  <center> Yelp Restaurant Photo Classification <center>","metadata":{}},{"cell_type":"markdown","source":"### **เป้าหมายในการศึกษา**\n\n#### \"Use less quantity of train data set but get high accuracy score\"\n1. จากจำนวนชุดข้อมูลทั้งหมด **239,000 datasets และ 9 features** เป้าหมายของเราคือการลดขนาดของชุดข้อมูลที่ใช้ฝึกแบบจำลอง  เนื่องด้วยข้อจำกัดด้านฮาร์ดแวร์ที่มีอยู่จำกัด การใช้แบบจำลองที่ต้องใช้ชุดข้อมูลขนาดใหญ่จึงต้องใช้ทรัพยากรจำนวนมากและใช้เวลานานในการประมวลผล\n2. เปรียบเทียบ **Accuracy score และเวลา**ที่ใช้ในการประมวลผลของแต่ละแบบจำลองที่ใช้จากการลดขนาดของชุดข้อมูล\n\n### **ชุดข้อมูลทั้งหมดที่ใช้การฝึกแบบจำลอง**\n1. train_photos : Directory ชุดข้อมูลสำหรับ Train\n2. test_photos : Directory ชุดข้อมูลสำหรับ Test\n3. sample_submission.csv : ไฟล์ตัวอย่างการส่ง submission\n4. train_photo_to_biz_ids.csv : ไฟล์","metadata":{}},{"cell_type":"markdown","source":"### **สมาชิกกลุ่ม** \n* 6524650030  นายธนารักษ์ ลีนนานนท์ \n* 6524650071  นายวัชรนันท์ พันมูล \n* 6524650089  นายศิรภพ จุลละภมร \n* 6524651012  นางสาวนริศรา กรวิรัตน์ \n* 6524651061  นายจีรัฎ โนดไธสง \n* 6524651277  นายนวัตกรณ์ แสงศิลา","metadata":{}},{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np; np.random.seed(101) \nfrom PIL import Image, ImageFilter\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n%matplotlib inline\nimport pandas as pd\nfrom itertools import chain\nfrom collections import Counter\nimport random\nfrom keras.layers import Dense,Flatten, Input\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom tqdm import tqdm # Enable progress bar\nfrom keras.applications.resnet50 import ResNet50, preprocess_input, decode_predictions # Load pre-trained model\nfrom keras.models import Model\nfrom keras.preprocessing import image\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-27T18:21:32.757883Z","iopub.execute_input":"2023-11-27T18:21:32.758719Z","iopub.status.idle":"2023-11-27T18:21:32.774727Z","shell.execute_reply.started":"2023-11-27T18:21:32.758683Z","shell.execute_reply":"2023-11-27T18:21:32.773709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### แตกไฟล์ด้วย pigz\n* เหตุผลที่เลือกใช้ pigz ในการแตกไฟล์ tgz เนื่องจากเป็นแพ็กเกจที่มีประสิทธิภาพในการแตกไฟล์ tgz ได้อย่างรวดเร็วและใช้ทรัพยากรน้อยที่สุด","metadata":{}},{"cell_type":"code","source":"%%capture\n# extract files\n!apt install pigz pv -y\n!pip install -U sentence-transformers \n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz | tar xf -","metadata":{"execution":{"iopub.status.busy":"2023-11-27T18:22:02.195061Z","iopub.execute_input":"2023-11-27T18:22:02.195725Z","iopub.status.idle":"2023-11-27T18:26:03.976893Z","shell.execute_reply.started":"2023-11-27T18:22:02.195690Z","shell.execute_reply":"2023-11-27T18:26:03.975568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  <center> Step 1 : Data preparing <center>\n    \n**ขั้นตอนนี้จะทำการวิเคราะห์ข้อมูลและเตรียมข้อมูลสำหรับการนำไปใช้กับแบบจำลอง**\n    \n* train_photos : Directory ของรูปภาพสำหรับ Train\n* test_photos : Directory ของรูปภาพสำหรับ Test\n* sample_submission.csv : ไฟล์ตัวอย่างการส่ง submission\n* train_photo_to_biz_ids.csv : ไฟล์","metadata":{}},{"cell_type":"code","source":"# Paths of train & test photos\ntrain_path = \"/kaggle/working/train_photos/\"\ntest_path = \"/kaggle/working/test_photos/\"\n\n# Paths of CSV files\ntrain_pid_bid = '/kaggle/working/train_photo_to_biz_ids.csv'\ntrain_bid_label = '/kaggle/working/train.csv'\ntest_pid_bid = '/kaggle/working/test_photo_to_biz.csv'\n\n# Make dataframes \nraw_train_photos = pd.read_csv(train_pid_bid)\nraw_train_label = pd.read_csv(train_bid_label)\nraw_test_photos = pd.read_csv(test_pid_bid)\n\n# Labels dictionary\nlabel_notation = {0: 'good_for_lunch', 1: 'good_for_dinner', 2: 'takes_reservations',  3: 'outdoor_seating',\n                  4: 'restaurant_is_expensive', 5: 'has_alcohol', 6: 'has_table_service', 7: 'ambience_is_classy',\n                  8: 'good_for_kids'}","metadata":{"execution":{"iopub.status.busy":"2023-11-27T18:26:03.979701Z","iopub.execute_input":"2023-11-27T18:26:03.980213Z","iopub.status.idle":"2023-11-27T18:26:04.488182Z","shell.execute_reply.started":"2023-11-27T18:26:03.980174Z","shell.execute_reply":"2023-11-27T18:26:04.487134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_photos.groupby('business_id').count().describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T18:26:04.504727Z","iopub.execute_input":"2023-11-27T18:26:04.505052Z","iopub.status.idle":"2023-11-27T18:26:04.527616Z","shell.execute_reply.started":"2023-11-27T18:26:04.505026Z","shell.execute_reply":"2023-11-27T18:26:04.526623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_test_photos.groupby('business_id').count().describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T18:26:04.528899Z","iopub.execute_input":"2023-11-27T18:26:04.529184Z","iopub.status.idle":"2023-11-27T18:26:04.692622Z","shell.execute_reply.started":"2023-11-27T18:26:04.529160Z","shell.execute_reply":"2023-11-27T18:26:04.691322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = raw_train_photos\nX_train = pd.concat( (X_train.groupby(['business_id'], as_index=False).nth(0),X_train.groupby(['business_id'], as_index=False).nth(1),X_train.groupby(['business_id'], as_index=False).nth(2)), axis=0 , ignore_index=True)\nX_train = pd.merge(X_train, raw_train_label, how='left', on='business_id')\ndisplay(X_train)\nX_train = pd.concat((X_train.groupby(['labels'], as_index=False).nth(0), X_train.groupby(['labels'], as_index=False).last()), axis=0, ignore_index=True)\n\ny_train = X_train[['business_id', 'labels']]\nX_train = X_train.drop(['labels'],axis=1)\nX_test = raw_test_photos.groupby(['business_id'], as_index=False).first()\nid_test = X_test[\"business_id\"]\n\nprint(len(X_train), len(y_train), len(X_test), len(id_test))\ndisplay(X_train)\ndisplay(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"### แสดงผลรูปภาพใน **business_ids** เพื่อดูตัวอย่างรูปภาพภายในชุดข้อมูล","metadata":{}},{"cell_type":"code","source":"# เลือก business_ids ที่ไม่ซ้ำกันจำนวน 5 รูปภาพ โดยการสุ่มจาก business_ids ที่ไม่ซ้ำกันที่มีอยู่ในชุดข้อมูล (`X_train`)\nsample_business_ids = random.sample(X_train['business_id'].unique().tolist(), 5)\n\n# แสดงรูปภาพสำหรับ business_ids แต่ละรายการ\nfor business_id in sample_business_ids:\n    # เลือก photo_id ทั้งหมดที่เกี่ยวข้องกับ business_id จากชุดข้อมูล\n    all_images = X_train[X_train['business_id'] == business_id]['photo_id']\n    \n    # กำหนดจำนวนภาพที่ต้องการแสดงผล ซึ่งก็คืออย่างน้อย 5 ภาพและจำนวนภาพที่เกี่ยวข้องกับ business_id \n    num_images_to_display = min(5, len(all_images))\n    \n    # สุ่มเลือก subset ของรูปภาพจากรูปภาพทั้งหมดที่เกี่ยวข้องกับ business_id โดยพิจารณาจากจำนวนรูปภาพที่กำหนดให้แสดงผล\n    sample_images = random.sample(all_images.tolist(), num_images_to_display)\n    \n    # แสดงรปภาพที่อย่ใน business_id \n    plt.figure(figsize=(15, 3))\n    for i, img_path in enumerate(sample_images):\n        plt.subplot(1, num_images_to_display, i + 1)\n        img = mpimg.imread(train_path+str(img_path)+'.jpg')\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(f'Business ID: {business_id}')\n    \n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"### **Split dataframe**\n\nเพื่อให้เห็นภาพรวมของโครงสร้างและเนื้อหาของชุดข้อมูลการฝึกอบรมและการทดสอบหลังการแยก และแสดงถึงรูปแบบของชุดข้อมูลก่อนที่จะประมวลผลหรือวิเคราะห์เพิ่มเติม","metadata":{}},{"cell_type":"code","source":"# คัดลอกและรีเซ็ต index ของ DataFrames\ntrain_photos = X_train.copy().reset_index(drop=True)\ntrain_label = y_train.copy().reset_index(drop=True)\ntest_photos = X_test.copy().reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แสดงข้อมูลเกี่ยวกับ dataframes `train_photos`, `train_label` และ `test_photos`\nprint(\"Train Photos:\", len(train_photos), len(train_photos.columns))\ndisplay(train_photos.head())\nprint(\"Train Label:\", len(train_label), len(train_label.columns))\ndisplay(train_label.head())\nprint(\"Test Photos:\", len(test_photos), len(test_photos.columns))\ndisplay(test_photos.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"### **แสดงรูปภาพตัวอย่าง ของชุดข้อมูลที่จะนำไปใช้ในการฝึกแบบจำลอง**","metadata":{}},{"cell_type":"code","source":"# `train_merge` คือ dataframe ใหม่ที่รวมคอลัมน์จาก `train_photos` และ `train_label` ตาม 'business_id'\ntrain_merge = pd.merge(train_photos,train_label, on='business_id',how='left')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"################\nprint(\"Train...\")\n################\n\n# แสดงตารางรูปภาพ 25 ภาพจากชุดข้อมูลที่ใช้ในการฝึกอบรม\nplt.rcParams['figure.figsize'] = (10.0, 10.0)\nplt.subplots_adjust(wspace=0, hspace=0)\n\nfor x in range(25):\n        plt.subplot(5, 5, x+1)\n        im = Image.open(train_path + str(train_photos.photo_id[x]) + '.jpg')\n        im = im.resize((192, 192), Image.LANCZOS)\n        plt.imshow(im)\n        plt.axis('off')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"################      \nprint(\"Test...\")\n################\n\n# แสดงตารางรูปภาพ 25 ภาพจากชุดข้อมูลที่ใช้ในการทดสอบ\nplt.rcParams['figure.figsize'] = (10.0, 10.0)\nplt.subplots_adjust(wspace=0, hspace=0)\n\nfor x in range(25):\n        plt.subplot(5, 5, x+1)\n        im = Image.open(test_path +  str(test_photos.photo_id[x]) + '.jpg')\n        im = im.resize((192, 192), Image.LANCZOS)\n        plt.imshow(im)\n        plt.axis('off')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แสดง 9 รูปภาพที่ไม่ซ้ำกันในแต่ละ label\nfor l in label_notation:\n    unique_images = set()  # เก็บรายการภาพที่ไม่ซ้ำกัน\n    plt.rcParams['figure.figsize'] = (6.0, 6.0)\n    plt.subplots_adjust(wspace=0, hspace=0)\n    fig = plt.figure()\n    fig.suptitle(label_notation[l])\n    \n    for idx, row in train_merge[train_merge['labels'].str.contains(str(l))==True].iterrows():\n        business_id = row['business_id']\n        photo_id = row['photo_id']\n        if business_id not in unique_images:\n            unique_images.add(business_id)\n            \n            im = Image.open(train_path + str(photo_id) + '.jpg')\n            im = im.resize((192, 192), Image.LANCZOS)\n            plt.subplot(3, 3, len(unique_images))\n            plt.imshow(im)\n            plt.axis('off')\n\n        # ถ้าเก็บภาพที่ไม่ซ้ำกันได้ 9 รูปเสร็จแล้ว ให้หยุดดำเนินการ \n        if len(unique_images) >= 9:\n            break\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"### **Extract bottlenect features**\n* ใช้โมเดล ResNet50 เพื่อสกัด bottleneck feature ของรูปภาพ ซึ่งเป็นลักษณะที่ถูกเลือกมาโดยโมเดล ResNet50 เพื่อทำนายไฟล์รูปภาพที่กำลังถูกป้อนเข้ามาในโมเดล ResNet50\n\n<br>\nอ้างอิงข้อมูลจาก : https://github.com/rouille/YelpKaggle/blob/master/bottleneckFeaturesExtraction.ipynb","metadata":{}},{"cell_type":"code","source":"# ใช้ ResNet50 เพื่อแยกคุณสมบัติ bottleneck features\nResNet_model = ResNet50(weights='imagenet', include_top=False,pooling=\"avg\")\n\n# สร้าง feature extractor ตามแบบจำลองที่ได้รับการ pre-trained model\ninput = Input(shape=(224, 224, 3), name='image_input')\nfeature_extractor = ResNet_model(input)\nflattener = Flatten()(feature_extractor)\n\n# เพิ่มชั้น Dense ที่มีหน่วยน้อยลงเพื่อลดขนาดของมิติ\n# reduced_features = Dense(128, activation='relu')(flattener)\n\nbottleneck_feature_extractor = Model(inputs=input, outputs=flattener)\n\n# Data augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=25,\n    width_shift_range=0.4,\n    height_shift_range=0.4,\n    shear_range=0.3,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    preprocessing_function=preprocess_input\n)\n\n# Empty arrays สำหรับจัดเก็บ feature ที่แยกออกมา\nX_train = []\nX_test = []\n\n# แยก bottleneck features ของภาพถ่ายเพื่อการฝึกแบบจำลอง\nfor i in tqdm(range(len(train_photos))):\n    img_path = train_path + str(train_photos.photo_id[i]) + '.jpg'\n    img = image.load_img(img_path, target_size=(224, 224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n    \n    img_gen = datagen.flow(x, batch_size=1)\n    bottleneck_features_train_raw = bottleneck_feature_extractor.predict(img_gen.next())\n    bottleneck_features_train_reduced = bottleneck_features_train_raw.squeeze()\n    X_train.append(bottleneck_features_train_reduced)\n\n# แยก bottleneck features ของภาพถ่ายเพื่อทดสอบแบบจำลอง\nfor i in tqdm(range(len(test_photos))):\n    img_path = test_path + str(test_photos.photo_id[i]) + '.jpg'\n    img = image.load_img(img_path, target_size=(224, 224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n    \n    img_gen = datagen.flow(x, batch_size=1)\n    bottleneck_features_test_raw = bottleneck_feature_extractor.predict(img_gen.next())\n    bottleneck_features_test_reduced = bottleneck_features_test_raw.squeeze()\n    X_test.append(bottleneck_features_test_reduced)\n","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# <center> **Step 2: Features Aggregation**  <center>\n    \n* ทำ data augmentation ด้วย ImageDataGenerator หมุนภาพ,เลื่อนรูปภาพ,ซูมรูปภาพ,พลิกภาพไปแนวนอน เพื่อทำให้มีความสามารถในการรับมือกับความแปรปรวนของข้อมูลได้ดีขึ้น\n* ในส่วนนี้เพิ่ม features ที่ได้มาจากการ extract เข้าไปใน dataset ทำการ group by business_id และ รวมค่า feature ของแต่ละภาพใน business_id เพื่อคำนวนค่าเฉลี่ย ในขั้นตอนนี้เราจะได้ feature เฉลี่ยของแต่ละ \n    \n<br>\n    อ้างอิงข้อมูลจาก : https://journalofbigdata.springeropen.com/articles/10.1186/s40537-021-00444-8\n","metadata":{}},{"cell_type":"code","source":"# สำหรับ extracted features ที่ได้ นำมาสร้างชุดข้อมูลสำหรับฝึกแบบจำลองใหม่โดยเชื่อมต่อกับ photo_id ในชุดข้อมูลสำหรับฝึกแบบจำลองเดิม \nbottleneck_features_train = pd.DataFrame({'features' : X_train})\ntrain_merge = pd.concat([bottleneck_features_train, train_photos], axis=1)\n\n# สำหรับ extracted features ที่ได้ นำมาสร้างชุดข้อมูลสำหรับทดสอบแบบจำลองใหม่โดยเชื่อมต่อกับ photo_id ในชุดข้อมูลสำหรับทดสอบแบบจำลอง\nbottleneck_features_test = pd.DataFrame({'features' : X_test})\ntest_merge = pd.concat([bottleneck_features_test, test_photos], axis=1)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# หาค่าเฉลี่ยทุก features ของรูปภาพในชุดข้อมูลสำหรับฝึกแบบจำลองที่ถูกกำหนดให้กับ business_id เดียวกัน\ntrain_bid_features = pd.DataFrame(train_merge.groupby('business_id')['features'].apply(np.mean))\ntrain_bid_features.reset_index(level=0, inplace=True)\n\n# หาค่าเฉลี่ยทุก features ของรูปภาพในชุดข้อมูลสำหรับทดสอบแบบจำลองที่ถูกกำหนดให้กับ business_id เดียวกัน\ntest_bid_features = pd.DataFrame(test_merge.groupby('business_id')['features'].apply(np.mean))\ntest_bid_features.reset_index(level=0, inplace=True)\n\n# เช็คว่าตารางข้อมูลถูกสร้างอย่างถูกต้องหรือไม่\ntest_bid_features.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# จัดเรียง Label ตาม business_id ตามลำดับจากน้อยไปมาก จากนั้นรวม Label เข้ากับ mean feature ของแต่ละ business_id\ntrain_label = train_label.sort_values('business_id')\ntrain_label = train_label.reset_index(drop=True)\ntrain_bid_features['labels'] = train_label['labels']\n\n# ค้นหาค่า NaN และลบแถวออก\nnan_label_indices = pd.isnull(train_bid_features).any().to_numpy().nonzero()[0]\ntrain_bid_features = train_bid_features.drop(train_bid_features.index[list(nan_label_indices)])\n\n# ตรวจสอบว่า data frame  ถูกสร้างอย่างถูกต้องหรือไม่\ntrain_bid_features.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center> **Step 3: Multi-label classification using a SVM classifier**  <center>\n* เป็นการแปลงคุณสมบัติและ Label ให้อยู่ในรูปแบบที่เหมาะสมสำหรับการฝึกแบบจำลอง Support Vector Machine \n* คุณสมบัติจะถูกเก็บไว้ในอาร์เรย์ NumPy (`X_train` และ `X_test`) และป้ายกำกับจะถูกเก็บไว้ในอาร์เรย์ NumPy (`Y_train`)","metadata":{}},{"cell_type":"code","source":"# นำเข้าชุดข้อมูลที่ทำการ manipulated แล้ว\ntrain_df = train_bid_features\ntest_df  = test_bid_features","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ฟังก์ชั่นสร้าง Label ใน data frame ให้เป็นรายการ (เช่น 0 8 => [0, 8])\ndef labels_to_list(labels): return list(map(int, labels.split()))\n\nmax_length = max(train_df['labels'].apply(lambda x: len(labels_to_list(x))))\n\n# แปลงฟีเจอร์ในชุดข้อมูล Train (`train_df`) เป็น NumPy Array (`X_train`)\nX_train = np.array([x for x in train_df['features']])\nY_train = np.array([labels_to_list(y) for y in train_df['labels']], dtype=object)\nX_test = np.array([x for x in test_df['features']])\n\nprint(\"X_train: \", X_train.shape)\nprint(\"Y_train: \", Y_train.shape)\nprint(\"X_test: \", X_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **SVM (Support Vector Machine)** \n* SVM เหมาะกับ Dataset ที่มี Feature จำนวนมากแต่มีปริมาณข้อมูลน้อยถึงปานกลาง เพื่อให้สอดคล้องกับเป้าหมายในการศึกษา\n* หากเลือกใช้งานอัลกอริทึม SVM กับชุดข้อมูลที่มีขนาดใหญ่ เวลาที่ใช้ในการฝึก (Training Time) จะเพิ่มขึ้นและอาจส่งผลลบต่อประสิทธิภาพของอัลกอริทึม\n\n<br>\nอ้างอิงข้อมูลจาก : https://web.stanford.edu/class/cs231a/prev_projects_2016/Final_Report%20%281%29.pdf\n\n            \n","metadata":{}},{"cell_type":"code","source":"# โหลด packages สำหรับการแยก train & validation set, SVM classifier, 1-of-K encoder\nfrom sklearn import svm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\n# Package สำหรับการแสดงผลเวลาที่ใช้\nimport time; t=time.time()\n\n# แปลงรายการที่อยู่ใน Label ให้เป็นไปตามรูปแบบ 1-of-K coding scheme\none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แยกชุดข้อมูลออกเป็น train set into 8:2 (train : validation)\nrandom_state = np.random.RandomState(101)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ทำการฝึกแบบจำลองด้วยโมเดล SVM classifier\nclassifier = OneVsRestClassifier(svm.SVC(kernel='rbf', probability=True))\nclassifier.fit(X_train_, Y_train_)\n\n# ทำนาย Label โดยใช้แบบจำลองที่ได้จากการฝึกอบรม\nY_predict = classifier.predict(X_test_)\n\n# แสดงผลเวลาที่ใช้ในการทำงาน\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score # นำเข้า package เพื่อใช้ในการคำนวณ F1 score \n\n# แสดงผล global F1 score & on-label F1 score\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## **MLP (Multilayer perceptron)**\n\n* สามารถใช้ในการจำแนกข้อมูลเป็นหลายประเภท โดยการเรียนรู้จากข้อมูลฝึกสอนและการให้ผลลัพธ์เป็นเปอร์เซ็นต์ความน่าจะเป็นสำหรับแต่ละประเภท ซึ่งทำให้ MLP มีประสิทธิภาพในการจำแนกข้อมูลที่มีความซับซ้อนและหลากหลาย \n\n<br>\nอ้างอิงข้อมูลจาก : https://github.com/vaseline555/Yelp-Restaurant-Photo-Classification/tree/master","metadata":{}},{"cell_type":"code","source":"# โหลด packages สำหรับการแยก train & validation set, SVM classifier, 1-of-K encoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.neural_network import MLPClassifier\n\n# Package สำหรับการแสดงผลเวลาที่ใช้\nimport time; t=time.time()\n\n# แปลงรายการที่อยู่ใน Label ให้เป็นไปตามรูปแบบ 1-of-K coding scheme\none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แยกชุดข้อมูลออกเป็น train set into 8:2 (train : validation)\nrandom_state = np.random.RandomState(101)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ทำการฝึกแบบจำลองด้วยโมเดล MLP classifier\nclassifier = MLPClassifier(solver='lbfgs', random_state=random_state, max_iter=100, hidden_layer_sizes=(1024,512,256,128,64,32,))\nclassifier.fit(X_train_, Y_train_)\n\n# ทำนาย Label โดยใช้แบบจำลองที่ได้จากการฝึกอบรม\nY_predict = classifier.predict(X_test_)\n\n# แสดงผลเวลาที่ใช้ในการทำงาน\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score # นำเข้า package เพื่อใช้ในการคำนวณ F1 score \n\n# แสดงผล global F1 score & on-label F1 score\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **XGBoost (Extreme Gradient Boosting)**\n\n* มีความสามารถด้านความเร็วและประสิทธิภาพในการจัดการชุดข้อมูลขนาดใหญ่ และจัดการค่าที่หายไปในชุดข้อมูล\n* ใช้ความสามารถที่แต่ละ decision tree จะเรียนรู้จาก error ของ tree ก่อนหน้า ทำให้ความแม่นยำของในการทำ prediction จะ แม่นยำมากขึ้นเรื่อยๆ เมื่อมีการเรียนรู้ของ tree ต่อเนื่องกันจนมีความลึกมากพอ\n\n<br>\nอ้างอิงข้อมูลจาก : https://github.com/vaseline555/Yelp-Restaurant-Photo-Classification/tree/master","metadata":{}},{"cell_type":"code","source":"# โหลด packages สำหรับการแยก train & validation set, SVM classifier, 1-of-K encoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom xgboost import XGBClassifier\n\n# Package สำหรับการแสดงผลเวลาที่ใช้\nimport time; t=time.time()\n\none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แยกชุดข้อมูลออกเป็น train set into 8:2 (train : validation)\nrandom_state = np.random.RandomState(101)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ทำการฝึกแบบจำลองด้วยโมเดล XGBoost classifier\nclassifier = OneVsRestClassifier(XGBClassifier(num_class=9, gamma=0.024, learning_rate=0.3, max_depth=5, nthread=4, n_estimators=10, objective=\"multi:softmax\"))\nclassifier.fit(X_train_, Y_train_)\n\n# ทำนาย Label โดยใช้แบบจำลองที่ได้จากการฝึกอบรม\nY_predict = classifier.predict(X_test_)\n\n# แสดงผลเวลาที่ใช้ในการทำงาน\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score # นำเข้า package เพื่อใช้ในการคำนวณ F1 score \n\n# แสดงผล global F1 score & on-label F1 score\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **LightGBM**\n\n* มีจุดเด่นที่ความเร็วในการฝึกแบบจำลอง เพื่อให้สอดคล้องกับเป้าหมายในการศึกษา \n* โมเดลจะทำการค้นหาตัวแปรต้นที่ส่งผลอย่างมีนัยสำคัญต่อค่าตัวแปรตามที่สนใจ (ในกรณีนี้คือ photo_id ที่จะอยู่ใในแต่ละ business_id)\n\n<br>\nอ้างอิงข้อมูลจาก : https://www.linkedin.com/pulse/boosting-techniques-battle-catboost-vs-xgboost-lightgbm-uttam-kumar/?trackingId=V2tFgiE7REyfmpUh2QRY6g%3D%3D/?trackingId=V2tFgiE7REyfmpUh2QRY6g==","metadata":{}},{"cell_type":"code","source":"# โหลด packages สำหรับการแยก train & validation set, SVM classifier, 1-of-K encoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.multiclass import OneVsRestClassifier\nimport lightgbm as lgb\nimport time\n\n# แปลงรายการที่อยู่ใน Label ให้เป็นไปตามรูปแบบ 1-of-K coding schemeone_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แยกชุดข้อมูลออกเป็น train set into 8:2 (train : validation)\nrandom_state = np.random.RandomState(101)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ทำการฝึกแบบจำลองด้วยโมเดล LightGBM classifier\nparams = {\n    'objective': 'multiclass',\n    'num_class': 9,\n    'learning_rate': 0.3,\n    'max_depth': 5,\n    'n_estimators': 10,\n    'n_jobs': 4,\n}\n\nclassifier = OneVsRestClassifier(lgb.LGBMClassifier(**params))\nclassifier.fit(X_train_, Y_train_)\n\n# ทำนาย Label โดยใช้แบบจำลองที่ได้จากการฝึกอบรม\nY_predict = classifier.predict(X_test_)\n\n# แสดงผลเวลาที่ใช้ในการทำงาน\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score # นำเข้า package เพื่อใช้ในการคำนวณ F1 score \n\n# แสดงผล global F1 score & on-label F1 score\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **CatBoost**\n\n* ใช้ความสามารถในด้านการจัดการกับข้อมูลแบบหมวดหมู่ (categorical data) \n* สามารถจัดการข้อมูลแบบหมวดหมู่ได้อัตโนมัติ (Categorical data) โดยที่ต้องมีการเข้ารหัส (encoding) ข้อมูลด้วยตัวเอง และสามารถปรับจูนพารามิเตอร์ของโมเดลและอัลกอรึทึมที่เหมาะสมได้โดยอัตโนมัติ \n\n<br>\nอ้างอิงข้อมูลจาก : https://www.linkedin.com/pulse/boosting-techniques-battle-catboost-vs-xgboost-lightgbm-uttam-kumar/?trackingId=V2tFgiE7REyfmpUh2QRY6g%3D%3D/?trackingId=V2tFgiE7REyfmpUh2QRY6g==","metadata":{}},{"cell_type":"code","source":"# โหลด packages สำหรับการแยก train & validation set, 1-of-K encoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom catboost import CatBoostClassifier\nimport time\n\n# แปลงรายการที่อยู่ใน Label ให้เป็นไปตามรูปแบบ 1-of-K coding scheme \nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แยกชุดข้อมูลออกเป็น train set into 8:2 (train : validation)\nrandom_state = np.random.RandomState(101)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ทำการฝึกแบบจำลองด้วยโมเดล CatBoost classifier\nparams = {\n    'iterations': 10,\n    'learning_rate': 0.3,\n    'depth': 5,\n    'loss_function': 'MultiClass',\n    'custom_metric': 'Accuracy',\n    'thread_count': 4,\n    'l2_leaf_reg': 0.024\n}\n\nclassifier = OneVsRestClassifier(CatBoostClassifier(**params))\nclassifier.fit(X_train_, Y_train_)\n\n# ทำนาย Label โดยใช้แบบจำลองที่ได้จากการฝึกอบรม\nY_predict = classifier.predict(X_test_)\n\n# แสดงผลเวลาที่ใช้ในการทำงาน\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")\n","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score # นำเข้า package เพื่อใช้ในการคำนวณ F1 score \n\n# แสดงผล global F1 score & on-label F1 score\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"# แสดงผลระยะเวลาในการดำเนินการ\nt = time.time()\n\n# แปลงรายการที่อยู่ใน Label ให้เป็นไปตามรูปแบบ 1-of-K coding scheme \none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# Train the SVM classifier again with a full train set\nrandom_state = np.random.RandomState(101)\nclassifier = OneVsRestClassifier(svm.SVC(kernel='rbf', probability=True))\nclassifier.fit(X_train, Y_train_)\n\nY_predict = classifier.predict(X_test)\nY_predict_label = one_of_K_encoder.inverse_transform(Y_predict)\n\nprint(\"Time passed: \", \"{0:.1f}\".format(time.time() - t), \"sec\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# สร้างกรอบข้อมูลสำหรับการส่ง (จับคู่ predicted label กับ business_id ในชุดทดสอบ)\nfinal_df = pd.DataFrame(columns=['business_id','labels'])\n\nfor i in range(len(test_df)):\n    biz = test_df.loc[i]['business_id']\n    label = Y_predict_label[i]\n    label = str(label)[1:-1].replace(\",\", \" \")\n    \n    final_df.loc[i] = [str(biz), label]\n\n# กำหนด submission file\nwith open(\"submission.csv\",'w') as file:\n    final_df.to_csv(file, index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r /kaggle/working/test_photos /kaggle/working/train_photos /kaggle/working/test_photo_to_biz.csv /kaggle/working/train.csv /kaggle/working/sample_submission.csv /kaggle/working/train_photo_to_biz_ids.csv","metadata":{},"execution_count":null,"outputs":[]}]}