{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"},{"sourceId":7030332,"sourceType":"datasetVersion","datasetId":4043743},{"sourceId":7030565,"sourceType":"datasetVersion","datasetId":4043891}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DSI206 Project: Yelp Restaurant Photo Classification\n# Introduction\nจุดมุ่งหมาย: เพื่อศึกษาการทำ Image Classification โดย Model สามารถทำนายรูปภาพของร้านอาหารต่างๆว่าร้านอาหารนั้นๆมีลักษณะและคุณสมบัติอย่างไรบ้าง ตาม label ทั้ง 9 attributes\n","metadata":{}},{"cell_type":"markdown","source":"\n1. { 0: good_for_lunch (เหมาะสำหรับอาหารกลางวัน)}\n2. {1: good_for_dinner (เหมาะสำหรับอาหารกลางคืน)}\n3. {2: takes_reservations (สามารถจองได้)}\n4. {3: outdoor_seating (มีที่นั่งด้านนอก)}\n5. {4: restaurant_is_expensive (ราคาแพง)}\n6. {5: has_alcohol (มีขายเครื่องดื่มแอลกอฮอล)}\n7. {6: has_table_service (บริการถึงโต๊ะ)}\n8. {7: ambience_is_classy (ร้านหรูมีระดับ)}\n9. {8: good_for_kids (เหมาะสำหรับเด็ก)}\n\nสมาชิกกลุ่ม:\n1. กชพร สิทธิไพศาล 6524651145\n2. กนกพร สะมะถะธัญกร 6524651152\n3. จิรภิญญา ธนันโสภณ 6524651186\n4. ซุลฮาซาล ยูโซะ 6524651228\n5. ภูมิภัทร จักษุจินดา  6524651335\n6. พัชรดา เชื้อใจ 6524651293\n\n","metadata":{}},{"cell_type":"markdown","source":"# Data Exploration\n- แตกไฟล์เข้ามาใช้ สร้าง dataframe และกำหนดค่า labels\n","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport tarfile\nimport os\n\ndef extract(tgz_file):\n    with tarfile.open(tgz_file, 'r:gz') as tar:\n        members = list(tar)\n        for member in tqdm(members, desc='Extracting'):\n            tar.extract(member)\n\ntry:\n    extract('../input/yelp-restaurant-photo-classification/test_photos.tgz')\n    print ('Done.')\nexcept Exception as e:\n    print(f'Error: {e}')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:10:24.766225Z","iopub.execute_input":"2023-11-27T22:10:24.766844Z","iopub.status.idle":"2023-11-27T22:14:15.251212Z","shell.execute_reply.started":"2023-11-27T22:10:24.766806Z","shell.execute_reply":"2023-11-27T22:14:15.250156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport tarfile\nimport os\n\ndef extract(tgz_file):\n    with tarfile.open(tgz_file, 'r:gz') as tar:\n        members = list(tar)\n        for member in tqdm(members, desc='Extracting'):\n            tar.extract(member)\n\ntry:\n    extract('../input/yelp-restaurant-photo-classification/train_photos.tgz')\n    print ('Done.')\nexcept Exception as e:\n    print(f'Error: {e}')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:14:15.252946Z","iopub.execute_input":"2023-11-27T22:14:15.253233Z","iopub.status.idle":"2023-11-27T22:18:15.382112Z","shell.execute_reply.started":"2023-11-27T22:14:15.253209Z","shell.execute_reply":"2023-11-27T22:18:15.381111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport tarfile\nimport os\n\ntry:\n    extract('../input/yelp-restaurant-photo-classification/train.csv.tgz')\n    extract('../input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz')\n    extract('../input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz')\n    \n    print ('Done.')\nexcept:\n    name = os.path.basename(sys.argv[0])\n    print('error')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:15.383336Z","iopub.execute_input":"2023-11-27T22:18:15.383649Z","iopub.status.idle":"2023-11-27T22:18:15.721431Z","shell.execute_reply.started":"2023-11-27T22:18:15.383623Z","shell.execute_reply":"2023-11-27T22:18:15.720556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> - โชว์ภาพตัวอย่าง test_photos ","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# ระบุที่อยู่ของไดเรกทอรีที่มีไฟล์รูปภาพ\nimage_directory = '/kaggle/working/test_photos'\n\n# ดึงรายชื่อไฟล์รูปภาพทั้งหมดที่มีชื่อขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# กำหนดขนาดของตารางรูปภาพ\nnum_rows = 3\nnum_cols = 3\n\n# สร้าง subplot\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n\n# แสดงรูปภาพทั้งหมดใน subplot\nfor i in range(num_rows):\n    for j in range(num_cols):\n        index = i * num_cols + j\n        if index < len(image_files):\n            image_file = image_files[index]\n            image_path = os.path.join(image_directory, image_file)\n            img = mpimg.imread(image_path)\n            axes[i, j].imshow(img)\n            axes[i, j].axis('off')\n\n# ปรับระยะห่างของ subplot\nplt.subplots_adjust(wspace=0.2, hspace=0.5)\n\n# แสดง subplot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:15.723585Z","iopub.execute_input":"2023-11-27T22:18:15.723882Z","iopub.status.idle":"2023-11-27T22:18:17.737576Z","shell.execute_reply.started":"2023-11-27T22:18:15.723857Z","shell.execute_reply":"2023-11-27T22:18:17.736413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> - โชว์ภาพตัวอย่าง train_photos","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom tqdm.notebook import tqdm\n\n# ระบุที่อยู่ของไดเรกทอรีที่มีไฟล์รูปภาพ\nimage_directory = '/kaggle/working/train_photos'\n\n# ดึงรายชื่อไฟล์รูปภาพทั้งหมดที่มีชื่อขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# กำหนดขนาดของตารางรูปภาพ\nnum_rows = 3\nnum_cols = 3\n\n# สร้าง subplot\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n\n# แสดงรูปภาพทั้งหมดใน subplot\nfor i in range(num_rows):\n    for j in range(num_cols):\n        index = i * num_cols + j\n        if index < len(image_files):\n            image_file = image_files[index]\n            image_path = os.path.join(image_directory, image_file)\n            img = mpimg.imread(image_path)\n            axes[i, j].imshow(img)\n            axes[i, j].axis('off')\n\n# ปรับระยะห่างของ subplot\nplt.subplots_adjust(wspace=0.2, hspace=0.5)\n\n# แสดง subplot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:17.738856Z","iopub.execute_input":"2023-11-27T22:18:17.739166Z","iopub.status.idle":"2023-11-27T22:18:19.166942Z","shell.execute_reply.started":"2023-11-27T22:18:17.739139Z","shell.execute_reply":"2023-11-27T22:18:19.165887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport tarfile\nfrom tqdm import tqdm\n\ndef extract(tar_url, extract_path='.'):\n    print(tar_url)\n    tar = tarfile.open(tar_url, 'r')\n    for item in tqdm(tar):\n        tar.extract(item, extract_path)\n        if item.name.find(\".tgz\") != -1 or item.name.find(\".tar\") != -1:\n            extract(item.name, \"./\" + item.name[:item.name.rfind('/')])","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:19.168257Z","iopub.execute_input":"2023-11-27T22:18:19.168629Z","iopub.status.idle":"2023-11-27T22:18:19.176052Z","shell.execute_reply.started":"2023-11-27T22:18:19.168597Z","shell.execute_reply":"2023-11-27T22:18:19.175124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np; np.random.seed(1040941203) # For reproducibility (+82-10-4094-1203)\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom PIL import Image\nfrom PIL import ImageFilter","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:19.177147Z","iopub.execute_input":"2023-11-27T22:18:19.177465Z","iopub.status.idle":"2023-11-27T22:18:19.954352Z","shell.execute_reply.started":"2023-11-27T22:18:19.177439Z","shell.execute_reply":"2023-11-27T22:18:19.953552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    Step 1. Data exploration & bottleneck features extraction\n\"\"\"\n\n# Paths of train & test photos\ntrain_path = \"/kaggle/working/train_photos\"\ntest_path = \"/kaggle/working/test_photos\"\n\n# Paths of CSV files\ntrain_pid_bid = '/kaggle/working/train_photo_to_biz_ids.csv'\ntrain_bid_label = '/kaggle/working/train.csv'\ntest_pid_bid = '/kaggle/working/test_photo_to_biz.csv'\n\n# Make dataframes \ntrain_photos = pd.read_csv(train_pid_bid)\ntrain_label = pd.read_csv(train_bid_label)\ntrain_id = pd.read_csv(train_pid_bid) \ntest_photos = pd.read_csv(test_pid_bid)\n\n# Labels dictionary\nlabel_notation = {0: 'good_for_lunch', 1: 'good_for_dinner', 2: 'takes_reservations',  3: 'outdoor_seating',\n                  4: 'restaurant_is_expensive', 5: 'has_alcohol', 6: 'has_table_service', 7: 'ambience_is_classy',\n                  8: 'good_for_kids'}","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:19.955361Z","iopub.execute_input":"2023-11-27T22:18:19.955784Z","iopub.status.idle":"2023-11-27T22:18:21.773659Z","shell.execute_reply.started":"2023-11-27T22:18:19.955756Z","shell.execute_reply":"2023-11-27T22:18:21.772563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- โชว์ภาพตัวอย่างของแต่ละ labels\n","metadata":{}},{"cell_type":"code","source":"# Show 9 images belonged to each label\nfor l in label_notation:\n    ids = train_label[train_label['labels'].str.contains(str(l))==True].business_id.tolist()[:9]\n    plt.rcParams['figure.figsize'] = (7.0, 7.0)\n    plt.subplots_adjust(wspace=0, hspace=0)  # ตั้งค่า wspace และ hspace เป็น 0\n    \n    # Create a new figure for each label\n    fig = plt.figure()\n    fig.suptitle(label_notation[l])\n    \n    for x in range(9):\n        file_path = os.path.join(train_path, str(train_photos.photo_id[ids[x]]) + '.jpg')\n        #print(\"File path:\", file_path)  # Add this line to print the file path\n        \n        # Open the image without resizing\n        im = Image.open(file_path)\n        \n        # Create a subplot for each image\n        plt.subplot(3, 3, x+1)\n        plt.imshow(im)\n        plt.axis('off')\nplt.show()\n# Display the figures","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:21.774984Z","iopub.execute_input":"2023-11-27T22:18:21.775285Z","iopub.status.idle":"2023-11-27T22:18:28.069339Z","shell.execute_reply.started":"2023-11-27T22:18:21.775258Z","shell.execute_reply":"2023-11-27T22:18:28.068458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preparation","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm # Enable progress bar\nfrom keras.applications.resnet50 import ResNet50, preprocess_input, decode_predictions # Load pre-trained model\nfrom keras.models import Model\nfrom keras.preprocessing import image\nfrom keras.layers import Flatten, Input\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:28.073115Z","iopub.execute_input":"2023-11-27T22:18:28.073419Z","iopub.status.idle":"2023-11-27T22:18:41.833434Z","shell.execute_reply.started":"2023-11-27T22:18:28.073393Z","shell.execute_reply":"2023-11-27T22:18:41.832557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- อ่านไฟล์ CSV\n\n- แสดง 5 แถวแรก","metadata":{}},{"cell_type":"code","source":"test_photo_to_biz_path = \"/kaggle/working/test_photo_to_biz.csv\"\ndf_test_photo_to_biz = pd.read_csv(test_photo_to_biz_path)\n\nprint(len(df_test_photo_to_biz))\ndf_test_photo_to_biz.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:41.834626Z","iopub.execute_input":"2023-11-27T22:18:41.835194Z","iopub.status.idle":"2023-11-27T22:18:42.207485Z","shell.execute_reply.started":"2023-11-27T22:18:41.835164Z","shell.execute_reply":"2023-11-27T22:18:42.206388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- อ่านไฟล์ CSV \n- แสดง 5 แถวแรก","metadata":{}},{"cell_type":"code","source":"# อ่านไฟล์ CSV\ntrain_photo_to_biz_path = \"/kaggle/working/train_photo_to_biz_ids.csv\"\ndf_train_photo_to_biz = pd.read_csv(train_photo_to_biz_path)\n\n# แสดง 5 แถวแรก\nprint(len(df_train_photo_to_biz))\ndf_train_photo_to_biz.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.208928Z","iopub.execute_input":"2023-11-27T22:18:42.209333Z","iopub.status.idle":"2023-11-27T22:18:42.282913Z","shell.execute_reply.started":"2023-11-27T22:18:42.209295Z","shell.execute_reply":"2023-11-27T22:18:42.281278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Check Duplicate","metadata":{}},{"cell_type":"code","source":"# ตรวจสอบจำนวนข้อมูลที่มี business_id ซ้ำ\nduplicate_counts = df_test_photo_to_biz['business_id'].value_counts()\n\n# แสดงจำนวนข้อมูลที่ business_id ซ้ำ\nprint(len(duplicate_counts))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.283993Z","iopub.execute_input":"2023-11-27T22:18:42.284292Z","iopub.status.idle":"2023-11-27T22:18:42.434323Z","shell.execute_reply.started":"2023-11-27T22:18:42.284262Z","shell.execute_reply":"2023-11-27T22:18:42.433135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- สร้าง DataFrame ที่เก็บตัวอย่างข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว","metadata":{}},{"cell_type":"code","source":"# สร้าง DataFrame ที่เก็บตัวอย่างข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว\ndf_sampled_test = df_test_photo_to_biz.drop_duplicates(subset='business_id', keep='first')\n\n# แสดงจำนวนแถวและข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว\nprint(len(df_sampled_test))\ndf_sampled_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.436164Z","iopub.execute_input":"2023-11-27T22:18:42.436694Z","iopub.status.idle":"2023-11-27T22:18:42.495599Z","shell.execute_reply.started":"2023-11-27T22:18:42.436647Z","shell.execute_reply":"2023-11-27T22:18:42.494572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- บันทึก DataFrame ลงในไฟล์ CSV","metadata":{}},{"cell_type":"code","source":"# บันทึก DataFrame ลงในไฟล์ CSV\noutput_path = '/kaggle/working/test_photo_to_biz_2.csv'\ndf_sampled_test.to_csv(output_path, index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.497086Z","iopub.execute_input":"2023-11-27T22:18:42.497478Z","iopub.status.idle":"2023-11-27T22:18:42.532696Z","shell.execute_reply.started":"2023-11-27T22:18:42.497448Z","shell.execute_reply":"2023-11-27T22:18:42.531693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- สร้าง DataFrame ที่เก็บตัวอย่างข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว\n\n- แสดงจำนวนแถวและข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว","metadata":{}},{"cell_type":"code","source":"# สร้าง DataFrame ที่เก็บตัวอย่างข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว\ndf_sampled_train = df_train_photo_to_biz.drop_duplicates(subset='business_id', keep='first')\n\n# แสดงจำนวนแถวและข้อมูลที่มีแต่ละ business_id มาแค่ 1 ตัว\nprint(len(df_sampled_train))\ndf_sampled_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.533835Z","iopub.execute_input":"2023-11-27T22:18:42.534123Z","iopub.status.idle":"2023-11-27T22:18:42.54755Z","shell.execute_reply.started":"2023-11-27T22:18:42.534099Z","shell.execute_reply":"2023-11-27T22:18:42.54661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- บันทึก DataFrame ลงในไฟล์ CSV","metadata":{}},{"cell_type":"code","source":"# บันทึก DataFrame ลงในไฟล์ CSV\noutput_path = '/kaggle/working/train_photo_to_biz_2.csv'\ndf_sampled_train.to_csv(output_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.549077Z","iopub.execute_input":"2023-11-27T22:18:42.549459Z","iopub.status.idle":"2023-11-27T22:18:42.560834Z","shell.execute_reply.started":"2023-11-27T22:18:42.549424Z","shell.execute_reply":"2023-11-27T22:18:42.559827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Load the CSV file\n\n- Display the first 5 rows of the DataFrame","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ndf_test_photo_to_biz_2 = pd.read_csv('/kaggle/working/test_photo_to_biz_2.csv')\n\n# Display the first 5 rows of the DataFrame\nprint(df_test_photo_to_biz_2)\nprint(df_test_photo_to_biz_2.info())","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.561903Z","iopub.execute_input":"2023-11-27T22:18:42.562146Z","iopub.status.idle":"2023-11-27T22:18:42.594812Z","shell.execute_reply.started":"2023-11-27T22:18:42.562124Z","shell.execute_reply":"2023-11-27T22:18:42.593881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Load the CSV file\n- Display the first 5 rows of the DataFrame","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ndf_train_photo_to_biz_2 = pd.read_csv('/kaggle/working/train_photo_to_biz_2.csv')\ndf_train_photo_to_biz_2['business_id'] = df_train_photo_to_biz_2['business_id'].astype('object')\n\n# Display the first 5 rows of the DataFrame\nprint(df_train_photo_to_biz_2)\nprint(df_train_photo_to_biz_2.info())","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.596122Z","iopub.execute_input":"2023-11-27T22:18:42.596495Z","iopub.status.idle":"2023-11-27T22:18:42.617221Z","shell.execute_reply.started":"2023-11-27T22:18:42.596459Z","shell.execute_reply":"2023-11-27T22:18:42.614984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- สร้างตัวเเแปรเก็บ Path ไฟล์","metadata":{}},{"cell_type":"code","source":"sample_test = '/kaggle/working/test_photo_to_biz_2.csv'\nsample_train = '/kaggle/working/train_photo_to_biz_2.csv'\n\ntest_photos2 = pd.read_csv(sample_test)\ntrain_photos2 = pd.read_csv(sample_train)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.618669Z","iopub.execute_input":"2023-11-27T22:18:42.619104Z","iopub.status.idle":"2023-11-27T22:18:42.638927Z","shell.execute_reply.started":"2023-11-27T22:18:42.619071Z","shell.execute_reply":"2023-11-27T22:18:42.636995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Extract bottleneck features โดยใช้ Resnet50 Model","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport tarfile\nimport os\n# Use ResNet50 model to extract bottleneck features\nResNet_model = ResNet50(weights='imagenet', include_top = False)\n\n# Construct a feature extractor based on pre-trained model\ninput = Input(shape=(224, 224, 3), name='image_input')\nfeature_extractor = ResNet_model(input)\nflattener = Flatten()(feature_extractor)\nbottleneck_feature_extractor = Model(inputs=input, outputs=flattener)\n\n# Empty arrays for storing extracted features\nX_train = []; X_test = []\n\n# Extract bottleneck features of photos for traninig\nfor i in tqdm(range(len(train_photos2))):\n    #img_path = train_path + str(train_photos.photo_id[i]) + '.jpg'\n    img_path = f\"{train_path}/{train_photos2.photo_id[i]}.jpg\"\n    img = image.load_img(img_path, target_size=(224, 224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n\n    bottleneck_features_train_raw = bottleneck_feature_extractor.predict(x, verbose = 0)\n    bottleneck_features_train_reduced =  bottleneck_features_train_raw.squeeze()\n    X_train.append(bottleneck_features_train_reduced)\n\n# Extract bottleneck features of photos for testing    \nfor i in tqdm(range(len(test_photos2))):\n    #img_path = test_path + str(test_photos.photo_id[i]) + '.jpg'\n    img_path = f\"{test_path}/{test_photos2.photo_id[i]}.jpg\"\n    img = image.load_img(img_path, target_size=(224, 224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n    \n    bottleneck_features_test_raw = bottleneck_feature_extractor.predict(x, verbose = 0)\n    bottleneck_features_test_reduced =  bottleneck_features_test_raw.squeeze()\n    X_test.append(bottleneck_features_test_reduced)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:18:42.640054Z","iopub.execute_input":"2023-11-27T22:18:42.640368Z","iopub.status.idle":"2023-11-27T22:34:07.211495Z","shell.execute_reply.started":"2023-11-27T22:18:42.640336Z","shell.execute_reply":"2023-11-27T22:34:07.2105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ลบไฟล์ train_photo_to_biz_ids.csv , test_photo_to_biz.csv เพราะพื้นที่ไม่เพียงพอ","metadata":{}},{"cell_type":"code","source":"import os\n\n# List of file paths to be removed\nfile_paths = [\n    '/kaggle/working/train_photo_to_biz_ids.csv',\n    '/kaggle/working/test_photo_to_biz.csv'\n]\n\n# Remove each file\nfor file_path in file_paths:\n    try:\n        os.remove(file_path)\n        print(f\"File {file_path} has been removed.\")\n    except FileNotFoundError:\n        print(f\"File {file_path} not found.\")\n    except Exception as e:\n        print(f\"An error occurred while removing {file_path}: {e}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:34:07.212581Z","iopub.execute_input":"2023-11-27T22:34:07.212834Z","iopub.status.idle":"2023-11-27T22:34:07.221894Z","shell.execute_reply.started":"2023-11-27T22:34:07.21281Z","shell.execute_reply":"2023-11-27T22:34:07.220932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ลบไฟล์ test_photos , train_photos เพราะพื้นที่ไม่เพียงพอ","metadata":{}},{"cell_type":"code","source":"import shutil\n\n# List of directories to be removed\ndirectories = [\n    '/kaggle/working/test_photos',\n    '/kaggle/working/train_photos'\n]\n\n# Remove each directory\nfor directory in directories:\n    try:\n        shutil.rmtree(directory)\n        print(f\"Directory {directory} has been removed.\")\n    except FileNotFoundError:\n        print(f\"Directory {directory} not found.\")\n    except Exception as e:\n        print(f\"An error occurred while removing {directory}: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:34:07.223026Z","iopub.execute_input":"2023-11-27T22:34:07.223408Z","iopub.status.idle":"2023-11-27T22:34:55.257469Z","shell.execute_reply.started":"2023-11-27T22:34:07.223369Z","shell.execute_reply":"2023-11-27T22:34:55.256338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เก็บค่า bottleneck features train กับ photo_id และ business_id ในไฟล์ train_photo_to_biz.csv โดยทำ dataframe เก็บไว้ในไฟล์ train_merge.csv**\n\n- เก็บค่า bottleneck features test กับ photo_id และ business_id  ในไฟล์ test_photo_to_biz.csv โดยทำ dataframe เก็บไว้ในไฟล์ test_merge.csv**","metadata":{}},{"cell_type":"code","source":"# With extracted features, make a new train set by concatenating with photo_id in an original train set \nbottleneck_features_train = pd.DataFrame({'features' : X_train})\ntrain_merge = pd.concat([bottleneck_features_train, train_photos2], axis=1)\ntrain_merge.to_pickle('train_merge.csv') # Save in pickle for preserving array format and considering its size\n\n# With extracted features, make a new test set by concatenating with photo_id in an original test set \nbottleneck_features_test = pd.DataFrame({'features' : X_test})\ntest_merge = pd.concat([bottleneck_features_test, test_photos2], axis=1)\ntest_merge.to_pickle('test_merge.csv') # Save in pickle for preserving array format and considering its size","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:34:55.258574Z","iopub.execute_input":"2023-11-27T22:34:55.258973Z","iopub.status.idle":"2023-11-27T22:35:12.209069Z","shell.execute_reply.started":"2023-11-27T22:34:55.258936Z","shell.execute_reply":"2023-11-27T22:35:12.20793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- pd.read_pickle('train_merge.csv'): อ่านไฟล์ pickle ที่ชื่อ 'train_merge.csv' และโหลดเนื้อหาลงใน DataFrame ของ Pandas ที่ชื่อ train_merge โดย Pickle เป็นรูปแบบการบันทึกข้อมูลลงไฟล์ที่สามารถนำกลับมาใช้ได้ เป็นวิธีที่ใช้เพื่อบันทึก DataFrame ไว้ก่อนหน้านี้**\n\n- pd.read_pickle('test_merge.csv'): ในทางเดียวกัน, มันอ่านไฟล์ pickle ที่ชื่อ 'test_merge.csv' และโหลดเนื้อหาลงใน DataFrame ของ Pandas ที่ชื่อ test_merge.**\n\n- เช็ค Type ของข้อมูลใน Column ว่าเป็นรูปแบบที่ถูกไหม","metadata":{}},{"cell_type":"markdown","source":"- เช็ค Type ของข้อมูลใน Column ว่าเป็นรูปแบบที่ถูกไหม","metadata":{}},{"cell_type":"code","source":"# Test loading data\ntrain_merge = pd.read_pickle('train_merge.csv')\ntest_merge = pd.read_pickle('test_merge.csv')\n\n# Check if the data frame is loaded normally\nprint(len(test_merge))\ntest_merge.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:12.210562Z","iopub.execute_input":"2023-11-27T22:35:12.21097Z","iopub.status.idle":"2023-11-27T22:35:29.342547Z","shell.execute_reply.started":"2023-11-27T22:35:12.210928Z","shell.execute_reply":"2023-11-27T22:35:29.341482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เช็ค Type ของข้อมูลใน Column ว่าเป็นรูปแบบที่ถูกไหม","metadata":{}},{"cell_type":"code","source":"train_merge.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:29.345266Z","iopub.execute_input":"2023-11-27T22:35:29.346248Z","iopub.status.idle":"2023-11-27T22:35:29.356302Z","shell.execute_reply.started":"2023-11-27T22:35:29.346204Z","shell.execute_reply":"2023-11-27T22:35:29.355241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ทำการเปลี่ยนรูปแบบข้อมูลให้เป็นข้อมูลที่ถูกต้อง","metadata":{}},{"cell_type":"code","source":"train_merge['business_id'] = train_merge['business_id'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:29.357976Z","iopub.execute_input":"2023-11-27T22:35:29.358252Z","iopub.status.idle":"2023-11-27T22:35:29.374386Z","shell.execute_reply.started":"2023-11-27T22:35:29.358223Z","shell.execute_reply":"2023-11-27T22:35:29.373554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ทำการเปลี่ยนรูปแบบข้อมูลให้เป็นข้อมูลที่ถูกต้อง","metadata":{}},{"cell_type":"code","source":"train_merge['photo_id'] = train_merge['photo_id'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:29.382765Z","iopub.execute_input":"2023-11-27T22:35:29.383053Z","iopub.status.idle":"2023-11-27T22:35:29.394113Z","shell.execute_reply.started":"2023-11-27T22:35:29.383029Z","shell.execute_reply":"2023-11-27T22:35:29.393307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เช็ค Type ของข้อมูลใน Column หลังจากทำการเปลี่ยนแล้ว","metadata":{}},{"cell_type":"code","source":"print(len(train_merge))\ntrain_merge.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:29.395223Z","iopub.execute_input":"2023-11-27T22:35:29.395487Z","iopub.status.idle":"2023-11-27T22:35:29.410258Z","shell.execute_reply.started":"2023-11-27T22:35:29.395465Z","shell.execute_reply":"2023-11-27T22:35:29.409303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- หาค่าเฉลี่ยของ features ของรูปภาพที่มี business_id เดียวกัน\n- train_merge.groupby('business_id')['features'].apply(np.mean): กรุ๊ปข้อมูลใน DataFrame train_merge ตามคอลัมน์ 'business_id' และคำนวณค่าเฉลี่ยของคุณลักษณะ (features) สำหรับแต่ละกลุ่ม (ธุรกิจ) โดยใช้ np.mean. ผลลัพธ์จะเป็น Series ที่มี index เป็น 'business_id' และค่าเฉลี่ยของ features ที่ได้จากการกลุ่มรวมกัน.\n\n- train_bid_features.reset_index(level=0, inplace=True): เพื่อรีเซ็ต index เพื่อให้ 'business_id' เป็นคอลัมน์ที่แยกออกมาเป็นคอลัมน์ใน DataFrame train_bid_features. inplace=True หมายถึงการแก้ไข DataFrame ตัวเองแทนที่จะสร้าง DataFrame ใหม่.\n\n- test_bid_features มีกระบวนการเดียวกันกับ train_bid_features โดยใช้ DataFrame test_merge แทนที่จะใช้ train_merge.\n\n- test_bid_features.head(): แสดงห้าแถวแรกของ DataFrame test_bid_features เพื่อตรวจสอบว่าได้สร้างข้อมูลได้อย่างถูกต้องหรือไม่.","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Step 2. Business feature manipulation\n\"\"\"\n# Take average of all features of photos in a train set that are assigned to the same business id\ntrain_bid_features = pd.DataFrame(train_merge.groupby('business_id')['features'].apply(np.mean))\ntrain_bid_features.reset_index(level=0, inplace=True)\n\n# Take average of all features of photos in a test set that are assigned to the same business id\ntest_bid_features = pd.DataFrame(test_merge.groupby('business_id')['features'].apply(np.mean))\ntest_bid_features.reset_index(level=0, inplace=True)\n\n# Check if the data frame is well constructed\ntest_bid_features.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:29.411407Z","iopub.execute_input":"2023-11-27T22:35:29.411862Z","iopub.status.idle":"2023-11-27T22:35:35.834799Z","shell.execute_reply.started":"2023-11-27T22:35:29.411816Z","shell.execute_reply":"2023-11-27T22:35:35.833674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เช็คข้อมูลใน test_bid_features","metadata":{}},{"cell_type":"code","source":"test_bid_features.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:35.836159Z","iopub.execute_input":"2023-11-27T22:35:35.836573Z","iopub.status.idle":"2023-11-27T22:35:35.858053Z","shell.execute_reply.started":"2023-11-27T22:35:35.836541Z","shell.execute_reply":"2023-11-27T22:35:35.857054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เช็คข้อมูลใน train_bid_features","metadata":{}},{"cell_type":"code","source":"print(len(train_bid_features))\ntrain_bid_features.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:35.860152Z","iopub.execute_input":"2023-11-27T22:35:35.860873Z","iopub.status.idle":"2023-11-27T22:35:35.881228Z","shell.execute_reply.started":"2023-11-27T22:35:35.860836Z","shell.execute_reply":"2023-11-27T22:35:35.88024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- ทำการเรียงข้อมูลใน DataFrame train_label ตามคอลัมน์ 'business_id' ในลำดับจากน้อยไปมาก นั่นคือ เรียงข้อมูลขึ้นตามลำดับของ 'business_id' ในเซตข้อมูลของการฝึก\n\n- การรีเซ็ตดัชนี (index) ของ DataFrame train_label เพื่อให้มันเริ่มต้นที่ 0 และเพิ่มคอลัมน์ 'index' ที่บอกถึงลำดับของแถวเมื่อ DataFrame นี้ถูกเรียง\n\n- เพิ่มคอลัมน์ 'labels' ลงใน DataFrame train_bid_features โดยนำข้อมูลที่อยู่ในคอลัมน์ 'labels' ของ train_label มาใส่\n\n- หาตำแหน่งที่มีค่า NaN ใน DataFrame train_bid_features โดยใช้ pd.isnull() เพื่อตรวจสอบว่าแต่ละเซลล์มีค่า NaN หรือไม่.\n\n- ทำการลบแถวที่มีค่า NaN ใน DataFrame train_bid_features.","metadata":{}},{"cell_type":"code","source":"# Sort labels by business_id in an ascending order, and then merge the label with mean feature of each business id\ntrain_label = train_label.sort_values('business_id')\ntrain_label = train_label.reset_index(drop=True)\ntrain_bid_features['labels'] = train_label['labels']\n\nnan_label_indices = pd.isnull(train_bid_features).any(axis=1).to_numpy().nonzero()[0]\ntrain_bid_features = train_bid_features.drop(train_bid_features.index[nan_label_indices])\n\n# Check if the data frame is well constructed\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:35.882383Z","iopub.execute_input":"2023-11-27T22:35:35.883244Z","iopub.status.idle":"2023-11-27T22:35:35.896006Z","shell.execute_reply.started":"2023-11-27T22:35:35.883205Z","shell.execute_reply":"2023-11-27T22:35:35.894912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- บันทึก DataFrame ของ Pandas ที่มีชื่อว่า train_bid_features และ test_bid_features ลงในไฟล์ในรูปแบบ Pickle (.pkl) ซึ่งเป็นวิธีที่นิยมในการบันทึกข้อมูลขนาดใหญ่ที่มีโครงสร้างที่ซับซ้อน โดยการใช้ Pickle จะช่วยในการรักษาโครงสร้างและประเภทของข้อมูล","metadata":{}},{"cell_type":"code","source":"# Save the manipulated train & test data sets in pickle for preserving array format and considering its size\ntrain_bid_features.to_pickle('train_set.csv')\ntest_bid_features.to_pickle('test_set.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:35.89759Z","iopub.execute_input":"2023-11-27T22:35:35.89804Z","iopub.status.idle":"2023-11-27T22:35:48.272722Z","shell.execute_reply.started":"2023-11-27T22:35:35.897987Z","shell.execute_reply":"2023-11-27T22:35:48.271816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- เช็คดูข้อมูลใน test_set และ train_set","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ntest_set = pd.read_pickle('/kaggle/working/test_set.csv')\nprint(\"First 5 rows of test_set:\")\nprint(test_set.head())\nprint(len(test_set))\n\ntrain_set = pd.read_pickle('/kaggle/working/train_set.csv')\nprint(\"\\nFirst 5 rows of train_set:\")\nprint(train_set.head())\nprint(len(train_set))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:35:48.274074Z","iopub.execute_input":"2023-11-27T22:35:48.274371Z","iopub.status.idle":"2023-11-27T22:36:09.079061Z","shell.execute_reply.started":"2023-11-27T22:35:48.274346Z","shell.execute_reply":"2023-11-27T22:36:09.077998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> \n> Step 3. Construct a classifier SVM\n>","metadata":{}},{"cell_type":"code","source":"import numpy as np; np.random.seed(1040941203) # For reproducibility (+82-10-4094-1203)\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:43:00.589789Z","iopub.execute_input":"2023-11-27T22:43:00.59055Z","iopub.status.idle":"2023-11-27T22:43:01.749707Z","shell.execute_reply.started":"2023-11-27T22:43:00.590499Z","shell.execute_reply":"2023-11-27T22:43:01.74886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- หลดข้อมูลที่ถูกจัดการแล้วจากไฟล์ Pickle ที่ชื่อ \"train_set.csv\" และ \"test_set.csv\" มาเป็น DataFrame ของ Pandas ","metadata":{}},{"cell_type":"code","source":"# Load manipulated data set\ntrain_df = pd.read_pickle(\"train_set.csv\")\ntest_df  = pd.read_pickle(\"test_set.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:43:02.042025Z","iopub.execute_input":"2023-11-27T22:43:02.042899Z","iopub.status.idle":"2023-11-27T22:43:21.361698Z","shell.execute_reply.started":"2023-11-27T22:43:02.042852Z","shell.execute_reply":"2023-11-27T22:43:21.360831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- แสดงข้อมูล 5 แถวแรกใน train_df","metadata":{}},{"cell_type":"code","source":"# Check if the data frame is loaded normally\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:43:21.363456Z","iopub.execute_input":"2023-11-27T22:43:21.363798Z","iopub.status.idle":"2023-11-27T22:43:21.392245Z","shell.execute_reply.started":"2023-11-27T22:43:21.363771Z","shell.execute_reply":"2023-11-27T22:43:21.391353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- แสดงข้อมูล 5 แถวแรกใน test_df","metadata":{}},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:43:21.393328Z","iopub.execute_input":"2023-11-27T22:43:21.393638Z","iopub.status.idle":"2023-11-27T22:43:21.409249Z","shell.execute_reply.started":"2023-11-27T22:43:21.393612Z","shell.execute_reply":"2023-11-27T22:43:21.40828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- X_train: pandas array ที่เก็บ features ของ train set \n- Y_train: pandas array ที่เก็บ labels ของ train set \n- X_test: pandas array ที่เก็บ features ของ test set","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Function to make labels in the data frame into a list (i.e. 0 8 => [0, 8])\ndef labels_to_list(labels): return list(map(int, labels.split()))\n\n# Assuming train_df and test_df are Pandas DataFrames\nX_train = np.array([x for x in train_df['features']])\nY_train = train_df['labels'].apply(labels_to_list).to_numpy()\nX_test = np.array([x for x in test_df['features']])\n\n# Check shape of array-format train & test set\nprint(\"X_train: \", X_train.shape)\nprint(\"Y_train: \", Y_train.shape)\nprint(\"X_test: \", X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:43:21.411812Z","iopub.execute_input":"2023-11-27T22:43:21.412295Z","iopub.status.idle":"2023-11-27T22:43:22.953199Z","shell.execute_reply.started":"2023-11-27T22:43:21.412261Z","shell.execute_reply":"2023-11-27T22:43:22.952095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Import Library","metadata":{}},{"cell_type":"code","source":"# Load packages for splitting train & validation set, SVM classifier, 1-of-K encoder\nfrom sklearn import svm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\n# Package for estimating a time taken\nimport time; t=time.time()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- การฝึกตัวจำแนก (classifier) แบบ One-vs-Rest SVM ด้วยใช้ OneVsRestClassifier ที่มี SVM (SVC) ทำหน้าที่ตัวจำแนกแต่ละกลุ่ม (label) ของป้ายกำกับ (multi-label classification)**\n\n- การฝึกโมเดลนี้บนชุดฝึกอบรม X_train_ และ Y_train_ ที่ได้จากขั้นตอนการแปลง labels ให้เป็นรูปแบบ One-hot encoding (MultiLabelBinarizer)ในแต่ละรอบของการฝึก (loop), โมเดลถูกฝึกด้วยข้อมูลตัวอย่างทีละตัว ([X_train_[i]], [Y_train_[i]]).\n\n- การแสดงกระบวนการฝึกโดยใช้ progress bar ที่ถูกสร้างโดย tqdm.","metadata":{}},{"cell_type":"code","source":"import pickle\nimport warnings\nfrom tqdm.notebook import tqdm\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.svm import SVC\nimport numpy as np\nimport time\n\n# Suppress warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\n# บันทึก test_merge ในรูปแบบ pickle\n#with open('test_merge.pkl', 'wb') as f:\n    #pickle.dump(test_merge, f)\n\n# แปลงลิสต์ของป้ายกำกับให้เป็นรูปแบบการเขียน 1 ของ K\none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# แบ่งชุดฝึกอบรมเป็น 8:2 (train: validation)\nrandom_state = np.random.RandomState(1040941203)\nX_train_, X_test_, Y_train_, Y_test_ = train_test_split(X_train, Y_train_, test_size=0.2, random_state=random_state)\n\n# ฝึกตัวจำแนก SVM\nclassifier = OneVsRestClassifier(SVC(kernel='rbf', random_state=random_state, C=1.0, probability=True, verbose=0))\n\n# Fit the classifier with tqdm progress bar\nt = time.time()\nwith tqdm(total=len(X_train_), desc=\"Training SVM\") as pbar:\n    for i in range(len(X_train_)):\n        classifier.fit([X_train_[i]], [Y_train_[i]])\n        pbar.update(1)\n\n# ทำนายป้ายกำกับโดยใช้โมเดลที่ฝึก\nY_predict = classifier.predict(X_test_)\n\n# แสดงเวลาที่ใช้\nprint(\"Time passed: \", \"{0:.3f}\".format(time.time() - t), \"sec\")","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:26.735703Z","iopub.execute_input":"2023-11-27T22:45:26.736558Z","iopub.status.idle":"2023-11-27T22:45:32.517225Z","shell.execute_reply.started":"2023-11-27T22:45:26.736521Z","shell.execute_reply":"2023-11-27T22:45:32.516148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- การแสดงตัวอย่างของป้ายกำกับที่ทำนายได้จากโมเดล (predicted labels) โดยให้แสดงข้อมูลที่ตัวอย่างที่ 2 ถึง 3 ของ Y_predict ในรูปแบบ One-hot encoding (1-of-K coding scheme).\n- การแสดงตัวอย่างของป้ายกำกับที่ทำนายได้จากโมเดล (predicted labels) โดยใช้ inverse_transform ของ one_of_K_encoder เพื่อแปลงกลับมาเป็นรูปแบบป้ายกำกับเดิม.\n- การแสดงจำนวนของตัวอย่างที่ทำนายใน Y_predict.\n- การแสดงตัวอย่างของป้ายกำกับที่ทำนายได้จากโมเดล, รวมถึงจำนวนของตัวอย่างที่ทำนายทั้งหมดในชุดทดสอบ.","metadata":{}},{"cell_type":"code","source":"\n# Show some predicted values\nprint(\"Samples of predicted labels (in 1-of-K coding scheme):\\n\", Y_predict[1:3])\nprint(\"\\nSamples of corresponding predicted labels:\\n\", one_of_K_encoder.inverse_transform(Y_predict[1:3]))\nprint(len(Y_predict))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:32.519199Z","iopub.execute_input":"2023-11-27T22:45:32.519559Z","iopub.status.idle":"2023-11-27T22:45:32.526833Z","shell.execute_reply.started":"2023-11-27T22:45:32.519502Z","shell.execute_reply":"2023-11-27T22:45:32.525645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- DataFrame stat ถูกสร้างโดยให้มี 11 คอลัมน์ (9 คอลัมน์สำหรับแต่ละ label และ 2 คอลัมน์สุดท้ายสำหรับผลรวมของป้ายกำกับและจำนวนธุรกิจทั้งหมด).\n\n- np.sum(Y_predict, axis=0) คำนวณผลรวมของทุกแถวใน Y_predict ที่มีแต่ละ label.\n\n- ตัว DataFrame stat ถูกแปลงให้แสดงเป็นร้อยละ (biz_percentage) โดยการคำนวณจาก biz_count.\n\n- pd.options.display.float_format = '{:.0f}%'.format กำหนดรูปแบบการแสดงผลของ DataFrame ให้เป็นร้อยละที่ไม่มีทศนิยม ผลลัพธ์ DataFrame stat ถูกแสดง.\n\n- สรุปคือเพื่อสร้างและแสดง DataFrame ที่แสดงรายละเอียดเกี่ยวกับอัตราส่วนของแต่ละป้ายกำกับที่ทำนายได้ในชุดทดสอบ.","metadata":{}},{"cell_type":"code","source":"# Construct a data frame to show ratio of each label in a predicted set\nstat = pd.DataFrame(columns=['label ' + str(i) for i in range(9)] + ['total_biz'], index = ['biz_count', 'biz_percentage'])\n\nstat.loc['biz_count'] = np.append(np.sum(Y_predict, axis=0), len(Y_predict))\nstat.loc['biz_percentage'] = stat.loc[\"biz_count\"] * 100 / len(Y_predict)\n\npd.options.display.float_format = '{:.0f}%'.format\n\nstat","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:33.007345Z","iopub.execute_input":"2023-11-27T22:45:33.007778Z","iopub.status.idle":"2023-11-27T22:45:33.025759Z","shell.execute_reply.started":"2023-11-27T22:45:33.007745Z","shell.execute_reply":"2023-11-27T22:45:33.024766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- from sklearn.metrics import f1_score นำเข้าฟังก์ชัน f1_score จาก scikit-learn.\n\n- f1_score(Y_test_, Y_predict, average='micro') คำนวณ F1 score ทั่วไปโดยใช้การเฉลี่ย micro ที่คำนวณ F1 score โดยให้ทุกรายการนับเท่ากัน ไม่ว่าจะเป็น class ใด.\n\n- f1_score(Y_test_, Y_predict, average=None) คำนวณ F1 score ของแต่ละ class โดยไม่ใช้การเฉลี่ย ซึ่งจะได้ผลลัพธ์เป็น array ที่มีค่า F1 score สำหรับแต่ละ class.\n\n- ผลลัพธ์ F1 score ทั่วไปและ F1 score ของแต่ละ class ถูกแสดงผล.\n\n- F1 score มีค่าระหว่าง 0 ถึง 1, โดยค่าที่ใกล้เคียง 1 แสดงถึงประสิทธิภาพที่ดีของโมเดลในการจำแนกป้ายกำกับ.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import f1_score # For measuring F1 score metrics\n\n# Show global F1 score & on-label F1 scoreV\nprint(\"Overall F1 score: \", f1_score(Y_test_, Y_predict, average='micro')) \nprint(\"F1 score of each label : \", f1_score(Y_test_, Y_predict, average=None))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:39.823835Z","iopub.execute_input":"2023-11-27T22:45:39.824953Z","iopub.status.idle":"2023-11-27T22:45:39.838942Z","shell.execute_reply.started":"2023-11-27T22:45:39.824915Z","shell.execute_reply":"2023-11-27T22:45:39.837813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Y_train_ = one_of_K_encoder.fit_transform(Y_train) - แปลงลิสต์ของป้ายกำกับ (Y_train) เป็นรูปแบบการเขียน 1 ของ K (one-hot encoding) โดยใช้ MultiLabelBinarizer.\n\n- classifier = OneVsRestClassifier(SVC(kernel='rbf', random_state=random_state, C=1.0, probability=True, verbose=0)) - สร้าง classifier โดยใช้ SVM (Support Vector Machine) แบบ one-vs-rest โดยกำหนดค่าพารามิเตอร์ของ SVM เช่น kernel เป็น 'rbf' (Radial basis function), C เป็น 1.0 (ค่า regularization), probability=True (เพื่อให้ SVM สามารถคำนวณความน่าจะเป็นได้) และ verbose=0 (ไม่แสดงข้อความบันทึก).\n\n- with tqdm(total=len(X_train_), desc=\"Training SVM\") as pbar: - ใช้ tqdm เพื่อสร้าง progress bar ในการฝึก classifier และให้ pbar.update(1) เพื่ออัพเดท progress bar ทุกครั้งที่ classifier ถูกฝึก.","metadata":{}},{"cell_type":"code","source":"import pickle\nimport warnings\nfrom tqdm.notebook import tqdm\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.svm import SVC\nimport numpy as np\nimport time\nt = time.time()\n\n# Convert list of labels to follow 1-of-K coding scheme\none_of_K_encoder = MultiLabelBinarizer()\nY_train_ = one_of_K_encoder.fit_transform(Y_train)\n\n# Create a tqdm instance to display a progress bar for training\nprogress_bar = tqdm(total=len(X_train) + len(X_test), desc=\"Training and Prediction Progress\")\n\n# Train the SVM classifier with a progress bar\nrandom_state = np.random.RandomState(0)\nclassifier = OneVsRestClassifier(SVC(kernel='rbf', probability=True))\n\nfor X_train_batch, Y_train_batch in zip(X_train, Y_train_):\n    classifier.fit([X_train_batch], [Y_train_batch])\n    progress_bar.update(1)\n\n# Predict labels using the trained model\nt_predict = time.time()\nY_predict = classifier.predict(X_test)\nY_predict_label = one_of_K_encoder.inverse_transform(Y_predict)\nprogress_bar.update(len(X_test))  # Update progress bar for prediction\n\n# Close the progress bar\nprogress_bar.close()\n\nprint(\"Time passed for training and prediction: \", \"{0:.1f}\".format(time.time() - t), \"sec\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:41.919782Z","iopub.execute_input":"2023-11-27T22:45:41.92061Z","iopub.status.idle":"2023-11-27T22:45:48.754977Z","shell.execute_reply.started":"2023-11-27T22:45:41.920555Z","shell.execute_reply":"2023-11-27T22:45:48.75402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- stat = pd.DataFrame(columns=['label ' + str(i) for i in range(9)] + ['total_biz'], index = ['biz_count', 'biz_percentage']) - สร้าง DataFrame ที่มีคอลัมน์เป็น 'label 0', 'label 1', ..., 'label 8', และ 'total_biz', และมี index เป็น 'biz_count' และ 'biz_percentage'.\n\n- stat.loc['biz_count'] = np.append(np.sum(Y_predict, axis=0), len(Y_predict)) - คำนวณจำนวนธุรกิจทั้งหมด (biz_count) โดยรวมตามป้ายกำกับที่ทำนายได้จาก Y_predict, และนับจำนวนทั้งหมดของการทำนาย (total_biz) จาก len(Y_predict) และนำไปเพิ่มใน DataFrame ในแถว 'biz_count'.\n\n- stat.loc['biz_percentage'] = stat.loc[\"biz_count\"] * 100 / len(Y_predict) - คำนวณร้อยละของจำนวนธุรกิจทั้งหมด (biz_percentage) โดยหารด้วย len(Y_predict) และเพิ่มค่านี้ใน DataFrame ในแถว 'biz_percentage'.\n\n- pd.options.display.float_format = '{:.0f}%'.format - กำหนดรูปแบบการแสดงผลของข้อมูลที่เป็นทศนิยมใน DataFrame เป็นเปอร์เซ็นต์โดยใช้ {:.0f}% (ไม่มีทศนิยม).\n\n- stat - แสดง DataFrame ที่ได้หลังจากการคำนวณ.","metadata":{}},{"cell_type":"code","source":"# Construct a data frame to show ratio of each label in a predicted set\nstat = pd.DataFrame(columns=['label ' + str(i) for i in range(9)] + ['total_biz'], index = ['biz_count', 'biz_percentage'])\n\nstat.loc['biz_count'] = np.append(np.sum(Y_predict, axis=0), len(Y_predict))\nstat.loc['biz_percentage'] = stat.loc[\"biz_count\"] * 100 / len(Y_predict)\n\npd.options.display.float_format = '{:.0f}%'.format\n\nstat","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:48.757183Z","iopub.execute_input":"2023-11-27T22:45:48.757903Z","iopub.status.idle":"2023-11-27T22:45:48.777047Z","shell.execute_reply.started":"2023-11-27T22:45:48.757864Z","shell.execute_reply":"2023-11-27T22:45:48.775992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Y_predict_label = one_of_K_encoder.inverse_transform(Y_predict) - แปลงกลับป้ายกำกับที่ถูกเข้ารหัสในรูปแบบ One-Hot Encoding (Y_predict) ให้กลับมาเป็นรูปแบบของป้ายกำกับเดิม.\n\n- final_df = pd.DataFrame(columns=['business_id', 'labels']) - สร้าง DataFrame ที่มีคอลัมน์ 'business_id' และ 'labels' เพื่อให้เก็บข้อมูลที่จะนำไปสร้างไฟล์ submission.\n\n- ลูปผ่านแต่ละแถวใน test_df เพื่อนำข้อมูลไปสร้างเป็นไฟล์ submission:\n\n- biz = test_df.loc[i]['business_id'] - ดึงข้อมูล business_id จากแถว i ใน test_df.\n\n- ตรวจสอบว่า index i อยู่ในขอบเขตของ Y_predict_label หรือไม่ และกำหนดค่า label โดยใช้ Y_predict_label[i] และแปลงเป็นสตริงที่คั่นด้วยช่องว่าง.\n\n- final_df.loc[i] = [str(biz), label] - เพิ่มข้อมูล business_id และ labels ลงใน DataFrame ที่สร้างไว้.\n\n- with open(\"submission_Seokju_Hahn_MLP.csv\", 'w') as file: final_df.to_csv(file, index=False) - เขียน DataFrame เป็นไฟล์ CSV ที่ชื่อ \"submission_Seokju_Hahn_MLP.csv\" โดยไม่รวม index ลงในไฟล์.\n\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"# Assuming Y_predict is one-hot encoded, convert it back to the original labels\nY_predict_label = one_of_K_encoder.inverse_transform(Y_predict)\n\n# Construct a data frame for submission (matching predicted label with business id in a test set)\nfinal_df = pd.DataFrame(columns=['business_id', 'labels'])\n\nfor i in range(len(test_df)):\n    biz = test_df.loc[i]['business_id']\n    \n    # Check if the index is within the range of Y_predict_label\n    if i < len(Y_predict_label):\n        label = Y_predict_label[i]\n        label = ' '.join(map(str, label))  # Convert the list to a space-separated string\n    else:\n        label = ''  # Set an empty string if the index is out of range\n    \n    final_df.loc[i] = [str(biz), label]\n\n# Write a submission file\nwith open(\"submission.csv\", 'w') as file:\n    final_df.to_csv(file, index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:48.7783Z","iopub.execute_input":"2023-11-27T22:45:48.778664Z","iopub.status.idle":"2023-11-27T22:45:59.67936Z","shell.execute_reply.started":"2023-11-27T22:45:48.778633Z","shell.execute_reply":"2023-11-27T22:45:59.67856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- final_df.info() ใช้เพื่อแสดงข้อมูลเกี่ยวกับ DataFrame final_df. ผลลัพธ์ที่ได้จะรวมถึงข้อมูลเชิงลึกเกี่ยวกับ DataFrame นี้เช่น จำนวนแถวและคอลัมน์ทั้งหมด, ประเภทข้อมูลของแต่ละคอลัมน์, จำนวนข้อมูลที่ไม่ใช่ค่าว่าง, และอื่นๆ.","metadata":{}},{"cell_type":"code","source":"print(final_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-27T22:45:59.680911Z","iopub.execute_input":"2023-11-27T22:45:59.681206Z","iopub.status.idle":"2023-11-27T22:45:59.687775Z","shell.execute_reply.started":"2023-11-27T22:45:59.68118Z","shell.execute_reply":"2023-11-27T22:45:59.686716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary\nจากการทดลองทำ Model กลุ่มของเราได้ทำ Model 2 Model คือ SVM และ RESNET-50 ซึ่งทางกลุ่มเราเลือก SVM เนื่องจาก SVM มีการทำนายที่แม่นยำ ไม่ต้องการข้อมูลมากเพราะ RAM CPU และ เวลามีไม่มากพอที่จะทำ Model RESNET-50 แม้ว่า RESNET-50 จะมีความเป็นไปได้ที่จะได้ผลลัพธ์ดีกว่า แต่เนื่องจาก Model RESNET-50 เรามี Layer น้อยและ fine-tune ได้ไม่ดีพอ หากมีโอกาสในอนาคตเราอาจจะเลือก RESNET-50 มาช่วยทำ Model ","metadata":{}},{"cell_type":"markdown","source":"- Reference\n> - By. Seok-Ju Hahn\n> - https://github.com/vaseline555/Yelp-Restaurant-Photo-Classification.git","metadata":{}}]}