{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DSI206 Final Project\n**Yelp Restaurant Photo Classification**\n","metadata":{}},{"cell_type":"markdown","source":"# Introduction\n**จุดมุ่งหมาย** : เพื่อศึกษา Image Classification โดยการจำแนกรูปภาพที่เกี่ยวข้องกับร้านอาหาร ตามlables จำนวน 9 lables ในชุดข้อมูล Yelp Restaurant Photo Classification โดยที่ labels จะแบ่งออกตามนี้\n* 0: good_for_lunch\n* 1: good_for_dinner\n* 2: takes_reservations\n* 3: outdoor_seating\n* 4: restaurant_is_expensive\n* 5: has_alcohol\n* 6: has_table_service\n* 7: ambience_is_classy\n* 8: good_for_kids\n\nใน ไฟล์ Yelp Restaurant Photo Classification จะมี\n* train_photos.tgz – ภาพ train_photos\n* test_photos.tgz – ภาพ test_photos\n* train_photo_to_biz_ids.csv – เชื่อม photo id กับ business id สำหรับ train\n* test_photo_to_biz_ids.csv - เชื่อม photo id กับ business id สำหรับ test\n* train.csv – dataset ที่เชื่อม business id กับ label. \n* sample_submission.csv - เป็นตัวอย่างการส่งข้อมูลกลับไปที่ Kaggle ที่ถูกต้อง  \n\nโดยที่ รูปภาพที่มี business_id เดียวกัน จะมี labels เดียวกัน และ ใน 1 รูปจะมีได้หลาย labels \n","metadata":{}},{"cell_type":"markdown","source":"# Members\n1. ปณิชา เตชะเศรษฐกุล 6524651020\n2. กุลิสรา อิทธิสิริกุลชัย 6524651178\n3. ชัญญานุช เจริญพนารัตน์ 6524651210\n4. เปมิกา จันทร์เขียว 6524651285\n5. ลลิลธร วิยา 6524651384\n6. ศศิชา พุฒิลือชา 6524651426\n","metadata":{}},{"cell_type":"markdown","source":"# Table of Contents\n1. Import Libraries\n2. Extract Files\n3. Load Data\n4. Model for Feature Extraction\n5. Feature Extraction\n6. Aggregate Features for Each Business\n7. Prepare Dataset for Classification\n8. KNN Implementation\n9. Project Summary\n10. Submit the work","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-27T15:17:52.781201Z","iopub.execute_input":"2023-11-27T15:17:52.781479Z","iopub.status.idle":"2023-11-27T15:17:53.133073Z","shell.execute_reply.started":"2023-11-27T15:17:52.781454Z","shell.execute_reply":"2023-11-27T15:17:53.132173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries\n* **นำเข้า Libraries ที่ต้องใช้**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tensorflow.keras.preprocessing.image import img_to_array, load_img\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom tqdm import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:17:53.134805Z","iopub.execute_input":"2023-11-27T15:17:53.135159Z","iopub.status.idle":"2023-11-27T15:18:05.432232Z","shell.execute_reply.started":"2023-11-27T15:17:53.135135Z","shell.execute_reply":"2023-11-27T15:18:05.431226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract Files\n* นำเข้าข้อมูลที่ต้องใช้","metadata":{}},{"cell_type":"code","source":"import tarfile\nimport os\nfrom tqdm import tqdm\n\n# list ของไฟล์ที่ต้องแตก\nfiles_to_extract = [\n    'sample_submission.csv.tgz',\n    'test_photo_to_biz.csv.tgz',\n    'test_photos.tgz',\n    'train.csv.tgz',\n    'train_photo_to_biz_ids.csv.tgz',\n    'train_photos.tgz'\n]\n\n# Directory ของไฟล์ที่จะแตก\ntarget_directory = '/kaggle/working/'\n\n# ตรวจสอบว่าไฟล์นั้นว่ามีอยู่ใน Directory ก่อนทำการแตกไฟล์\nfor file_to_extract in files_to_extract:\n    target_file_path = os.path.join(target_directory, file_to_extract.replace('.tgz', ''))\n    if not os.path.exists(target_file_path):\n        file_path = f'/kaggle/input/yelp-restaurant-photo-classification/{file_to_extract}'\n        try:\n            with tarfile.open(file_path, 'r:gz') as tar:\n                members = list(tar.getmembers())\n                for member in tqdm(iterable=members, desc=f\"Extracting {file_to_extract}\", total=len(members)):\n                    tar.extract(member, target_directory)\n        except Exception as e:\n            print(f\"Error extracting {file_to_extract}: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:18:05.433846Z","iopub.execute_input":"2023-11-27T15:18:05.434604Z","iopub.status.idle":"2023-11-27T15:26:07.032280Z","shell.execute_reply.started":"2023-11-27T15:18:05.434563Z","shell.execute_reply":"2023-11-27T15:26:07.031310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data\n* load data\n* เชื่อม train_photos กับ train_labels ","metadata":{}},{"cell_type":"code","source":"# Load data\ntrain_photos = pd.read_csv('/kaggle/working/train_photo_to_biz_ids.csv')\ntrain_labels = pd.read_csv('/kaggle/working/train.csv')\n\n# เชื่อม train_photos กับ train_labels ด้วย business_id\ndata = pd.merge(train_photos, train_labels, on='business_id')\n\n# ลบ rows ที่ไม่มี labels ออก และ แปลง labels เป็น lists\ndata.dropna(subset=['labels'], inplace=True)\ndata['labels'] = data['labels'].apply(lambda x: x.split(' '))\n\n# แปลง photo_id เป็น str และเพิ่ม .jpg\ndata['photo_id'] = data['photo_id'].astype(str) + '.jpg'\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:07.034547Z","iopub.execute_input":"2023-11-27T15:26:07.034870Z","iopub.status.idle":"2023-11-27T15:26:07.951100Z","shell.execute_reply.started":"2023-11-27T15:26:07.034845Z","shell.execute_reply":"2023-11-27T15:26:07.950363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:07.952313Z","iopub.execute_input":"2023-11-27T15:26:07.952610Z","iopub.status.idle":"2023-11-27T15:26:07.969415Z","shell.execute_reply.started":"2023-11-27T15:26:07.952585Z","shell.execute_reply":"2023-11-27T15:26:07.968387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_photos","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:07.970586Z","iopub.execute_input":"2023-11-27T15:26:07.970857Z","iopub.status.idle":"2023-11-27T15:26:07.981093Z","shell.execute_reply.started":"2023-11-27T15:26:07.970834Z","shell.execute_reply":"2023-11-27T15:26:07.980177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:07.982308Z","iopub.execute_input":"2023-11-27T15:26:07.982593Z","iopub.status.idle":"2023-11-27T15:26:08.000165Z","shell.execute_reply.started":"2023-11-27T15:26:07.982568Z","shell.execute_reply":"2023-11-27T15:26:07.999304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model for Feature Extraction\n* ใช้ CNN ในการ extract feature รูปภาพ\n* นำ base model มาใช้\n* เพิ่ม layers ให้กับ base model\n* สร้าง model สำหรับ Extract features รูปภาพ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalMaxPooling2D, Dropout, Flatten\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\n\n# โหลด pre-trained model ชื่อ ResNet50 กับข้อมูล imagenet และลบ top ของ base model ออก\nbase_model = ResNet50(weights='imagenet', include_top=False)\nx = base_model.output\n  \n\n# Flatten the output\nx = Flatten()(x)  \n\n# สร้าง final model\nmodel = Model(inputs=base_model.input, outputs=x)\n\n# Freeze layers ของ base model\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# Model Extract features รูปภาพ\ndef extract_features(img_path, model):\n    try:\n        if not os.path.exists(img_path):\n            print(f\"Image not found: {img_path}\")\n            return None\n        img = load_img(img_path, target_size=(256, 256))\n        img_array = img_to_array(img)\n        img_array = np.expand_dims(img_array, axis=0)\n        img_array = preprocess_input(img_array)\n        features = model.predict(img_array)\n        return features.flatten()  \n    except Exception as e:\n        print(f\"Error processing image {img_path}: {e}\")\n        return None\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:08.001206Z","iopub.execute_input":"2023-11-27T15:26:08.001580Z","iopub.status.idle":"2023-11-27T15:26:13.932968Z","shell.execute_reply.started":"2023-11-27T15:26:08.001537Z","shell.execute_reply":"2023-11-27T15:26:13.931978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Extraction\n* นำ model มา Extract features รูปภาพ จำนวน 20,000 รูป โดยแบ่งเป็น 4 cell เพื่อลดการทำงานของ RAM","metadata":{}},{"cell_type":"code","source":"# กำหนดช่วง 0 - 5,000\nstart_idx = 0\nend_idx = 5000\n\n# Extract features รูปภาพ\nfeature_dict_chunk1 = {}\nlimited_data_chunk1 = data[start_idx:end_idx]\n\nfor _, row in tqdm(limited_data_chunk1.iterrows(), total=limited_data_chunk1.shape[0], desc=\"Extracting features\"):\n    img_path = os.path.join('/kaggle/working/train_photos', row['photo_id'])\n    biz_id = row['business_id']\n    \n    features = extract_features(img_path,model)\n    if features is not None:\n        if biz_id not in feature_dict_chunk1:\n            feature_dict_chunk1[biz_id] = []\n        feature_dict_chunk1[biz_id].append(features)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:26:13.934289Z","iopub.execute_input":"2023-11-27T15:26:13.935049Z","iopub.status.idle":"2023-11-27T15:32:19.012943Z","shell.execute_reply.started":"2023-11-27T15:26:13.935013Z","shell.execute_reply":"2023-11-27T15:32:19.012012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# กำหนดช่วง 5,000 - 10,000\nstart_idx = 5000\nend_idx = 10000\n\n# Extract features รูปภาพ\nfeature_dict_chunk2 = {}\nlimited_data_chunk2 = data[start_idx:end_idx]\n\nfor _, row in tqdm(limited_data_chunk2.iterrows(), total=limited_data_chunk2.shape[0], desc=\"Extracting features\"):\n    img_path = os.path.join('/kaggle/working/train_photos', row['photo_id'])\n    biz_id = row['business_id']\n    \n    features = extract_features(img_path,model)\n    if features is not None:\n        if biz_id not in feature_dict_chunk2:\n            feature_dict_chunk2[biz_id] = []\n        feature_dict_chunk2[biz_id].append(features)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T15:32:19.014380Z","iopub.execute_input":"2023-11-27T15:32:19.014786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# กำหนดช่วง 10,000 - 15,000\nstart_idx = 10000\nend_idx = 15000\n\n# Extract features รูปภาพ\nfeature_dict_chunk3 = {}\nlimited_data_chunk3 = data[start_idx:end_idx]\n\nfor _, row in tqdm(limited_data_chunk3.iterrows(), total=limited_data_chunk3.shape[0], desc=\"Extracting features\"):\n    img_path = os.path.join('/kaggle/working/train_photos', row['photo_id'])\n    biz_id = row['business_id']\n    \n    features = extract_features(img_path,model)\n    if features is not None:\n        if biz_id not in feature_dict_chunk3:\n            feature_dict_chunk3[biz_id] = []\n        feature_dict_chunk3[biz_id].append(features)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# กำหนดช่วง 15,000 - 20,000\nstart_idx = 15000\nend_idx = 20000\n\n# Extract features รูปภาพ\nfeature_dict_chunk4 = {}\nlimited_data_chunk4 = data[start_idx:end_idx]\n\nfor _, row in tqdm(limited_data_chunk4.iterrows(), total=limited_data_chunk4.shape[0], desc=\"Extracting features\"):\n    img_path = os.path.join('/kaggle/working/train_photos', row['photo_id'])\n    biz_id = row['business_id']\n    \n    features = extract_features(img_path,model)\n    if features is not None:\n        if biz_id not in feature_dict_chunk4:\n            feature_dict_chunk4[biz_id] = []\n        feature_dict_chunk4[biz_id].append(features)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# รวมข้อมูลที่ Extract features รูปภาพ\nfeature_dict = {**feature_dict_chunk1, **feature_dict_chunk2, **feature_dict_chunk3,**feature_dict_chunk4}\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Aggregate Features for Each Business\n* สร้าง dataframe ที่เก็บ mean ของ feature รูปภาพในแต่ละ business_id\n","metadata":{}},{"cell_type":"code","source":"# loop เพื่อเก็บ features ของรูปภาพที่รวมกันในแต่ละ business_id และ คำนวณค่า mean ของ feature ในแต่ละ business_id\naggregated_data = []\n\nfor biz_id, features in feature_dict.items():\n    if features:\n        mean_feature = np.mean(features, axis=0)\n        aggregated_data.append({\n            'business_id': biz_id, \n            'aggregated_features': mean_feature.tolist()\n        })\n\n# สร้าง DataFrame\naggregated_features_df = pd.DataFrame(aggregated_data)\n\n# บันทึกข้อมูลลงไฟล์ csv\naggregated_features_df.to_csv('/kaggle/working/aggregated_features.csv', index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aggregated_features_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Dataset for Classification\n* เตรียม dataset ที่จะใช้ใน model Classification โดยการทำให้ features ของ business_id และ labels อยู่ใน dataset เดียวกัน\n* กำหนด x,y เพื่อนนำไปใช้ในการ train","metadata":{}},{"cell_type":"code","source":"# ดึง business_id จาก aggregated_features_df และ Filter business_id ที่มีใน DataFrame train_labels และ aggregated_features_df เนื่องจากค่า labels อยู่ใน DataFrame train_labels\nbiz_ids = aggregated_features_df['business_id'].values\nbiz_ids = [biz_id for biz_id in biz_ids if biz_id in train_labels.index]\n\n# Filter DataFrame ที่รวมเฉพาะ business IDs ที่ filter แล้ว\nfiltered_aggregated_features_df = aggregated_features_df[aggregated_features_df['business_id'].isin(biz_ids)]\n\n# กำหนดตัวแปร X ที่เก็บ mean ของ features รูปภาพในแต่ละ business_id แล้วแปลงเป็น numpy array\nX = np.array(filtered_aggregated_features_df['aggregated_features'].tolist())\n\n# กำหนดตัวแปร y ที่เก็บ labels ในแต่ละ business_id\ny = train_labels.loc[biz_ids]['labels'].apply(lambda x: x.split(' '))\n\n# แปลง label ใน y ให้เป็น Binary\nmlb = MultiLabelBinarizer()\ny_binarized = mlb.fit_transform(y)\nprint(\"Binarized labels shape:\", y_binarized.shape)\nprint(\"Label classes after binarization:\", mlb.classes_)\n\n# แบ่ง dataset ออกเป็น train,test\nX_train, X_test, y_train, y_test = train_test_split(X, y_binarized, test_size=0.2, random_state=42)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN Implementation\n* ใช้ model KNN เพื่อการ Classification image\n* หาค่า k ที่ดีที่สุด","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn.neighbors import KNeighborsClassifier\n\nk_range = range(1, 10)\nk_scores = []\n\n# Loop หาค่า K\nfor k in k_range:\n    knn = KNeighborsClassifier(n_neighbors=k)\n    # Perform cross-validation\n    scores = cross_val_score(knn, X_train, y_train, cv=10, scoring='f1_micro')  # Change scoring to f1_micro, f1_macro, or f1_weighted\n    k_scores.append(scores.mean())\n\n# Best k value\nbest_k = k_range[np.argmax(k_scores)]\nprint(\"Best k value:\", best_k)\n\n# Plot กราฟ\nimport matplotlib.pyplot as plt\n\nplt.figure()\nplt.xlabel('k')\nplt.ylabel('F1 Score')\nplt.scatter(k_range, k_scores)\nplt.xticks([0, 5, 10, 15, 20, 25, 30])\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom sklearn.preprocessing import StandardScaler\n\n# กำหนดค่า k \nknn_classifier = KNeighborsClassifier(n_neighbors=best_k)  \n\n# train model\nknn_classifier.fit(X_train, y_train)\n\n# สร้างตัวแปรเพื่อใส่ Predict test data\ny_pred_knn = knn_classifier.predict(X_test)\n\n# คำนวณ accuracy\naccuracy = accuracy_score(y_test, y_pred_knn)\nprint(\"Accuracy:\", accuracy)\n\n# คำนวณ F1 score for multi-label classification\nf1 = f1_score(y_test, y_pred_knn, average='micro') \nprint(\"F1 Score:\", f1)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Project Summary\nในโปรเจกต์นี้เราได้มีการลองผิดลองถูกในหลายขั้นตอนทำให้พวกเราได้ผลลัพธ์ที่พึ่งพอใจที่สุด โดยขั้นตอนที่เราทำจะมีดังนี้\n\n**Process that don’t work**\n\n1. เราได้ใช้ CNN โดยที่เราลองทำ sequential ของเราเอง ทำให้ได้ค่า f1 ที่ต่ำมากๆ epoch รอบแรก ได้ค่า f1 = 0.03 และ ค่า loss ที่สูงมากๆ\n2. เราได้ใช้ CNN โดยที่เราได้ใช้ base model (Resnet50) แต่ก็คงมีค่า f1 ที่ต่ำ ,ค่า loss ที่สูงพอๆกับไม่ได้ใช้ base model โดยในการรันแต่ละรอบของ epoch ใช้เวลานาน ( 1 epoch ใช้เวลา ครึ่งชั่วโมง โดยเรากำหนดทั้งหมด 20 epoch) และ epoch ในรอบถัดๆไป ค่า f1 เพิ่มขึ้นเพียงเล็กน้อยเท่านั้น [ตัวอย่าง code](https://www.kaggle.com/code/kulisara/cnn-resnet50)\n\nเราจึงเลือกที่จะไม่ใช้วิธีนี้ในการทำ model เนื่องจาก ตอนแรกเราไปหาข้อมูลแล้วค้นพบว่า CNN เป็น  model ที่เหมาะกับการทำ image classification แต่หลังจากลองทำ model ก็พบว่า เราไม่มีความเข้าใจใน model นี้มากพอที่จะมาประยุกต์ใช้กับ project นี้ และเรายังเข้าใจ data ไม่ดีพอ โดยในตอนแรกเราเข้าใจว่า เราสามารถนำรูปภาพไป predict labels ได้เลย \n\n**Final process**\n\nหลังจากไปหาข้อมูลเพิ่มเติม ก็พบว่า รูปภาพไม่สามารถบอก labels ได้โดยตรง แต่ business_id จะเป็นตัวบอก labels โดยที่ รูปภาพที่มี business_id เดียวกัน จะมี labels เดียวกัน\n\n[reference code](https://web.stanford.edu/class/cs231a/prev_projects_2016/Final_Report%20(1).pdf)\n\n3 step ในการแก้ปัญหานี้ \n1. แยก features รูปภาพ เนื่องจากส่วนใหญ่เป็นภาพอาหาร แต่ก็มีภาพเกี่ยวกับร้านอาหารด้วย ดังนั้นเราจึง extract  features ออกมา และ features นั้นควรจะสามารถจัดการกับวัตถุประเภทต่างๆ ในภาพได้ ทั้งอาหาร เครื่องดื่ม เมนู คน ตาราง อาคาร และข้อมูลอื่นๆ โดยเราได้ใช้ CNN เป็น model เพื่อทำ convolutional เพื่อ extract feature(vector) ของรูป จำนวน 20,000 รูป เราใช้ Resnet50 เป็น base model เนื่องจาก มีถึง 50 layer ,ถูก train มากับชุดข้อมูลขนาดใหญ่ และสามารถแยก feature ได้เป็นอย่างดี ปัญหาที่พบในขั้นตอนนี้คือ ในตอนแรกเราได้ใช้รูปจำนวน 50,000 รูป แต่เนื่องด้วย Kaggle Kernel มี RAM ไม่เพียงพอจนไม่สามารถเปิดใช้ GPU เข้ามาช่วยได้ โดยเราแก้ปัญยหาด้วยการลดจำนวนภาพให้เหลือ 20,000 รูป และแบ่ง extract ภาพ เป็นหลายๆcell \n2. รวม features รูปภาพ ที่อยู่ใน business_id เดียวกัน ให้เป็น features ของ business_id นั้นโดยการหา mean ของ feature รูปภาพ จะได้ทั้งหมด 58 business_id ที่มีรูปภายใน business_id ครบ\n3. classify business features ที่ได้จากstep 2  โดยเราจะใช้ KNN ในการทำ model เนื่องจาก เป็น model ที่ใช้เวลา predict สั้นและมี performance ที่ดี และได้กำหนดค่า k =5 โดยเราได้แบ่ง train เป็น 80% ,test เป็น 20%\n    * X_train = features ของแต่ละ business_id \n    * y_train= labels ในแต่ละ business_id \n    * X_test = features รูปภาพ ที่ input เข้ามาใหม่ \n    * y_test = labels ที่ตรงกับ x_test\n\n**Result**\n\nจากการ predict เราได้ accuracy 0.0 และ f1 score 0.65 ที่เราได้ผลลัพธ์เป็นแบบนี้เนื่องจาก ถ้า model predict ค่าจาก x_test ไม่ตรงกับ y_test ทั้งหมด จะถือว่า predict ผิด ซึ่งทำให้ค่า accuracy  เท่ากับ 0.0 และค่า f1 score สูง เนื่องจาก f1 score ให้ความสำคัญกับ labels ที่ predict ถูก โดยไม่จำเป็นต้อง predict labels ถูกทั้งหมด\n","metadata":{}},{"cell_type":"markdown","source":"# Submit the work\n* test model ที่มาจากไฟล์ test_photo แล้วส่งให้ kaggle ตรวจ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport random\n\n# Load datasets\nsample_submission = pd.read_csv('./sample_submission.csv')\ntest_photo_to_biz = pd.read_csv('./test_photo_to_biz.csv')\n\n# ดึงรายชื่อรูปภาพจาก test_photos \nimage_filenames = os.listdir('./test_photos')\n","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# เชื่อม test_photo_to_biz กับ sample_submission ด้วย business IDs\nmerged_data = pd.merge(test_photo_to_biz, sample_submission , on='business_id',how = 'left')\n\n# Group by business_id และ random 1 รูปภาพ มาจาก business_id นั้น\nselected_images = merged_data.groupby('business_id', as_index = False).first()\nselected_images = selected_images.drop(columns=['labels'])\nselected_images","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract features รูปภาพของ test เพื่อนำไปใช้ใน model\ndef extract_features(img_path, model):\n    try:\n        if not os.path.exists(img_path):\n            print(f\"Image not found: {img_path}\")\n            return None\n        img = load_img(img_path, target_size=(256, 256))\n        img_array = img_to_array(img)\n        img_array = np.expand_dims(img_array, axis=0)\n        img_array = preprocess_input(img_array)\n        features = model.predict(img_array)\n        return features.flatten()  # Ensure features are flattened\n    except Exception as e:\n        print(f\"Error processing image {img_path}: {e}\")\n        return None\n\n\nfeature_dict_test = {}\n\nfor _, row in tqdm(selected_images.iterrows(), total=selected_images.shape[0], desc=\"Extracting features\"):\n    img_path = os.path.join('/kaggle/working/test_photos', str(row['photo_id']) + '.jpg')\n    \n    features = extract_features(img_path, model)\n    if features is not None:\n        biz_id = row['business_id']\n        if biz_id not in feature_dict_test:\n            feature_dict_test[biz_id] = []\n        feature_dict_test[biz_id].append(features)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_features = []\ntest_biz_ids = []\nfor biz_id, features_list in feature_dict_test.items():\n    for features in features_list:\n        test_features.append(features)  \n        test_biz_ids.append(biz_id)  \n        \ntest_features_array = np.array(test_features)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict test โดยใช้ KNN model ที่ train แล้ว\npredicted_labels = knn_classifier.predict(test_features_array)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แปลง binary labels เป็น labels จริง\ndef binary_to_indices(binary_labels):\n    return [i for i, label in enumerate(binary_labels) if label == 1]\n\nlabel_indices = [binary_to_indices(row) for row in predicted_labels]\nformatted_labels = [' '.join(map(str, indices)) for indices in label_indices]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# สร้าง dataframe สำหรับ submission\nsubmission_df = pd.DataFrame({\n    'business_id': test_biz_ids,\n    'labels': formatted_labels\n})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sort the DataFrame by business_id if necessary\nsubmission_df = submission_df.sort_values(by='business_id')\n\n# Save to CSV without an index\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# แสดงตัวอย่างที่ predict \nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import load_img  \n\nrandom_entries = selected_images.sample(5)\n\nfor _, row in random_entries.iterrows():\n    img_path = os.path.join('/kaggle/working/test_photos', str(row['photo_id']) + '.jpg')\n    biz_id = row['business_id']\n\n    predicted_label = submission_df[submission_df['business_id'] == biz_id]['labels'].values[0]\n\n    img = load_img(img_path, target_size=(256, 256))\n    plt.imshow(img)\n    plt.title(f\"Business ID: {biz_id}\\nPredicted Labels: {predicted_label}\")\n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**คะแนนหลังจากส่งให้ kaggle**\n* Private score:0.54762\n* Public score:0.54696","metadata":{}}]}