{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-27T16:13:25.002050Z","iopub.execute_input":"2023-11-27T16:13:25.002345Z","iopub.status.idle":"2023-11-27T16:13:25.353144Z","shell.execute_reply.started":"2023-11-27T16:13:25.002319Z","shell.execute_reply":"2023-11-27T16:13:25.352277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# extract files\n!apt install pigz\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz | tar xf -\n!pigz -dc /kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz | tar xf -","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:13:25.355211Z","iopub.execute_input":"2023-11-27T16:13:25.356007Z","iopub.status.idle":"2023-11-27T16:17:20.619395Z","shell.execute_reply.started":"2023-11-27T16:13:25.355962Z","shell.execute_reply":"2023-11-27T16:17:20.618230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training data that maps business ID to labels\ntrain = pd.read_csv('train.csv')\ndisplay(train.head())\nprint('Shape of train data:', train.shape)\nprint('Number of unique businesses:', train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:20.621320Z","iopub.execute_input":"2023-11-27T16:17:20.622298Z","iopub.status.idle":"2023-11-27T16:17:20.665081Z","shell.execute_reply.started":"2023-11-27T16:17:20.622258Z","shell.execute_reply":"2023-11-27T16:17:20.664164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training data that maps photos to business ID\ntrain_photo_to_id = pd.read_csv('train_photo_to_biz_ids.csv')\ndisplay(train_photo_to_id.head())\nprint('Shape of train_photo_to_id:', train_photo_to_id.shape)\nprint('Number of images in training set:', train_photo_to_id.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:20.667467Z","iopub.execute_input":"2023-11-27T16:17:20.667756Z","iopub.status.idle":"2023-11-27T16:17:20.754532Z","shell.execute_reply.started":"2023-11-27T16:17:20.667730Z","shell.execute_reply":"2023-11-27T16:17:20.753519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain_biz=pd.read_csv('./train_photo_to_biz_ids.csv')\ntest_biz=pd.read_csv('./test_photo_to_biz.csv')\ntrain=pd.read_csv('./train.csv')\nsub=pd.read_csv('./sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:20.755662Z","iopub.execute_input":"2023-11-27T16:17:20.755942Z","iopub.status.idle":"2023-11-27T16:17:21.224722Z","shell.execute_reply.started":"2023-11-27T16:17:20.755916Z","shell.execute_reply":"2023-11-27T16:17:21.223716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_biz=train_biz.groupby(\"business_id\").last()\nprint(train_biz.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:21.226119Z","iopub.execute_input":"2023-11-27T16:17:21.226772Z","iopub.status.idle":"2023-11-27T16:17:21.263285Z","shell.execute_reply.started":"2023-11-27T16:17:21.226733Z","shell.execute_reply":"2023-11-27T16:17:21.262364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train.merge(train_biz,on=\"business_id\")","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:21.264407Z","iopub.execute_input":"2023-11-27T16:17:21.264777Z","iopub.status.idle":"2023-11-27T16:17:21.282695Z","shell.execute_reply.started":"2023-11-27T16:17:21.264745Z","shell.execute_reply":"2023-11-27T16:17:21.282003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['labs']=train['labels'].apply(lambda x:str(x).split(' '))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:21.283643Z","iopub.execute_input":"2023-11-27T16:17:21.283907Z","iopub.status.idle":"2023-11-27T16:17:21.292195Z","shell.execute_reply.started":"2023-11-27T16:17:21.283882Z","shell.execute_reply":"2023-11-27T16:17:21.291181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=test_biz.groupby(\"business_id\").last()\ntest","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:21.293492Z","iopub.execute_input":"2023-11-27T16:17:21.293781Z","iopub.status.idle":"2023-11-27T16:17:21.447198Z","shell.execute_reply.started":"2023-11-27T16:17:21.293755Z","shell.execute_reply":"2023-11-27T16:17:21.446275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications import VGG19\nfrom tensorflow.keras.applications.vgg16 import preprocess_input \nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, Flatten\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\n\n# Load VGG19 as the base model\nbase_model = VGG19(weights='imagenet', include_top=False, input_shape=(600, 600, 3))\nn_filters = 32\nkernel_size = 3\npool_size = 2\ninputs = Input(shape=(600, 600, 3))\n\n# VGG19 feature extraction\nbase_features = base_model(inputs, training=False)\n\n# Additional convolutional layers for dimensionality reduction\nx = Conv2D(n_filters, kernel_size, activation='relu')(base_features)\nx = Conv2D(n_filters, kernel_size, activation='relu')(x)\nx = Flatten()(x)\n\n# Create the feature extraction model\nmodel_fet = Model(inputs, x)\n\ndef load_and_preprocess_image(file_path):\n    img = image.load_img(file_path, target_size=(600, 600))\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array = preprocess_input(img_array)\n    return img_array\n\ndef extract_features(file_paths):\n    features = []\n   \n    for i, path in enumerate(file_paths):\n        path = \"/kaggle/working/train_photos/\" + str(path) + \".jpg\"\n        img_array = load_and_preprocess_image(path)\n        feature = model_fet.predict(img_array)\n        feature = np.squeeze(feature)\n        features.append(feature)\n        print(i)\n    return np.array(features)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-11-27T16:17:21.451359Z","iopub.execute_input":"2023-11-27T16:17:21.451652Z","iopub.status.idle":"2023-11-27T16:17:41.672481Z","shell.execute_reply.started":"2023-11-27T16:17:21.451626Z","shell.execute_reply":"2023-11-27T16:17:41.671470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display summary for the feature extraction model\nprint(\"\\nFeature Extraction Model Summary:\")\nmodel_fet.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:17:41.673816Z","iopub.execute_input":"2023-11-27T16:17:41.674198Z","iopub.status.idle":"2023-11-27T16:17:41.703774Z","shell.execute_reply.started":"2023-11-27T16:17:41.674162Z","shell.execute_reply":"2023-11-27T16:17:41.702916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = extract_features(train['photo_id'].tolist())","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-11-27T16:17:41.705268Z","iopub.execute_input":"2023-11-27T16:17:41.705519Z","iopub.status.idle":"2023-11-27T16:21:29.442992Z","shell.execute_reply.started":"2023-11-27T16:17:41.705497Z","shell.execute_reply":"2023-11-27T16:21:29.441957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nmlb = MultiLabelBinarizer()\none_hot_labels = mlb.fit_transform(train['labs'])\none_hot_labels=one_hot_labels[:,:-1]","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:29.445241Z","iopub.execute_input":"2023-11-27T16:21:29.445570Z","iopub.status.idle":"2023-11-27T16:21:29.458428Z","shell.execute_reply.started":"2023-11-27T16:21:29.445543Z","shell.execute_reply":"2023-11-27T16:21:29.457578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, classification_report\nX_train, X_test, y_train, y_test = train_test_split(features, one_hot_labels, test_size=0.2, random_state=42)\nbase_classifier = LogisticRegression(C=0.001)\nclassifier = MultiOutputClassifier(base_classifier)\nclassifier.fit(X_train, y_train)\npredictions = classifier.predict(X_test)\nprint(\"\\nClassification Report for VGG19-based Model:\\n\", classification_report(y_test, predictions))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:29.459535Z","iopub.execute_input":"2023-11-27T16:21:29.459825Z","iopub.status.idle":"2023-11-27T16:21:38.108557Z","shell.execute_reply.started":"2023-11-27T16:21:29.459791Z","shell.execute_reply":"2023-11-27T16:21:38.107266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pipline_predict(photo_id):\n    path=\"/kaggle/working/test_photos/\"+ str(photo_id)+\".jpg\"\n    img = image.load_img(path, target_size=(600, 600))\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array = preprocess_input(img_array)\n    feature = model_fet.predict(img_array)\n    feature = np.squeeze(feature)\n    y_pred=classifier.predict([feature])\n    y_pred=np.where(y_pred > 0.5)[1]\n    return ' '.join(map(str, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:38.110656Z","iopub.execute_input":"2023-11-27T16:21:38.111496Z","iopub.status.idle":"2023-11-27T16:21:38.126442Z","shell.execute_reply.started":"2023-11-27T16:21:38.111446Z","shell.execute_reply":"2023-11-27T16:21:38.124799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pipline_predict_proba(photo_id):\n    path=\"/kaggle/working/test_photos/\"+ str(photo_id)+\".jpg\"\n    img = image.load_img(path, target_size=(600, 600))\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array = preprocess_input(img_array)\n    feature = model_fet.predict(img_array)\n    feature = np.squeeze(feature)\n    y_pred=classifier.predict_proba([feature])\n    return y_pred","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:38.132317Z","iopub.execute_input":"2023-11-27T16:21:38.132996Z","iopub.status.idle":"2023-11-27T16:21:38.143199Z","shell.execute_reply.started":"2023-11-27T16:21:38.132942Z","shell.execute_reply":"2023-11-27T16:21:38.141949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipline_predict(405544)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:38.145191Z","iopub.execute_input":"2023-11-27T16:21:38.146150Z","iopub.status.idle":"2023-11-27T16:21:38.334891Z","shell.execute_reply.started":"2023-11-27T16:21:38.146100Z","shell.execute_reply":"2023-11-27T16:21:38.333995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=test.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:21:38.335990Z","iopub.execute_input":"2023-11-27T16:21:38.336317Z","iopub.status.idle":"2023-11-27T16:21:38.342073Z","shell.execute_reply.started":"2023-11-27T16:21:38.336289Z","shell.execute_reply":"2023-11-27T16:21:38.340825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"labels\"]=np.vectorize(pipline_predict)(test[\"photo_id\"])","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-11-27T16:21:38.343904Z","iopub.execute_input":"2023-11-27T16:21:38.344262Z","iopub.status.idle":"2023-11-27T16:40:37.385557Z","shell.execute_reply.started":"2023-11-27T16:21:38.344212Z","shell.execute_reply":"2023-11-27T16:40:37.384534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display, Image\nimport random \n\n# สุ่มเลือก 3 รูปภาพ\nrandom_photo_ids = random.sample(test['photo_id'].tolist(), 3)\n\n# วนลูปแสดงผลลัพธ์สำหรับทุกรูปภาพที่สุ่มมา\nfor photo_id in random_photo_ids:\n    prediction = pipline_predict(photo_id)\n    prediction_proba = pipline_predict_proba(photo_id)\n    \n    # โหลดรูปภาพ\n    img_path = \"/kaggle/working/test_photos/\" + str(photo_id) + \".jpg\"\n    img = Image(filename=img_path, width=300, height=300)\n    \n    # แสดงรูปภาพ\n    display(img)\n\n    print(\"Photo ID:\", photo_id)\n    print(\"Prediction:\", prediction)\n    print(\"Predicted Probabilities:\")\n    print(prediction_proba)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:40:37.386808Z","iopub.execute_input":"2023-11-27T16:40:37.387090Z","iopub.status.idle":"2023-11-27T16:40:38.067403Z","shell.execute_reply.started":"2023-11-27T16:40:37.387064Z","shell.execute_reply":"2023-11-27T16:40:38.066460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[[\"business_id\",\"labels\"]].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-27T16:40:38.068730Z","iopub.execute_input":"2023-11-27T16:40:38.069130Z","iopub.status.idle":"2023-11-27T16:40:38.106551Z","shell.execute_reply.started":"2023-11-27T16:40:38.069093Z","shell.execute_reply":"2023-11-27T16:40:38.105633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}