{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-22T11:12:26.071087Z","iopub.execute_input":"2023-11-22T11:12:26.071824Z","iopub.status.idle":"2023-11-22T11:12:26.08507Z","shell.execute_reply.started":"2023-11-22T11:12:26.071761Z","shell.execute_reply":"2023-11-22T11:12:26.083621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Import Library**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.decomposition import PCA\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn import metrics\nfrom skimage import io\nimport os\nfrom tqdm import tqdm\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:12:26.08853Z","iopub.execute_input":"2023-11-22T11:12:26.090342Z","iopub.status.idle":"2023-11-22T11:12:26.103634Z","shell.execute_reply.started":"2023-11-22T11:12:26.090286Z","shell.execute_reply":"2023-11-22T11:12:26.102347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tarfile\n\n#Specify the path to your TGZ file\ntrain_photos = \"/kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz\"\ntest_photos = \"/kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz\"\ntrain_csv = \"/kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz\"\ntrain_biz = \"/kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz\"\ntest_biz = \"/kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz\"\nsample_sub = \"/kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz\"\n\n#Specify the directory where you want to extract the contents\nextracted_dir_path = \"/kaggle/working/data_source\"\n\n#Create the directory if it doesn't exist\n!mkdir -p {extracted_dir_path}\n\n#Open the TGZ file\nwith tarfile.open(train_photos, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\nwith tarfile.open(test_photos, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\nwith tarfile.open(train_csv, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\nwith tarfile.open(train_biz, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\nwith tarfile.open(test_biz, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\nwith tarfile.open(sample_sub, 'r:gz') as tar:\n    # Extract all contents to the specified directory\n    tar.extractall(path=extracted_dir_path)\n\n#Check the extracted files\n!ls {extracted_dir_path}\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:12:26.105279Z","iopub.execute_input":"2023-11-22T11:12:26.106388Z","iopub.status.idle":"2023-11-22T11:21:27.141222Z","shell.execute_reply.started":"2023-11-22T11:12:26.106341Z","shell.execute_reply":"2023-11-22T11:21:27.139495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Set the path to the data\ndata_path = '/kaggle/working/data_source'\n\n# Load the training data\ntrain_data = pd.read_csv(os.path.join(data_path, 'train.csv'))\ntrain_photo_to_biz_ids = pd.read_csv(os.path.join(data_path, 'train_photo_to_biz_ids.csv'))\n\n# Display basic information about the training data\nprint(train_data.info())\nprint(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:21:27.143377Z","iopub.execute_input":"2023-11-22T11:21:27.143882Z","iopub.status.idle":"2023-11-22T11:21:27.275253Z","shell.execute_reply.started":"2023-11-22T11:21:27.143837Z","shell.execute_reply":"2023-11-22T11:21:27.274189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom PIL import Image\nimport os\nfrom tqdm import tqdm\n\n# Set the path to the data\ndata_path = '/kaggle/working/data_source/'\n\n# Common size for resizing the images\ncommon_size = (224, 224)\n\n# Function to load and preprocess images from a directory\ndef load_and_preprocess_images(image_ids, image_path):\n    images = []\n    for img_id in tqdm(image_ids):\n        img_path = os.path.join(image_path, f'{img_id}.jpg')\n        img = Image.open(img_path)\n        \n        # Resize the image to common size\n        img = img.resize(common_size)\n        \n        # Convert to numpy array\n        img_array = np.array(img)\n        \n        # Perform any preprocessing if necessary (resizing, normalization, etc.)\n        images.append(img_array)\n    return np.array(images)\n\n# Load the training data\ntrain_data = pd.read_csv(os.path.join(data_path, 'train.csv'))\ntrain_photo_to_biz_ids = pd.read_csv(os.path.join(data_path, 'train_photo_to_biz_ids.csv'))\n\n# Load and preprocess training images in batches of 50,000\nbatch_size = 25000\nnum_batches = len(train_photo_to_biz_ids) // batch_size\n\nfor i in range(num_batches):\n    start_idx = i * batch_size\n    end_idx = (i + 1) * batch_size\n    batch_image_ids = train_photo_to_biz_ids['photo_id'].iloc[start_idx:end_idx]\n    \n    # Load and preprocess images directly from the directory\n    batch_images = load_and_preprocess_images(batch_image_ids, os.path.join(data_path, 'train_photos'))\n    \n    # Perform further processing or training with the batch_images and train_data\n    # ...\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:21:27.278309Z","iopub.execute_input":"2023-11-22T11:21:27.279345Z","iopub.status.idle":"2023-11-22T11:47:38.605407Z","shell.execute_reply.started":"2023-11-22T11:21:27.279306Z","shell.execute_reply":"2023-11-22T11:47:38.602974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA LOADING ","metadata":{}},{"cell_type":"code","source":"display(train_data.head())\nprint('shape of train data :',train_data.shape)\nprint('Nomber of unique business :',train_data.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.608208Z","iopub.execute_input":"2023-11-22T11:47:38.60875Z","iopub.status.idle":"2023-11-22T11:47:38.651974Z","shell.execute_reply.started":"2023-11-22T11:47:38.608707Z","shell.execute_reply":"2023-11-22T11:47:38.650501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3=pd.DataFrame(train_data)\ndf3","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.655208Z","iopub.execute_input":"2023-11-22T11:47:38.656421Z","iopub.status.idle":"2023-11-22T11:47:38.677432Z","shell.execute_reply.started":"2023-11-22T11:47:38.656344Z","shell.execute_reply":"2023-11-22T11:47:38.675777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test=pd.merge(train_data, train_photo_to_biz_ids, on='business_id',how='left') \ndf2=pd.DataFrame(data_test)\nprint('Number of training images:', len(train_data))\ndf2","metadata":{"execution":{"iopub.status.busy":"2023-11-22T12:32:23.314546Z","iopub.execute_input":"2023-11-22T12:32:23.315263Z","iopub.status.idle":"2023-11-22T12:32:23.378157Z","shell.execute_reply.started":"2023-11-22T12:32:23.315216Z","shell.execute_reply":"2023-11-22T12:32:23.376448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:41:17.26236Z","iopub.execute_input":"2023-11-22T13:41:17.263671Z","iopub.status.idle":"2023-11-22T13:41:18.20179Z","shell.execute_reply.started":"2023-11-22T13:41:17.263607Z","shell.execute_reply":"2023-11-22T13:41:18.200302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.759448Z","iopub.execute_input":"2023-11-22T11:47:38.760672Z","iopub.status.idle":"2023-11-22T11:47:38.769291Z","shell.execute_reply.started":"2023-11-22T11:47:38.760618Z","shell.execute_reply":"2023-11-22T11:47:38.768075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1=pd.DataFrame(train_data1)\ndf1","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:39:37.977225Z","iopub.execute_input":"2023-11-22T13:39:37.978624Z","iopub.status.idle":"2023-11-22T13:39:38.042984Z","shell.execute_reply.started":"2023-11-22T13:39:37.978572Z","shell.execute_reply":"2023-11-22T13:39:38.041306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(train_data)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.841176Z","iopub.status.idle":"2023-11-22T11:47:38.842745Z","shell.execute_reply.started":"2023-11-22T11:47:38.842324Z","shell.execute_reply":"2023-11-22T11:47:38.842365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Business id to labels dataframe\nprint('Total number of missing labels:', train_data['labels'].isnull().sum())\ndisplay(train_data[train_data['labels'].isnull()])","metadata":{"execution":{"iopub.status.busy":"2023-11-22T12:32:31.185748Z","iopub.execute_input":"2023-11-22T12:32:31.186675Z","iopub.status.idle":"2023-11-22T12:32:31.204096Z","shell.execute_reply.started":"2023-11-22T12:32:31.18663Z","shell.execute_reply":"2023-11-22T12:32:31.202692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2['labs']=df2['labels'].apply(lambda x:str(x).split(' '))\ndf2","metadata":{"execution":{"iopub.status.busy":"2023-11-22T12:33:16.217812Z","iopub.execute_input":"2023-11-22T12:33:16.218904Z","iopub.status.idle":"2023-11-22T12:33:17.08349Z","shell.execute_reply.started":"2023-11-22T12:33:16.218839Z","shell.execute_reply":"2023-11-22T12:33:17.082146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ลองเปิดรุป","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport plotly\nimport plotly.graph_objs as go\nimport cv2\nimport random\nimport os\n%matplotlib inline\n\n# Init plotly for offline plotting\nplotly.offline.init_notebook_mode(connected=True)\n\nprint('Pandas version:', pd.__version__)\nprint('Numpy version:', np.__version__)\nprint('OpenCV version:', cv2.__version__)\nprint('Plotly version:', plotly.__version__)\nprint(os.listdir(\"/kaggle/working/data_source\"))","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:27:54.174678Z","iopub.execute_input":"2023-11-22T13:27:54.175193Z","iopub.status.idle":"2023-11-22T13:27:54.191377Z","shell.execute_reply.started":"2023-11-22T13:27:54.17516Z","shell.execute_reply":"2023-11-22T13:27:54.189905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_images()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:22:51.728966Z","iopub.execute_input":"2023-11-22T13:22:51.73059Z","iopub.status.idle":"2023-11-22T13:22:51.795601Z","shell.execute_reply.started":"2023-11-22T13:22:51.730543Z","shell.execute_reply":"2023-11-22T13:22:51.793588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '/kaggle/working/data_source/train_photos'\ntrain_imgs = os.listdir(train_dir)\n\ntest_dir = '/kaggle/working/data_source/test_photos'\ntest_imgs = os.listdir(test_dir)\n\nprint('Number of training images:', len(train_imgs))\nprint('Number of testing images:', len(test_imgs))","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:30:30.359217Z","iopub.execute_input":"2023-11-22T13:30:30.360428Z","iopub.status.idle":"2023-11-22T13:30:31.07322Z","shell.execute_reply.started":"2023-11-22T13:30:30.360342Z","shell.execute_reply":"2023-11-22T13:30:31.071878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomly sample 8 images\nimgs_samples = random.sample(train_imgs, 8)\n\n# Plot random sample of 8 images\nplt.figure(figsize=(15, 10))\nfor i in range(len(imgs_samples)):\n    # OpenCV2 reads images in BGR format\n    img = cv2.imread(os.path.join(train_dir, imgs_samples[i]))\n    # Switch color channels to RGB to make compatible with matplotlib imshow func\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    # Grab image's business ID and labels\n    business = train_photo_to_id.loc[train_photo_to_id['photo_id'] == int(imgs_samples[i][:-4]), 'business_id']\n    labels = train.loc[train['business_id'] == business.values[0], 'labels']\n    # Annotate each image with image ID, business ID, and labels\n    title = \"Image ID: \" + imgs_samples[i] + ' Business: ' + str(business.values[0]) + '\\nLabels: ' + ''.join(labels.values)\n    # Plot the image\n    plt.subplot(2, 4, i+1)\n    plt.tight_layout(pad=0.4, w_pad=0.5, h_pad=1.0)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(title)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:31:18.93206Z","iopub.execute_input":"2023-11-22T13:31:18.932657Z","iopub.status.idle":"2023-11-22T13:31:19.216469Z","shell.execute_reply.started":"2023-11-22T13:31:18.932609Z","shell.execute_reply":"2023-11-22T13:31:19.214499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"จบการลอง","metadata":{}},{"cell_type":"markdown","source":"ทดลองอีกอัน","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch.utils.tensorboard import SummaryWriter\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:36:27.461297Z","iopub.execute_input":"2023-11-22T13:36:27.462235Z","iopub.status.idle":"2023-11-22T13:36:27.523882Z","shell.execute_reply.started":"2023-11-22T13:36:27.462192Z","shell.execute_reply":"2023-11-22T13:36:27.522384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.manual_seed(2020)\ntorch.cuda.manual_seed(2020)\nnp.random.seed(2020)\nrandom.seed(2020)\ntorch.backends.cudnn.deterministic = True","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:36:29.398416Z","iopub.execute_input":"2023-11-22T13:36:29.398858Z","iopub.status.idle":"2023-11-22T13:36:29.413964Z","shell.execute_reply.started":"2023-11-22T13:36:29.398821Z","shell.execute_reply":"2023-11-22T13:36:29.41269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labelsint=df2.labs.tolist()\nfor i in tqdm(labelsint):\n    for j in range(len(i)):\n        #print(j)\n        #break\n        i[j]=int(i[j])\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:38:15.209142Z","iopub.execute_input":"2023-11-22T13:38:15.209718Z","iopub.status.idle":"2023-11-22T13:38:15.451117Z","shell.execute_reply.started":"2023-11-22T13:38:15.209676Z","shell.execute_reply":"2023-11-22T13:38:15.44903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2['labsint']=labelsint\ndf2","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:43:23.826284Z","iopub.execute_input":"2023-11-22T13:43:23.827807Z","iopub.status.idle":"2023-11-22T13:43:23.91477Z","shell.execute_reply.started":"2023-11-22T13:43:23.827735Z","shell.execute_reply":"2023-11-22T13:43:23.912941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport time\nimport numpy as np\nfrom PIL import Image\nfrom torch.utils.data.dataset import Dataset\nfrom tqdm import tqdm\nfrom torchvision import transforms\nfrom torchvision import models\nimport torch\nfrom torch.utils.tensorboard import SummaryWriter\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom torch import nn\nfrom torch.utils.data.dataloader import DataLoader\nfrom matplotlib import pyplot as plt\nfrom numpy import printoptions\nimport requests\nimport tarfile\nimport random\nimport json\nfrom shutil import copyfile","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:45:48.660253Z","iopub.execute_input":"2023-11-22T13:45:48.660894Z","iopub.status.idle":"2023-11-22T13:45:49.091832Z","shell.execute_reply.started":"2023-11-22T13:45:48.660849Z","shell.execute_reply":"2023-11-22T13:45:49.090568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics, model_selection, preprocessing\ndf_train, df_valid = model_selection.train_test_split(\n        df2, test_size=0.1, random_state=42\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:43:32.820007Z","iopub.execute_input":"2023-11-22T13:43:32.820591Z","iopub.status.idle":"2023-11-22T13:43:32.956482Z","shell.execute_reply.started":"2023-11-22T13:43:32.820547Z","shell.execute_reply":"2023-11-22T13:43:32.95484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NusDataset(Dataset):\n    def __init__(self, data_path, data, transforms):\n        self.transforms = transforms\n        data=data\n        samples = data['photo_id'].tolist()\n        labs=data['labsint'].tolist()\n        self.classes = [0,1,2,3,4,5,6,7,8]\n\n        self.imgs = []\n        self.annos = []\n        self.data_path = data_path\n        #print('loading', anno_path)\n        for sample in samples:\n            self.imgs.append(sample)\n        for lab in labs:\n            self.annos.append(lab)\n            \n        for item_id in range(len(self.annos)):\n            item = self.annos[item_id]\n            vector = [cls in item for cls in self.classes]\n            self.annos[item_id] = np.array(vector, dtype=float)\n\n    def __getitem__(self, item):\n        anno = self.annos[item]\n        img_path = os.path.join(self.data_path, str(self.imgs[item])+'.jpg')\n        img = Image.open(img_path)\n        if self.transforms is not None:\n            img = self.transforms(img)\n        return img, anno\n\n        \n\n    def __len__(self):\n        return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:45:53.277814Z","iopub.execute_input":"2023-11-22T13:45:53.278289Z","iopub.status.idle":"2023-11-22T13:45:53.294482Z","shell.execute_reply.started":"2023-11-22T13:45:53.278253Z","shell.execute_reply":"2023-11-22T13:45:53.292845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_val = NusDataset('/kaggle/working/data_source/train_photos', df_valid, None)\ndataset_train = NusDataset('/kaggle/working/data_source/test_photos', df_train, None)\n\n\nclass Resnext50(nn.Module):\n    def __init__(self, n_classes):\n        super().__init__()\n        resnet = models.resnext50_32x4d(pretrained=True)\n        resnet.fc = nn.Sequential(\n            nn.Dropout(p=0.2),\n            nn.Linear(in_features=resnet.fc.in_features, out_features=n_classes)\n        )\n        self.base_model = resnet\n        self.sigm = nn.Sigmoid()\n\n    def forward(self, x):\n        return self.sigm(self.base_model(x))","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:45:56.865172Z","iopub.execute_input":"2023-11-22T13:45:56.865693Z","iopub.status.idle":"2023-11-22T13:45:58.319353Z","shell.execute_reply.started":"2023-11-22T13:45:56.865652Z","shell.execute_reply":"2023-11-22T13:45:58.317818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_photos_new = '/kaggle/working/data_source/train_photos'","metadata":{"execution":{"iopub.status.busy":"2023-11-22T14:22:01.857641Z","iopub.execute_input":"2023-11-22T14:22:01.858802Z","iopub.status.idle":"2023-11-22T14:22:01.869018Z","shell.execute_reply.started":"2023-11-22T14:22:01.858736Z","shell.execute_reply":"2023-11-22T14:22:01.864399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Directory path\ndirectory_path = '/kaggle/working/data_source/train_photos'\n\ntry:\n    # List all files and directories in the specified path\n    contents = os.listdir(directory_path)\n\n    # Print the contents\n    print(f\"Contents of {directory_path}:\")\n    for item in contents:\n        print(item)\nexcept FileNotFoundError:\n    print(f\"Error: The directory {directory_path} does not exist.\")\nexcept Exception as e:\n    print(f\"Error: {e}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-22T15:33:05.684644Z","iopub.execute_input":"2023-11-22T15:33:05.685359Z","iopub.status.idle":"2023-11-22T15:33:09.055479Z","shell.execute_reply.started":"2023-11-22T15:33:05.68532Z","shell.execute_reply":"2023-11-22T15:33:09.05419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n# LOAD AN IMAGE USING 'IMREAD'\nimg = cv2.imread('/kaggle/working/data_source/train_photos/80603.jpg')\n# DISPLAY\ncv2.imshow('/kaggle/working/data_source/train_photos/80603.jpg')\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T14:42:11.23735Z","iopub.execute_input":"2023-11-22T14:42:11.23786Z","iopub.status.idle":"2023-11-22T14:42:11.304374Z","shell.execute_reply.started":"2023-11-22T14:42:11.237816Z","shell.execute_reply":"2023-11-22T14:42:11.302251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_metrics(pred, target, threshold=0.5):\n    pred = np.array(pred > threshold, dtype=float)\n    return {'micro/precision': precision_score(y_true=target, y_pred=pred, average='micro'),\n            'micro/recall': recall_score(y_true=target, y_pred=pred, average='micro'),\n            'micro/f1': f1_score(y_true=target, y_pred=pred, average='micro'),\n            'macro/precision': precision_score(y_true=target, y_pred=pred, average='macro'),\n            'macro/recall': recall_score(y_true=target, y_pred=pred, average='macro'),\n            'macro/f1': f1_score(y_true=target, y_pred=pred, average='macro'),\n            'samples/precision': precision_score(y_true=target, y_pred=pred, average='samples'),\n            'samples/recall': recall_score(y_true=target, y_pred=pred, average='samples'),\n            'samples/f1': f1_score(y_true=target, y_pred=pred, average='samples'),\n            }","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:47:32.305987Z","iopub.execute_input":"2023-11-22T13:47:32.306558Z","iopub.status.idle":"2023-11-22T13:47:32.317665Z","shell.execute_reply.started":"2023-11-22T13:47:32.306511Z","shell.execute_reply":"2023-11-22T13:47:32.316162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the training parameters.\nnum_workers = 8 \nlr = 1e-4 # Learning rate\nbatch_size = 32\nsave_freq = 35 # checkpoint frequency (epochs)\ntest_freq = 200 # Test model frequency (iterations)\nmax_epoch_number = 15 # Number of epochs for training \n\n\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\n\ndevice = torch.device('cuda')\n# Save path for checkpoints\nsave_path = 'chekpoints/'\n# Save path for logs\nlogdir = 'logs/'","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:47:43.169245Z","iopub.execute_input":"2023-11-22T13:47:43.169852Z","iopub.status.idle":"2023-11-22T13:47:43.179674Z","shell.execute_reply.started":"2023-11-22T13:47:43.169805Z","shell.execute_reply":"2023-11-22T13:47:43.177661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def checkpoint_save(model, save_path, epoch):\n    f = os.path.join(save_path, 'checkpoint-{:06d}.pth'.format(epoch))\n    if 'module' in dir(model):\n        torch.save(model.module.state_dict(), f)\n    else:\n        torch.save(model.state_dict(), f)\n    print('saved checkpoint:', f)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:47:50.299819Z","iopub.execute_input":"2023-11-22T13:47:50.300306Z","iopub.status.idle":"2023-11-22T13:47:50.309533Z","shell.execute_reply.started":"2023-11-22T13:47:50.300269Z","shell.execute_reply":"2023-11-22T13:47:50.307691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_transform = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])\nprint(tuple(np.array(np.array(mean)*255).tolist()))\n\n# Train preprocessing\ntrain_transform = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.RandomHorizontalFlip(),\n    transforms.ColorJitter(),\n    transforms.RandomAffine(degrees=20, translate=(0.2, 0.2), scale=(0.5, 1.5),\n                            shear=None, resample=False, \n                            fillcolor=tuple(np.array(np.array(mean)*255).astype(int).tolist())),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:48:00.716057Z","iopub.execute_input":"2023-11-22T13:48:00.716554Z","iopub.status.idle":"2023-11-22T13:48:00.82776Z","shell.execute_reply.started":"2023-11-22T13:48:00.716515Z","shell.execute_reply":"2023-11-22T13:48:00.825693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_annotations = os.path.join(img_folder, 'small_test.json')\n#train_annotations = os.path.join(img_folder, 'small_train.json')\n\ntest_dataset = NusDataset('./train_photos', df_valid, val_transform)\ntrain_dataset = NusDataset('./train_photos', df_train, train_transform)\n\n\n\ntrain_dataloader = DataLoader(train_dataset, batch_size=batch_size, num_workers=num_workers, shuffle=True,\n                              drop_last=True)\ntest_dataloader = DataLoader(test_dataset, batch_size=batch_size, num_workers=num_workers)\n\nnum_train_batches = int(np.ceil(len(train_dataset) / batch_size))\n\n# Initialize the model\nmodel = Resnext50(len(train_dataset.classes))\n# Switch model to the training mode and move it to GPU.\nmodel.train()\nmodel = model.to(device)\n\noptimizer = torch.optim.Adam(model.parameters(), lr=lr)\n\n\nif torch.cuda.device_count() > 1:\n    model = nn.DataParallel(model)\n\nos.makedirs(save_path, exist_ok=True)\n\n# Loss function\ncriterion = nn.BCELoss()\n# Tensoboard logger\nlogger = SummaryWriter(logdir)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T13:48:22.353724Z","iopub.execute_input":"2023-11-22T13:48:22.354902Z","iopub.status.idle":"2023-11-22T13:48:22.559349Z","shell.execute_reply.started":"2023-11-22T13:48:22.354848Z","shell.execute_reply":"2023-11-22T13:48:22.557465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nmodel.eval()\nlabel_notation = {0: 'good_for_lunch', 1: 'good_for_dinner', 2: 'takes_reservations',  3: 'outdoor_seating',\n                  4: 'restaurant_is_expensive', 5: 'has_alcohol', 6: 'has_table_service', 7: 'ambience_is_classy',\n                  8: 'good_for_kids'}\nif torch.cuda.is_available():\n    model.cuda()\nfor sample_id in [1,2,3,4,6,7,8,9,10,11]:\n    test_img, test_labels = test_dataset[sample_id]\n    print(type(test_img))\n    test_img = test_img.cuda()\n    test_img_path = os.path.join(train_photos, str(test_dataset.imgs[sample_id])+'.jpg')\n    with torch.no_grad():\n        raw_pred = model(test_img.unsqueeze(0)).cpu().numpy()[0]\n        raw_pred = np.array(raw_pred > 0.5, dtype=float)\n\n    predicted_labels = np.array(dataset_val.classes)[np.argwhere(raw_pred > 0)[:, 0]]\n    if not len(predicted_labels):\n        predicted_labels = ['no predictions']\n    img_labels = np.array(dataset_val.classes)[np.argwhere(test_labels > 0)[:, 0]]\n    \n    result = [label_notation[p] for p in predicted_labels]\n    expected = [label_notation[p] for p in img_labels]\n    plt.imshow(Image.open(test_img_path))\n    print(result)\n    print(expected)\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.862072Z","iopub.status.idle":"2023-11-22T11:47:38.862627Z","shell.execute_reply.started":"2023-11-22T11:47:38.862374Z","shell.execute_reply":"2023-11-22T11:47:38.862411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomly sample 8 images\nimport random\nimgs_samples = random.sample(train_photos, 8)\n\n# Plot random sample of 8 images\nplt.figure(figsize=(15, 10))\nfor i in range(len(imgs_samples)):\n    # OpenCV2 reads images in BGR format\n    img = cv2.imread(os.path.join(train_dir, imgs_samples[i]))\n    # Switch color channels to RGB to make compatible with matplotlib imshow func\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    # Grab image's business ID and labels\n    business = train_photo_to_id.loc[train_photo_to_id['photo_id'] == int(imgs_samples[i][:-4]), 'business_id']\n    labels = train.loc[train['business_id'] == business.values[0], 'labels']\n    # Annotate each image with image ID, business ID, and labels\n    title = \"Image ID: \" + imgs_samples[i] + ' Business: ' + str(business.values[0]) + '\\nLabels: ' + ''.join(labels.values)\n    # Plot the image\n    plt.subplot(2, 4, i+1)\n    plt.tight_layout(pad=0.4, w_pad=0.5, h_pad=1.0)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(title)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.863999Z","iopub.status.idle":"2023-11-22T11:47:38.864481Z","shell.execute_reply.started":"2023-11-22T11:47:38.864245Z","shell.execute_reply":"2023-11-22T11:47:38.864265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training data that maps photos to business ID\ndisplay(train_data_merg.head())\nprint('Shape of train_photo_to_id:',train_data_merg.shape)\nprint('Number of images in training set:', train_data_merg.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.865985Z","iopub.status.idle":"2023-11-22T11:47:38.866411Z","shell.execute_reply.started":"2023-11-22T11:47:38.866195Z","shell.execute_reply":"2023-11-22T11:47:38.866214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.867726Z","iopub.status.idle":"2023-11-22T11:47:38.868144Z","shell.execute_reply.started":"2023-11-22T11:47:38.867936Z","shell.execute_reply":"2023-11-22T11:47:38.867955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, (3, 3), input_shape=(img_height, img_width, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(num_classes, activation='sigmoid'))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.870023Z","iopub.status.idle":"2023-11-22T11:47:38.87048Z","shell.execute_reply.started":"2023-11-22T11:47:38.870253Z","shell.execute_reply":"2023-11-22T11:47:38.870274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.872177Z","iopub.status.idle":"2023-11-22T11:47:38.872584Z","shell.execute_reply.started":"2023-11-22T11:47:38.872371Z","shell.execute_reply":"2023-11-22T11:47:38.872406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model.summary()) \n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.873825Z","iopub.status.idle":"2023-11-22T11:47:38.874224Z","shell.execute_reply.started":"2023-11-22T11:47:38.874023Z","shell.execute_reply":"2023-11-22T11:47:38.874042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_images, train_labels, epochs=10, batch_size=32, validation_data=(val_images, val_labels)) \n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.875216Z","iopub.status.idle":"2023-11-22T11:47:38.875621Z","shell.execute_reply.started":"2023-11-22T11:47:38.875422Z","shell.execute_reply":"2023-11-22T11:47:38.875441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_photo_to_biz_ids = pd.read_csv(os.path.join(data_path, 'test_photo_to_biz_ids.csv'))\ntest_image_ids = test_photo_to_biz_ids['photo_id']\ntest_images = load_and_preprocess_images(test_image_ids, os.path.join(data_path, 'test_photos'))","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.876727Z","iopub.status.idle":"2023-11-22T11:47:38.877131Z","shell.execute_reply.started":"2023-11-22T11:47:38.876925Z","shell.execute_reply":"2023-11-22T11:47:38.876944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = model.predict(test_images)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.878247Z","iopub.status.idle":"2023-11-22T11:47:38.878669Z","shell.execute_reply.started":"2023-11-22T11:47:38.878467Z","shell.execute_reply":"2023-11-22T11:47:38.878486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({'business_id': test_photo_to_biz_ids['business_id']})\nsubmission_df[attribute_columns] = (test_predictions > threshold).astype(int)\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.879875Z","iopub.status.idle":"2023-11-22T11:47:38.880307Z","shell.execute_reply.started":"2023-11-22T11:47:38.880069Z","shell.execute_reply":"2023-11-22T11:47:38.880087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_photos)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.881721Z","iopub.status.idle":"2023-11-22T11:47:38.882118Z","shell.execute_reply.started":"2023-11-22T11:47:38.881913Z","shell.execute_reply":"2023-11-22T11:47:38.881932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# > -----------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"ทดลอง MODEL จ้า","metadata":{}},{"cell_type":"code","source":"!pip install tensorflow\n\n!pip install tensorflow-gpu\n     ","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.883275Z","iopub.status.idle":"2023-11-22T11:47:38.883733Z","shell.execute_reply.started":"2023-11-22T11:47:38.883505Z","shell.execute_reply":"2023-11-22T11:47:38.883524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout, BatchNormalization, Conv2D, MaxPool2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing import image\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.886413Z","iopub.status.idle":"2023-11-22T11:47:38.886896Z","shell.execute_reply.started":"2023-11-22T11:47:38.886671Z","shell.execute_reply":"2023-11-22T11:47:38.886693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.889062Z","iopub.status.idle":"2023-11-22T11:47:38.889525Z","shell.execute_reply.started":"2023-11-22T11:47:38.8893Z","shell.execute_reply":"2023-11-22T11:47:38.889319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_photos.getname()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.891641Z","iopub.status.idle":"2023-11-22T11:47:38.892081Z","shell.execute_reply.started":"2023-11-22T11:47:38.891874Z","shell.execute_reply":"2023-11-22T11:47:38.891893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(train_photos)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.893233Z","iopub.status.idle":"2023-11-22T11:47:38.893687Z","shell.execute_reply.started":"2023-11-22T11:47:38.893486Z","shell.execute_reply":"2023-11-22T11:47:38.893506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.895021Z","iopub.status.idle":"2023-11-22T11:47:38.895475Z","shell.execute_reply.started":"2023-11-22T11:47:38.895255Z","shell.execute_reply":"2023-11-22T11:47:38.895275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_width = 350\nimg_height = 350\n\nX = []\n\nfor i in tqdm(range(train_data.shape[0])):\n  path = '/kaggle/input/yelp-restaurant-photo-classification/test_photos' + train_data['Id'][i] + '.jpg'\n  img = image.load_img(path, target_size=(img_width, img_height, 3))\n  img = image.img_to_array(img)\n  img = img/255.0\n  X.append(img)\n\nX = np.array(train_photos)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.897087Z","iopub.status.idle":"2023-11-22T11:47:38.897533Z","shell.execute_reply.started":"2023-11-22T11:47:38.89731Z","shell.execute_reply":"2023-11-22T11:47:38.89733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt install -y caffe-cpu","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.899637Z","iopub.status.idle":"2023-11-22T11:47:38.900074Z","shell.execute_reply.started":"2023-11-22T11:47:38.899866Z","shell.execute_reply":"2023-11-22T11:47:38.899886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%writefile training_image_features.py\nimport numpy as np\nimport pandas as pd\nimport tarfile\nimport skimage\nimport io\nimport h5py\nimport os\nimport caffe\nimport time\n\n# Paths\nCAFFE_HOME = \"/content/drive/My Drive/Yelp-Restaurant-Classification/Model/caffe/\"\nDATA_HOME = \"/content/drive/My Drive/Yelp-Restaurant-Classification/Model/data/\"\nFEATURES_HOME = '/content/drive/My Drive/Yelp-Restaurant-Classification/Model/features/'\nDATA_ = \"/content/\"\n# Model creation\n# Using bvlc_reference_caffenet model for training\nimport os\nif os.path.isfile(CAFFE_HOME + 'models/bvlc_reference_caffenet/bvlc_reference_caffenet.caffemodel'):\n    print('CaffeNet found.')\nelse:\n    print('Downloading pre-trained CaffeNet model...')\n    #os.system('/caffe/scripts/download_model_binary.py /caffe/models/bvlc_reference_caffenet')\n    !python /content/drive/My\\ Drive/Yelp-Restaurant-Classification/Model/caffe/scripts/download_model_binary.py /content/drive/My\\ Drive/Yelp-Restaurant-Classification/Model/caffe//models/bvlc_reference_caffenet\n\nmodel_def = CAFFE_HOME + 'models/bvlc_reference_caffenet/deploy.prototxt'\nmodel_weights = CAFFE_HOME + 'models/bvlc_reference_caffenet/bvlc_reference_caffenet.caffemodel'\n\n# Create a net object \nmodel = caffe.Net(model_def,      # defines the structure of the model\n            model_weights,  # contains the trained weights\n            caffe.TEST)     # use test mode (e.g., don't perform dropout)\n\n# set up transformer - creates transformer object\ntransformer = caffe.io.Transformer({'data': model.blobs['data'].data.shape})\n# transpose image from HxWxC to CxHxW \ntransformer.set_transpose('data', (2, 0, 1))\ntransformer.set_mean('data', np.load(CAFFE_HOME + 'python/caffe/imagenet/ilsvrc_2012_mean.npy').mean(1).mean(1))\n# set raw_scale = 255 to multiply with the values loaded with caffe.io.load_image \ntransformer.set_raw_scale('data', 255)\n# swap image channels from RGB to BGR \ntransformer.set_channel_swap('data', (2, 1, 0))\n\n\ndef extract_features(image_paths):\n    \"\"\"\n        This function is used to extract feature from the current batch of photos.\n        Features are extracted using the pretrained bvlc_reference_caffenet\n        Instead of returning 1000-dim vector from SoftMax layer, using fc7 as the final layer to get 4096-dim vector\n    \"\"\"\n    test_size = len(image_paths)\n    model.blobs['data'].reshape(test_size, 3, 227, 227)\n    model.blobs['data'].data[...] = list(map(lambda x: transformer.preprocess('data', skimage.img_as_float(skimage.io.imread(x)).astype(np.float32) ), image_paths))\n    out = model.forward()\n    return model.blobs['fc7'].data\n\nif not os.path.isfile(FEATURES_HOME + 'train_features.h5'):\n    \"\"\"\n        If this file doesn't exist, create a new one and set up two columns: photoId, feature\n    \"\"\"\n    file = h5py.File(FEATURES_HOME + 'train_features.h5', 'w')\n    photoId = file.create_dataset('photoId', (0,), maxshape=(None,), dtype='|S54')\n    feature = file.create_dataset('feature', (0, 4096), maxshape=(None, 4096), dtype=np.dtype('int16'))\n    file.close()\n\n# If this file exists, then track how many of the images are already done.\nfile = h5py.File(FEATURES_HOME + 'train_features.h5', 'r+')\nalready_extracted_images = len(file['photoId'])\nfile.close()\n\n# Get training images and their business ids\ntrain_data = pd.read_csv(DATA_ + 'train_photo_to_biz_ids.csv')\ntrain_photo_paths = [os.path.join(DATA_ + 'train_photos/', str(photo_id) + '.jpg') for photo_id in\n                     train_data['photo_id']]\n\n# Each batch will have 500 images for feature extraction\ntrain_size = len(train_photo_paths)\nbatch_size = 500\nbatch_number = round(already_extracted_images / batch_size + 1,3)\nhours_elapsed = 0\n\nprint(\"Total images:\", train_size)\nprint(\"already_done_images: \", already_extracted_images)\n\n# Feature extraction of the train dataset\nfor image_count in range(already_extracted_images, train_size, batch_size):\n    start_time = round(time.time(),3)\n    # Get the paths for images in the current batch\n    image_paths = train_photo_paths[image_count: min(image_count + batch_size, train_size)]\n\n    # Feature extraction for the current batch\n    features = extract_features(image_paths)\n\n    # Update the total count of images done so far\n    total_done_images = image_count + features.shape[0]\n\n    # Storing the features in h5 file\n    file = h5py.File(FEATURES_HOME + 'train_features.h5', 'r+')\n    file['photoId'].resize((total_done_images,))\n    file['photoId'][image_count: total_done_images] = np.array(image_paths,dtype='|S54')\n    file['feature'].resize((total_done_images, features.shape[1]))\n    file['feature'][image_count: total_done_images, :] = features\n    file.close()\n\n    print(\"Batch No:\", batch_number, \"\\tStart:\", image_count, \"\\tEnd:\", image_count + batch_size, \"\\tTime elapsed:\", hours_elapsed, \"hrs\", \"\\tCompleted:\", round(float(\n        image_count + batch_size) / float(train_size) * 100,3), \"%\")\n    batch_number += 1\n    hours_elapsed += round(((time.time() - start_time)/60)/60,3)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T11:47:38.902128Z","iopub.status.idle":"2023-11-22T11:47:38.902616Z","shell.execute_reply.started":"2023-11-22T11:47:38.902373Z","shell.execute_reply":"2023-11-22T11:47:38.902411Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}