{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-20T07:23:27.232624Z","iopub.execute_input":"2023-11-20T07:23:27.233411Z","iopub.status.idle":"2023-11-20T07:23:27.770588Z","shell.execute_reply.started":"2023-11-20T07:23:27.233365Z","shell.execute_reply":"2023-11-20T07:23:27.769112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport tarfile\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:23:27.773713Z","iopub.execute_input":"2023-11-20T07:23:27.774359Z","iopub.status.idle":"2023-11-20T07:23:44.382601Z","shell.execute_reply.started":"2023-11-20T07:23:27.774306Z","shell.execute_reply":"2023-11-20T07:23:44.381139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# List and Extract Data Files","metadata":{}},{"cell_type":"code","source":"import tarfile\nimport os\nfrom tqdm import tqdm\n\n# List of .tgz files to extract\nfiles_to_extract = [\n    'sample_submission.csv.tgz',\n    'test_photo_to_biz.csv.tgz',\n    'test_photos.tgz',\n    'train.csv.tgz',\n    'train_photo_to_biz_ids.csv.tgz',\n    'train_photos.tgz'\n]\n\n# Directory where you want to extract the contents\ntarget_directory = '/kaggle/working/'\n\n# Check if the files already exist in the target directory before extraction\nfor file_to_extract in files_to_extract:\n    target_file_path = os.path.join(target_directory, file_to_extract.replace('.tgz', ''))\n    if not os.path.exists(target_file_path):\n        file_path = f'/kaggle/input/yelp-restaurant-photo-classification/{file_to_extract}'\n        try:\n            with tarfile.open(file_path, 'r:gz') as tar:\n                members = list(tar.getmembers())\n                for member in tqdm(iterable=members, desc=f\"Extracting {file_to_extract}\", total=len(members)):\n                    tar.extract(member, target_directory)\n        except Exception as e:\n            print(f\"Error extracting {file_to_extract}: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:23:44.384224Z","iopub.execute_input":"2023-11-20T07:23:44.385074Z","iopub.status.idle":"2023-11-20T07:34:22.014275Z","shell.execute_reply.started":"2023-11-20T07:23:44.385023Z","shell.execute_reply":"2023-11-20T07:34:22.012248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ntrain_biz=pd.read_csv('/kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz')\ntrain_biz=train_biz.head(10000)\ntest_biz_biz=pd.read_csv('./test_photo_to_biz.csv')\ntest_biz_biz=test_biz_biz.head(10000)\ntrain=pd.read_csv('./train.csv')\ntrain=train.head(10000)\nsub=pd.read_csv('./sample_submission.csv')\ntrain_biz.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:34:22.017610Z","iopub.execute_input":"2023-11-20T07:34:22.018870Z","iopub.status.idle":"2023-11-20T07:34:22.918972Z","shell.execute_reply.started":"2023-11-20T07:34:22.018801Z","shell.execute_reply":"2023-11-20T07:34:22.917446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# train_biz=pd.read_csv('./train_photo_to_biz_ids.csv')\n# train_biz=train_biz.head(10000)\n# test_biz_biz=pd.read_csv('./test_photo_to_biz.csv')\n# test_biz_biz=test_biz_biz.head(10000)\n# train=pd.read_csv('./train.csv')\n# train=train.head(10000)\n# sub=pd.read_csv('./sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:34:22.924105Z","iopub.execute_input":"2023-11-20T07:34:22.925324Z","iopub.status.idle":"2023-11-20T07:34:22.933854Z","shell.execute_reply.started":"2023-11-20T07:34:22.925247Z","shell.execute_reply":"2023-11-20T07:34:22.931491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming train_biz and train are already defined\ndata = train_biz.merge(train , on='business_id')\ndata = data.dropna(subset=['labels'])\ndata['labels'] = data['labels'].apply(lambda x: str(x).split(' '))  # Convert to list of strings\n\n# Convert photo_id to strings and add file extension if necessary\ndata['photo_id'] = data['photo_id'].astype(str) + '.jpg'  # Add the correct file extension\n\ndf_train, df_valid = train_test_split(data, test_size=0.1, random_state=42)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:34:22.936382Z","iopub.execute_input":"2023-11-20T07:34:22.937756Z","iopub.status.idle":"2023-11-20T07:34:24.376599Z","shell.execute_reply.started":"2023-11-20T07:34:22.937696Z","shell.execute_reply":"2023-11-20T07:34:24.375343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\n\n# Convert labels to binary format\nmlb = MultiLabelBinarizer()\ndata_labels = mlb.fit_transform(data['labels'])\nlabels_df = pd.DataFrame(data_labels, columns=mlb.classes_)\ndata = pd.concat([data.reset_index(drop=True), labels_df.reset_index(drop=True)], axis=1)\n\n# Split into training and validation sets\ndf_train, df_valid = train_test_split(data, test_size=0.1, random_state=42)\n\ndef create_generator(df, directory, x_col, y_cols, batch_size, img_size):\n    datagen = ImageDataGenerator(rescale=1./255)\n    i = 0\n    while True:\n        batch_x = df[i * batch_size:(i + 1) * batch_size]\n        images = []\n        labels = []\n        for _, row in batch_x.iterrows():\n            img_path = os.path.join(directory, row[x_col])\n            img = load_img(img_path, target_size=img_size)\n            img_array = img_to_array(img)\n            images.append(img_array)\n            labels.append(row[y_cols].values)\n        i = (i + 1) % (df.shape[0] // batch_size)\n        yield np.array(images), np.array(labels)\n\nlabel_columns = mlb.classes_\n\n# Create training and validation generators\ntrain_generator = create_generator(df_train, './train_photos', 'photo_id', label_columns, 32, (256, 256))\nvalidation_generator = create_generator(df_valid, './train_photos', 'photo_id', label_columns, 32, (256, 256))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:36:19.553430Z","iopub.execute_input":"2023-11-20T07:36:19.554174Z","iopub.status.idle":"2023-11-20T07:36:20.303142Z","shell.execute_reply.started":"2023-11-20T07:36:19.554113Z","shell.execute_reply":"2023-11-20T07:36:20.302002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:36:23.134873Z","iopub.execute_input":"2023-11-20T07:36:23.135740Z","iopub.status.idle":"2023-11-20T07:36:23.163127Z","shell.execute_reply.started":"2023-11-20T07:36:23.135683Z","shell.execute_reply":"2023-11-20T07:36:23.161443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Data Generator and Data Loading","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.5,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,  # Increased from 20 to 40\n    width_shift_range=0.3,  # Increased from 0.2 to 0.3\n    height_shift_range=0.3,  # Increased from 0.2 to 0.3\n    shear_range=0.3,  # Increased from 0.2 to 0.3\n    zoom_range=0.6,  # Increased from 0.5 to 0.6\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=df_train,\n    directory='./train_photos',\n    x_col='photo_id',\n    y_col='labels',\n    target_size=(256, 256),\n    batch_size=32,  # or another batch size suitable for your training\n    class_mode='categorical'\n)\n# Validation data should not be augmented, only rescaled\ntest_datagen = ImageDataGenerator(rescale=1./255)\nvalidation_generator = test_datagen.flow_from_dataframe(\n    dataframe=df_valid,\n    directory='./train_photos',  # Update this path if different\n    x_col='photo_id',\n    y_col='labels',\n    target_size=(256, 256),\n    batch_size=128,\n    class_mode='categorical'  # Adjust based on your label format\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:36:30.822525Z","iopub.execute_input":"2023-11-20T07:36:30.823039Z","iopub.status.idle":"2023-11-20T07:36:36.016147Z","shell.execute_reply.started":"2023-11-20T07:36:30.823001Z","shell.execute_reply":"2023-11-20T07:36:36.014827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build and Summarize the Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import models\nfrom tensorflow.keras import layers  # Import the 'layers' module\n\nnum_classes = 9  # Update based on your dataset\n\nbase_model = tf.keras.applications.ResNet50(weights='imagenet', include_top=False, input_shape=(256, 256, 3))\nbase_model.trainable = False  # Freeze base model layers\n\nmodel = tf.keras.Sequential([\n    base_model,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.BatchNormalization(),  # Added Batch Normalization layer\n    tf.keras.layers.Dropout(0.4),  # Adjusted dropout rate\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.BatchNormalization(),  # Added Batch Normalization layer\n    tf.keras.layers.Dropout(0.3),  # Adjusted dropout rate\n    tf.keras.layers.Dense(num_classes, activation='softmax')\n])\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:36:39.926098Z","iopub.execute_input":"2023-11-20T07:36:39.927460Z","iopub.status.idle":"2023-11-20T07:36:43.541579Z","shell.execute_reply.started":"2023-11-20T07:36:39.927412Z","shell.execute_reply":"2023-11-20T07:36:43.540153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and Train the Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau  # Import the 'ReduceLROnPlateau' callback\n# Custom learning rate scheduler\ndef scheduler(epoch, lr):\n    if epoch < 10:\n        return lr\n    else:\n        return lr * tf.math.exp(-0.1)\n\n# Compile the model\nmodel.compile(\n    optimizer=tf.keras.optimizers.SGD(learning_rate=0.001, momentum=0.9),  # Using SGD with momentum\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nepochs = 20  # Adjusted number of epochs\nsteps_per_epoch = 200\ncallbacks = [\n    tf.keras.callbacks.LearningRateScheduler(scheduler),  # Using learning rate scheduler\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True),\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, min_lr=0.001)\n]\n\nhistory = model.fit(\n    train_generator,\n    epochs=epochs,\n    steps_per_epoch = steps_per_epoch,\n    validation_data=validation_generator,\n    callbacks=callbacks\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:36:47.812103Z","iopub.execute_input":"2023-11-20T07:36:47.812635Z","iopub.status.idle":"2023-11-20T08:04:01.010986Z","shell.execute_reply.started":"2023-11-20T07:36:47.812595Z","shell.execute_reply":"2023-11-20T08:04:01.008375Z"},"trusted":true},"execution_count":null,"outputs":[]}]}