{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11466444,"sourceType":"datasetVersion","datasetId":7185586}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:48:28.090881Z","iopub.execute_input":"2025-05-15T06:48:28.091566Z","iopub.status.idle":"2025-05-15T06:48:41.964656Z","shell.execute_reply.started":"2025-05-15T06:48:28.091535Z","shell.execute_reply":"2025-05-15T06:48:41.964085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntraindf = pd.read_csv(\"/kaggle/input/landmark-recognition-2021/train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:48:41.965798Z","iopub.execute_input":"2025-05-15T06:48:41.966270Z","iopub.status.idle":"2025-05-15T06:48:43.464172Z","shell.execute_reply.started":"2025-05-15T06:48:41.966249Z","shell.execute_reply":"2025-05-15T06:48:43.463614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"traindf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:48:43.464950Z","iopub.execute_input":"2025-05-15T06:48:43.465155Z","iopub.status.idle":"2025-05-15T06:48:43.483262Z","shell.execute_reply.started":"2025-05-15T06:48:43.465138Z","shell.execute_reply":"2025-05-15T06:48:43.482610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_unique = traindf['landmark_id'].unique()[0:60]\nimage_ids = []\nlabels = []\ntemp_labels = []\n\nfor i, id_ in enumerate(landmark_unique):\n    for iid in traindf['id'][traindf['landmark_id'] == id_]:\n        image_ids.append(iid)\n        labels.append(id_)\n        temp_labels.append(i)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:48:43.484870Z","iopub.execute_input":"2025-05-15T06:48:43.485077Z","iopub.status.idle":"2025-05-15T06:48:43.619291Z","shell.execute_reply.started":"2025-05-15T06:48:43.485061Z","shell.execute_reply":"2025-05-15T06:48:43.618725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mainpath = '../input/landmark-recognition-2021/train'\nimages_pixels = []\n\nfor iid in image_ids:\n    first_dir = os.path.join(mainpath, iid[0])\n    second_dir = os.path.join(first_dir, iid[1])\n    third_dir = os.path.join(second_dir, iid[2])\n    finalpath = os.path.join(third_dir, iid + '.jpg')\n    \n    img_pix = cv2.imread(finalpath, 1)\n    images_pixels.append(cv2.resize(img_pix, (100, 100)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:48:43.619978Z","iopub.execute_input":"2025-05-15T06:48:43.620226Z","iopub.status.idle":"2025-05-15T06:49:07.599888Z","shell.execute_reply.started":"2025-05-15T06:48:43.620203Z","shell.execute_reply":"2025-05-15T06:49:07.599301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nX_data = np.array(images_pixels) / 255.0\nY_data = to_categorical(temp_labels, num_classes=60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:49:07.600540Z","iopub.execute_input":"2025-05-15T06:49:07.600726Z","iopub.status.idle":"2025-05-15T06:49:07.750155Z","shell.execute_reply.started":"2025-05-15T06:49:07.600709Z","shell.execute_reply":"2025-05-15T06:49:07.749573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, Y_train, Y_val = train_test_split(X_data, Y_data, test_size=0.3, random_state=101)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:49:07.750806Z","iopub.execute_input":"2025-05-15T06:49:07.751000Z","iopub.status.idle":"2025-05-15T06:49:07.859954Z","shell.execute_reply.started":"2025-05-15T06:49:07.750984Z","shell.execute_reply":"2025-05-15T06:49:07.859167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_val.shape)\nprint(Y_train.shape)\nprint(Y_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:49:07.860866Z","iopub.execute_input":"2025-05-15T06:49:07.861131Z","iopub.status.idle":"2025-05-15T06:49:07.865576Z","shell.execute_reply.started":"2025-05-15T06:49:07.861108Z","shell.execute_reply":"2025-05-15T06:49:07.864985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(12, 8))\nfor i in range(16):\n    plt.subplot(4, 4, i + 1)\n    plt.imshow(X_train[i])\n    plt.title(f\"Label: {np.argmax(Y_train[i])}\")\n    plt.axis('off')\n    \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:49:07.866423Z","iopub.execute_input":"2025-05-15T06:49:07.866806Z","iopub.status.idle":"2025-05-15T06:49:08.872081Z","shell.execute_reply.started":"2025-05-15T06:49:07.866755Z","shell.execute_reply":"2025-05-15T06:49:08.871321Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DenseNet","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:40:40.314509Z","iopub.execute_input":"2025-04-19T02:40:40.314821Z","iopub.status.idle":"2025-04-19T02:40:40.319662Z","shell.execute_reply.started":"2025-04-19T02:40:40.314799Z","shell.execute_reply":"2025-04-19T02:40:40.318732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pretrained_model = tf.keras.applications.DenseNet201(input_shape=(100,100,3),\n                                                      include_top=False,\n                                                      weights='imagenet',\n                                                      pooling='avg')\npretrained_model.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:40:41.990316Z","iopub.execute_input":"2025-04-19T02:40:41.990611Z","iopub.status.idle":"2025-04-19T02:40:45.051976Z","shell.execute_reply.started":"2025-04-19T02:40:41.990590Z","shell.execute_reply":"2025-04-19T02:40:45.051251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs = pretrained_model.input\ndrop_layer = tf.keras.layers.Dropout(0.25)(pretrained_model.output)\nx_layer = tf.keras.layers.Dense(512, activation='relu')(drop_layer)\nx_layer1 = tf.keras.layers.Dense(128, activation='relu')(x_layer)\ndrop_layer1 = tf.keras.layers.Dropout(0.20)(x_layer1)\noutputs = tf.keras.layers.Dense(60, activation='softmax')(drop_layer1)\n\n\nmodel2 = tf.keras.Model(inputs=inputs, outputs=outputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:40:46.982569Z","iopub.execute_input":"2025-04-19T02:40:46.982859Z","iopub.status.idle":"2025-04-19T02:40:47.067603Z","shell.execute_reply.started":"2025-04-19T02:40:46.982839Z","shell.execute_reply":"2025-04-19T02:40:47.066804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datagen = ImageDataGenerator(horizontal_flip=False,\n                             vertical_flip=False,\n                             rotation_range=0,\n                             zoom_range=0.2,\n                             width_shift_range=0,\n                             height_shift_range=0,\n                             shear_range=0,\n                             fill_mode=\"nearest\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:40:48.885883Z","iopub.execute_input":"2025-04-19T02:40:48.886731Z","iopub.status.idle":"2025-04-19T02:40:48.891228Z","shell.execute_reply.started":"2025-04-19T02:40:48.886705Z","shell.execute_reply":"2025-04-19T02:40:48.890327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\nmodel2.compile(optimizer=optimizer,loss='categorical_crossentropy',metrics=['acc'])\nhistory = model2.fit(datagen.flow(X_train,Y_train,batch_size=32),validation_data=(X_val,Y_val),epochs=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:40:57.822757Z","iopub.execute_input":"2025-04-19T02:40:57.823080Z","iopub.status.idle":"2025-04-19T02:58:26.615811Z","shell.execute_reply.started":"2025-04-19T02:40:57.823059Z","shell.execute_reply":"2025-04-19T02:58:26.615034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MobileNet","metadata":{}},{"cell_type":"code","source":"base_model = MobileNetV2(input_shape=(100,100,3), include_top=False, weights='imagenet', pooling='avg')\nbase_model.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:26.617373Z","iopub.execute_input":"2025-04-19T02:58:26.617653Z","iopub.status.idle":"2025-04-19T02:58:27.607312Z","shell.execute_reply.started":"2025-04-19T02:58:26.617631Z","shell.execute_reply":"2025-04-19T02:58:27.606635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = base_model.output\nx = Dropout(0.25)(x)\nx = Dense(128, activation='relu')(x)\npredictions = Dense(60, activation='softmax')(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:28.282770Z","iopub.execute_input":"2025-04-19T02:58:28.283126Z","iopub.status.idle":"2025-04-19T02:58:28.308663Z","shell.execute_reply.started":"2025-04-19T02:58:28.283102Z","shell.execute_reply":"2025-04-19T02:58:28.307777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Model(inputs=base_model.input, outputs=predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:29.612571Z","iopub.execute_input":"2025-04-19T02:58:29.612882Z","iopub.status.idle":"2025-04-19T02:58:29.626415Z","shell.execute_reply.started":"2025-04-19T02:58:29.612860Z","shell.execute_reply":"2025-04-19T02:58:29.625421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = Adam(learning_rate=0.001)\nmodel.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=['acc'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:32.922799Z","iopub.execute_input":"2025-04-19T02:58:32.923520Z","iopub.status.idle":"2025-04-19T02:58:32.932921Z","shell.execute_reply.started":"2025-04-19T02:58:32.923490Z","shell.execute_reply":"2025-04-19T02:58:32.932215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datagen = ImageDataGenerator(horizontal_flip=False,\n                             vertical_flip=False,\n                             rotation_range=0,\n                             zoom_range=0.2,\n                             width_shift_range=0,\n                             height_shift_range=0,\n                             shear_range=0,\n                             fill_mode=\"nearest\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:34.380372Z","iopub.execute_input":"2025-04-19T02:58:34.380680Z","iopub.status.idle":"2025-04-19T02:58:34.385641Z","shell.execute_reply.started":"2025-04-19T02:58:34.380659Z","shell.execute_reply":"2025-04-19T02:58:34.384636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(datagen.flow(X_train, Y_train, batch_size=32),\n                    validation_data=(X_val, Y_val),\n                    epochs=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T02:58:37.128281Z","iopub.execute_input":"2025-04-19T02:58:37.128935Z","iopub.status.idle":"2025-04-19T03:02:22.756918Z","shell.execute_reply.started":"2025-04-19T02:58:37.128908Z","shell.execute_reply":"2025-04-19T03:02:22.756242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EVALUATION","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:02:25.518373Z","iopub.execute_input":"2025-04-19T03:02:25.518686Z","iopub.status.idle":"2025-04-19T03:02:25.522952Z","shell.execute_reply.started":"2025-04-19T03:02:25.518663Z","shell.execute_reply":"2025-04-19T03:02:25.522070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:02:27.613745Z","iopub.execute_input":"2025-04-19T03:02:27.614088Z","iopub.status.idle":"2025-04-19T03:02:32.807968Z","shell.execute_reply.started":"2025-04-19T03:02:27.614064Z","shell.execute_reply":"2025-04-19T03:02:32.807084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_val.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:02:34.476001Z","iopub.execute_input":"2025-04-19T03:02:34.476383Z","iopub.status.idle":"2025-04-19T03:02:34.481983Z","shell.execute_reply.started":"2025-04-19T03:02:34.476360Z","shell.execute_reply":"2025-04-19T03:02:34.481291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plotting training and validation accuracy\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['acc'], label='Training Accuracy')\nplt.plot(history.history['val_acc'], label='Validation Accuracy')\nplt.legend()\nplt.title('Training and Validation Accuracy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:02:36.264581Z","iopub.execute_input":"2025-04-19T03:02:36.264901Z","iopub.status.idle":"2025-04-19T03:02:36.500600Z","shell.execute_reply.started":"2025-04-19T03:02:36.264878Z","shell.execute_reply":"2025-04-19T03:02:36.499797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.legend()\nplt.title('Training and Validation Loss')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:02:38.765057Z","iopub.execute_input":"2025-04-19T03:02:38.765354Z","iopub.status.idle":"2025-04-19T03:02:38.968787Z","shell.execute_reply.started":"2025-04-19T03:02:38.765334Z","shell.execute_reply":"2025-04-19T03:02:38.967863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Efficient Net","metadata":{}},{"cell_type":"code","source":"pip install timm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T02:05:08.482922Z","iopub.execute_input":"2025-05-15T02:05:08.483551Z","iopub.status.idle":"2025-05-15T02:06:21.515639Z","shell.execute_reply.started":"2025-05-15T02:05:08.483527Z","shell.execute_reply":"2025-05-15T02:06:21.514810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, GlobalAveragePooling2D, Dense\n\ninput_shape = (224, 224, 3)\n\nbase_model = EfficientNetB0(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=input_shape\n)\n\nbase_model.trainable = True\n\ninputs = Input(shape=input_shape)\nx = base_model(inputs, training=True)  # no preprocess_input here!\nx = GlobalAveragePooling2D()(x)\noutputs = Dense(60, activation=\"softmax\")(x)\n\nmodel = Model(inputs, outputs)\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T03:05:37.562906Z","iopub.execute_input":"2025-05-15T03:05:37.563753Z","iopub.status.idle":"2025-05-15T03:05:40.787727Z","shell.execute_reply.started":"2025-05-15T03:05:37.563723Z","shell.execute_reply":"2025-05-15T03:05:40.786394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport numpy as np\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\n\n# Resize to 224x224 and normalize using EfficientNet preprocessing\nX_data_resized = np.array([cv2.resize(img, (224, 224)) for img in images_pixels])\nX_data_resized = preprocess_input(X_data_resized)\n\n# Labels to one-hot\nY_data = to_categorical(temp_labels, num_classes=60)\n\n# Split\nX_train, X_val, Y_train, Y_val = train_test_split(X_data_resized, Y_data, test_size=0.3, random_state=101)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T03:05:40.788840Z","iopub.execute_input":"2025-05-15T03:05:40.789224Z","iopub.status.idle":"2025-05-15T03:05:41.319401Z","shell.execute_reply.started":"2025-05-15T03:05:40.789187Z","shell.execute_reply":"2025-05-15T03:05:41.318522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\nearly_stop = EarlyStopping(patience=5, restore_best_weights=True)\ncheckpoint = ModelCheckpoint(\"best_model.keras\", monitor='val_loss', save_best_only=True)\n\nhistory = model.fit(\n    X_train, Y_train,\n    validation_data=(X_val, Y_val),\n    epochs=10,\n    batch_size=32,\n    callbacks=[early_stop, checkpoint],\n    verbose=1\n)\n# with tf.device('/GPU:0'):\n#     model = Model(inputs, outputs)\n#     model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n#     history = model.fit(\n#         X_train, Y_train,\n#         validation_data=(X_val, Y_val),\n#         epochs=30,\n#         batch_size=32,\n#         callbacks=[early_stop, checkpoint],\n#         verbose=1\n#     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T03:05:41.320344Z","iopub.execute_input":"2025-05-15T03:05:41.320652Z","iopub.status.idle":"2025-05-15T03:36:48.976344Z","shell.execute_reply.started":"2025-05-15T03:05:41.320620Z","shell.execute_reply":"2025-05-15T03:36:48.975049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import average_precision_score\nimport numpy as np\n\n# Predict\ny_pred_probs = model.predict(X_val)\ny_true = Y_val\n\n# Compute GAP\ngap_score = average_precision_score(y_true, y_pred_probs, average='macro')\nprint(f\"GAP score: {gap_score:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T03:40:05.922277Z","iopub.execute_input":"2025-05-15T03:40:05.922639Z","iopub.status.idle":"2025-05-15T03:40:27.617075Z","shell.execute_reply.started":"2025-05-15T03:40:05.922613Z","shell.execute_reply":"2025-05-15T03:40:27.616231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SWIN Transformer","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport numpy as np\n\nclass NumpyImageDataset(Dataset):\n    def __init__(self, images, labels, transform=None):\n        self.images = images\n        self.labels = labels\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img = self.images[idx]\n        label = self.labels[idx]\n\n        # Convert to uint8 if needed for PIL, else normalize here\n        img = img.astype(np.uint8)\n        if self.transform:\n            img = self.transform(img)\n\n        return img, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T04:59:47.034556Z","iopub.execute_input":"2025-05-15T04:59:47.034905Z","iopub.status.idle":"2025-05-15T04:59:56.760611Z","shell.execute_reply.started":"2025-05-15T04:59:47.034870Z","shell.execute_reply":"2025-05-15T04:59:56.759806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision.transforms import ToTensor, ToPILImage, Resize, Normalize, Compose\n\ntransform = Compose([\n    ToPILImage(),\n    Resize((224, 224)),\n    ToTensor(),\n    Normalize([0.5] * 3, [0.5] * 3)\n])\n\n# One-hot to label index if needed\nif Y_train.ndim == 2:  # (N, C)\n    Y_train = np.argmax(Y_train, axis=1)\n    Y_val = np.argmax(Y_val, axis=1)\n\ntrain_dataset = NumpyImageDataset(X_train, Y_train, transform)\nval_dataset = NumpyImageDataset(X_val, Y_val, transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T05:02:39.440457Z","iopub.execute_input":"2025-05-15T05:02:39.441420Z","iopub.status.idle":"2025-05-15T05:02:39.450900Z","shell.execute_reply.started":"2025-05-15T05:02:39.441381Z","shell.execute_reply":"2025-05-15T05:02:39.449672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import timm\nfrom torch import nn\nimport torch\nfrom torch.nn import functional as F\nfrom torch import nn\nimport math\nfrom torch.nn import functional as F\nfrom torch.nn.parameter import Parameter\nimport numpy as np\n\n\nclass Swish(torch.autograd.Function):\n\n    @staticmethod\n    def forward(ctx, i):\n        result = i * torch.sigmoid(i)\n        ctx.save_for_backward(i)\n        return result\n\n    @staticmethod\n    def backward(ctx, grad_output):\n        i = ctx.saved_variables[0]\n        sigmoid_i = torch.sigmoid(i)\n        return grad_output * (sigmoid_i * (1 + i * (1 - sigmoid_i)))\n\n\nclass Swish_module(nn.Module):\n    def forward(self, x):\n        return Swish.apply(x)\n\n\nclass CrossEntropyLossWithLabelSmoothing(nn.Module):\n    def __init__(self, n_dim, ls_=0.9):\n        super().__init__()\n        self.n_dim = n_dim\n        self.ls_ = ls_\n\n    def forward(self, x, target):\n        target = F.one_hot(target, self.n_dim).float()\n        target *= self.ls_\n        target += (1 - self.ls_) / self.n_dim\n\n        logprobs = torch.nn.functional.log_softmax(x, dim=-1)\n        loss = -logprobs * target\n        loss = loss.sum(-1)\n        return loss.mean()\n\n\nclass DenseCrossEntropy(nn.Module):\n    def forward(self, x, target):\n        x = x.float()\n        target = target.float()\n        logprobs = torch.nn.functional.log_softmax(x, dim=-1)\n\n        loss = -logprobs * target\n        loss = loss.sum(-1)\n        return loss.mean()\n\n\nclass ArcMarginProduct_subcenter(nn.Module):\n    def __init__(self, in_features, out_features, k=3):\n        super().__init__()\n        self.weight = nn.Parameter(torch.FloatTensor(out_features*k, in_features))\n        self.reset_parameters()\n        self.k = k\n        self.out_features = out_features\n        \n    def reset_parameters(self):\n        stdv = 1. / math.sqrt(self.weight.size(1))\n        self.weight.data.uniform_(-stdv, stdv)\n        \n    def forward(self, features):\n        cosine_all = F.linear(F.normalize(features), F.normalize(self.weight))\n        cosine_all = cosine_all.view(-1, self.out_features, self.k)\n        cosine, _ = torch.max(cosine_all, dim=2)\n        return cosine   \n\n\nclass ArcFaceLossAdaptiveMargin(nn.modules.Module):\n    def __init__(self, margins, n_classes, s=30.0):\n        super().__init__()\n        self.crit = DenseCrossEntropy()\n        self.s = s\n        self.margins = margins\n        self.out_dim =n_classes\n            \n    def forward(self, logits, labels):\n        ms = []\n        ms = self.margins[labels.cpu().numpy()]\n        cos_m = torch.from_numpy(np.cos(ms)).float().cuda()\n        sin_m = torch.from_numpy(np.sin(ms)).float().cuda()\n        th = torch.from_numpy(np.cos(math.pi - ms)).float().cuda()\n        mm = torch.from_numpy(np.sin(math.pi - ms) * ms).float().cuda()\n        labels = F.one_hot(labels, self.out_dim).float()\n        logits = logits.float()\n        cosine = logits\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * cos_m.view(-1,1) - sine * sin_m.view(-1,1)\n        phi = torch.where(cosine > th.view(-1,1), phi, cosine - mm.view(-1,1))\n        output = (labels * phi) + ((1.0 - labels) * cosine)\n        output *= self.s\n        loss = self.crit(output, labels)\n        return loss     \n\n\n\nclass ArcMarginProduct(nn.Module):\n    def __init__(self, in_features, out_features):\n        super().__init__()\n        self.weight = nn.Parameter(torch.Tensor(out_features, in_features))\n        self.reset_parameters()\n\n    def reset_parameters(self):\n        nn.init.xavier_uniform_(self.weight)\n        # stdv = 1. / math.sqrt(self.weight.size(1))\n        # self.weight.data.uniform_(-stdv, stdv)\n\n    def forward(self, features):\n        cosine = F.linear(F.normalize(features), F.normalize(self.weight))\n        return cosine\n\n\nclass ArcFaceLoss(nn.modules.Module):\n    def __init__(self, s=45.0, m=0.1, crit=\"bce\", weight=None, reduction=\"mean\",class_weights_norm=None ):\n        super().__init__()\n\n        self.weight = weight\n        self.reduction = reduction\n        self.class_weights_norm = class_weights_norm\n        \n        if crit == \"focal\":\n            self.crit = FocalLoss(gamma=args.focal_loss_gamma)\n        elif crit == \"bce\":\n            self.crit = nn.CrossEntropyLoss(reduction=\"none\")   \n        elif crit == \"label_smoothing\":\n            self.crit = LabelSmoothingLoss(classes=args.n_classes)   \n\n        if s is None:\n            self.s = torch.nn.Parameter(torch.tensor([45.], requires_grad=True, device='cuda'))\n        else:\n            self.s = s\n\n        \n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.th = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n        \n    def forward(self, logits, labels):\n        #print(self.weight[labels])\n        #print(self.s)\n        logits = logits.float()\n        cosine = logits\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n\n#         labels2 = torch.nn.functional.one_hot(labels, num_classes=args.n_classes+2)\n#         labels2 = labels2[:,:args.n_classes+1]\n        labels2 = torch.zeros_like(cosine)\n        labels2.scatter_(1, labels.view(-1, 1).long(), 1)\n        output = (labels2 * phi) + ((1.0 - labels2) * cosine)\n\n        s = self.s\n\n        output = output * s\n        loss = self.crit(output, labels)\n\n        if self.weight is not None:\n            w = self.weight[labels].to(logits.device)\n\n            loss = loss * w\n            if self.class_weights_norm == \"batch\":\n                loss = loss.sum() / w.sum()\n            if self.class_weights_norm == \"global\":\n                loss = loss.mean()\n            else:\n                loss = loss.mean()\n            \n            return loss\n\n        if self.reduction == \"mean\":\n            loss = loss.mean()\n        elif self.reduction == \"sum\":\n            loss = loss.sum()\n        return loss    \n\ndef gem(x, p=3, eps=1e-6):\n    return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1./p)\n\nclass GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6, p_trainable=False):\n        super(GeM,self).__init__()\n        if p_trainable:\n            self.p = Parameter(torch.ones(1)*p)\n        else:\n            self.p = p\n        self.eps = eps\n\n    def forward(self, x):\n        ret = gem(x, p=self.p, eps=self.eps)   \n        return ret\n    def __repr__(self):\n        return self.__class__.__name__ + '(' + 'p=' + '{:.4f}'.format(self.p.data.tolist()[0]) + ', ' + 'eps=' + str(self.eps) + ')'\n\n    \n    \nfrom timm.models.vision_transformer_hybrid import HybridEmbed    \n\nclass Net(nn.Module):\n    def __init__(self, cfg, dataset):\n        super(Net, self).__init__()\n\n        self.cfg = cfg\n        self.n_classes = self.cfg.n_classes\n        \n        self.backbone = timm.create_model(cfg.backbone, \n                                          pretrained=cfg.pretrained, \n                                          num_classes=0, \n                                          in_chans=self.cfg.in_channels)\n        embedder = timm.create_model(cfg.embedder, \n                                          pretrained=cfg.pretrained, \n                                          in_chans=self.cfg.in_channels,features_only=True, out_indices=[1])\n\n        \n        self.backbone.patch_embed = HybridEmbed(embedder,img_size=cfg.img_size[0], \n                                              patch_size=1, \n                                              feature_size=self.backbone.patch_embed.grid_size, \n                                              in_chans=3, \n                                              embed_dim=self.backbone.embed_dim)\n#         if 'efficientnet' in cfg.backbone:\n#             backbone_out = self.backbone.num_features\n#         else:\n#             backbone_out = self.backbone.feature_info[-1]['num_chs']\n\n        if cfg.pool == \"gem\":\n            self.global_pool = GeM(p_trainable=cfg.gem_p_trainable)\n        elif cfg.pool == \"identity\":\n            self.global_pool = torch.nn.Identity()\n        elif cfg.pool == \"avg\":\n            self.global_pool = nn.AdaptiveAvgPool2d(1)\n            \n            \n        if \"xcit_small_24_p16\" in cfg.backbone:\n            backbone_out = 384\n        elif \"xcit_medium_24_p16\" in cfg.backbone:\n            backbone_out = 512\n        elif \"xcit_small_12_p16\" in cfg.backbone:\n            backbone_out = 384\n        elif \"xcit_medium_12_p16\" in cfg.backbone:\n            backbone_out = 512   \n        elif \"swin\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        elif \"vit\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        elif \"cait\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        else:\n            backbone_out = 2048 \n\n        self.embedding_size = cfg.embedding_size\n\n        # https://www.groundai.com/project/arcface-additive-angular-margin-loss-for-deep-face-recognition\n        if cfg.neck == \"option-D\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size, bias=True),\n                nn.BatchNorm1d(self.embedding_size),\n                torch.nn.PReLU()\n            )\n        elif cfg.neck == \"option-F\":\n            self.neck = nn.Sequential(\n                nn.Dropout(0.3),\n                nn.Linear(backbone_out, self.embedding_size, bias=True),\n                nn.BatchNorm1d(self.embedding_size),\n                torch.nn.PReLU()\n            )\n        elif cfg.neck == \"option-X\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size, bias=False),\n                nn.BatchNorm1d(self.embedding_size),\n            )\n            \n        elif cfg.neck == \"option-S\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size),\n                Swish_module()\n            )\n\n        if not self.cfg.headless:    \n            self.head_in_units = self.embedding_size\n            self.head = ArcMarginProduct_subcenter(self.embedding_size, self.n_classes)\n        if self.cfg.loss == 'adaptive_arcface':\n            self.loss_fn = ArcFaceLossAdaptiveMargin(dataset.margins,self.n_classes,cfg.arcface_s)\n        elif self.cfg.loss == 'arcface':\n            self.loss_fn = ArcFaceLoss(cfg.arcface_s,cfg.arcface_m)\n        else:\n            pass\n        \n        if cfg.freeze_backbone_head:\n            for name, param in self.named_parameters():\n                if not 'patch_embed' in name:\n                    param.requires_grad = False\n\n    def forward(self, batch):\n\n        x = batch['input']\n\n        x = self.backbone(x)\n\n        x_emb = self.neck(x)\n\n        if self.cfg.headless:\n            return {\"target\": batch['target'],'embeddings': x_emb}\n        \n        logits = self.head(x_emb)\n#         loss = self.loss_fn(logits, batch['target'].long(), self.n_classes)\n        preds = logits.softmax(1)\n        preds_conf, preds_cls = preds.max(1)\n        if self.training:\n            loss = self.loss_fn(logits, batch['target'].long())\n            return {'loss': loss, \"target\": batch['target'], \"preds_conf\":preds_conf,'preds_cls':preds_cls}\n        else:\n            loss = torch.zeros((1),device=x.device)\n            return {'loss': loss, \"target\": batch['target'],\"preds_conf\":preds_conf,'preds_cls':preds_cls,\n                    'embeddings': x_emb\n                   }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:54:47.118326Z","iopub.execute_input":"2025-05-15T06:54:47.118961Z","iopub.status.idle":"2025-05-15T06:54:47.150591Z","shell.execute_reply.started":"2025-05-15T06:54:47.118939Z","shell.execute_reply":"2025-05-15T06:54:47.150028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    n_classes = 60\n    backbone = \"swin_base_patch4_window7_224\"\n    embedder = \"efficientnet_b0\"\n    pretrained = True\n    in_channels = 3\n    img_size = (100, 100)\n    pool = \"avg\"\n    embedding_size = 512\n    neck = \"option-D\"\n    headless = False\n    loss = \"arcface\"  # or \"adaptive_arcface\" if you have per-class margins\n    arcface_s = 45.0\n    arcface_m = 0.3\n    freeze_backbone_head = False\n    gem_p_trainable = False\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:54:57.963597Z","iopub.execute_input":"2025-05-15T06:54:57.963866Z","iopub.status.idle":"2025-05-15T06:54:57.968154Z","shell.execute_reply.started":"2025-05-15T06:54:57.963848Z","shell.execute_reply":"2025-05-15T06:54:57.967426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cfg = CFG()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T06:56:04.134106Z","iopub.execute_input":"2025-05-15T06:56:04.134741Z","iopub.status.idle":"2025-05-15T06:56:04.137877Z","shell.execute_reply.started":"2025-05-15T06:56:04.134720Z","shell.execute_reply":"2025-05-15T06:56:04.137195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Imports and Config\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn.parameter import Parameter\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nimport timm\nimport math\n\nclass CFG:\n    n_classes = 60\n    backbone = \"swin_base_patch4_window7_224\"\n    embedder = \"efficientnet_b0\"\n    pretrained = True\n    in_channels = 3\n    img_size = (100, 100)\n    pool = \"avg\"\n    embedding_size = 512\n    neck = \"option-D\"\n    headless = False\n    loss = \"arcface\"\n    arcface_s = 45.0\n    arcface_m = 0.3\n    freeze_backbone_head = False\n    gem_p_trainable = False\n    batch_size = 16\n    num_workers = 2\n    lr = 1e-4\n    epochs = 3\n\ncfg = CFG()\n\nX_data_resized = np.array([cv2.resize(img, (224, 224)) for img in images_pixels]).astype(np.float32)\nX_data_resized /= 255.0\n\nmean = np.array([0.5, 0.5, 0.5], dtype=np.float32)\nstd = np.array([0.5, 0.5, 0.5], dtype=np.float32)\nX_data_resized = (X_data_resized - mean) / std\n\nX_train, X_val, Y_train, Y_val = train_test_split(X_data_resized, Y_data, test_size=0.3, random_state=101)\n\n\n\nclass LandmarkDataset(Dataset):\n    def __init__(self, images, labels):\n        self.images = torch.tensor(images).permute(0, 3, 1, 2)\n        self.labels = torch.tensor(labels).argmax(1).long()\n\n    def __len__(self): return len(self.labels)\n    def __getitem__(self, idx): return {'input': self.images[idx], 'target': self.labels[idx]}\n\ntrain_loader = DataLoader(LandmarkDataset(X_train, Y_train), batch_size=cfg.batch_size, shuffle=True, num_workers=cfg.num_workers)\n\n# 3. ArcMargin and Loss\nclass ArcMarginProduct(nn.Module):\n    def __init__(self, in_features, out_features):\n        super().__init__()\n        self.weight = nn.Parameter(torch.Tensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n    def forward(self, x): return F.linear(F.normalize(x), F.normalize(self.weight))\n\nclass ArcFaceLoss(nn.Module):\n    def __init__(self, s=45.0, m=0.1):\n        super().__init__()\n        self.crit = nn.CrossEntropyLoss()\n        self.s = s\n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.th = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n\n    def forward(self, logits, labels):\n        cosine = logits\n        sine = torch.sqrt(1.0 - cosine**2)\n        phi = cosine * self.cos_m - sine * self.sin_m\n        phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n        one_hot = torch.zeros_like(cosine).scatter(1, labels.unsqueeze(1), 1)\n        output = (one_hot * phi + (1 - one_hot) * cosine) * self.s\n        return self.crit(output, labels)\n\n# 4. Full Model\nclass Net(nn.Module):\n    def __init__(self, cfg):\n        super().__init__()\n        self.backbone = timm.create_model(cfg.backbone, pretrained=cfg.pretrained, num_classes=0)\n        out_dim = self.backbone.num_features\n        self.neck = nn.Sequential(nn.Linear(out_dim, cfg.embedding_size), nn.BatchNorm1d(cfg.embedding_size), nn.ReLU())\n        self.head = ArcMarginProduct(cfg.embedding_size, cfg.n_classes)\n        self.loss_fn = ArcFaceLoss(cfg.arcface_s, cfg.arcface_m)\n\n    def forward(self, batch):\n        x, y = batch['input'].cuda(), batch['target'].cuda()\n        feat = self.backbone(x)\n        emb = self.neck(feat)\n        logits = self.head(emb)\n        loss = self.loss_fn(logits, y)\n        preds = logits.softmax(1).argmax(1)\n        return {\"loss\": loss, \"preds_cls\": preds, \"target\": y}\n\n# 5. Train\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = Net(cfg).to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=cfg.lr)\n\nfor epoch in range(cfg.epochs):\n    model.train()\n    total_loss, correct = 0, 0\n    for batch in train_loader:\n        optimizer.zero_grad()\n        out = model(batch)\n        out['loss'].backward()\n        optimizer.step()\n        total_loss += out['loss'].item() * batch['input'].size(0)\n        correct += (out['preds_cls'] == out['target']).sum().item()\n    print(f\"Epoch {epoch+1} | Loss: {total_loss:.4f} | Accuracy: {correct/len(train_loader.dataset):.4f}\")\n    \n   \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T07:10:28.452580Z","iopub.execute_input":"2025-05-15T07:10:28.453320Z","iopub.status.idle":"2025-05-15T07:11:37.617997Z","shell.execute_reply.started":"2025-05-15T07:10:28.453286Z","shell.execute_reply":"2025-05-15T07:11:37.617158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# hybrid swin trasnformer attempt","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom tensorflow.keras.utils import to_categorical\n\nmainpath = '../input/landmark-recognition-2021/train'\nimages_pixels = []\nvalid_labels = []\n\nfor i, iid in enumerate(image_ids):\n    iid = str(iid).strip()  # Ensure iid is a string and remove spaces\n\n    # Skip invalid ids\n    if len(iid) < 3:\n        print(f\"Skipping ID with <3 characters: {iid}\")\n        continue\n\n    try:\n        # Build path: /train/a/b/c/abc123.jpg\n        finalpath = os.path.join(mainpath, iid[0], iid[1], iid[2], iid + '.jpg')\n\n        if os.path.exists(finalpath):\n            img_pix = cv2.imread(finalpath, cv2.IMREAD_COLOR)\n            if img_pix is not None:\n                resized = cv2.resize(img_pix, (100, 100))\n                images_pixels.append(resized)\n                valid_labels.append(temp_labels[i])\n            else:\n                print(f\"Warning: Failed to load image at {finalpath}\")\n        else:\n            print(f\"Warning: File not found: {finalpath}\")\n\n    except Exception as e:\n        print(f\"Error with image ID {iid}: {e}\")\n\n# Final processing\nX_dataa = np.array(images_pixels, dtype=np.float32) / 255.0\nY_dataa = to_categorical(valid_labels, num_classes=60)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T08:02:18.916255Z","iopub.execute_input":"2025-05-15T08:02:18.916572Z","iopub.status.idle":"2025-05-15T08:02:27.917788Z","shell.execute_reply.started":"2025-05-15T08:02:18.916549Z","shell.execute_reply":"2025-05-15T08:02:27.917006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.optim as optim\nfrom sklearn.model_selection import train_test_split\n\n# ================================\n# Dataset Wrapper\n# ================================\nclass LandmarkDataset(Dataset):\n    def __init__(self, images, labels):\n        self.images = torch.tensor(images).permute(0, 3, 1, 2).float()  # (N, 3, 100, 100)\n        self.labels = torch.tensor(labels).argmax(dim=1).long()         # one-hot to class index\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        return {\n            'input': self.images[idx],\n            'target': self.labels[idx]\n        }\n\n# ================================\n# Train/Test Split and Loaders\n# ================================\nx_train, x_val, y_train, y_val = train_test_split(X_dataa, Y_dataa, test_size=0.1, random_state=42)\n\ntrain_ds = LandmarkDataset(x_train, y_train)\nval_ds = LandmarkDataset(x_val, y_val)\n\ntrain_loader = DataLoader(train_ds, batch_size=32, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_ds, batch_size=32, shuffle=False, num_workers=2)\n\n# ================================\n# Config (same as yours)\n# ================================\nclass CFG:\n    n_classes = 60\n    backbone = \"swin_base_patch4_window7_224\"\n    embedder = \"efficientnet_b0\"\n    pretrained = True\n    in_channels = 3\n    img_size = (100, 100)\n    pool = \"avg\"\n    embedding_size = 512\n    neck = \"option-D\"\n    headless = False\n    loss = \"arcface\"\n    arcface_s = 45.0\n    arcface_m = 0.3\n    freeze_backbone_head = False\n    gem_p_trainable = False\n    lr = 1e-4\n    epochs = 5\n\ncfg = CFG()\n\n# Dummy dataset object with margins for loss init\nclass Dummy:\n    margins = np.ones(cfg.n_classes) * 0.3\ndataset = Dummy()\n\n# ================================\n# Model and Optimizer\n# ================================\nmodel = Net(cfg, dataset).cuda()\noptimizer = optim.AdamW(model.parameters(), lr=cfg.lr)\n\n# ================================\n# Training Loop\n# ================================\nfor epoch in range(cfg.epochs):\n    model.train()\n    total_loss = 0.0\n    for batch in train_loader:\n        batch = {k: v.cuda() for k, v in batch.items()}\n        optimizer.zero_grad()\n        output = model(batch)\n        output['loss'].backward()\n        optimizer.step()\n        total_loss += output['loss'].item()\n\n    print(f\"Epoch [{epoch+1}/{cfg.epochs}] - Train Loss: {total_loss/len(train_loader):.4f}\")\n\n    # Validation\n    model.eval()\n    val_loss = 0.0\n    correct = 0\n    total = 0\n    with torch.no_grad():\n        for batch in val_loader:\n            batch = {k: v.cuda() for k, v in batch.items()}\n            output = model(batch)\n            val_loss += output['loss'].item()\n            correct += (output['preds_cls'] == batch['target']).sum().item()\n            total += batch['target'].size(0)\n    val_acc = correct / total\n    print(f\"          Val Loss: {val_loss/len(val_loader):.4f}, Acc: {val_acc*100:.2f}%\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import timm\nfrom torch import nn\nimport torch\nfrom torch.nn import functional as F\nfrom torch import nn\nimport math\nfrom torch.nn import functional as F\nfrom torch.nn.parameter import Parameter\nimport numpy as np\n\n\nclass Swish(torch.autograd.Function):\n\n    @staticmethod\n    def forward(ctx, i):\n        result = i * torch.sigmoid(i)\n        ctx.save_for_backward(i)\n        return result\n\n    @staticmethod\n    def backward(ctx, grad_output):\n        i = ctx.saved_variables[0]\n        sigmoid_i = torch.sigmoid(i)\n        return grad_output * (sigmoid_i * (1 + i * (1 - sigmoid_i)))\n\n\nclass Swish_module(nn.Module):\n    def forward(self, x):\n        return Swish.apply(x)\n\n\nclass CrossEntropyLossWithLabelSmoothing(nn.Module):\n    def __init__(self, n_dim, ls_=0.9):\n        super().__init__()\n        self.n_dim = n_dim\n        self.ls_ = ls_\n\n    def forward(self, x, target):\n        target = F.one_hot(target, self.n_dim).float()\n        target *= self.ls_\n        target += (1 - self.ls_) / self.n_dim\n\n        logprobs = torch.nn.functional.log_softmax(x, dim=-1)\n        loss = -logprobs * target\n        loss = loss.sum(-1)\n        return loss.mean()\n\n\nclass DenseCrossEntropy(nn.Module):\n    def forward(self, x, target):\n        x = x.float()\n        target = target.float()\n        logprobs = torch.nn.functional.log_softmax(x, dim=-1)\n\n        loss = -logprobs * target\n        loss = loss.sum(-1)\n        return loss.mean()\n\n\nclass ArcMarginProduct_subcenter(nn.Module):\n    def __init__(self, in_features, out_features, k=3):\n        super().__init__()\n        self.weight = nn.Parameter(torch.FloatTensor(out_features*k, in_features))\n        self.reset_parameters()\n        self.k = k\n        self.out_features = out_features\n        \n    def reset_parameters(self):\n        stdv = 1. / math.sqrt(self.weight.size(1))\n        self.weight.data.uniform_(-stdv, stdv)\n        \n    def forward(self, features):\n        cosine_all = F.linear(F.normalize(features), F.normalize(self.weight))\n        cosine_all = cosine_all.view(-1, self.out_features, self.k)\n        cosine, _ = torch.max(cosine_all, dim=2)\n        return cosine   \n\n\nclass ArcFaceLossAdaptiveMargin(nn.modules.Module):\n    def __init__(self, margins, n_classes, s=30.0):\n        super().__init__()\n        self.crit = DenseCrossEntropy()\n        self.s = s\n        self.margins = margins\n        self.out_dim =n_classes\n            \n    def forward(self, logits, labels):\n        ms = []\n        ms = self.margins[labels.cpu().numpy()]\n        cos_m = torch.from_numpy(np.cos(ms)).float().cuda()\n        sin_m = torch.from_numpy(np.sin(ms)).float().cuda()\n        th = torch.from_numpy(np.cos(math.pi - ms)).float().cuda()\n        mm = torch.from_numpy(np.sin(math.pi - ms) * ms).float().cuda()\n        labels = F.one_hot(labels, self.out_dim).float()\n        logits = logits.float()\n        cosine = logits\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * cos_m.view(-1,1) - sine * sin_m.view(-1,1)\n        phi = torch.where(cosine > th.view(-1,1), phi, cosine - mm.view(-1,1))\n        output = (labels * phi) + ((1.0 - labels) * cosine)\n        output *= self.s\n        loss = self.crit(output, labels)\n        return loss     \n\n\n\nclass ArcMarginProduct(nn.Module):\n    def __init__(self, in_features, out_features):\n        super().__init__()\n        self.weight = nn.Parameter(torch.Tensor(out_features, in_features))\n        self.reset_parameters()\n\n    def reset_parameters(self):\n        nn.init.xavier_uniform_(self.weight)\n        # stdv = 1. / math.sqrt(self.weight.size(1))\n        # self.weight.data.uniform_(-stdv, stdv)\n\n    def forward(self, features):\n        cosine = F.linear(F.normalize(features), F.normalize(self.weight))\n        return cosine\n\n\nclass ArcFaceLoss(nn.modules.Module):\n    def __init__(self, s=45.0, m=0.1, crit=\"bce\", weight=None, reduction=\"mean\",class_weights_norm=None ):\n        super().__init__()\n\n        self.weight = weight\n        self.reduction = reduction\n        self.class_weights_norm = class_weights_norm\n        \n        if crit == \"focal\":\n            self.crit = FocalLoss(gamma=args.focal_loss_gamma)\n        elif crit == \"bce\":\n            self.crit = nn.CrossEntropyLoss(reduction=\"none\")   \n        elif crit == \"label_smoothing\":\n            self.crit = LabelSmoothingLoss(classes=args.n_classes)   \n\n        if s is None:\n            self.s = torch.nn.Parameter(torch.tensor([45.], requires_grad=True, device='cuda'))\n        else:\n            self.s = s\n\n        \n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.th = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n        \n    def forward(self, logits, labels):\n        #print(self.weight[labels])\n        #print(self.s)\n        logits = logits.float()\n        cosine = logits\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n\n#         labels2 = torch.nn.functional.one_hot(labels, num_classes=args.n_classes+2)\n#         labels2 = labels2[:,:args.n_classes+1]\n        labels2 = torch.zeros_like(cosine)\n        labels2.scatter_(1, labels.view(-1, 1).long(), 1)\n        output = (labels2 * phi) + ((1.0 - labels2) * cosine)\n\n        s = self.s\n\n        output = output * s\n        loss = self.crit(output, labels)\n\n        if self.weight is not None:\n            w = self.weight[labels].to(logits.device)\n\n            loss = loss * w\n            if self.class_weights_norm == \"batch\":\n                loss = loss.sum() / w.sum()\n            if self.class_weights_norm == \"global\":\n                loss = loss.mean()\n            else:\n                loss = loss.mean()\n            \n            return loss\n\n        if self.reduction == \"mean\":\n            loss = loss.mean()\n        elif self.reduction == \"sum\":\n            loss = loss.sum()\n        return loss    \n\ndef gem(x, p=3, eps=1e-6):\n    return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1./p)\n\nclass GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6, p_trainable=False):\n        super(GeM,self).__init__()\n        if p_trainable:\n            self.p = Parameter(torch.ones(1)*p)\n        else:\n            self.p = p\n        self.eps = eps\n\n    def forward(self, x):\n        ret = gem(x, p=self.p, eps=self.eps)   \n        return ret\n    def __repr__(self):\n        return self.__class__.__name__ + '(' + 'p=' + '{:.4f}'.format(self.p.data.tolist()[0]) + ', ' + 'eps=' + str(self.eps) + ')'\n\n    \n    \nfrom timm.models.vision_transformer_hybrid import HybridEmbed    \n\nclass Net(nn.Module):\n    def __init__(self, cfg, dataset):\n        super(Net, self).__init__()\n\n        self.cfg = cfg\n        self.n_classes = self.cfg.n_classes\n        \n        self.backbone = timm.create_model(cfg.backbone, \n                                          pretrained=cfg.pretrained, \n                                          num_classes=0, \n                                          in_chans=self.cfg.in_channels)\n        embedder = timm.create_model(cfg.embedder, \n                                          pretrained=cfg.pretrained, \n                                          in_chans=self.cfg.in_channels,features_only=True, out_indices=[1])\n\n        \n        self.backbone.patch_embed = HybridEmbed(embedder,img_size=cfg.img_size[0], \n                                              patch_size=1, \n                                              feature_size=self.backbone.patch_embed.grid_size, \n                                              in_chans=3, \n                                              embed_dim=self.backbone.embed_dim)\n#         if 'efficientnet' in cfg.backbone:\n#             backbone_out = self.backbone.num_features\n#         else:\n#             backbone_out = self.backbone.feature_info[-1]['num_chs']\n\n        if cfg.pool == \"gem\":\n            self.global_pool = GeM(p_trainable=cfg.gem_p_trainable)\n        elif cfg.pool == \"identity\":\n            self.global_pool = torch.nn.Identity()\n        elif cfg.pool == \"avg\":\n            self.global_pool = nn.AdaptiveAvgPool2d(1)\n            \n            \n        if \"xcit_small_24_p16\" in cfg.backbone:\n            backbone_out = 384\n        elif \"xcit_medium_24_p16\" in cfg.backbone:\n            backbone_out = 512\n        elif \"xcit_small_12_p16\" in cfg.backbone:\n            backbone_out = 384\n        elif \"xcit_medium_12_p16\" in cfg.backbone:\n            backbone_out = 512   \n        elif \"swin\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        elif \"vit\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        elif \"cait\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        else:\n            backbone_out = 2048 \n\n        self.embedding_size = cfg.embedding_size\n\n        # https://www.groundai.com/project/arcface-additive-angular-margin-loss-for-deep-face-recognition\n        if cfg.neck == \"option-D\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size, bias=True),\n                nn.BatchNorm1d(self.embedding_size),\n                torch.nn.PReLU()\n            )\n        elif cfg.neck == \"option-F\":\n            self.neck = nn.Sequential(\n                nn.Dropout(0.3),\n                nn.Linear(backbone_out, self.embedding_size, bias=True),\n                nn.BatchNorm1d(self.embedding_size),\n                torch.nn.PReLU()\n            )\n        elif cfg.neck == \"option-X\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size, bias=False),\n                nn.BatchNorm1d(self.embedding_size),\n            )\n            \n        elif cfg.neck == \"option-S\":\n            self.neck = nn.Sequential(\n                nn.Linear(backbone_out, self.embedding_size),\n                Swish_module()\n            )\n\n        if not self.cfg.headless:    \n            self.head_in_units = self.embedding_size\n            self.head = ArcMarginProduct_subcenter(self.embedding_size, self.n_classes)\n        if self.cfg.loss == 'adaptive_arcface':\n            self.loss_fn = ArcFaceLossAdaptiveMargin(dataset.margins,self.n_classes,cfg.arcface_s)\n        elif self.cfg.loss == 'arcface':\n            self.loss_fn = ArcFaceLoss(cfg.arcface_s,cfg.arcface_m)\n        else:\n            pass\n        \n        if cfg.freeze_backbone_head:\n            for name, param in self.named_parameters():\n                if not 'patch_embed' in name:\n                    param.requires_grad = False\n\n    def forward(self, batch):\n\n        x = batch['input']\n\n        x = self.backbone(x)\n\n        x_emb = self.neck(x)\n\n        if self.cfg.headless:\n            return {\"target\": batch['target'],'embeddings': x_emb}\n        \n        logits = self.head(x_emb)\n#         loss = self.loss_fn(logits, batch['target'].long(), self.n_classes)\n        preds = logits.softmax(1)\n        preds_conf, preds_cls = preds.max(1)\n        if self.training:\n            loss = self.loss_fn(logits, batch['target'].long())\n            return {'loss': loss, \"target\": batch['target'], \"preds_conf\":preds_conf,'preds_cls':preds_cls}\n        else:\n            loss = torch.zeros((1),device=x.device)\n            return {'loss': loss, \"target\": batch['target'],\"preds_conf\":preds_conf,'preds_cls':preds_cls,\n                    'embeddings': x_emb\n                   }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T08:00:00.877789Z","iopub.execute_input":"2025-05-15T08:00:00.878485Z","iopub.status.idle":"2025-05-15T08:00:00.910231Z","shell.execute_reply.started":"2025-05-15T08:00:00.878461Z","shell.execute_reply":"2025-05-15T08:00:00.909627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}