{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T04:07:23.945585Z","iopub.execute_input":"2023-04-19T04:07:23.945975Z","iopub.status.idle":"2023-04-19T04:07:25.618166Z","shell.execute_reply.started":"2023-04-19T04:07:23.945939Z","shell.execute_reply":"2023-04-19T04:07:25.616566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_data(data_dir, labels_path):\n    # Load labels CSV\n    labels_df = pd.read_csv(labels_path)\n\n    # Preprocess images and labels\n    images = []\n    poses = []\n    for _, row in labels_df.iterrows():\n        img_path = os.path.join(data_dir, row['image_path'])\n        img = cv2.imread(img_path)\n\n        if img is None:\n            print(f\"Failed to load image: {img_path}\")\n            continue\n\n        img = cv2.resize(img, (224, 224))  # Resize image\n        img = img / 255.0  # Normalize pixel values\n\n        rotation = np.array(row['rotation_matrix'].split(';'), dtype=float).reshape(3, 3)\n        translation = np.array(row['translation_vector'].split(';'), dtype=float)\n\n        pose = np.hstack((rotation, translation.reshape(-1, 1)))\n        pose = pose.flatten()  # Flatten the pose matrix to match the model output shape\n\n        images.append(img)\n        poses.append(pose)\n\n    images = np.array(images)\n    poses = np.array(poses)\n\n    return images, poses","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:11:30.349778Z","iopub.execute_input":"2023-04-19T04:11:30.350202Z","iopub.status.idle":"2023-04-19T04:11:30.362481Z","shell.execute_reply.started":"2023-04-19T04:11:30.350153Z","shell.execute_reply":"2023-04-19T04:11:30.361092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = \"/kaggle/input/image-matching-challenge-2023/train\"\nlabels_path = \"/kaggle/input/image-matching-challenge-2023/train/train_labels.csv\"\n\nimages, poses = preprocess_data(data_dir, labels_path)\nX_train, X_val, y_train, y_val = train_test_split(images, poses, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:12:12.297277Z","iopub.execute_input":"2023-04-19T04:12:12.297762Z","iopub.status.idle":"2023-04-19T04:13:08.538263Z","shell.execute_reply.started":"2023-04-19T04:12:12.297724Z","shell.execute_reply":"2023-04-19T04:13:08.537208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport glob\nimport os\nimport re\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport pydicom as dicom\nimport torch\nimport torchvision as tv\nfrom sklearn.model_selection import GroupKFold\nfrom torch.cuda.amp import GradScaler, autocast\nfrom torchvision.models.feature_extraction import create_feature_extractor\nfrom tqdm.notebook import tqdm\n\nimport wandb\n\nplt.rcParams['figure.figsize'] = (20, 5)\npd.set_option('display.max_rows', 100)\npd.set_option('display.max_columns', 1000)\n\n# Effnet\nWEIGHTS = tv.models.efficientnet.EfficientNet_V2_S_Weights.DEFAULT\nRSNA_2022_PATH = '/kaggle/input/image-matching-challenge-2023'\nTRAIN_IMAGES_PATH = f'{RSNA_2022_PATH}/train'\nTEST_IMAGES_PATH = f'{RSNA_2022_PATH}/test'\nEFFNET_MAX_TRAIN_BATCHES = 4000\nEFFNET_MAX_EVAL_BATCHES = 200\nONE_CYCLE_MAX_LR = 0.0001\nONE_CYCLE_PCT_START = 0.3\nSAVE_CHECKPOINT_EVERY_STEP = 1000\nEFFNET_CHECKPOINTS_PATH = '../input/rsna-2022-base-effnetv2'\nFRAC_LOSS_WEIGHT = 2.\nN_FOLDS = 3\nMETADATA_PATH = '../input/vertebrae-detection-checkpoints'\n\nPREDICT_MAX_BATCHES = 1e9\n\n# Common\ntry:\n    from kaggle_secrets import UserSecretsClient\n    IS_KAGGLE = True\nexcept:\n    IS_KAGGLE = False\n\nos.environ[\"WANDB_MODE\"] = \"online\"\nif os.environ[\"WANDB_MODE\"] == \"online\":\n    if IS_KAGGLE:\n        os.environ['WANDB_API_KEY'] = UserSecretsClient().get_secret(\"WANDB_API_KEY\")\n\nif not IS_KAGGLE:\n    print('Running locally')\n    RSNA_2022_PATH = '/mnt/rsna2022'\n    TRAIN_IMAGES_PATH = '/mnt/rsna2022/train_images'\n    TEST_IMAGES_PATH = '/mnt/rsna2022/test_images'\n    METADATA_PATH = '/home/vslaykovsky/Downloads/'\n    EFFNET_CHECKPOINTS_PATH = 'frac_checkpoints'\n    os.environ['WANDB_API_KEY'] = 'yourkeyhere'\n\nDEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nif DEVICE == 'cuda':\n    BATCH_SIZE = 32\nelse:\n    BATCH_SIZE = 2","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:16:52.223174Z","iopub.execute_input":"2023-04-19T04:16:52.223662Z","iopub.status.idle":"2023-04-19T04:16:52.515229Z","shell.execute_reply.started":"2023-04-19T04:16:52.223619Z","shell.execute_reply":"2023-04-19T04:16:52.514270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n! pip install efficientnet_pytorch\nfrom efficientnet_pytorch import EfficientNet\n\nclass MultiHeadAttention(nn.Module):\n    def __init__(self, d_model, num_heads):\n        super().__init__()\n        self.d_model = d_model\n        self.num_heads = num_heads\n        \n        assert d_model % num_heads == 0\n        \n        self.d_k = d_model // num_heads\n        \n        self.q_linear = nn.Linear(d_model, d_model)\n        self.v_linear = nn.Linear(d_model, d_model)\n        self.k_linear = nn.Linear(d_model, d_model)\n        \n        self.dropout = nn.Dropout(0.1)\n        self.out_linear = nn.Linear(d_model, d_model)\n        \n    def forward(self, q, k, v, mask=None):\n        bs = q.size(0)\n        \n        # perform linear operation and split into heads\n        q = self.q_linear(q).view(bs, -1, self.num_heads, self.d_k)\n        k = self.k_linear(k).view(bs, -1, self.num_heads, self.d_k)\n        v = self.v_linear(v).view(bs, -1, self.num_heads, self.d_k)\n        \n        # transpose to get dimensions bs * num_heads * sl * d_model\n        q = q.transpose(1,2)\n        k = k.transpose(1,2)\n        v = v.transpose(1,2)\n        \n        # calculate attention using function we will define next\n        scores = self.attention(q, k, v, self.d_k, mask, self.dropout)\n        \n        # concatenate heads and put through final linear layer\n        concat = scores.transpose(1,2).contiguous().view(bs, -1, self.d_model)\n        output = self.out_linear(concat)\n        return output\n    \n    def attention(self, q, k, v, d_k, mask=None, dropout=None):\n        scores = torch.matmul(q, k.transpose(-2, -1)) /  math.sqrt(d_k)\n        if mask is not None:\n            mask = mask.unsqueeze(1)\n            scores = scores.masked_fill(mask == 0, -1e9)\n        scores = nn.functional.softmax(scores, dim=-1)\n        if dropout is not None:\n            scores = dropout(scores)\n        output = torch.matmul(scores, v)\n        return output\n\nclass EfficientNetWithAttention(nn.Module):\n    def __init__(self, num_classes, d_model, num_heads):\n        super().__init__()\n        \n        self.efficient_net = EfficientNet.from_pretrained('efficientnet-b0',in_channels=2)\n        \n        self.pooling = nn.AdaptiveAvgPool2d(1) # add global average pooling layer\n        self.linear1 = nn.Linear(1280, d_model) # add linear layer to adjust input size\n        self.attention = MultiHeadAttention(d_model, num_heads)\n        self.classifier = nn.Linear(d_model, num_classes)\n        \n    def forward(self, x):\n        x = self.efficient_net.extract_features(x)\n        x = self.pooling(x)\n        x = x.view(x.size(0), -1) # flatten\n        x = self.linear1(x) # adjust input size\n        x = self.attention(x, x, x)\n        x = self.classifier(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:23:54.852688Z","iopub.execute_input":"2023-04-19T04:23:54.853138Z","iopub.status.idle":"2023-04-19T04:24:11.099768Z","shell.execute_reply.started":"2023-04-19T04:23:54.853102Z","shell.execute_reply":"2023-04-19T04:24:11.097781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport math\ncustom_model = EfficientNetWithAttention(num_classes=12, d_model=256, num_heads=4)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:25:52.672532Z","iopub.execute_input":"2023-04-19T04:25:52.673884Z","iopub.status.idle":"2023-04-19T04:25:53.812014Z","shell.execute_reply.started":"2023-04-19T04:25:52.673823Z","shell.execute_reply":"2023-04-19T04:25:53.810555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:48:07.280736Z","iopub.execute_input":"2023-04-19T04:48:07.281845Z","iopub.status.idle":"2023-04-19T04:48:07.288996Z","shell.execute_reply.started":"2023-04-19T04:48:07.281796Z","shell.execute_reply":"2023-04-19T04:48:07.287483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom keras.layers import Input","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:42:26.762093Z","iopub.execute_input":"2023-04-19T04:42:26.762555Z","iopub.status.idle":"2023-04-19T04:42:38.900118Z","shell.execute_reply.started":"2023-04-19T04:42:26.762515Z","shell.execute_reply":"2023-04-19T04:42:38.898453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n    print(\"Device:\", tpu.master())\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"Not connected to a TPU runtime. Using CPU/GPU strategy\")\n    strategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:54:34.323201Z","iopub.execute_input":"2023-04-19T04:54:34.323627Z","iopub.status.idle":"2023-04-19T04:54:34.335374Z","shell.execute_reply.started":"2023-04-19T04:54:34.323591Z","shell.execute_reply":"2023-04-19T04:54:34.333936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras import layers\n\ndef build_model(num_classes):\n    inputs = layers.Input(shape=(224, 224, 3))\n    #x = img_augmentation(inputs)\n    model = EfficientNetB0(include_top=False, input_tensor=inputs, weights=\"imagenet\")\n\n    # Freeze the pretrained weights\n    model.trainable = False\n\n    # Rebuild top\n    x = layers.GlobalAveragePooling2D(name=\"avg_pool\")(model.output)\n    x = layers.BatchNormalization()(x)\n\n    top_dropout_rate = 0.2\n    x = layers.Dropout(top_dropout_rate, name=\"top_dropout\")(x)\n    outputs = layers.Dense(12, activation=\"softmax\", name=\"pred\")(x)\n\n    # Compile\n    model = tf.keras.Model(inputs, outputs, name=\"EfficientNet\")\n    optimizer = tf.keras.optimizers.Adam(learning_rate=1e-4)\n    model.compile(\n        optimizer=optimizer, loss=\"mean_squared_error\", metrics=[\"accuracy\"]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:59:47.149215Z","iopub.execute_input":"2023-04-19T04:59:47.149625Z","iopub.status.idle":"2023-04-19T04:59:47.162132Z","shell.execute_reply.started":"2023-04-19T04:59:47.149590Z","shell.execute_reply":"2023-04-19T04:59:47.160764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model(num_classes=12)\n\nhistory = model.fit(data_gen.flow(X_train, y_train, batch_size=1),\n                    validation_data=(X_val, y_val),\n                    epochs=75,\n                    steps_per_epoch=len(X_train) // 1)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T04:59:48.885244Z","iopub.execute_input":"2023-04-19T04:59:48.885658Z","iopub.status.idle":"2023-04-19T05:37:09.286103Z","shell.execute_reply.started":"2023-04-19T04:59:48.885625Z","shell.execute_reply":"2023-04-19T05:37:09.284063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('/kaggle/working/model.ckpt')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('/kaggle/working/model.ckpt')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_test_data(data_dir, image_paths):\n    images = []\n    for image_path in image_paths:\n        img_path = os.path.join(data_dir, image_path)\n        img = cv2.imread(img_path)\n\n        if img is None:\n            print(f\"Failed to load image: {img_path}\")\n            continue\n\n        img = cv2.resize(img, (224, 224))  # Resize image\n        img = img / 255.0  # Normalize pixel values\n\n        images.append(img)\n\n    images = np.array(images)\n\n    return images","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:37:09.292354Z","iopub.execute_input":"2023-04-19T05:37:09.292816Z","iopub.status.idle":"2023-04-19T05:37:09.306126Z","shell.execute_reply.started":"2023-04-19T05:37:09.292765Z","shell.execute_reply":"2023-04-19T05:37:09.304279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the sample submission file\nsample_submission_path = \"/kaggle/input/image-matching-challenge-2023/sample_submission.csv\"\nsample_submission_df = pd.read_csv(sample_submission_path)\ntest_image_paths = sample_submission_df['image_path'].values\n\n# Preprocess the test data\ntest_data_dir = \"/kaggle/input/image-matching-challenge-2023/test\"\ntest_images_processed = preprocess_test_data(test_data_dir, test_image_paths)\n\n# Predict the poses using the trained model\npredicted_poses = model.predict(test_images_processed)\n\n# Create a submission file with the predicted poses\nsample_submission_df.iloc[:, 1:] = predicted_poses\nsample_submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:37:09.308538Z","iopub.execute_input":"2023-04-19T05:37:09.309406Z","iopub.status.idle":"2023-04-19T05:37:09.569684Z","shell.execute_reply.started":"2023-04-19T05:37:09.309340Z","shell.execute_reply":"2023-04-19T05:37:09.567162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}