{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":9797223,"sourceType":"datasetVersion","datasetId":982170},{"sourceId":13336877,"sourceType":"datasetVersion","datasetId":8456669},{"sourceId":261002435,"sourceType":"kernelVersion"},{"sourceId":265636816,"sourceType":"kernelVersion"},{"sourceId":267120005,"sourceType":"kernelVersion"},{"sourceId":574351,"sourceType":"modelInstanceVersion","modelInstanceId":429902,"modelId":446851},{"sourceId":574363,"sourceType":"modelInstanceVersion","modelInstanceId":429911,"modelId":446860}],"dockerImageVersionId":31090,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nsys.path.append(\"../input/tez-lib/\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.624141Z","iopub.execute_input":"2025-10-11T15:02:34.624695Z","iopub.status.idle":"2025-10-11T15:02:34.628193Z","shell.execute_reply.started":"2025-10-11T15:02:34.624671Z","shell.execute_reply":"2025-10-11T15:02:34.627461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tez\nfrom tez import Tez, TezConfig\nimport albumentations\nimport pandas as pd\nimport cv2\nimport numpy as np\nimport timm\nimport torch\nimport torch.nn as nn\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.629518Z","iopub.execute_input":"2025-10-11T15:02:34.629744Z","iopub.status.idle":"2025-10-11T15:02:34.643121Z","shell.execute_reply.started":"2025-10-11T15:02:34.629714Z","shell.execute_reply":"2025-10-11T15:02:34.642203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class args:\n    batch_size = 64\n    image_size = 384","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.643800Z","iopub.execute_input":"2025-10-11T15:02:34.644058Z","iopub.status.idle":"2025-10-11T15:02:34.657569Z","shell.execute_reply.started":"2025-10-11T15:02:34.644034Z","shell.execute_reply":"2025-10-11T15:02:34.656892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BEST_MODEL_NAME = \"inception_v3\"\nINFERENCE_FOLD = 0\nBASE_IMAGE_PATH = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/test/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.658396Z","iopub.execute_input":"2025-10-11T15:02:34.659123Z","iopub.status.idle":"2025-10-11T15:02:34.672187Z","shell.execute_reply.started":"2025-10-11T15:02:34.659092Z","shell.execute_reply":"2025-10-11T15:02:34.671517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class targetDataset:\n    def __init__(self, image_paths, dense_features, targets, augmentations):\n        self.image_paths = image_paths\n        self.dense_features = dense_features\n        self.targets = targets \n        self.augmentations = augmentations\n        \n    def __len__(self):\n        return len(self.image_paths)\n    \n    def __getitem__(self, item):\n        image_path = self.image_paths[item]\n        image = cv2.imread(image_path)\n        \n       \n        if image is None:\n            raise FileNotFoundError(f\"Could not read image file: {image_path}. Please check the path.\")\n            \n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        if self.augmentations is not None:\n            augmented = self.augmentations(image=image)\n            image = augmented[\"image\"]\n            \n        image = np.transpose(image, (2, 0, 1)).astype(np.float32)\n        features = self.dense_features[item, :]\n        \n        \n        return {\n            \"image\": torch.tensor(image, dtype=torch.float),\n            \"features\": torch.tensor(features, dtype=torch.float),\n        }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.673753Z","iopub.execute_input":"2025-10-11T15:02:34.674166Z","iopub.status.idle":"2025-10-11T15:02:34.688372Z","shell.execute_reply.started":"2025-10-11T15:02:34.674148Z","shell.execute_reply":"2025-10-11T15:02:34.687791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class targetModel(nn.Module):\n    def __init__(self, model_name=\"resnet50\", dense_dim=12, pretrained=False):\n        super().__init__()\n        self.model = timm.create_model(model_name, pretrained=pretrained, in_chans=3)\n        n_features = self.model.get_classifier().in_features\n        self.model.reset_classifier(0)\n        self.out = nn.Linear(n_features + dense_dim, 1)\n\n    def forward(self, image, features, targets=None):\n        image_features = self.model(image)\n        x = torch.cat([image_features, features], dim=1)\n        output = self.out(x)\n        return output, 0, {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.689096Z","iopub.execute_input":"2025-10-11T15:02:34.689321Z","iopub.status.idle":"2025-10-11T15:02:34.706266Z","shell.execute_reply.started":"2025-10-11T15:02:34.689295Z","shell.execute_reply":"2025-10-11T15:02:34.705614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_aug = albumentations.Compose([\n    albumentations.LongestMaxSize(args.image_size, p=1),\n    albumentations.PadIfNeeded(args.image_size, args.image_size, p=1, border_mode=cv2.BORDER_CONSTANT),\n    albumentations.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], max_pixel_value=255.0, p=1.0),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.706988Z","iopub.execute_input":"2025-10-11T15:02:34.707273Z","iopub.status.idle":"2025-10-11T15:02:34.721221Z","shell.execute_reply.started":"2025-10-11T15:02:34.707248Z","shell.execute_reply":"2025-10-11T15:02:34.720607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.722154Z","iopub.execute_input":"2025-10-11T15:02:34.722422Z","iopub.status.idle":"2025-10-11T15:02:34.795832Z","shell.execute_reply.started":"2025-10-11T15:02:34.722396Z","shell.execute_reply":"2025-10-11T15:02:34.795251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = ['sex', 'anatom_site_general_challenge']\ndf_train = pd.get_dummies(df_train, columns=categorical_features, dummy_na=False)\ndf_test = pd.get_dummies(df_test, columns=categorical_features, dummy_na=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:34.796532Z","iopub.execute_input":"2025-10-11T15:02:34.796712Z","iopub.status.idle":"2025-10-11T15:02:34.818212Z","shell.execute_reply.started":"2025-10-11T15:02:34.796697Z","shell.execute_reply":"2025-10-11T15:02:34.817611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"one_hot_cols = [col for col in df_train.columns if any(f\"_{cat}\" in col for cat in categorical_features)]\nnumerical_features = ['age_approx']\ndense_features_list = numerical_features + one_hot_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:02:55.383245Z","iopub.execute_input":"2025-10-11T15:02:55.383563Z","iopub.status.idle":"2025-10-11T15:02:55.387977Z","shell.execute_reply.started":"2025-10-11T15:02:55.383522Z","shell.execute_reply":"2025-10-11T15:02:55.387169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = set(df_train.columns)\nfor col in dense_features_list:\n    if col not in df_test.columns:\n        df_test[col] = 0\ndf_test = df_test[df_train.columns.intersection(df_test.columns)]\nprint(f\"Final dense features being used ({len(dense_features_list)}): {dense_features_list}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:03:14.696909Z","iopub.execute_input":"2025-10-11T15:03:14.697525Z","iopub.status.idle":"2025-10-11T15:03:14.706477Z","shell.execute_reply.started":"2025-10-11T15:03:14.697497Z","shell.execute_reply":"2025-10-11T15:03:14.705702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"\\n--- Generating predictions using the optimal model: {BEST_MODEL_NAME.upper()} ---\")\n\n# Initialize model with chosen architecture\nbest_model = targetModel(\n    model_name=BEST_MODEL_NAME, \n    dense_dim=len(dense_features_list), \n    pretrained=False\n)\nbest_model = Tez(best_model)\n\n# Load the pretrained weights\nweights_file = \"/kaggle/input/modela/model_melf0.bin\"\nbest_model.load(weights_file, weights_only=True)\n\n# Prepare test image paths and dataset\ntest_image_files = [f\"{BASE_IMAGE_PATH}{img}.jpg\" for img in df_test[\"image_name\"].values]\ntest_data = targetDataset(\n    image_paths=test_image_files,\n    dense_features=df_test[dense_features_list].values,\n    targets=np.ones(len(test_image_files)),\n    augmentations=test_aug,\n)\n\n# Run prediction\nraw_predictions = best_model.predict(\n    test_data,\n    batch_size=2 * args.batch_size,\n    n_jobs=-1\n)\n\n# Collect logits from predictions\nlogit_outputs = []\nfor output in tqdm(raw_predictions):\n    logit_outputs.extend(output.ravel().tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:05:30.127485Z","iopub.execute_input":"2025-10-11T15:05:30.128035Z","iopub.status.idle":"2025-10-11T15:12:38.162592Z","shell.execute_reply.started":"2025-10-11T15:05:30.128000Z","shell.execute_reply":"2025-10-11T15:12:38.161484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nPreparing submission file using the top-performing model...\")\n\n# Sigmoid function to convert logits into probabilities\ndef sigmoid_fn(z):\n    return 1 / (1 + np.exp(-z))\n\n# Convert logits to probabilities\nlogits_array = np.array(logit_outputs)\nprobabilities = sigmoid_fn(logits_array)\n\n# Attach predictions to dataframe and save\ndf_test[\"target\"] = probabilities\nsubmission = df_test[[\"image_name\", \"target\"]]\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file has been generated successfully!\")\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T15:12:56.183245Z","iopub.execute_input":"2025-10-11T15:12:56.183950Z","iopub.status.idle":"2025-10-11T15:12:56.248287Z","shell.execute_reply.started":"2025-10-11T15:12:56.183914Z","shell.execute_reply":"2025-10-11T15:12:56.247662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}