{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Explore 🕵 provdided data","metadata":{}},{"cell_type":"code","source":"!ls -l /kaggle/input/happy-whale-and-dolphin\n\nPATH_DATASET = \"/kaggle/input/happy-whale-and-dolphin\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-08T15:53:09.669901Z","iopub.execute_input":"2022-03-08T15:53:09.670772Z","iopub.status.idle":"2022-03-08T15:53:10.455105Z","shell.execute_reply.started":"2022-03-08T15:53:09.670661Z","shell.execute_reply":"2022-03-08T15:53:10.453893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Browsing the metadata","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport seaborn as sn\nimport matplotlib.pyplot as plt\n\nsn.set()\n\ndf_train = pd.read_csv(os.path.join(PATH_DATASET, \"train.csv\"))\ndisplay(df_train.head())\nprint(f\"Dataset size: {len(df_train)}\")\nprint(f\"Unique ids: {len(df_train['individual_id'].unique())}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:53:10.457722Z","iopub.execute_input":"2022-03-08T15:53:10.458026Z","iopub.status.idle":"2022-03-08T15:53:11.602480Z","shell.execute_reply.started":"2022-03-08T15:53:10.457979Z","shell.execute_reply":"2022-03-08T15:53:11.601409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see how many speaced we have in the database...","metadata":{}},{"cell_type":"code","source":"counts_imgs = df_train[\"species\"].value_counts()\ncounts_inds = df_train.drop_duplicates(\"individual_id\")[\"species\"].value_counts()\n\nax = pd.concat({\"per Images\": counts_imgs, \"per Individuals\": counts_inds}, axis=1).plot.barh(grid=True, figsize=(7, 10))\nax.set_xscale('log')","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:53:11.604543Z","iopub.execute_input":"2022-03-08T15:53:11.604858Z","iopub.status.idle":"2022-03-08T15:53:13.766597Z","shell.execute_reply.started":"2022-03-08T15:53:11.604814Z","shell.execute_reply":"2022-03-08T15:53:13.765637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And compare they with unique individuals... \n\n**Note:** that the counts are in log scale","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom pprint import pprint\n\nspecies_individuals = {}\nfor name, dfg in df_train.groupby(\"species\"):\n    species_individuals[name] = dfg[\"individual_id\"].value_counts()\n\nsi_max = max(list(map(len, species_individuals.values())))\nsi = {n: [0] * si_max for n in species_individuals}\nfor n, counts in species_individuals.items():\n    si[n][:len(counts)] = list(np.log(counts))\nsi = pd.DataFrame(si)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:53:13.772480Z","iopub.execute_input":"2022-03-08T15:53:13.775153Z","iopub.status.idle":"2022-03-08T15:53:13.906287Z","shell.execute_reply.started":"2022-03-08T15:53:13.775108Z","shell.execute_reply":"2022-03-08T15:53:13.905349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\n\nfig = plt.figure(figsize=(10, 8))\nax = sn.heatmap(si[:500].T, cmap=\"BuGn\", ax=fig.gca())","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:53:13.911099Z","iopub.execute_input":"2022-03-08T15:53:13.914118Z","iopub.status.idle":"2022-03-08T15:53:15.581493Z","shell.execute_reply.started":"2022-03-08T15:53:13.914071Z","shell.execute_reply":"2022-03-08T15:53:15.580580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline: embedding with Lightning⚡Flash\n\nFollow the example: https://lightning-flash.readthedocs.io/en/stable/reference/image_embedder.html","metadata":{}},{"cell_type":"code","source":"!pip install -q vissl fairscale 'lightning-flash[image]'\n# temp fix untill it is merged to master & released...\n!pip install -q -U \"https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/master.zip\"\n!pip uninstall -y wandb","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-03-08T15:53:15.582584Z","iopub.execute_input":"2022-03-08T15:53:15.582890Z","iopub.status.idle":"2022-03-08T15:55:15.005090Z","shell.execute_reply.started":"2022-03-08T15:53:15.582840Z","shell.execute_reply":"2022-03-08T15:55:15.004057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip download -q vissl fairscale 'lightning-flash[image]' --dest frozen_packages --prefer-binary\n!pip wheel -q \"https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/master.zip\" --wheel-dir frozen_packages\n!rm frozen_packages/torch-*\n!ls -l frozen_packages","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-03-08T15:55:15.007235Z","iopub.execute_input":"2022-03-08T15:55:15.008124Z","iopub.status.idle":"2022-03-08T15:58:08.200211Z","shell.execute_reply.started":"2022-03-08T15:55:15.008072Z","shell.execute_reply":"2022-03-08T15:58:08.199071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nimport flash\nfrom flash.core.data.utils import download_data\nfrom flash.image import ImageClassificationData, ImageEmbedder","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:08.202352Z","iopub.execute_input":"2022-03-08T15:58:08.203920Z","iopub.status.idle":"2022-03-08T15:58:19.715737Z","shell.execute_reply.started":"2022-03-08T15:58:08.203863Z","shell.execute_reply":"2022-03-08T15:58:19.714766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Create the Dataset 🗄️","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nfrom torch.utils.data import Dataset\n\n\nclass HappyWhaleDataset(Dataset):\n    def __init__(self, df: pd.DataFrame, path_folder: str, transform = None):\n        self.df = df\n        self.transform = transform\n\n        self.image_names = self.df[\"image\"].values\n        self.image_paths = [os.path.join(path_folder, n) for n in self.image_names]\n        self.targets = list(self.df[\"individual_id\"])\n        self.uq_targets = sorted(set(self.targets))\n        lut = {v: k for k, v in dict(enumerate(self.uq_targets)).items()}\n        self.labels = [lut[ind] for ind in self.targets]\n\n    def __getitem__(self, idx: int) -> tuple:\n        img_path = self.image_paths[idx]\n        img = Image.open(img_path).convert('RGB')\n\n        if self.transform:\n            img = self.transform(img)\n\n        lb = torch.tensor(self.labels[idx], dtype=torch.long)\n        return img, lb\n\n    def __len__(self) -> int:\n        return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:19.717540Z","iopub.execute_input":"2022-03-08T15:58:19.717880Z","iopub.status.idle":"2022-03-08T15:58:19.729331Z","shell.execute_reply.started":"2022-03-08T15:58:19.717796Z","shell.execute_reply":"2022-03-08T15:58:19.728458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = HappyWhaleDataset(\n    # df=df_train,\n    # ToDo: use full dataset\n    df=df_train[:int(len(df_train) * 0.6)],\n    path_folder=f\"{PATH_DATASET}/train_images\",\n)\n\nfig, axarr = plt.subplots(nrows=2, ncols=5, figsize=(14, 4))\nfor i in range(10):\n    img, ind = dataset[i]\n    axarr[i // 5, i % 5].imshow(img)\n    axarr[i // 5, i % 5].set_title(ind)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:19.735106Z","iopub.execute_input":"2022-03-08T15:58:19.735617Z","iopub.status.idle":"2022-03-08T15:58:29.321654Z","shell.execute_reply.started":"2022-03-08T15:58:19.735574Z","shell.execute_reply":"2022-03-08T15:58:29.320586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datamodule = ImageClassificationData.from_datasets(\n    train_dataset=dataset,\n    batch_size=64,\n    num_workers=6,\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:29.323664Z","iopub.execute_input":"2022-03-08T15:58:29.324291Z","iopub.status.idle":"2022-03-08T15:58:29.337947Z","shell.execute_reply.started":"2022-03-08T15:58:29.324236Z","shell.execute_reply":"2022-03-08T15:58:29.336884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Build the task ⚙️","metadata":{}},{"cell_type":"code","source":"embedder = ImageEmbedder(\n    backbone=\"resnet\",\n    training_strategy=\"barlow_twins\",\n    head=\"simclr_head\",\n    pretraining_transform=\"barlow_twins_transform\",\n    training_strategy_kwargs={\"latent_embedding_dim\": 256},\n    pretraining_transform_kwargs={\"size_crops\": [196]},\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:29.339836Z","iopub.execute_input":"2022-03-08T15:58:29.340482Z","iopub.status.idle":"2022-03-08T15:58:30.806050Z","shell.execute_reply.started":"2022-03-08T15:58:29.340436Z","shell.execute_reply":"2022-03-08T15:58:30.803589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Finetune the model 🛠️","metadata":{}},{"cell_type":"code","source":"from pytorch_lightning.loggers import CSVLogger\n# from pytorch_lightning.callbacks import StochasticWeightAveraging\n\n# Trainer Args\nGPUS = int(torch.cuda.is_available())  # Set to 1 if GPU is enabled for notebook\n\n# swa = StochasticWeightAveraging(swa_epoch_start=0.6)\nlogger = CSVLogger(save_dir='logs/')\n\ntrainer = flash.Trainer(\n    max_epochs=5,\n    # gradient_clip_val=0.01,\n    gpus=GPUS,\n    precision=16 if GPUS else 32,\n    logger=logger,\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T15:58:30.807890Z","iopub.execute_input":"2022-03-08T15:58:30.808184Z","iopub.status.idle":"2022-03-08T15:58:30.824044Z","shell.execute_reply.started":"2022-03-08T15:58:30.808141Z","shell.execute_reply":"2022-03-08T15:58:30.822956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.fit(embedder, datamodule=datamodule)\n\ntrainer.save_checkpoint(\"image_embedder_model.pt\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-03-08T15:58:30.825222Z","iopub.execute_input":"2022-03-08T15:58:30.825462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics = pd.read_csv(f'{trainer.logger.log_dir}/metrics.csv')\ndel metrics[\"step\"]\nmetrics.set_index(\"epoch\", inplace=True)\ndisplay(metrics.dropna(axis=1, how=\"all\").head())\ng = sn.relplot(data=metrics, kind=\"line\")\nplt.gcf().set_size_inches(15, 5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Run predictions 🎉","metadata":{}},{"cell_type":"code","source":"import glob\n\nimgs = glob.glob(f\"{PATH_DATASET}/test_images/*.jpg\")\ndatamodule = ImageClassificationData.from_files(\n    predict_files=imgs[:5],\n    batch_size=12\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedder.input_transform = None\nembeddings = trainer.predict(embedder, datamodule=datamodule)\n\n# list of embeddings for images sent to the predict function\npprint(embeddings)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}