{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torch.utils.data import DataLoader,Dataset\nfrom torchvision import datasets\nfrom torchvision.transforms import ToTensor\nimport os\n\nimport pandas as pd\nimport numpy as np\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n\nimport cv2\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:26.77626Z","iopub.execute_input":"2022-10-06T14:24:26.77685Z","iopub.status.idle":"2022-10-06T14:24:26.786722Z","shell.execute_reply.started":"2022-10-06T14:24:26.776812Z","shell.execute_reply":"2022-10-06T14:24:26.785056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"mrckkk\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RESIZE_SIZE = 224\nBATCH_SIZE = 50\nEPOCHS = 3\nLEARNING_RATE = 0.001\nROOT_DIR = '../input/happy-whale-and-dolphin'\nTRAIN_DIR = '../input/happy-whale-and-dolphin/train_images'\nTEST_DIR = '../input/happy-whale-and-dolphin/test_images'\n\ndef get_train_file_path(id):\n    return f\"{TRAIN_DIR}/{id}\"","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:26.788288Z","iopub.execute_input":"2022-10-06T14:24:26.788969Z","iopub.status.idle":"2022-10-06T14:24:26.798086Z","shell.execute_reply.started":"2022-10-06T14:24:26.788933Z","shell.execute_reply":"2022-10-06T14:24:26.797129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(42)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:26.801086Z","iopub.execute_input":"2022-10-06T14:24:26.801507Z","iopub.status.idle":"2022-10-06T14:24:26.816363Z","shell.execute_reply.started":"2022-10-06T14:24:26.801473Z","shell.execute_reply":"2022-10-06T14:24:26.815429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\ndf['file_path'] = df['image'].apply(get_train_file_path)\ndf['species_id']=df['species']\nencoder = LabelEncoder()\ndf['individual_id'] = encoder.fit_transform(df['individual_id'])\ndf['species_id'] = encoder.fit_transform(df['species_id'])","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:26.818947Z","iopub.execute_input":"2022-10-06T14:24:26.819709Z","iopub.status.idle":"2022-10-06T14:24:26.987437Z","shell.execute_reply.started":"2022-10-06T14:24:26.819661Z","shell.execute_reply":"2022-10-06T14:24:26.986477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5)\n\nfor fold, ( _, val_) in enumerate(skf.split(X=df, y=df.species_id)):\n      df.loc[val_ , \"kfold\"] = fold\ndf.to_csv('adjusted_train2.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:26.989005Z","iopub.execute_input":"2022-10-06T14:24:26.989381Z","iopub.status.idle":"2022-10-06T14:24:27.179904Z","shell.execute_reply.started":"2022-10-06T14:24:26.989342Z","shell.execute_reply":"2022-10-06T14:24:27.178807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HappyWhaleDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n        self.labels = df['species_id'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        label = self.labels[index]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            'label': torch.tensor(label, dtype=torch.long)\n        }","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.181504Z","iopub.execute_input":"2022-10-06T14:24:27.181855Z","iopub.status.idle":"2022-10-06T14:24:27.189895Z","shell.execute_reply.started":"2022-10-06T14:24:27.181817Z","shell.execute_reply":"2022-10-06T14:24:27.188848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transforms = {\n    \"train\": A.Compose([\n        A.Resize(RESIZE_SIZE, RESIZE_SIZE),\n        A.ShiftScaleRotate(shift_limit=0.1, \n                           scale_limit=0.15, \n                           rotate_limit=60, \n                           p=0.5),\n\n        A.RandomBrightnessContrast(\n                brightness_limit=(-0.1,0.1), \n                contrast_limit=(-0.1, 0.1), \n                p=0.5\n            ),\n        A.Normalize(\n                mean=[0.485], \n                std=[0.229], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.),\n    \n    \"valid\": A.Compose([\n        A.Resize(RESIZE_SIZE, RESIZE_SIZE),\n        A.Normalize(\n                mean=[0.485], \n                std=[0.229], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.191728Z","iopub.execute_input":"2022-10-06T14:24:27.192079Z","iopub.status.idle":"2022-10-06T14:24:27.202083Z","shell.execute_reply.started":"2022-10-06T14:24:27.192045Z","shell.execute_reply":"2022-10-06T14:24:27.201135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_loaders(df, fold):\n    df_train = df[df.kfold != fold].reset_index(drop=True)\n    df_valid = df[df.kfold == fold].reset_index(drop=True)\n    \n    train_dataset = HappyWhaleDataset(df_train, transforms=data_transforms[\"train\"])\n    valid_dataset = HappyWhaleDataset(df_valid, transforms=data_transforms[\"valid\"])\n\n    train_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, \n                              num_workers=2, shuffle=True, pin_memory=True, drop_last=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=BATCH_SIZE, \n                              num_workers=2, shuffle=False, pin_memory=True)\n    \n    return train_loader, valid_loader","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.204002Z","iopub.execute_input":"2022-10-06T14:24:27.204434Z","iopub.status.idle":"2022-10-06T14:24:27.214903Z","shell.execute_reply.started":"2022-10-06T14:24:27.20431Z","shell.execute_reply":"2022-10-06T14:24:27.214253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass FeedForwardNet(nn.Module):\n\n    def __init__(self):\n        super().__init__()\n        self.flatten = nn.Flatten()\n        self.dense_layers = nn.Sequential(\n            nn.Linear(RESIZE_SIZE*RESIZE_SIZE, 700),\n            nn.ReLU(),\n            nn.Linear(700, 300),\n            nn.ReLU(),\n            nn.Linear(300, 30)\n        )\n        self.softmax = nn.Softmax(dim=1)\n        \n    def forward(self, input_data):\n        x = self.flatten(input_data)\n        logits = self.dense_layers(x)\n        predictions = self.softmax(logits)\n\n        return predictions","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.219274Z","iopub.execute_input":"2022-10-06T14:24:27.220138Z","iopub.status.idle":"2022-10-06T14:24:27.228124Z","shell.execute_reply.started":"2022-10-06T14:24:27.220096Z","shell.execute_reply":"2022-10-06T14:24:27.226948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_single_epoch(model, data_loader, loss_fn, optimiser, device):\n\n    model.train()\n    \n    for data in data_loader:\n        input = data['image'].to(device)\n        target = data['label'].to(device)\n\n        # calculate loss\n        prediction = model(input)\n        loss = loss_fn(prediction, target)\n\n        # backpropagate error and update weights\n        optimiser.zero_grad()\n        loss.backward()\n        optimiser.step()\n    print(f\"loss: {loss.item()}\")","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.229832Z","iopub.execute_input":"2022-10-06T14:24:27.230246Z","iopub.status.idle":"2022-10-06T14:24:27.241554Z","shell.execute_reply.started":"2022-10-06T14:24:27.230145Z","shell.execute_reply":"2022-10-06T14:24:27.240587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, data_loader, loss_fn, optimiser, device, epochs):\n    for i in range(epochs):\n        print(f\"Epoch {i+1}\")\n        train_single_epoch(model, data_loader, loss_fn, optimiser, device)\n        print(\"---------------------------\")\n    print(\"Finished training\")","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.243539Z","iopub.execute_input":"2022-10-06T14:24:27.243975Z","iopub.status.idle":"2022-10-06T14:24:27.252167Z","shell.execute_reply.started":"2022-10-06T14:24:27.243939Z","shell.execute_reply":"2022-10-06T14:24:27.251228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create data loader\ntrain_dataloader,_ = prepare_loaders(df=df,fold=0)\n\n# construct model and assign it to device\nif torch.cuda.is_available():\n    device = \"cuda\"\nelse:\n    device = \"cpu\"\nprint(f\"Using {device}\")\nfeed_forward_net = FeedForwardNet().to(device)\n\n# loss funtion + optimiser\nloss_fn = nn.CrossEntropyLoss()\noptimiser = torch.optim.Adam(feed_forward_net.parameters(),lr=LEARNING_RATE)\n\n# train model\n\ntrain(feed_forward_net, train_dataloader, loss_fn, optimiser, device, EPOCHS)\n\n# save model\ntorch.save(feed_forward_net.state_dict(), \"feedforwardnet_whale_species.pth\")\nprint(\"Trained feed forward net saved at feedforwardnet_whale_species.pth\")","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:24:27.253935Z","iopub.execute_input":"2022-10-06T14:24:27.25437Z","iopub.status.idle":"2022-10-06T16:11:33.211144Z","shell.execute_reply.started":"2022-10-06T14:24:27.254289Z","shell.execute_reply":"2022-10-06T16:11:33.209668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}