{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://i.imgur.com/Kk8L8Ei.png)","metadata":{}},{"cell_type":"markdown","source":"# Import libraries 📚","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport albumentations as A\nimport wandb\n\nfrom termcolor import colored\nfrom colorama import Fore, Back, Style\n# colored output\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:46:55.961876Z","iopub.execute_input":"2022-02-18T06:46:55.962432Z","iopub.status.idle":"2022-02-18T06:47:00.193201Z","shell.execute_reply.started":"2022-02-18T06:46:55.962333Z","shell.execute_reply":"2022-02-18T06:47:00.191912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<center><img src=\"https://camo.githubusercontent.com/dd842f7b0be57140e68b2ab9cb007992acd131c48284eaf6b1aca758bfea358b/68747470733a2f2f692e696d6775722e636f6d2f52557469567a482e706e67\"></center>\n\nI will be integrating ```W&B``` for ```visualizations``` and ```logging artifacts```!\n\n[Happywhale - Whale and Dolphin Identification Project on W&B Dashboard](https://wandb.ai/ruchi798/happywhale?workspace=user-ruchi798) 🏋️‍♀️\n\n* To get the API key, an account is to be created on the website first.\n* Next, use secrets to use API Keys more securely🤫","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\napi_key = user_secrets.get_secret(\"api_key\")\n\nCONFIG = {'competition': 'happywhale', '_wandb_kernel': 'ruch'}\n\nos.environ[\"WANDB_SILENT\"] = \"true\"","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:00.196199Z","iopub.execute_input":"2022-02-18T06:47:00.196604Z","iopub.status.idle":"2022-02-18T06:47:00.350532Z","shell.execute_reply.started":"2022-02-18T06:47:00.196553Z","shell.execute_reply":"2022-02-18T06:47:00.349046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! wandb login $api_key","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:00.351618Z","iopub.execute_input":"2022-02-18T06:47:00.351882Z","iopub.status.idle":"2022-02-18T06:47:03.111937Z","shell.execute_reply.started":"2022-02-18T06:47:00.351852Z","shell.execute_reply":"2022-02-18T06:47:03.110668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:03.114472Z","iopub.execute_input":"2022-02-18T06:47:03.114730Z","iopub.status.idle":"2022-02-18T06:47:03.234431Z","shell.execute_reply.started":"2022-02-18T06:47:03.114702Z","shell.execute_reply":"2022-02-18T06:47:03.233808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unique species and Species names 🐋 🐬","metadata":{}},{"cell_type":"code","source":"print(colored(\"Before fixing duplicate labels:\", 'green'))\nprint(\"Number of unique species: \",train_df['species'].nunique())\nprint(\"\\nSpecies names: \" ,train_df[\"species\"].unique())\n\n# fixing duplicate labels\ntrain_df['species'] = train_df['species'].str.replace('bottlenose_dolpin','bottlenose_dolphin')\ntrain_df['species'] = train_df['species'].str.replace('kiler_whale','killer_whale')\n\nprint(colored(\"\\nAfter fixing duplicate labels:\", 'green'))\nprint(\"Number of unique species: \",train_df['species'].nunique())\nprint(\"\\nSpecies names: \" ,train_df[\"species\"].unique())\n\n# append _whale to beluga and globis\ntrain_df[\"species\"].replace({\"beluga\": \"beluga_whale\", \"globis\": \"globis_whale\"}, inplace=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:47:03.235275Z","iopub.execute_input":"2022-02-18T06:47:03.235486Z","iopub.status.idle":"2022-02-18T06:47:03.371279Z","shell.execute_reply.started":"2022-02-18T06:47:03.235457Z","shell.execute_reply":"2022-02-18T06:47:03.370397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# specifying directory paths\n\ntrain_jpg_directory = '../input/happy-whale-and-dolphin/train_images'\ntest_jpg_directory = '../input/happy-whale-and-dolphin/test_images'\n\n# function to get image paths from train and test directory\n\ndef getImagePaths(path):\n    image_names = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_names.append(fullpath)\n    return image_names\n\ntrain_images_path = getImagePaths(train_jpg_directory)\ntest_images_path = getImagePaths(test_jpg_directory)\n\nprint(f\"{y_}Number of train images: {g_} {len(train_images_path)}\\n\")\nprint(f\"{y_}Number of test images: {g_} {len(test_images_path)}\\n\")\n\nrun = wandb.init(project='happywhale', name='count',config = CONFIG)\n\nun_ID = train_df.individual_id.nunique()\nun_sp = train_df['species'].nunique()\nwandb.log({'Training samples': len(train_images_path), \n          'Test samples': len(test_images_path),\n          'Number of individual IDs': un_ID,\n          'Number of unique species': un_sp,\n          })\n\nrun.finish()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:03.373582Z","iopub.execute_input":"2022-02-18T06:47:03.373943Z","iopub.status.idle":"2022-02-18T06:47:48.330078Z","shell.execute_reply.started":"2022-02-18T06:47:03.373898Z","shell.execute_reply":"2022-02-18T06:47:48.328760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getShape(data, images_paths):\n    shape = cv2.imread(images_paths[0]).shape\n    \n    for image_path in images_paths:\n        image_shape=cv2.imread(image_path).shape\n        if (image_shape!=shape):\n            flag = False\n            break;\n              \n    if (flag): return (data +\" - Same image shape \" + str(shape))\n    else: return (data +\" - Different image shape\")      \n        \nprint(getShape('train images', train_images_path))\nprint(getShape('test images', test_images_path))","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:48.331890Z","iopub.execute_input":"2022-02-18T06:47:48.332192Z","iopub.status.idle":"2022-02-18T06:47:49.135754Z","shell.execute_reply.started":"2022-02-18T06:47:48.332157Z","shell.execute_reply":"2022-02-18T06:47:49.134554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train and test images 📷","metadata":{}},{"cell_type":"code","source":"# function to display multiple images\n\ndef display_multiple_img(images_paths, rows, cols,title):\n    \n    figure, ax = plt.subplots(nrows=rows,ncols=cols,figsize=(16,8))\n    plt.suptitle(title, fontsize=20)\n    for ind,image_path in enumerate(images_paths):\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) \n        try:\n            ax.ravel()[ind].imshow(image)\n            ax.ravel()[ind].set_axis_off()\n        except:\n            continue;\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:49.137511Z","iopub.execute_input":"2022-02-18T06:47:49.137900Z","iopub.status.idle":"2022-02-18T06:47:49.147105Z","shell.execute_reply.started":"2022-02-18T06:47:49.137856Z","shell.execute_reply":"2022-02-18T06:47:49.145989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_multiple_img(train_images_path[0:25], 5, 5,\"Train images\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:47:49.148721Z","iopub.execute_input":"2022-02-18T06:47:49.149070Z","iopub.status.idle":"2022-02-18T06:48:02.052791Z","shell.execute_reply.started":"2022-02-18T06:47:49.149027Z","shell.execute_reply":"2022-02-18T06:48:02.052091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_multiple_img(test_images_path[0:25], 5, 5,\"Test images\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:02.055542Z","iopub.execute_input":"2022-02-18T06:48:02.056086Z","iopub.status.idle":"2022-02-18T06:48:15.214603Z","shell.execute_reply.started":"2022-02-18T06:48:02.056037Z","shell.execute_reply":"2022-02-18T06:48:15.213410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Whales and Dolphins Distribution 🐋 🐬 ","metadata":{}},{"cell_type":"code","source":"#====== Function to plot WandB bar chart ======\ndef plot_wb_bar(df,col1,col2, title): \n    run = wandb.init(project='happywhale', job_type='image-visualization',name=col1,config = CONFIG, anonymous=\"allow\")\n    \n    dt = [[label, val] for (label, val) in zip(df[col1], df[col2])]\n    table = wandb.Table(data=dt, columns = [col1,col2])\n    wandb.log({col1 : wandb.plot.bar(table, col1,col2,title=title)})\n    run.finish()\n    \n#====== Function to create a dataframe of value counts ======\ndef count_values(df,col,top=False):\n    df = pd.DataFrame(df[col].value_counts().reset_index().values,columns=[col, \"counts\"])\n    if top==True: df=df[:10]\n    return df\n\n#====== Function to create a dataframe ======\ndef intermediate_df(col, labels, sizes):\n    d = pd.DataFrame()\n    d[col] = labels\n    d['counts'] = sizes\n    return d","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:48:15.216029Z","iopub.execute_input":"2022-02-18T06:48:15.216305Z","iopub.status.idle":"2022-02-18T06:48:15.226618Z","shell.execute_reply.started":"2022-02-18T06:48:15.216274Z","shell.execute_reply":"2022-02-18T06:48:15.225700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a new column \ntrain_df['label'] = train_df.species.map(lambda x: 'dolphin' if 'dolphin' in x else 'whale')\n\nfig, ax  = plt.subplots(figsize=(16, 8))\nfig.suptitle('Whales and Dolphins ', size = 20, font=\"Serif\")\nexplode = (0.05, 0.05)\nlabels = list(train_df.label.value_counts().index)\nsizes = train_df.label.value_counts().values\nax.pie(sizes, explode=explode,startangle=60, labels=labels,autopct='%1.0f%%', pctdistance=0.7, colors=[\"#0077b6\",\"#90e0ef\"])\nax.add_artist(plt.Circle((0,0),0.4,fc='white'))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:15.227913Z","iopub.execute_input":"2022-02-18T06:48:15.228169Z","iopub.status.idle":"2022-02-18T06:48:15.440034Z","shell.execute_reply.started":"2022-02-18T06:48:15.228138Z","shell.execute_reply":"2022-02-18T06:48:15.438720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_wb_bar(intermediate_df('label', labels, sizes),\"label\", 'counts', \"Whales and Dolphins Distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:15.442395Z","iopub.execute_input":"2022-02-18T06:48:15.442842Z","iopub.status.idle":"2022-02-18T06:48:28.722036Z","shell.execute_reply.started":"2022-02-18T06:48:15.442789Z","shell.execute_reply":"2022-02-18T06:48:28.721199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Species Distribution 🐋 🐬","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nplt.yticks(fontsize=16)\nsns.countplot(y=\"species\",data=train_df,order=train_df.iloc[0:][\"species\"].value_counts().index,palette=\"PuBu\",linewidth=3)\nplt.title(\"Species Distribution\",font=\"Serif\", size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:28.723244Z","iopub.execute_input":"2022-02-18T06:48:28.723499Z","iopub.status.idle":"2022-02-18T06:48:29.271255Z","shell.execute_reply.started":"2022-02-18T06:48:28.723467Z","shell.execute_reply":"2022-02-18T06:48:29.270279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_wb_bar(count_values(train_df,\"species\", top=True),\"species\", 'counts', \"Most frequent species\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:29.272735Z","iopub.execute_input":"2022-02-18T06:48:29.273280Z","iopub.status.idle":"2022-02-18T06:48:42.415237Z","shell.execute_reply.started":"2022-02-18T06:48:29.273242Z","shell.execute_reply":"2022-02-18T06:48:42.414174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of training images: \",train_df.shape[0])\nprint(\"\\nNumber of individual IDs: \" ,train_df.individual_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:48:42.417312Z","iopub.execute_input":"2022-02-18T06:48:42.417574Z","iopub.status.idle":"2022-02-18T06:48:42.434784Z","shell.execute_reply.started":"2022-02-18T06:48:42.417542Z","shell.execute_reply":"2022-02-18T06:48:42.433640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frequency(df, col, freq):\n    n = 5\n    if freq == \"Most\":\n        return df[col].value_counts()[:n].index.tolist()\n    elif freq == \"Least\":\n        return df[col].value_counts()[-n:].index.tolist()\n    \nm_freq_species = frequency(train_df,\"species\", \"Most\")\nl_freq_species = frequency(train_df,\"species\", \"Least\")\nm_freq_ID = frequency(train_df,\"individual_id\", \"Most\")\nl_freq_ID = frequency(train_df,\"individual_id\", \"Least\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:48:42.436417Z","iopub.execute_input":"2022-02-18T06:48:42.436971Z","iopub.status.idle":"2022-02-18T06:48:42.490278Z","shell.execute_reply.started":"2022-02-18T06:48:42.436879Z","shell.execute_reply":"2022-02-18T06:48:42.489294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def path(df,group,group_type):\n    PATH = \"../input/happy-whale-and-dolphin/train_images\"\n    \n    #species\n    if group_type=='sp':\n        z = df['image'][df['species']==group].values \n    \n    #ID\n    if group_type=='id':\n        z = df['image'][df['individual_id']==group].values \n   \n    image_names = []\n    for filename in z:\n        fullpath = os.path.join(PATH, filename)\n        image_names.append(fullpath)\n    return image_names","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:48:42.491612Z","iopub.execute_input":"2022-02-18T06:48:42.492493Z","iopub.status.idle":"2022-02-18T06:48:42.499976Z","shell.execute_reply.started":"2022-02-18T06:48:42.492455Z","shell.execute_reply":"2022-02-18T06:48:42.498796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_groups(df, group_type, lst):\n    for item in lst:\n        display_multiple_img(path(df,item,group_type)[:9], 3, 3,item)\n        \ndef map_species(df,group_type, lst, name, table_name):\n    run = wandb.init(project='happywhale', job_type='image-visualization',name=name,config = CONFIG, anonymous=\"allow\")\n\n    # Initialize an empty W&B Table\n    data_table = wandb.Table(columns=['species', 'img1', 'img2', 'img_3', 'img_4', 'img_5'])\n\n    for item in lst: \n        paths = path(df,item,group_type)[:5]\n        # Add data to the table row-wise\n        data_table.add_data(item,\n                                wandb.Image(paths[0]),\n                                wandb.Image(paths[1]),\n                                wandb.Image(paths[2]),\n                                wandb.Image(paths[3]),\n                                wandb.Image(paths[4]))\n\n    # Log the table\n    wandb.log({table_name: data_table})\n\n    # Finish the run\n    wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T07:20:19.593698Z","iopub.execute_input":"2022-02-18T07:20:19.594507Z","iopub.status.idle":"2022-02-18T07:20:19.604141Z","shell.execute_reply.started":"2022-02-18T07:20:19.594465Z","shell.execute_reply":"2022-02-18T07:20:19.603109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Most Frequent Species 🐋 🐬","metadata":{}},{"cell_type":"code","source":"display_groups(train_df,'sp', m_freq_species)\nmap_species(train_df,'sp', m_freq_species, \"Most Frequent Species\", \"most_freq_species\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T07:20:20.711464Z","iopub.execute_input":"2022-02-18T07:20:20.712391Z","iopub.status.idle":"2022-02-18T07:21:00.359760Z","shell.execute_reply.started":"2022-02-18T07:20:20.712349Z","shell.execute_reply":"2022-02-18T07:21:00.358782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://i.imgur.com/pOVECHZ.png)","metadata":{}},{"cell_type":"markdown","source":"# Least Frequent Species 🐋 🐬","metadata":{}},{"cell_type":"code","source":"display_groups(train_df,'sp', l_freq_species)\nmap_species(train_df,'sp', l_freq_species, \"Least Frequent Species\", \"least_freq_species\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T07:21:00.361594Z","iopub.execute_input":"2022-02-18T07:21:00.361893Z","iopub.status.idle":"2022-02-18T07:21:29.454999Z","shell.execute_reply.started":"2022-02-18T07:21:00.361855Z","shell.execute_reply":"2022-02-18T07:21:29.453988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://i.imgur.com/O7CqPbl.png)","metadata":{}},{"cell_type":"markdown","source":"# Most frequent whales & dolphins 🐋 🐬","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(1,2,figsize=(16,8))\n\nwhales = train_df[train_df['label']=='whale']\ndolphins = train_df[train_df['label']!='whale']\nwhales = whales.rename(columns = {\"species\":\"species_whales\"})\ndolphins = dolphins.rename(columns = {\"species\":\"species_dolphins\"})\n\nsns.countplot(y=\"species_whales\", data=whales, order=whales.iloc[0:][\"species_whales\"].value_counts().index, ax=ax[0], color = \"#0077b6\")\nax[0].set_title('Most frequent whales')\nax[0].set_ylabel(None)\n    \nsns.countplot(y=\"species_dolphins\", data=dolphins,order=dolphins.iloc[0:][\"species_dolphins\"].value_counts().index, ax=ax[1], color = \"#90e0ef\")\nax[1].set_title('Most frequent dolphins')\nax[1].set_ylabel(None)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:49:21.427112Z","iopub.execute_input":"2022-02-18T06:49:21.427410Z","iopub.status.idle":"2022-02-18T06:49:22.048257Z","shell.execute_reply.started":"2022-02-18T06:49:21.427374Z","shell.execute_reply":"2022-02-18T06:49:22.047267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_wb_bar(count_values(whales,\"species_whales\", top=True),\"species_whales\", 'counts', \"Most frequent whales\")\nplot_wb_bar(count_values(dolphins,\"species_dolphins\", top=True),\"species_dolphins\", 'counts', \"Most frequent dolphins\")","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:49:22.049667Z","iopub.execute_input":"2022-02-18T06:49:22.049988Z","iopub.status.idle":"2022-02-18T06:49:49.651779Z","shell.execute_reply.started":"2022-02-18T06:49:22.049939Z","shell.execute_reply":"2022-02-18T06:49:49.650741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Most frequent whales 🐋","metadata":{}},{"cell_type":"code","source":"m_freq_species_whales = frequency(whales,\"species_whales\", \"Most\")\nwhales = whales.rename(columns = {\"species_whales\":\"species\"})\ndisplay_groups(whales,'sp', m_freq_species_whales)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:49:49.653409Z","iopub.execute_input":"2022-02-18T06:49:49.653664Z","iopub.status.idle":"2022-02-18T06:50:13.481203Z","shell.execute_reply.started":"2022-02-18T06:49:49.653634Z","shell.execute_reply":"2022-02-18T06:50:13.480177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Most frequent dolphins 🐬","metadata":{}},{"cell_type":"code","source":"m_freq_species_dolphins = frequency(dolphins,\"species_dolphins\", \"Most\")\ndolphins = dolphins.rename(columns = {\"species_dolphins\":\"species\"})\ndisplay_groups(dolphins,'sp', m_freq_species_dolphins)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:50:13.482386Z","iopub.execute_input":"2022-02-18T06:50:13.482605Z","iopub.status.idle":"2022-02-18T06:50:34.537627Z","shell.execute_reply.started":"2022-02-18T06:50:13.482579Z","shell.execute_reply":"2022-02-18T06:50:34.536827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Most Frequent Individual IDs 🐋 🐬","metadata":{}},{"cell_type":"code","source":"display_groups(train_df,'id', m_freq_ID)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T06:50:34.539163Z","iopub.execute_input":"2022-02-18T06:50:34.539609Z","iopub.status.idle":"2022-02-18T06:51:20.948373Z","shell.execute_reply.started":"2022-02-18T06:50:34.539569Z","shell.execute_reply":"2022-02-18T06:51:20.947113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Augmentation ➕","metadata":{}},{"cell_type":"code","source":"def plot_augmentations(images, titles, sup_title):\n    fig, axes = plt.subplots(figsize=(20, 16), nrows=3, ncols=4, squeeze=False)\n    \n    for indx, (img, title) in enumerate(zip(images, titles)):\n        axes[indx // 4][indx % 4].imshow(img)\n        axes[indx // 4][indx % 4].set_title(title, fontsize=15)\n        \n    plt.tight_layout()\n    fig.suptitle(sup_title, fontsize = 20)\n    fig.subplots_adjust(wspace=0.2, hspace=0.2, top=0.93)\n    axes[2,2].set_visible(False)\n    axes[2,3].set_visible(False)\n    plt.show()\n    \ndef augment(paths, data):\n    \n    # list of albumentations\n    albumentations = [A.RandomSunFlare(p=0.02), A.RandomFog(p=1), A.RandomBrightness(p=1),\n                              A.Rotate(p=1, limit=90),\n                              A.RGBShift(p=1), A.RandomSnow(p=0.02),\n                              A.HorizontalFlip(p=1), A.RandomContrast(limit = 0.5,p = 1),\n                              A.HueSaturationValue(p=1,hue_shift_limit=20, sat_shift_limit=30, val_shift_limit=50)]\n    \n    # image titles\n    titles = [\"RandomSunFlare\",\"RandomFog\",\"RandomBrightnessContrast\",\n                       \"Rotate\", \"RGBShift\", \"RandomSnow\",\"HorizontalFlip\", \"RandomContrast\",\"HSV\"]\n    \n    for i in paths:\n        image_path = i\n        \n        # getting image name from path\n        image_name = image_path.split(\"/\")[4].split(\".\")[0]\n        \n        # reading image\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) \n        \n        # resizing the image\n        image = cv2.resize(image, (224, 224))\n        \n        # list of images\n        images = []\n        \n        # creating image augmentations\n        for augmentation_type in albumentations:\n            augmented_img = augmentation_type(image = image)['image']\n            images.append(augmented_img)\n\n        # original image\n        titles.insert(0, \"Original\")\n        images.insert(0,image)  \n        \n        sup_title = \"Image Augmentation for \" + data + \" - \" + image_name\n        plot_augmentations(images, titles, sup_title)\n        \n        titles.remove(\"Original\")\n        \naugment(train_images_path[0:2],'train')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-18T06:51:20.950483Z","iopub.execute_input":"2022-02-18T06:51:20.950886Z","iopub.status.idle":"2022-02-18T06:51:26.204425Z","shell.execute_reply.started":"2022-02-18T06:51:20.950836Z","shell.execute_reply":"2022-02-18T06:51:26.203568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is what my [project](https://wandb.ai/ruchi798/happywhale?workspace=user-ruchi798) looks like on the W&B dashboard ⬇️\n\n![](https://i.imgur.com/CzsCPux.png)","metadata":{}}]}