{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-26T05:22:12.687899Z","iopub.execute_input":"2023-08-26T05:22:12.688284Z","iopub.status.idle":"2023-08-26T05:22:12.69495Z","shell.execute_reply.started":"2023-08-26T05:22:12.688252Z","shell.execute_reply":"2023-08-26T05:22:12.693958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Read through the following notebooks and links. Thanks all for their briliant ideas.\n\nhttps://www.kaggle.com/c/hpa-single-cell-image-classification/discussion/221550 \n\nhttps://www.kaggle.com/code/dragonzhang/fastai-cell-tile-prototyping-training/notebook  \n\nhttps://www.kaggle.com/code/p4rallax/using-albumentations-with-fast-ai-datablock-api  ","metadata":{}},{"cell_type":"markdown","source":"## Use fastai framework\n## Load datasets, build dataloaders\n## Use load_from_df\n## Integrate albumentation. Use seamless augmentation. ","metadata":{}},{"cell_type":"code","source":"!pip install fastai -Uqq ","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:22:12.696837Z","iopub.execute_input":"2023-08-26T05:22:12.697177Z","iopub.status.idle":"2023-08-26T05:22:25.454672Z","shell.execute_reply.started":"2023-08-26T05:22:12.697146Z","shell.execute_reply":"2023-08-26T05:22:25.45304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport PIL.Image as Image\nfrom fastai.data.all import *\nfrom fastai.vision.all import *\nfrom fastai.tabular.all import *\n\nimport albumentations as A\n\nimport warnings\n# Ignore warning messages\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:22:25.45856Z","iopub.execute_input":"2023-08-26T05:22:25.469791Z","iopub.status.idle":"2023-08-26T05:22:32.622524Z","shell.execute_reply.started":"2023-08-26T05:22:25.469743Z","shell.execute_reply":"2023-08-26T05:22:32.621534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_source = Path('/kaggle/input/hpa-single-cell-image-classification')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:22:32.623942Z","iopub.execute_input":"2023-08-26T05:22:32.624305Z","iopub.status.idle":"2023-08-26T05:22:32.630205Z","shell.execute_reply.started":"2023-08-26T05:22:32.624272Z","shell.execute_reply":"2023-08-26T05:22:32.628665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build an image file meta data frame. \n%%time\nmeta_df = pd.DataFrame({\n    'image':(\n        *get_image_files(protein_source),\n    )\n})\n\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:22:32.633085Z","iopub.execute_input":"2023-08-26T05:22:32.634187Z","iopub.status.idle":"2023-08-26T05:24:18.823296Z","shell.execute_reply.started":"2023-08-26T05:22:32.634155Z","shell.execute_reply":"2023-08-26T05:24:18.822363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the parent folder name of each image path\nimage_paths = meta_df[\"image\"].tolist()\nparent_folder_names = []\nfor image_path in image_paths:\n    parent_folder_name = os.path.dirname(image_path)\n    label_value = parent_folder_name.split(\"/\")[-1]\n    parent_folder_names.append(label_value)\n\n# Add the \"trn_tst\" column to the DataFrame\nmeta_df[\"trn_tst\"] = parent_folder_names\n\n# check for labels\nmeta_df['trn_tst'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:18.824618Z","iopub.execute_input":"2023-08-26T05:24:18.825736Z","iopub.status.idle":"2023-08-26T05:24:19.306263Z","shell.execute_reply.started":"2023-08-26T05:24:18.825702Z","shell.execute_reply":"2023-08-26T05:24:19.305308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get image id","metadata":{}},{"cell_type":"code","source":"def get_image_id(image_file): return str(image_file).split('_')[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.307662Z","iopub.execute_input":"2023-08-26T05:24:19.308638Z","iopub.status.idle":"2023-08-26T05:24:19.313912Z","shell.execute_reply.started":"2023-08-26T05:24:19.308604Z","shell.execute_reply":"2023-08-26T05:24:19.312773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmeta_df['image_id'] = meta_df['image'].apply(get_image_id)\n\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.315429Z","iopub.execute_input":"2023-08-26T05:24:19.315817Z","iopub.status.idle":"2023-08-26T05:24:19.412453Z","shell.execute_reply.started":"2023-08-26T05:24:19.315783Z","shell.execute_reply":"2023-08-26T05:24:19.411401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for unique image id in train dataset\nmeta_df[meta_df['trn_tst'] == 'train']['image_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.413874Z","iopub.execute_input":"2023-08-26T05:24:19.4143Z","iopub.status.idle":"2023-08-26T05:24:19.480471Z","shell.execute_reply.started":"2023-08-26T05:24:19.414267Z","shell.execute_reply":"2023-08-26T05:24:19.479544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for unique image id in train dataset\nmeta_df[meta_df['trn_tst'] == 'test']['image_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.48199Z","iopub.execute_input":"2023-08-26T05:24:19.482366Z","iopub.status.idle":"2023-08-26T05:24:19.50464Z","shell.execute_reply.started":"2023-08-26T05:24:19.482333Z","shell.execute_reply":"2023-08-26T05:24:19.503332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(meta_df) / 4  #all images have R G B Yellow .png files","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.5097Z","iopub.execute_input":"2023-08-26T05:24:19.510006Z","iopub.status.idle":"2023-08-26T05:24:19.516309Z","shell.execute_reply.started":"2023-08-26T05:24:19.509981Z","shell.execute_reply":"2023-08-26T05:24:19.515424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce to a small set.\n\nmeta_df = meta_df[meta_df['trn_tst'] == 'train']\nmeta_small = meta_df.iloc[:400, :].copy()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.517873Z","iopub.execute_input":"2023-08-26T05:24:19.518519Z","iopub.status.idle":"2023-08-26T05:24:19.55431Z","shell.execute_reply.started":"2023-08-26T05:24:19.51847Z","shell.execute_reply":"2023-08-26T05:24:19.553231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_shape(image_file):\n    image = Image.open(image_file)\n    return image.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.556628Z","iopub.execute_input":"2023-08-26T05:24:19.556908Z","iopub.status.idle":"2023-08-26T05:24:19.563626Z","shell.execute_reply.started":"2023-08-26T05:24:19.556885Z","shell.execute_reply":"2023-08-26T05:24:19.56254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmeta_small['shape'] = meta_small['image'].apply(get_shape)\nmeta_small.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:19.566006Z","iopub.execute_input":"2023-08-26T05:24:19.566643Z","iopub.status.idle":"2023-08-26T05:24:28.308723Z","shell.execute_reply.started":"2023-08-26T05:24:19.566611Z","shell.execute_reply":"2023-08-26T05:24:28.307769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stack R, B, G, Yellow  \n","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/datasets/philculliton/hpa-challenge-2021-extra-train-images/code  ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\n\nfrom skimage.io import imsave, imread\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:28.310189Z","iopub.execute_input":"2023-08-26T05:24:28.310804Z","iopub.status.idle":"2023-08-26T05:24:28.746054Z","shell.execute_reply.started":"2023-08-26T05:24:28.310768Z","shell.execute_reply":"2023-08-26T05:24:28.745036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_small.iloc[0, 2].split('/')[-1]  #demo","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:28.747429Z","iopub.execute_input":"2023-08-26T05:24:28.747816Z","iopub.status.idle":"2023-08-26T05:24:28.758722Z","shell.execute_reply.started":"2023-08-26T05:24:28.74778Z","shell.execute_reply":"2023-08-26T05:24:28.757755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Looking for way to put item transformation into GPU","metadata":{}},{"cell_type":"code","source":"%%time\nSZ = 2048\ni=0\nfor iid in tqdm.tqdm(meta_small['image_id']):\n    r = cv2.imread(f'{iid}_red.png', 0)\n    g = cv2.imread(f'{iid}_green.png', 0)\n    b = cv2.imread(f'{iid}_blue.png', 0)\n    a = cv2.imread(f'{iid}_yellow.png', 0)\n    if not r.shape[0] == SZ:\n        r = cv2.resize(r, (SZ, SZ))\n        g = cv2.resize(g, (SZ, SZ))\n        b = cv2.resize(b, (SZ, SZ))\n        a = cv2.resize(a, (SZ, SZ))\n    img = np.stack([r, g, b, a], -1)\n    fname = meta_small.iloc[i, 2].split('/')[-1]\n    imsave(f'./{fname}_.png', img)\n    i += 1","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:24:28.760208Z","iopub.execute_input":"2023-08-26T05:24:28.760725Z","iopub.status.idle":"2023-08-26T05:38:22.534524Z","shell.execute_reply.started":"2023-08-26T05:24:28.760693Z","shell.execute_reply":"2023-08-26T05:38:22.533562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build an image file meta data frame. \nmeta_sml_wk = pd.DataFrame({\n    'image':(\n        *get_image_files('/kaggle/working/'),\n    )\n})\n\nmeta_sml_wk.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.536058Z","iopub.execute_input":"2023-08-26T05:38:22.53667Z","iopub.status.idle":"2023-08-26T05:38:22.553756Z","shell.execute_reply.started":"2023-08-26T05:38:22.536634Z","shell.execute_reply":"2023-08-26T05:38:22.552527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_id_wk(image_file): \n    tmp = str(image_file).split('_')[0]\n    return tmp.split('/')[-1]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.554946Z","iopub.execute_input":"2023-08-26T05:38:22.55574Z","iopub.status.idle":"2023-08-26T05:38:22.561034Z","shell.execute_reply.started":"2023-08-26T05:38:22.555706Z","shell.execute_reply":"2023-08-26T05:38:22.56007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_sml_wk['ID'] = meta_sml_wk['image'].apply(get_image_id_wk)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.562366Z","iopub.execute_input":"2023-08-26T05:38:22.562791Z","iopub.status.idle":"2023-08-26T05:38:22.575155Z","shell.execute_reply.started":"2023-08-26T05:38:22.562758Z","shell.execute_reply":"2023-08-26T05:38:22.574152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/hpa-single-cell-image-classification/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.576577Z","iopub.execute_input":"2023-08-26T05:38:22.576929Z","iopub.status.idle":"2023-08-26T05:38:22.631408Z","shell.execute_reply.started":"2023-08-26T05:38:22.576898Z","shell.execute_reply":"2023-08-26T05:38:22.630514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge = pd.merge(meta_sml_wk, df_train, left_on='ID', right_on='ID', how='outer')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.632806Z","iopub.execute_input":"2023-08-26T05:38:22.633128Z","iopub.status.idle":"2023-08-26T05:38:22.655846Z","shell.execute_reply.started":"2023-08-26T05:38:22.633097Z","shell.execute_reply":"2023-08-26T05:38:22.654783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge = df_merge.dropna()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.657425Z","iopub.execute_input":"2023-08-26T05:38:22.658084Z","iopub.status.idle":"2023-08-26T05:38:22.68309Z","shell.execute_reply.started":"2023-08-26T05:38:22.65805Z","shell.execute_reply":"2023-08-26T05:38:22.682237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process label. Convert integer back to a string","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/lnhtrang/hpa-public-data-download-and-hpacellseg/notebook  ","metadata":{}},{"cell_type":"code","source":"# All label names in the public HPA and their corresponding index. Reduce the number of class into 19.\nall_labels = {\n    \"Nucleoplasm\": 0,\n    \"Nuclear membrane\": 1,\n    \"Nucleoli\": 2,\n    \"Nucleoli fibrillar center\": 3,\n    \"Nuclear speckles\": 4,\n    \"Nuclear bodies\": 5,\n    \"Endoplasmic reticulum\": 6,\n    \"Golgi apparatus\": 7,\n    \"Intermediate filaments\": 8,\n#     \"Actin filaments\": 9,\n    \"Focal adhesion sites\": 9,\n    \"Microtubules\": 10,\n    \"Mitotic spindle\": 11,\n    \"Centrosome\": 12,\n#     \"Centriolar satellite\": 12,\n    \"Plasma membrane\": 13,\n#     \"Cell Junctions\": 13,\n    \"Mitochondria\": 14,\n    \"Aggresome\": 15,\n    \"Cytosol\": 16,\n    \"Vesicles\": 17,\n#     \"Peroxisomes\": 17,\n#     \"Endosomes\": 17,\n#     \"Lysosomes\": 17,\n#     \"Lipid droplets\": 17,\n#     \"Cytoplasmic bodies\": 17,\n    \"No staining\": 18\n}","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.684503Z","iopub.execute_input":"2023-08-26T05:38:22.684872Z","iopub.status.idle":"2023-08-26T05:38:22.691355Z","shell.execute_reply.started":"2023-08-26T05:38:22.68484Z","shell.execute_reply":"2023-08-26T05:38:22.690206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_labels.items()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.692834Z","iopub.execute_input":"2023-08-26T05:38:22.69317Z","iopub.status.idle":"2023-08-26T05:38:22.708946Z","shell.execute_reply.started":"2023-08-26T05:38:22.69314Z","shell.execute_reply":"2023-08-26T05:38:22.707499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge['Label']","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.711694Z","iopub.execute_input":"2023-08-26T05:38:22.712418Z","iopub.status.idle":"2023-08-26T05:38:22.722657Z","shell.execute_reply.started":"2023-08-26T05:38:22.712386Z","shell.execute_reply":"2023-08-26T05:38:22.721667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def str2list(df_merge):\n    for i in range(len(df_merge)):\n        label = df_merge.loc[i, 'Label']\n        if '|' in label:\n            df_merge.loc[i, 'Label'] = label.split('|')\n        else:\n            df_merge.loc[i, 'Label'] = label\n    return df_merge\n\ndf_merge = str2list(df_merge.copy())\nprint(df_merge['Label'].head())","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.723803Z","iopub.execute_input":"2023-08-26T05:38:22.724574Z","iopub.status.idle":"2023-08-26T05:38:22.784995Z","shell.execute_reply.started":"2023-08-26T05:38:22.724542Z","shell.execute_reply":"2023-08-26T05:38:22.783732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge['Label'][0]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.787014Z","iopub.execute_input":"2023-08-26T05:38:22.787571Z","iopub.status.idle":"2023-08-26T05:38:22.796434Z","shell.execute_reply.started":"2023-08-26T05:38:22.787536Z","shell.execute_reply":"2023-08-26T05:38:22.795499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Concat list to one string to use fastai datablocks\ndf_list = []\nfor i in range(len(df_merge)):\n  row_list = []\n  for point_elem in df_merge['Label'][i]:\n    print('index is', i)\n    value = [i for i in all_labels if all_labels[i] == int(point_elem)]\n    print('value is', value)\n    value = str(value[0])\n    row_list.append(value)\n\n  df_list.append(row_list)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.803059Z","iopub.execute_input":"2023-08-26T05:38:22.803328Z","iopub.status.idle":"2023-08-26T05:38:22.83714Z","shell.execute_reply.started":"2023-08-26T05:38:22.803304Z","shell.execute_reply":"2023-08-26T05:38:22.836179Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(df_merge)  == len(df_list)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.838574Z","iopub.execute_input":"2023-08-26T05:38:22.838926Z","iopub.status.idle":"2023-08-26T05:38:22.843609Z","shell.execute_reply.started":"2023-08-26T05:38:22.838894Z","shell.execute_reply":"2023-08-26T05:38:22.842426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge['label_str'] = df_list","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.845203Z","iopub.execute_input":"2023-08-26T05:38:22.845554Z","iopub.status.idle":"2023-08-26T05:38:22.855818Z","shell.execute_reply.started":"2023-08-26T05:38:22.845523Z","shell.execute_reply":"2023-08-26T05:38:22.854927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.858423Z","iopub.execute_input":"2023-08-26T05:38:22.858744Z","iopub.status.idle":"2023-08-26T05:38:22.877763Z","shell.execute_reply.started":"2023-08-26T05:38:22.85872Z","shell.execute_reply":"2023-08-26T05:38:22.876775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge = df_merge.drop(['ID','Label'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.879225Z","iopub.execute_input":"2023-08-26T05:38:22.879931Z","iopub.status.idle":"2023-08-26T05:38:22.889937Z","shell.execute_reply.started":"2023-08-26T05:38:22.8799Z","shell.execute_reply":"2023-08-26T05:38:22.888871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge['label_str'] = df_merge['label_str'].apply(lambda x: ','.join(x) )","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.891396Z","iopub.execute_input":"2023-08-26T05:38:22.892012Z","iopub.status.idle":"2023-08-26T05:38:22.901438Z","shell.execute_reply.started":"2023-08-26T05:38:22.891981Z","shell.execute_reply":"2023-08-26T05:38:22.900621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merge.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.902986Z","iopub.execute_input":"2023-08-26T05:38:22.903755Z","iopub.status.idle":"2023-08-26T05:38:22.919883Z","shell.execute_reply.started":"2023-08-26T05:38:22.903626Z","shell.execute_reply":"2023-08-26T05:38:22.919012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Use albumentation","metadata":{}},{"cell_type":"code","source":"class AlbumentationsTransform(DisplayedTransform):\n    split_idx,order=0,2\n    def __init__(self, train_aug): store_attr()\n    \n    def encodes(self, img: PILImage):\n        aug_img = self.train_aug(image=np.array(img))['image']\n        return PILImage.create(aug_img)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.922359Z","iopub.execute_input":"2023-08-26T05:38:22.922998Z","iopub.status.idle":"2023-08-26T05:38:22.931403Z","shell.execute_reply.started":"2023-08-26T05:38:22.922967Z","shell.execute_reply":"2023-08-26T05:38:22.930403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Integrate albumentation. Use seamless augmentation","metadata":{}},{"cell_type":"code","source":"def get_train_aug(): return A.Compose([\n            A.CoarseDropout(p=0.5),\n            A.RandomContrast(p = 0.6),\n            A.ElasticTransform(p=1, alpha=120, sigma=120 * 0.05, alpha_affine=120 * 0.03),\n            A.RandomRotate90(p=1),\n    \n])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.93296Z","iopub.execute_input":"2023-08-26T05:38:22.933664Z","iopub.status.idle":"2023-08-26T05:38:22.94305Z","shell.execute_reply.started":"2023-08-26T05:38:22.93363Z","shell.execute_reply":"2023-08-26T05:38:22.942153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_tfms = [RandomResizedCrop(224, min_scale=0.75, ratio=(1.,1.)),\n             AlbumentationsTransform(get_train_aug())]\nbatch_tfms = [*aug_transforms(flip_vert=True, size=128, max_warp=0),  \n              Normalize.from_stats(*imagenet_stats)]\nbs=16","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:22.945167Z","iopub.execute_input":"2023-08-26T05:38:22.94675Z","iopub.status.idle":"2023-08-26T05:38:26.023167Z","shell.execute_reply.started":"2023-08-26T05:38:22.946725Z","shell.execute_reply":"2023-08-26T05:38:26.022174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dblock = DataBlock(\n                blocks = (ImageBlock, MultiCategoryBlock),\n                splitter = RandomSplitter(),\n                get_x = ColReader('image'),\n                get_y = ColReader('label_str', label_delim=','),\n                item_tfms = item_tfms,\n                batch_tfms = batch_tfms\n                )\ndls = dblock.dataloaders(df_merge, bs=bs)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:26.024468Z","iopub.execute_input":"2023-08-26T05:38:26.024875Z","iopub.status.idle":"2023-08-26T05:38:28.373677Z","shell.execute_reply.started":"2023-08-26T05:38:26.024842Z","shell.execute_reply":"2023-08-26T05:38:28.372655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch(nrows=3, ncols=3)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:38:28.375193Z","iopub.execute_input":"2023-08-26T05:38:28.375584Z","iopub.status.idle":"2023-08-26T05:38:32.975878Z","shell.execute_reply.started":"2023-08-26T05:38:28.375546Z","shell.execute_reply":"2023-08-26T05:38:32.974969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dls.vocab) #Seems missing one of classes","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:39:53.041611Z","iopub.execute_input":"2023-08-26T05:39:53.042Z","iopub.status.idle":"2023-08-26T05:39:53.048986Z","shell.execute_reply.started":"2023-08-26T05:39:53.04197Z","shell.execute_reply":"2023-08-26T05:39:53.048009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add training loop.","metadata":{}},{"cell_type":"markdown","source":"https://docs.fast.ai/tutorial.vision.html#multi-label-classification ","metadata":{}},{"cell_type":"code","source":"f1_macro = F1ScoreMulti(thresh=0.5, average='macro')\nf1_macro.name = 'F1(macro)'\nf1_samples = F1ScoreMulti(thresh=0.5, average='samples')\nf1_samples.name = 'F1(samples)'\n\nlearn = vision_learner(dls, resnet50, metrics=[partial(accuracy_multi, thresh=0.5), f1_macro, f1_samples])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:45:35.997926Z","iopub.execute_input":"2023-08-26T05:45:35.998315Z","iopub.status.idle":"2023-08-26T05:45:39.323586Z","shell.execute_reply.started":"2023-08-26T05:45:35.998283Z","shell.execute_reply":"2023-08-26T05:45:39.322535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:45:46.90943Z","iopub.execute_input":"2023-08-26T05:45:46.910476Z","iopub.status.idle":"2023-08-26T05:49:32.771415Z","shell.execute_reply.started":"2023-08-26T05:45:46.910431Z","shell.execute_reply":"2023-08-26T05:49:32.770429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(2, 1e-3)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T05:50:29.074861Z","iopub.execute_input":"2023-08-26T05:50:29.075245Z","iopub.status.idle":"2023-08-26T05:53:21.036941Z","shell.execute_reply.started":"2023-08-26T05:50:29.075216Z","shell.execute_reply":"2023-08-26T05:53:21.035796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(12, slice(1e-3, 1e-2))  #Training for more than 20 epochs","metadata":{"execution":{"iopub.status.busy":"2023-08-26T06:15:11.948006Z","iopub.execute_input":"2023-08-26T06:15:11.948472Z","iopub.status.idle":"2023-08-26T06:26:34.738608Z","shell.execute_reply.started":"2023-08-26T06:15:11.948436Z","shell.execute_reply":"2023-08-26T06:26:34.737443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.show_results()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T06:26:34.741423Z","iopub.execute_input":"2023-08-26T06:26:34.741831Z","iopub.status.idle":"2023-08-26T06:26:38.965084Z","shell.execute_reply.started":"2023-08-26T06:26:34.741795Z","shell.execute_reply":"2023-08-26T06:26:38.96358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = Interpretation.from_learner(learn)\ninterp.plot_top_losses(9)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T06:26:38.96666Z","iopub.execute_input":"2023-08-26T06:26:38.967452Z","iopub.status.idle":"2023-08-26T06:26:54.137022Z","shell.execute_reply.started":"2023-08-26T06:26:38.967416Z","shell.execute_reply":"2023-08-26T06:26:54.132818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Move the entire pipe line to GCP","metadata":{}}]}