{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## SPR X-Ray Age Prediction Challeng","metadata":{"papermill":{"duration":10.118795,"end_time":"2023-03-19T11:01:28.219781","exception":false,"start_time":"2023-03-19T11:01:18.100986","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-26T07:47:40.061814Z","iopub.execute_input":"2023-03-26T07:47:40.062616Z","iopub.status.idle":"2023-03-26T07:47:50.335334Z","shell.execute_reply.started":"2023-03-26T07:47:40.062563Z","shell.execute_reply":"2023-03-26T07:47:50.334068Z"}}},{"cell_type":"markdown","source":"#### The objective of this work is to create an AI model capable of predicting the patient's age through chest X-rays.","metadata":{}},{"cell_type":"code","source":"# Imports\nimport io\nimport os\nimport cv2\nimport numpy as np\nimport random\nimport imageio\nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport keras\nimport tensorflow\nfrom glob import glob\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# Imports for image manipulation\nimport os\nimport cv2\nimport itertools\nimport shutil\nimport imageio\nimport skimage\nimport skimage.io\nimport imgaug.augmenters as iaa\nimport skimage.transform\nfrom pathlib import Path\n\n# Imports for Deep Learning\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D\nfrom keras import optimizers\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, Callback, ReduceLROnPlateau\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,img_to_array, array_to_img, load_img\nfrom tensorflow.keras.metrics import categorical_crossentropy\nfrom tensorflow.keras.metrics import binary_accuracy\nfrom tensorflow.keras.applications.resnet import ResNet152\nfrom tensorflow.keras.applications.efficientnet_v2 import EfficientNetV2L\nfrom tensorflow.keras import layers\n\n# Imports for calculating metrics and other tasks\nimport sklearn\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:08:03.033119Z","iopub.execute_input":"2023-04-23T21:08:03.033857Z","iopub.status.idle":"2023-04-23T21:08:13.745904Z","shell.execute_reply.started":"2023-04-23T21:08:03.033825Z","shell.execute_reply":"2023-04-23T21:08:13.744828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Images and Labels","metadata":{"papermill":{"duration":0.014618,"end_time":"2023-03-19T11:01:28.250187","exception":false,"start_time":"2023-03-19T11:01:28.235569","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"To reinforce the learning process, I will also use the public Random Sample NIH Chest X-ray dataset from the Kaggle repository.","metadata":{}},{"cell_type":"code","source":"# Path to csv file with SPR X-Ray Age labels\npath_dataset_1 = \"/kaggle/input/spr-x-ray-age\"\n# Path to csv file with NIH Chest X-ray Dataset Sample labels\npath_dataset_2= \"/kaggle/input/sample\"","metadata":{"papermill":{"duration":0.022845,"end_time":"2023-03-19T11:01:28.289155","exception":false,"start_time":"2023-03-19T11:01:28.266310","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:20.097839Z","iopub.execute_input":"2023-04-23T21:08:20.098553Z","iopub.status.idle":"2023-04-23T21:08:20.104335Z","shell.execute_reply.started":"2023-04-23T21:08:20.098514Z","shell.execute_reply":"2023-04-23T21:08:20.103137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating variable to identify the path of training and test images\npath_img_train_1 = '/kaggle/input/spr-x-ray-age/kaggle/kaggle/train'\npath_img_train_2 = '/kaggle/input/sample/sample/images'\npath_img_test  = '/kaggle/input/spr-x-ray-age/kaggle/kaggle/test'","metadata":{"papermill":{"duration":0.022235,"end_time":"2023-03-19T11:01:28.325827","exception":false,"start_time":"2023-03-19T11:01:28.303592","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:24.197436Z","iopub.execute_input":"2023-04-23T21:08:24.197836Z","iopub.status.idle":"2023-04-23T21:08:24.203371Z","shell.execute_reply.started":"2023-04-23T21:08:24.197802Z","shell.execute_reply":"2023-04-23T21:08:24.202152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading SPR X-Ray Age labels for training and testing\ndf_csv_train_1 = pd.read_csv(path_dataset_1 + '/train_age.csv', dtype = str)","metadata":{"papermill":{"duration":0.04116,"end_time":"2023-03-19T11:01:28.381245","exception":false,"start_time":"2023-03-19T11:01:28.340085","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:27.651982Z","iopub.execute_input":"2023-04-23T21:08:27.652649Z","iopub.status.idle":"2023-04-23T21:08:27.683480Z","shell.execute_reply.started":"2023-04-23T21:08:27.652595Z","shell.execute_reply":"2023-04-23T21:08:27.682411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading NIH Chest X-ray Dataset Sample labels for training and testing\ndf_csv_train_2 = pd.read_csv(path_dataset_2 + '/sample_labels.csv', dtype = str)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:08:30.853216Z","iopub.execute_input":"2023-04-23T21:08:30.853733Z","iopub.status.idle":"2023-04-23T21:08:30.891477Z","shell.execute_reply.started":"2023-04-23T21:08:30.853686Z","shell.execute_reply":"2023-04-23T21:08:30.890381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# viewing the data\ndf_csv_train_1.info()\ndf_csv_train_1.head()","metadata":{"papermill":{"duration":0.056894,"end_time":"2023-03-19T11:01:28.452937","exception":false,"start_time":"2023-03-19T11:01:28.396043","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:33.787603Z","iopub.execute_input":"2023-04-23T21:08:33.788647Z","iopub.status.idle":"2023-04-23T21:08:33.821320Z","shell.execute_reply.started":"2023-04-23T21:08:33.788596Z","shell.execute_reply":"2023-04-23T21:08:33.820171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# viewing the data\ndf_csv_train_2.info()\ndf_csv_train_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:08:37.306607Z","iopub.execute_input":"2023-04-23T21:08:37.307612Z","iopub.status.idle":"2023-04-23T21:08:37.335465Z","shell.execute_reply.started":"2023-04-23T21:08:37.307571Z","shell.execute_reply":"2023-04-23T21:08:37.334385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Understanding the Dataset","metadata":{"papermill":{"duration":0.014627,"end_time":"2023-03-19T11:01:28.483809","exception":false,"start_time":"2023-03-19T11:01:28.469182","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Shape labels training dataset SPR X-Ray Age\ndf_csv_train_1.shape","metadata":{"papermill":{"duration":0.023747,"end_time":"2023-03-19T11:01:28.522257","exception":false,"start_time":"2023-03-19T11:01:28.498510","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:42.096766Z","iopub.execute_input":"2023-04-23T21:08:42.097703Z","iopub.status.idle":"2023-04-23T21:08:42.106333Z","shell.execute_reply.started":"2023-04-23T21:08:42.097665Z","shell.execute_reply":"2023-04-23T21:08:42.104614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape labels training dataset Sample NIH Chest X-ray\ndf_csv_train_2.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:08:45.400534Z","iopub.execute_input":"2023-04-23T21:08:45.401222Z","iopub.status.idle":"2023-04-23T21:08:45.407623Z","shell.execute_reply.started":"2023-04-23T21:08:45.401184Z","shell.execute_reply":"2023-04-23T21:08:45.406456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading training image list SPR X-Ray Age\nimages_train_1 = os.listdir(path_img_train_1)\nimages_train_1.sort()","metadata":{"papermill":{"duration":0.398978,"end_time":"2023-03-19T11:01:28.935610","exception":false,"start_time":"2023-03-19T11:01:28.536632","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:48.642022Z","iopub.execute_input":"2023-04-23T21:08:48.642474Z","iopub.status.idle":"2023-04-23T21:08:48.988956Z","shell.execute_reply.started":"2023-04-23T21:08:48.642435Z","shell.execute_reply":"2023-04-23T21:08:48.987793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading training image list Sample NIH Chest X-ray\nimages_train_2 = os.listdir(path_img_train_2)\nimages_train_2.sort()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:08:52.447110Z","iopub.execute_input":"2023-04-23T21:08:52.447691Z","iopub.status.idle":"2023-04-23T21:08:52.936426Z","shell.execute_reply.started":"2023-04-23T21:08:52.447654Z","shell.execute_reply":"2023-04-23T21:08:52.935407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading test image list\nimages_test = os.listdir(path_img_test)\nimages_test.sort()","metadata":{"papermill":{"duration":0.357908,"end_time":"2023-03-19T11:01:29.309274","exception":false,"start_time":"2023-03-19T11:01:28.951366","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:55.626110Z","iopub.execute_input":"2023-04-23T21:08:55.627020Z","iopub.status.idle":"2023-04-23T21:08:56.241304Z","shell.execute_reply.started":"2023-04-23T21:08:55.626966Z","shell.execute_reply":"2023-04-23T21:08:56.240254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape training images SPR X-Ray Age \nlen(images_train_1)","metadata":{"papermill":{"duration":0.02622,"end_time":"2023-03-19T11:01:29.351768","exception":false,"start_time":"2023-03-19T11:01:29.325548","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:08:58.639200Z","iopub.execute_input":"2023-04-23T21:08:58.640104Z","iopub.status.idle":"2023-04-23T21:08:58.646341Z","shell.execute_reply.started":"2023-04-23T21:08:58.640067Z","shell.execute_reply":"2023-04-23T21:08:58.645180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape training images Sample NIH Chest X-ray\nlen(images_train_2)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:01.767096Z","iopub.execute_input":"2023-04-23T21:09:01.768046Z","iopub.status.idle":"2023-04-23T21:09:01.775274Z","shell.execute_reply.started":"2023-04-23T21:09:01.767993Z","shell.execute_reply":"2023-04-23T21:09:01.774192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape test images\nlen(images_test)","metadata":{"papermill":{"duration":0.026409,"end_time":"2023-03-19T11:01:29.393526","exception":false,"start_time":"2023-03-19T11:01:29.367117","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:06.955285Z","iopub.execute_input":"2023-04-23T21:09:06.955888Z","iopub.status.idle":"2023-04-23T21:09:06.962825Z","shell.execute_reply.started":"2023-04-23T21:09:06.955852Z","shell.execute_reply":"2023-04-23T21:09:06.961788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the training DataSet, the SPR X-Ray Age data contains 10702 labeled cases out of 10702 images and in Random Sample NIH Chest X-ray Dataset contains 5606 labeled cases out of 5606 images. This indicates that the data is complete.","metadata":{"papermill":{"duration":0.015119,"end_time":"2023-03-19T11:01:29.424554","exception":false,"start_time":"2023-03-19T11:01:29.409435","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Checking the ratio of labels per age\ndf_csv_train_1['age'].value_counts()","metadata":{"papermill":{"duration":0.028913,"end_time":"2023-03-19T11:01:29.469042","exception":false,"start_time":"2023-03-19T11:01:29.440129","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:10.513181Z","iopub.execute_input":"2023-04-23T21:09:10.513557Z","iopub.status.idle":"2023-04-23T21:09:10.523788Z","shell.execute_reply.started":"2023-04-23T21:09:10.513523Z","shell.execute_reply":"2023-04-23T21:09:10.522532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the ratio of labels per age\ndf_csv_train_2['Patient Age'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:13.875081Z","iopub.execute_input":"2023-04-23T21:09:13.876008Z","iopub.status.idle":"2023-04-23T21:09:13.888524Z","shell.execute_reply.started":"2023-04-23T21:09:13.875956Z","shell.execute_reply":"2023-04-23T21:09:13.887548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attribute Engineering","metadata":{"papermill":{"duration":0.015234,"end_time":"2023-03-19T11:01:29.499881","exception":false,"start_time":"2023-03-19T11:01:29.484647","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Excluir a linha com idade fora de contexto\ndf_csv_train_2 = df_csv_train_2.drop(df_csv_train_2[df_csv_train_2['Patient Age'] == '411Y'].index)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:17.373067Z","iopub.execute_input":"2023-04-23T21:09:17.373430Z","iopub.status.idle":"2023-04-23T21:09:17.381249Z","shell.execute_reply.started":"2023-04-23T21:09:17.373398Z","shell.execute_reply":"2023-04-23T21:09:17.380209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extrair apenas os números da coluna1 usando expressão regular\ndf_csv_train_2['age'] = df_csv_train_2['Patient Age'].str.extract(r'(\\d+)').astype(float)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:20.997257Z","iopub.execute_input":"2023-04-23T21:09:20.997639Z","iopub.status.idle":"2023-04-23T21:09:21.013427Z","shell.execute_reply.started":"2023-04-23T21:09:20.997605Z","shell.execute_reply":"2023-04-23T21:09:21.012373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the ratio of labels per class\ndf_csv_train_2['age'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:24.254307Z","iopub.execute_input":"2023-04-23T21:09:24.254668Z","iopub.status.idle":"2023-04-23T21:09:24.265426Z","shell.execute_reply.started":"2023-04-23T21:09:24.254636Z","shell.execute_reply":"2023-04-23T21:09:24.264407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataset with training images path\ndf_img_train_1 = pd.DataFrame(columns=[\"imageId\",\"img_name\",\"img_path\"])\ndf_img_train_1.set_index('imageId')","metadata":{"papermill":{"duration":0.028904,"end_time":"2023-03-19T11:01:29.543591","exception":false,"start_time":"2023-03-19T11:01:29.514687","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:27.921227Z","iopub.execute_input":"2023-04-23T21:09:27.921998Z","iopub.status.idle":"2023-04-23T21:09:27.935272Z","shell.execute_reply.started":"2023-04-23T21:09:27.921959Z","shell.execute_reply":"2023-04-23T21:09:27.934220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataset with training images path\ndf_img_train_2 = pd.DataFrame(columns=[\"imageId\",\"img_name\",\"img_path\"])\ndf_img_train_2.set_index('imageId')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:09:30.911542Z","iopub.execute_input":"2023-04-23T21:09:30.912227Z","iopub.status.idle":"2023-04-23T21:09:30.923554Z","shell.execute_reply.started":"2023-04-23T21:09:30.912188Z","shell.execute_reply":"2023-04-23T21:09:30.922322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataset with test images path\ndf_img_test = pd.DataFrame(columns=[\"imageId\",\"img_name\", \"img_path\",])\ndf_img_test.set_index('imageId')","metadata":{"papermill":{"duration":0.028325,"end_time":"2023-03-19T11:01:29.586400","exception":false,"start_time":"2023-03-19T11:01:29.558075","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:33.820306Z","iopub.execute_input":"2023-04-23T21:09:33.820670Z","iopub.status.idle":"2023-04-23T21:09:33.831703Z","shell.execute_reply.started":"2023-04-23T21:09:33.820639Z","shell.execute_reply":"2023-04-23T21:09:33.830690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tuning and Organizing the X-Ray Image Dataset","metadata":{}},{"cell_type":"code","source":"# Function to add information to metadata\ndef loads_df_img(df, images, path_img):\n    for img in tqdm(images):\n        # check if it is an image file\n        if img.endswith(\".png\"):\n            # extract label from file name\n            imageId = int(img.split(\".\")[0])\n            # add file path and id to dataframe\n            df = df.append({\"imageId\": imageId, \"img_name\": img, \"img_path\": os.path.join(path_img)}, ignore_index=True)\n    return df","metadata":{"papermill":{"duration":0.024784,"end_time":"2023-03-19T11:01:29.626555","exception":false,"start_time":"2023-03-19T11:01:29.601771","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:38.279908Z","iopub.execute_input":"2023-04-23T21:09:38.280628Z","iopub.status.idle":"2023-04-23T21:09:38.286791Z","shell.execute_reply.started":"2023-04-23T21:09:38.280589Z","shell.execute_reply":"2023-04-23T21:09:38.285701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appends information to the training DataFrame\ndf_img_train_1 = loads_df_img(df_img_train_1, images_train_1, path_img_train_1)","metadata":{"papermill":{"duration":24.216222,"end_time":"2023-03-19T11:01:53.858379","exception":false,"start_time":"2023-03-19T11:01:29.642157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:09:43.509878Z","iopub.execute_input":"2023-04-23T21:09:43.510457Z","iopub.status.idle":"2023-04-23T21:10:06.137744Z","shell.execute_reply.started":"2023-04-23T21:09:43.510419Z","shell.execute_reply":"2023-04-23T21:10:06.136674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appends information to the training DataFrame\ndf_img_train_2 = loads_df_img(df_img_train_2, images_train_2, path_img_train_2)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:11:07.915696Z","iopub.execute_input":"2023-04-23T21:11:07.916405Z","iopub.status.idle":"2023-04-23T21:11:19.387034Z","shell.execute_reply.started":"2023-04-23T21:11:07.916366Z","shell.execute_reply":"2023-04-23T21:11:19.385966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appends information to the test DataFrame\ndf_img_test = loads_df_img(df_img_test, images_test, path_img_test)","metadata":{"papermill":{"duration":26.276386,"end_time":"2023-03-19T11:02:20.150789","exception":false,"start_time":"2023-03-19T11:01:53.874403","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:11:29.263446Z","iopub.execute_input":"2023-04-23T21:11:29.264166Z","iopub.status.idle":"2023-04-23T21:11:53.582403Z","shell.execute_reply.started":"2023-04-23T21:11:29.264126Z","shell.execute_reply":"2023-04-23T21:11:53.581308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View training dataset\ndf_img_train_1.head()","metadata":{"papermill":{"duration":0.028689,"end_time":"2023-03-19T11:02:20.194673","exception":false,"start_time":"2023-03-19T11:02:20.165984","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:06.337687Z","iopub.execute_input":"2023-04-23T21:15:06.338752Z","iopub.status.idle":"2023-04-23T21:15:06.349030Z","shell.execute_reply.started":"2023-04-23T21:15:06.338692Z","shell.execute_reply":"2023-04-23T21:15:06.348004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View training dataset\ndf_img_train_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:15:10.221268Z","iopub.execute_input":"2023-04-23T21:15:10.222278Z","iopub.status.idle":"2023-04-23T21:15:10.234593Z","shell.execute_reply.started":"2023-04-23T21:15:10.222237Z","shell.execute_reply":"2023-04-23T21:15:10.233447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View test dataset\ndf_img_test.head()","metadata":{"papermill":{"duration":0.027367,"end_time":"2023-03-19T11:02:20.237976","exception":false,"start_time":"2023-03-19T11:02:20.210609","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:14.864574Z","iopub.execute_input":"2023-04-23T21:15:14.865260Z","iopub.status.idle":"2023-04-23T21:15:14.876065Z","shell.execute_reply.started":"2023-04-23T21:15:14.865221Z","shell.execute_reply":"2023-04-23T21:15:14.874670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataset with path of images associated with labels\ndf_img_labels_train_1 = pd.concat([df_img_train_1, df_csv_train_1['age']], axis=1, ignore_index=False)","metadata":{"papermill":{"duration":0.024238,"end_time":"2023-03-19T11:02:20.323135","exception":false,"start_time":"2023-03-19T11:02:20.298897","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:18.017404Z","iopub.execute_input":"2023-04-23T21:15:18.018090Z","iopub.status.idle":"2023-04-23T21:15:18.025137Z","shell.execute_reply.started":"2023-04-23T21:15:18.018051Z","shell.execute_reply":"2023-04-23T21:15:18.023932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataset with path of images associated with labels\ndf_img_labels_train_2 = pd.concat([df_img_train_2, df_csv_train_2['age']], axis=1, ignore_index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:15:20.977843Z","iopub.execute_input":"2023-04-23T21:15:20.978424Z","iopub.status.idle":"2023-04-23T21:15:20.991435Z","shell.execute_reply.started":"2023-04-23T21:15:20.978386Z","shell.execute_reply":"2023-04-23T21:15:20.990349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View new dataset\ndf_img_labels_train_1.head()","metadata":{"papermill":{"duration":0.02957,"end_time":"2023-03-19T11:02:20.368335","exception":false,"start_time":"2023-03-19T11:02:20.338765","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:24.116068Z","iopub.execute_input":"2023-04-23T21:15:24.117030Z","iopub.status.idle":"2023-04-23T21:15:24.129533Z","shell.execute_reply.started":"2023-04-23T21:15:24.116976Z","shell.execute_reply":"2023-04-23T21:15:24.128391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View new dataset\ndf_img_labels_train_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:15:27.493489Z","iopub.execute_input":"2023-04-23T21:15:27.494206Z","iopub.status.idle":"2023-04-23T21:15:27.506267Z","shell.execute_reply.started":"2023-04-23T21:15:27.494166Z","shell.execute_reply":"2023-04-23T21:15:27.505071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checks if there is a null value in the 'age' column\ndf_img_labels_train_1['age'].isnull().values.any()","metadata":{"papermill":{"duration":0.027552,"end_time":"2023-03-19T11:02:20.412479","exception":false,"start_time":"2023-03-19T11:02:20.384927","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:44.936954Z","iopub.execute_input":"2023-04-23T21:15:44.938031Z","iopub.status.idle":"2023-04-23T21:15:44.946976Z","shell.execute_reply.started":"2023-04-23T21:15:44.937985Z","shell.execute_reply":"2023-04-23T21:15:44.945951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checks if there is a null value in the 'age' column\ndf_img_labels_train_2['age'].isnull().values.any()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:15:48.110537Z","iopub.execute_input":"2023-04-23T21:15:48.111255Z","iopub.status.idle":"2023-04-23T21:15:48.120186Z","shell.execute_reply.started":"2023-04-23T21:15:48.111216Z","shell.execute_reply":"2023-04-23T21:15:48.118881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Excluir a linhas com idade nula\ndf_img_labels_train_2 = df_img_labels_train_2.drop(df_img_labels_train_2[df_img_labels_train_2['age'].isnull()].index)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:15:51.231251Z","iopub.execute_input":"2023-04-23T21:15:51.231938Z","iopub.status.idle":"2023-04-23T21:15:51.240598Z","shell.execute_reply.started":"2023-04-23T21:15:51.231900Z","shell.execute_reply":"2023-04-23T21:15:51.239428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_test['age'] = -1","metadata":{"papermill":{"duration":0.024207,"end_time":"2023-03-19T11:02:20.452468","exception":false,"start_time":"2023-03-19T11:02:20.428261","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:53.711961Z","iopub.execute_input":"2023-04-23T21:15:53.712452Z","iopub.status.idle":"2023-04-23T21:15:53.719634Z","shell.execute_reply.started":"2023-04-23T21:15:53.712408Z","shell.execute_reply":"2023-04-23T21:15:53.718411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train_1['age'].min()","metadata":{"papermill":{"duration":0.026347,"end_time":"2023-03-19T11:02:20.494518","exception":false,"start_time":"2023-03-19T11:02:20.468171","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:15:58.637140Z","iopub.execute_input":"2023-04-23T21:15:58.638284Z","iopub.status.idle":"2023-04-23T21:15:58.646174Z","shell.execute_reply.started":"2023-04-23T21:15:58.638227Z","shell.execute_reply":"2023-04-23T21:15:58.645009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train_2['age'].min()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:01.637906Z","iopub.execute_input":"2023-04-23T21:16:01.638898Z","iopub.status.idle":"2023-04-23T21:16:01.646095Z","shell.execute_reply.started":"2023-04-23T21:16:01.638859Z","shell.execute_reply":"2023-04-23T21:16:01.645025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train_1['age'].max()","metadata":{"papermill":{"duration":0.026568,"end_time":"2023-03-19T11:02:20.537750","exception":false,"start_time":"2023-03-19T11:02:20.511182","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:16:04.516682Z","iopub.execute_input":"2023-04-23T21:16:04.517772Z","iopub.status.idle":"2023-04-23T21:16:04.526319Z","shell.execute_reply.started":"2023-04-23T21:16:04.517707Z","shell.execute_reply":"2023-04-23T21:16:04.525355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train_2['age'].max()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:06.747812Z","iopub.execute_input":"2023-04-23T21:16:06.748410Z","iopub.status.idle":"2023-04-23T21:16:06.755612Z","shell.execute_reply.started":"2023-04-23T21:16:06.748372Z","shell.execute_reply":"2023-04-23T21:16:06.754465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train_1['age'] = df_img_labels_train_1['age'].astype(float)","metadata":{"papermill":{"duration":0.028922,"end_time":"2023-03-19T11:02:20.583930","exception":false,"start_time":"2023-03-19T11:02:20.555008","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:16:10.151490Z","iopub.execute_input":"2023-04-23T21:16:10.151984Z","iopub.status.idle":"2023-04-23T21:16:10.158552Z","shell.execute_reply.started":"2023-04-23T21:16:10.151949Z","shell.execute_reply":"2023-04-23T21:16:10.157275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a new \"age group\" column with 10 age categories\ndf_img_labels_train_1['age_group'] = pd.cut(df_img_labels_train_1['age'].astype(float).astype(int), bins=[0, 10, 20, 30, 40, 50, 60, 70 ,80, 90])","metadata":{"papermill":{"duration":0.037497,"end_time":"2023-03-19T11:02:20.639290","exception":false,"start_time":"2023-03-19T11:02:20.601793","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:16:12.927573Z","iopub.execute_input":"2023-04-23T21:16:12.928290Z","iopub.status.idle":"2023-04-23T21:16:12.943680Z","shell.execute_reply.started":"2023-04-23T21:16:12.928245Z","shell.execute_reply":"2023-04-23T21:16:12.942623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a new \"age group\" column with 10 age categories\ndf_img_labels_train_2['age_group'] = pd.cut(df_img_labels_train_2['age'].astype(float).astype(int), bins=[0, 10, 20, 30, 40, 50, 60, 70 ,80, 90])","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:15.809978Z","iopub.execute_input":"2023-04-23T21:16:15.810932Z","iopub.status.idle":"2023-04-23T21:16:15.820590Z","shell.execute_reply.started":"2023-04-23T21:16:15.810877Z","shell.execute_reply":"2023-04-23T21:16:15.819550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploratory Analysis","metadata":{"papermill":{"duration":0.016227,"end_time":"2023-03-19T11:02:20.671908","exception":false,"start_time":"2023-03-19T11:02:20.655681","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Creating the histogram with the age column\n\nage = df_img_labels_train_1.sort_values(\"age\")\n# Adding title and axis labels\nplt.title(\"Distribution by age group\")\nplt.xlabel(\"Age Group\")\nplt.ylabel(\"Frequency\")\nplt.hist(age['age'].values, bins=[0, 10, 20, 30, 40, 50, 60, 70 ,80, 90], rwidth=0.9)\n\n# Exiba o histograma\nplt.show()","metadata":{"papermill":{"duration":0.254577,"end_time":"2023-03-19T11:02:20.990570","exception":false,"start_time":"2023-03-19T11:02:20.735993","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:16:18.516160Z","iopub.execute_input":"2023-04-23T21:16:18.516805Z","iopub.status.idle":"2023-04-23T21:16:18.737598Z","shell.execute_reply.started":"2023-04-23T21:16:18.516764Z","shell.execute_reply":"2023-04-23T21:16:18.736573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating the histogram with the age column\n\nage = df_img_labels_train_2.sort_values(\"age\")\n# Adding title and axis labels\nplt.title(\"Distribution by age group\")\nplt.xlabel(\"Age Group\")\nplt.ylabel(\"Frequency\")\nplt.hist(age['age'].values, bins=[0, 10, 20, 30, 40, 50, 60, 70 ,80, 90], rwidth=0.9)\n\n# Exiba o histograma\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:23.015662Z","iopub.execute_input":"2023-04-23T21:16:23.016344Z","iopub.status.idle":"2023-04-23T21:16:23.334051Z","shell.execute_reply.started":"2023-04-23T21:16:23.016308Z","shell.execute_reply":"2023-04-23T21:16:23.332769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_img_labels_train_1['age_group']","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:26.772596Z","iopub.execute_input":"2023-04-23T21:16:26.773326Z","iopub.status.idle":"2023-04-23T21:16:26.780476Z","shell.execute_reply.started":"2023-04-23T21:16:26.773281Z","shell.execute_reply":"2023-04-23T21:16:26.777556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_img_labels_train_2['age_group']","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:29.609460Z","iopub.execute_input":"2023-04-23T21:16:29.610003Z","iopub.status.idle":"2023-04-23T21:16:29.619359Z","shell.execute_reply.started":"2023-04-23T21:16:29.609967Z","shell.execute_reply":"2023-04-23T21:16:29.618149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to view images\ndef view_images(col_name, figure_cols, df):\n\n    # Define the categories\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    simple_cat = random.sample(list(categories), figure_cols)\n    # Prepare the subplots\n    f, ax = plt.subplots(nrows = len(simple_cat), \n                         ncols = figure_cols, \n                         figsize = (4 * figure_cols, 4 * len(simple_cat))) \n    \n    # Draw the images\n    for i, cat in tqdm(enumerate(simple_cat)):\n        \n        # Extracts a sample\n        sample = df[df[col_name] == cat].sample(figure_cols) \n        \n        # Loop through the columns of the image\n        for j in range(0, figure_cols):\n            \n            # Extract image name\n            file = sample.iloc[j]['img_path'] +'/' + sample.iloc[j]['img_name']\n            \n            # Read image\n            #im = imageio.imread(file)\n            im = cv2.imread(file, cv2.IMREAD_GRAYSCALE)\n            \n            # Apply a histogram equalization to improve contrast\n            im = cv2.equalizeHist(im)\n            \n            # Show the image in gray (black and white)\n            ax[i, j].imshow(im, resample = True, cmap = 'gray')\n            ax[i, j].set_title(cat, fontsize = 14)  \n            \n    plt.tight_layout()\n    plt.show()","metadata":{"papermill":{"duration":0.028458,"end_time":"2023-03-19T11:02:21.036001","exception":false,"start_time":"2023-03-19T11:02:21.007543","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:16:31.835846Z","iopub.execute_input":"2023-04-23T21:16:31.836885Z","iopub.status.idle":"2023-04-23T21:16:31.846097Z","shell.execute_reply.started":"2023-04-23T21:16:31.836842Z","shell.execute_reply":"2023-04-23T21:16:31.844919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View a sample of training images of SPR X-Ray Age\nview_images('age', 4, df_img_labels_train_1)","metadata":{"papermill":{"duration":4.966234,"end_time":"2023-03-19T11:02:26.018472","exception":false,"start_time":"2023-03-19T11:02:21.052238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-22T10:07:50.039052Z","iopub.execute_input":"2023-04-22T10:07:50.040022Z","iopub.status.idle":"2023-04-22T10:07:53.853301Z","shell.execute_reply.started":"2023-04-22T10:07:50.039968Z","shell.execute_reply":"2023-04-22T10:07:53.850775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View a sample of training images of Sample NIH Chest X-ray\n#view_images('age', 4, df_img_labels_train_2)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:16:52.077810Z","iopub.execute_input":"2023-04-23T21:16:52.078398Z","iopub.status.idle":"2023-04-23T21:16:55.796376Z","shell.execute_reply.started":"2023-04-23T21:16:52.078360Z","shell.execute_reply":"2023-04-23T21:16:55.795009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Adjusting and Organizing Images","metadata":{"papermill":{"duration":0.033301,"end_time":"2023-03-19T11:02:26.086056","exception":false,"start_time":"2023-03-19T11:02:26.052755","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_img_labels_train = pd.concat([df_img_labels_train_1, df_img_labels_train_2], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:17:06.875560Z","iopub.execute_input":"2023-04-23T21:17:06.876575Z","iopub.status.idle":"2023-04-23T21:17:06.885157Z","shell.execute_reply.started":"2023-04-23T21:17:06.876522Z","shell.execute_reply":"2023-04-23T21:17:06.884151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining y (output)\ny = df_img_labels_train['age']","metadata":{"papermill":{"duration":0.045435,"end_time":"2023-03-19T11:02:26.165269","exception":false,"start_time":"2023-03-19T11:02:26.119834","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:10.647050Z","iopub.execute_input":"2023-04-23T21:17:10.647772Z","iopub.status.idle":"2023-04-23T21:17:10.653277Z","shell.execute_reply.started":"2023-04-23T21:17:10.647710Z","shell.execute_reply":"2023-04-23T21:17:10.651666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_labels_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:17:13.376550Z","iopub.execute_input":"2023-04-23T21:17:13.377315Z","iopub.status.idle":"2023-04-23T21:17:13.383991Z","shell.execute_reply.started":"2023-04-23T21:17:13.377273Z","shell.execute_reply":"2023-04-23T21:17:13.382868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining training and validation data\ndf_train, df_val = train_test_split(df_img_labels_train, test_size = 0.15, random_state = 101)","metadata":{"papermill":{"duration":0.052149,"end_time":"2023-03-19T11:02:26.252195","exception":false,"start_time":"2023-03-19T11:02:26.200046","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:16.457816Z","iopub.execute_input":"2023-04-23T21:17:16.458185Z","iopub.status.idle":"2023-04-23T21:17:16.468644Z","shell.execute_reply.started":"2023-04-23T21:17:16.458152Z","shell.execute_reply":"2023-04-23T21:17:16.467573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"papermill":{"duration":0.056954,"end_time":"2023-03-19T11:02:26.343046","exception":false,"start_time":"2023-03-19T11:02:26.286092","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:19.641460Z","iopub.execute_input":"2023-04-23T21:17:19.641935Z","iopub.status.idle":"2023-04-23T21:17:19.658793Z","shell.execute_reply.started":"2023-04-23T21:17:19.641898Z","shell.execute_reply":"2023-04-23T21:17:19.657402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"papermill":{"duration":0.053666,"end_time":"2023-03-19T11:02:26.431889","exception":false,"start_time":"2023-03-19T11:02:26.378223","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:23.247138Z","iopub.execute_input":"2023-04-23T21:17:23.247769Z","iopub.status.idle":"2023-04-23T21:17:23.264874Z","shell.execute_reply.started":"2023-04-23T21:17:23.247706Z","shell.execute_reply":"2023-04-23T21:17:23.263914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the training image name\nfile = df_train.iloc[1]['img_path']+'/'+df_train.iloc[1]['img_name']","metadata":{"papermill":{"duration":0.044826,"end_time":"2023-03-19T11:02:26.511691","exception":false,"start_time":"2023-03-19T11:02:26.466865","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:26.690473Z","iopub.execute_input":"2023-04-23T21:17:26.691009Z","iopub.status.idle":"2023-04-23T21:17:26.696987Z","shell.execute_reply.started":"2023-04-23T21:17:26.690972Z","shell.execute_reply":"2023-04-23T21:17:26.695676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Open the image\nimage = Image.open(file)","metadata":{"papermill":{"duration":0.055719,"end_time":"2023-03-19T11:02:26.603541","exception":false,"start_time":"2023-03-19T11:02:26.547822","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:28.957102Z","iopub.execute_input":"2023-04-23T21:17:28.957467Z","iopub.status.idle":"2023-04-23T21:17:28.993920Z","shell.execute_reply.started":"2023-04-23T21:17:28.957434Z","shell.execute_reply":"2023-04-23T21:17:28.992961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the training image size\nIMAGE_WIDTH, IMAGE_HEIGHT = image.size\n\nprint(\"Image width:\", IMAGE_WIDTH)\nprint(\"Image height:\", IMAGE_HEIGHT)","metadata":{"papermill":{"duration":0.046102,"end_time":"2023-03-19T11:02:26.686680","exception":false,"start_time":"2023-03-19T11:02:26.640578","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:31.648790Z","iopub.execute_input":"2023-04-23T21:17:31.649480Z","iopub.status.idle":"2023-04-23T21:17:31.655162Z","shell.execute_reply.started":"2023-04-23T21:17:31.649443Z","shell.execute_reply":"2023-04-23T21:17:31.653796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the test image name\nfile = df_img_test.iloc[1]['img_path']+'/'+df_img_test.iloc[1]['img_name']","metadata":{"papermill":{"duration":0.043437,"end_time":"2023-03-19T11:02:26.764029","exception":false,"start_time":"2023-03-19T11:02:26.720592","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:35.349318Z","iopub.execute_input":"2023-04-23T21:17:35.349683Z","iopub.status.idle":"2023-04-23T21:17:35.355857Z","shell.execute_reply.started":"2023-04-23T21:17:35.349649Z","shell.execute_reply":"2023-04-23T21:17:35.354615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the test image size\nIMAGE_WIDTH, IMAGE_HEIGHT = image.size\n\nprint(\"Largura da imagem:\", IMAGE_WIDTH)\nprint(\"Altura da imagem:\", IMAGE_HEIGHT)","metadata":{"papermill":{"duration":0.044739,"end_time":"2023-03-19T11:02:26.842667","exception":false,"start_time":"2023-03-19T11:02:26.797928","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:38.796279Z","iopub.execute_input":"2023-04-23T21:17:38.796691Z","iopub.status.idle":"2023-04-23T21:17:38.807290Z","shell.execute_reply.started":"2023-04-23T21:17:38.796656Z","shell.execute_reply":"2023-04-23T21:17:38.806135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/working/'","metadata":{"papermill":{"duration":0.043232,"end_time":"2023-03-19T11:02:26.919391","exception":false,"start_time":"2023-03-19T11:02:26.876159","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:42.154501Z","iopub.execute_input":"2023-04-23T21:17:42.155054Z","iopub.status.idle":"2023-04-23T21:17:42.159762Z","shell.execute_reply.started":"2023-04-23T21:17:42.155015Z","shell.execute_reply":"2023-04-23T21:17:42.158560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will create the folders below in the base directory:\n\n- training_data\n    \n- val_data\n\n- test_data","metadata":{"papermill":{"duration":0.03498,"end_time":"2023-03-19T11:02:26.987964","exception":false,"start_time":"2023-03-19T11:02:26.952984","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Preparing to create the path with training data\ntraining_data = os.path.join(base_dir, 'training_data/')\n\n# Creating the PATH\npath_training = Path(training_data)\n\n# Checking if the path already exists and if it doesn't, we create\nif path_training.exists():\n    print('The path already exists. Delete it and try again.')\n    shutil.rmtree(path_training)\nelse:\n    os.mkdir(training_data)","metadata":{"papermill":{"duration":0.044364,"end_time":"2023-03-19T11:02:27.067081","exception":false,"start_time":"2023-03-19T11:02:27.022717","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:45.725557Z","iopub.execute_input":"2023-04-23T21:17:45.726259Z","iopub.status.idle":"2023-04-23T21:17:45.732347Z","shell.execute_reply.started":"2023-04-23T21:17:45.726219Z","shell.execute_reply":"2023-04-23T21:17:45.731204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing path creation with validation data\nval_data = os.path.join(base_dir, 'val_data/')\n\n# Creating the PATH\npath_val = Path(val_data)\n\n# Checking if the path already exists and if it doesn't, we create\nif path_val.exists():\n    print('The path already exists. Delete it and try again.')\n    shutil.rmtree(path_val) \nelse:\n    os.mkdir(val_data)","metadata":{"papermill":{"duration":0.045188,"end_time":"2023-03-19T11:02:27.146650","exception":false,"start_time":"2023-03-19T11:02:27.101462","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:48.917551Z","iopub.execute_input":"2023-04-23T21:17:48.918510Z","iopub.status.idle":"2023-04-23T21:17:48.923810Z","shell.execute_reply.started":"2023-04-23T21:17:48.918469Z","shell.execute_reply":"2023-04-23T21:17:48.922550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing path creation with validation data\ntest_data = os.path.join(base_dir, 'test_data/')\n\n# Creating the PATH\npath_test = Path(test_data)\n\n# Checking if the path already exists and if it doesn't, we create\nif path_test.exists():\n    print('O diretório já existe. Delete no SO e tente novamente.')\n    shutil.rmtree(path_test) \nelse:\n    os.mkdir(test_data)","metadata":{"papermill":{"duration":0.04396,"end_time":"2023-03-19T11:02:27.225631","exception":false,"start_time":"2023-03-19T11:02:27.181671","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:51.799142Z","iopub.execute_input":"2023-04-23T21:17:51.799508Z","iopub.status.idle":"2023-04-23T21:17:51.806188Z","shell.execute_reply.started":"2023-04-23T21:17:51.799476Z","shell.execute_reply":"2023-04-23T21:17:51.805004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-Processing of Images","metadata":{"papermill":{"duration":0.033618,"end_time":"2023-03-19T11:02:27.292898","exception":false,"start_time":"2023-03-19T11:02:27.259280","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# images resize\nIMAGE_HEIGHT = 224\nIMAGE_WIDTH = 224","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:17:55.379033Z","iopub.execute_input":"2023-04-23T21:17:55.379398Z","iopub.status.idle":"2023-04-23T21:17:55.384767Z","shell.execute_reply.started":"2023-04-23T21:17:55.379366Z","shell.execute_reply":"2023-04-23T21:17:55.383636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gets a list of training, validation and test images\nlist_imagens_train = list(df_train['img_path']+'/'+df_train['img_name'])\nlist_imagens_val = list(df_val['img_path']+'/'+df_val['img_name'])\nlist_imagens_test = list(df_img_test['img_path']+'/'+df_img_test['img_name'])","metadata":{"papermill":{"duration":0.058355,"end_time":"2023-03-19T11:02:27.385620","exception":false,"start_time":"2023-03-19T11:02:27.327265","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:17:57.335968Z","iopub.execute_input":"2023-04-23T21:17:57.336663Z","iopub.status.idle":"2023-04-23T21:17:57.355762Z","shell.execute_reply.started":"2023-04-23T21:17:57.336624Z","shell.execute_reply":"2023-04-23T21:17:57.354684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function for pre-processing images\ndef process_imagens(lista_imagens, path_dados, image_height, image_width):\n\n    print('\\nData pre-processing! Wait...')\n\n    # Loop through the list of images\n\n    for image in tqdm(lista_imagens):\n    \n        # Image name\n        fname = image\n    \n        # Image source folder\n        src = os.path.join(fname)\n       \n        # Image destination folder\n        Iname = image[image.rfind('/')+1:]\n        dst = os.path.join(path_dados, Iname)\n        \n        # Copy the image\n        image = cv2.imread(src, cv2.IMREAD_GRAYSCALE)\n\n        # Apply a histogram equalization to improve contrast\n        image = cv2.equalizeHist(image)\n        \n        # Apply the resize\n        image = cv2.resize(image, (image_height, image_width))\n        \n        # Save the image in the destination folder\n        cv2.imwrite(dst, image)\n\n    print('\\nData is ready!')   ","metadata":{"papermill":{"duration":0.0455,"end_time":"2023-03-19T11:02:27.545575","exception":false,"start_time":"2023-03-19T11:02:27.500075","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:18:00.305903Z","iopub.execute_input":"2023-04-23T21:18:00.306270Z","iopub.status.idle":"2023-04-23T21:18:00.313683Z","shell.execute_reply.started":"2023-04-23T21:18:00.306236Z","shell.execute_reply":"2023-04-23T21:18:00.312461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transfer the pre-processed training images to the new folder\nprocess_imagens(list_imagens_train, training_data, IMAGE_HEIGHT, IMAGE_WIDTH)","metadata":{"papermill":{"duration":322.196909,"end_time":"2023-03-19T11:07:49.776112","exception":false,"start_time":"2023-03-19T11:02:27.579203","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:18:04.501395Z","iopub.execute_input":"2023-04-23T21:18:04.501777Z","iopub.status.idle":"2023-04-23T21:25:48.240554Z","shell.execute_reply.started":"2023-04-23T21:18:04.501732Z","shell.execute_reply":"2023-04-23T21:25:48.239087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transfer the pre-processed validation images to the new folder\nprocess_imagens(list_imagens_val, val_data, IMAGE_HEIGHT, IMAGE_WIDTH)","metadata":{"papermill":{"duration":57.503714,"end_time":"2023-03-19T11:08:47.314834","exception":false,"start_time":"2023-03-19T11:07:49.811120","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:25:48.242786Z","iopub.execute_input":"2023-04-23T21:25:48.243150Z","iopub.status.idle":"2023-04-23T21:27:09.883425Z","shell.execute_reply.started":"2023-04-23T21:25:48.243111Z","shell.execute_reply":"2023-04-23T21:27:09.882328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transfer the pre-processed test images to the new folder\nprocess_imagens(list_imagens_test, test_data, IMAGE_HEIGHT, IMAGE_WIDTH)","metadata":{"papermill":{"duration":399.782818,"end_time":"2023-03-19T11:15:27.134103","exception":false,"start_time":"2023-03-19T11:08:47.351285","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:27:09.884814Z","iopub.execute_input":"2023-04-23T21:27:09.885425Z","iopub.status.idle":"2023-04-23T21:35:19.203460Z","shell.execute_reply.started":"2023-04-23T21:27:09.885385Z","shell.execute_reply":"2023-04-23T21:35:19.202410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We check how many images we have now in each folder\nprint(len(os.listdir(training_data)))\nprint(len(os.listdir(val_data)))\nprint(len(os.listdir(test_data)))","metadata":{"papermill":{"duration":0.060592,"end_time":"2023-03-19T11:15:27.230970","exception":false,"start_time":"2023-03-19T11:15:27.170378","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:36.972250Z","iopub.execute_input":"2023-04-23T21:36:36.973218Z","iopub.status.idle":"2023-04-23T21:36:36.995336Z","shell.execute_reply.started":"2023-04-23T21:36:36.973161Z","shell.execute_reply":"2023-04-23T21:36:36.994380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:36:41.083989Z","iopub.execute_input":"2023-04-23T21:36:41.084574Z","iopub.status.idle":"2023-04-23T21:36:41.099267Z","shell.execute_reply.started":"2023-04-23T21:36:41.084535Z","shell.execute_reply":"2023-04-23T21:36:41.098231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Building","metadata":{}},{"cell_type":"code","source":"# Number of training samples\nnum_sample_train = len(df_train)\n\n# NNumber of validation samples\nnum_sample_val = len(df_val)\n\n# Number of test samples\nnum_sample_test = len(df_img_test)\n\n# Training batch size\nbatch_size_train = 10\n\n# Validation batch size\nbatch_size_val = 10","metadata":{"papermill":{"duration":0.047479,"end_time":"2023-03-19T11:15:27.313926","exception":false,"start_time":"2023-03-19T11:15:27.266447","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:44.401247Z","iopub.execute_input":"2023-04-23T21:36:44.401965Z","iopub.status.idle":"2023-04-23T21:36:44.408142Z","shell.execute_reply.started":"2023-04-23T21:36:44.401927Z","shell.execute_reply":"2023-04-23T21:36:44.407009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here we define the number of steps\nstep_train = np.ceil(num_sample_train / batch_size_train)\nstep_val = np.ceil(num_sample_val / batch_size_val)\nstep_test = np.ceil(num_sample_test / batch_size_val)","metadata":{"papermill":{"duration":0.046488,"end_time":"2023-03-19T11:15:27.396145","exception":false,"start_time":"2023-03-19T11:15:27.349657","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:47.573291Z","iopub.execute_input":"2023-04-23T21:36:47.573944Z","iopub.status.idle":"2023-04-23T21:36:47.579642Z","shell.execute_reply.started":"2023-04-23T21:36:47.573902Z","shell.execute_reply":"2023-04-23T21:36:47.578571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function for creating the data generator\ndef regression_data_generator(datagen, df, data_path, y, batch_size, target_size,  shuffle):\n    #datagen = ImageDataGenerator(rescale=1./255)\n    data_generator = datagen.flow_from_dataframe(dataframe=df, directory=data_path, \n                                                 x_col=\"img_name\", y_col=y, has_ext=True, \n                                                 class_mode=\"other\", target_size=target_size, \n                                                 batch_size=batch_size,\n                                                 shuffle = shuffle)\n    return data_generator","metadata":{"papermill":{"duration":0.046913,"end_time":"2023-03-19T11:15:27.639787","exception":false,"start_time":"2023-03-19T11:15:27.592874","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:50.155570Z","iopub.execute_input":"2023-04-23T21:36:50.156254Z","iopub.status.idle":"2023-04-23T21:36:50.162188Z","shell.execute_reply.started":"2023-04-23T21:36:50.156217Z","shell.execute_reply":"2023-04-23T21:36:50.161021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aqui geramos os batches de dados\n#datagen = ImageDataGenerator(rescale = 1.0/255)\n# Criar um gerador de imagens aumentadas para o conjunto de treinamento\n#datagen_train = ImageDataGenerator(\n#    rescale=1./255,  # normalização dos valores de pixel para o intervalo [0, 1]\n#    rotation_range=10,\n#    width_shift_range=0.1,\n#    height_shift_range=0.1,\n#    shear_range=0.2,\n#    zoom_range=0.2,\n#    horizontal_flip=True,\n#    fill_mode='nearest')\n\n#Here we generate batches of data\n\ndatagen_train = ImageDataGenerator(rescale=1./255, shear_range=0.2, zoom_range=0.2, horizontal_flip=True)\n\ndatagen = ImageDataGenerator(rescale=1./255)\n\n# Generate training batches\ngen_treino = regression_data_generator(datagen_train,\n                                       df_train,\n                                       training_data, \n                                       y='age',\n                                       batch_size=batch_size_train, \n                                       target_size=(IMAGE_HEIGHT, IMAGE_WIDTH),\n                                       shuffle = True)\n# Generate validation batches\ngen_val    = regression_data_generator(datagen,\n                                       df_val,\n                                       val_data,\n                                       y='age',\n                                       batch_size=batch_size_val, \n                                       target_size=(IMAGE_HEIGHT, IMAGE_WIDTH),\n                                       shuffle = False)\n\n# Generate test batches\n# Note: shuffle=False causes the test dataset not to be \"shuffled\"\n\ngen_teste  = regression_data_generator(datagen,\n                                       df_img_test, \n                                       test_data,\n                                       y='age',\n                                       batch_size=batch_size_val, \n                                       target_size=(IMAGE_HEIGHT, IMAGE_WIDTH),\n                                       shuffle = False)\n","metadata":{"papermill":{"duration":0.246403,"end_time":"2023-03-19T11:15:27.920933","exception":false,"start_time":"2023-03-19T11:15:27.674530","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:53.665428Z","iopub.execute_input":"2023-04-23T21:36:53.666161Z","iopub.status.idle":"2023-04-23T21:36:53.906538Z","shell.execute_reply.started":"2023-04-23T21:36:53.666119Z","shell.execute_reply":"2023-04-23T21:36:53.905583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of neurons in the custom layer\nnum_neurons_1 = 128\nnum_neurons_2 = 256\nnum_neurons_3 = 512\nnum_neurons_4 = 1024\n\n# Dense layer dropout rate\ndropout_dense = 0.3\n\n# Learning rate\ntaxa_aprendizado = 0.0001\n\n# Number of training epochs\nnum_epochs = 50","metadata":{"papermill":{"duration":0.045995,"end_time":"2023-03-19T11:15:28.001707","exception":false,"start_time":"2023-03-19T11:15:27.955712","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:36:57.719848Z","iopub.execute_input":"2023-04-23T21:36:57.720913Z","iopub.status.idle":"2023-04-23T21:36:57.725477Z","shell.execute_reply.started":"2023-04-23T21:36:57.720854Z","shell.execute_reply":"2023-04-23T21:36:57.724107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The EfficientNet architecture, proposed in 2019, stands out for its superior performance and lower consumption of computational resources compared to other deep neural network architectures. The architecture was developed to find the best balance between the number of parameters and performance in image classification, using a technique called \"composite scaling\". EfficientNet uses building blocks called \"MBConv\", which combine regular convolutions with depth-separable convolutions to increase network efficiency. The architecture is scalable and can be easily adapted for different image sizes and datasets, making it highly flexible and suitable for a wide range of deep learning applications.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import Model\n# Loads the pre-trained ResNet-152 model without the fully connected (top) layers\nbackbone = EfficientNetV2L(input_shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3), weights='imagenet', include_top=False)\n\n\n# Freeze the backbone\nbackbone.trainable = False\n\n# Adds custom fully connected (top) layers\nx = backbone.output\n#layer_inp = tf.keras.layers.Input(shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3))\n#x = backbone(layer_inp, training=False)\nx = GlobalAveragePooling2D()(x)  \nx = Dense(num_neurons_4, activation='relu')(x) \nx = Dropout(dropout_dense)(x)\npredictions = Dense(1, activation='linear')(x)\n\n# Create the final model\nmodel = Model(inputs=backbone.input, outputs=predictions)\n    \n# Output layer\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:37:02.156229Z","iopub.execute_input":"2023-04-23T21:37:02.156842Z","iopub.status.idle":"2023-04-23T21:37:20.320346Z","shell.execute_reply.started":"2023-04-23T21:37:02.156802Z","shell.execute_reply":"2023-04-23T21:37:20.319267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create optimizer\noptimizer = optimizers.Adam(learning_rate=taxa_aprendizado)\n# model compilation\nmodel.compile(optimizer, \n              loss = 'mse', \n              metrics = ['mae'], \n              sample_weight_mode = None)","metadata":{"papermill":{"duration":0.07367,"end_time":"2023-03-19T11:15:40.898806","exception":false,"start_time":"2023-03-19T11:15:40.825136","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:37:27.787111Z","iopub.execute_input":"2023-04-23T21:37:27.788060Z","iopub.status.idle":"2023-04-23T21:37:27.822554Z","shell.execute_reply.started":"2023-04-23T21:37:27.788007Z","shell.execute_reply":"2023-04-23T21:37:27.821627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce on Plateau\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', \n                              factor = 0.05, \n                              patience = 2, \n                              verbose = 1, \n                              min_delta=0.0001,\n                              mode = 'min', \n                              min_lr = 0.00001)","metadata":{"papermill":{"duration":0.054628,"end_time":"2023-03-19T11:15:40.992450","exception":false,"start_time":"2023-03-19T11:15:40.937822","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:37:30.760509Z","iopub.execute_input":"2023-04-23T21:37:30.761503Z","iopub.status.idle":"2023-04-23T21:37:30.767590Z","shell.execute_reply.started":"2023-04-23T21:37:30.761463Z","shell.execute_reply":"2023-04-23T21:37:30.766123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Erarly Stop\nearly_stop = EarlyStopping(monitor='val_loss', \n                           min_delta=0.0001, \n                           patience=10, \n                           verbose=1, \n                           mode='auto')","metadata":{"execution":{"iopub.status.busy":"2023-04-23T21:37:33.256667Z","iopub.execute_input":"2023-04-23T21:37:33.257070Z","iopub.status.idle":"2023-04-23T21:37:33.262872Z","shell.execute_reply.started":"2023-04-23T21:37:33.257037Z","shell.execute_reply":"2023-04-23T21:37:33.261579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creates the callbacks that will be used in training\ncallbacks_list = [reduce_lr, early_stop]","metadata":{"papermill":{"duration":0.047216,"end_time":"2023-03-19T11:15:41.077436","exception":false,"start_time":"2023-03-19T11:15:41.030220","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:37:36.739375Z","iopub.execute_input":"2023-04-23T21:37:36.740353Z","iopub.status.idle":"2023-04-23T21:37:36.744859Z","shell.execute_reply.started":"2023-04-23T21:37:36.740311Z","shell.execute_reply":"2023-04-23T21:37:36.743159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# Model training\nhistory = model.fit(gen_treino, \n                    steps_per_epoch = step_train, \n                    validation_data = gen_val,\n                    validation_steps = step_val,\n                    epochs = 10, \n                    verbose = 1,\n                    callbacks = callbacks_list)","metadata":{"papermill":{"duration":15544.993454,"end_time":"2023-03-19T15:34:46.109382","exception":false,"start_time":"2023-03-19T11:15:41.115928","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-23T21:37:40.413861Z","iopub.execute_input":"2023-04-23T21:37:40.414237Z","iopub.status.idle":"2023-04-23T22:15:04.152531Z","shell.execute_reply.started":"2023-04-23T21:37:40.414204Z","shell.execute_reply":"2023-04-23T22:15:04.151553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = optimizer.learning_rate.numpy()\nif lr < taxa_aprendizado:\n    taxa_aprendizado = lr\nelse:\n    taxa_aprendizado = taxa_aprendizado/10\n\nbackbone.trainable = True\n\n# model compilation\nmodel.compile(optimizer, \n              loss = 'mse', \n              metrics = ['mae'], \n              sample_weight_mode = None)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-23T22:15:16.627009Z","iopub.execute_input":"2023-04-23T22:15:16.627644Z","iopub.status.idle":"2023-04-23T22:15:16.774689Z","shell.execute_reply.started":"2023-04-23T22:15:16.627604Z","shell.execute_reply":"2023-04-23T22:15:16.773565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model training\nhistory = model.fit(gen_treino, \n                    steps_per_epoch = step_train, \n                    validation_data = gen_val,\n                    validation_steps = step_val,\n                    epochs = num_epochs, \n                    verbose = 1,\n                    callbacks = callbacks_list)","metadata":{"execution":{"iopub.status.busy":"2023-04-23T22:15:25.368198Z","iopub.execute_input":"2023-04-23T22:15:25.368574Z","iopub.status.idle":"2023-04-23T22:25:50.936105Z","shell.execute_reply.started":"2023-04-23T22:15:25.368541Z","shell.execute_reply":"2023-04-23T22:25:50.934675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gets the names of the model's metrics\nmodel.metrics_names","metadata":{"papermill":{"duration":2.776345,"end_time":"2023-03-19T15:34:51.567017","exception":false,"start_time":"2023-03-19T15:34:48.790672","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-22T10:50:11.163302Z","iopub.execute_input":"2023-04-22T10:50:11.163812Z","iopub.status.idle":"2023-04-22T10:50:11.175246Z","shell.execute_reply.started":"2023-04-22T10:50:11.163768Z","shell.execute_reply":"2023-04-22T10:50:11.173886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting training metrics\nval_loss, val_mae = model.evaluate(gen_val, steps = step_val)","metadata":{"papermill":{"duration":23.051777,"end_time":"2023-03-19T15:35:17.347249","exception":false,"start_time":"2023-03-19T15:34:54.295472","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-22T10:50:11.177076Z","iopub.execute_input":"2023-04-22T10:50:11.177527Z","iopub.status.idle":"2023-04-22T10:50:52.227998Z","shell.execute_reply.started":"2023-04-22T10:50:11.177484Z","shell.execute_reply":"2023-04-22T10:50:52.226919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the metrics\nprint('\\nErro do Modelo em Validação (val_loss):', val_loss)\nprint('Acurácia do Modelo em Validação (val_mae):', val_mae)","metadata":{"papermill":{"duration":2.723557,"end_time":"2023-03-19T15:35:22.808210","exception":false,"start_time":"2023-03-19T15:35:20.084653","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-22T10:50:52.239473Z","iopub.execute_input":"2023-04-22T10:50:52.240702Z","iopub.status.idle":"2023-04-22T10:50:52.251164Z","shell.execute_reply.started":"2023-04-22T10:50:52.240674Z","shell.execute_reply":"2023-04-22T10:50:52.250119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the metrics\nmae = history.history['mae']\nval_mae = history.history['mae']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(1, len(mae) + 1)","metadata":{"papermill":{"duration":2.713438,"end_time":"2023-03-19T15:35:28.172733","exception":false,"start_time":"2023-03-19T15:35:25.459295","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot\n\nplt.plot(epochs, mae, '-', label = 'mse em Treinamento', color = 'blue')\nplt.title('mse em Treinamento')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs, loss, '-', label = 'Erro em Treinamento', color = 'red')\nplt.title('Erro em Treinamento')\nplt.legend()\nplt.figure()","metadata":{"papermill":{"duration":3.26463,"end_time":"2023-03-19T15:35:33.945201","exception":false,"start_time":"2023-03-19T15:35:30.680571","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making predictions on test data\ny_pred = model.predict(gen_teste)","metadata":{"papermill":{"duration":126.39358,"end_time":"2023-03-19T15:37:43.056999","exception":false,"start_time":"2023-03-19T15:35:36.663419","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"papermill":{"duration":2.845253,"end_time":"2023-03-19T15:37:49.210833","exception":false,"start_time":"2023-03-19T15:37:46.365580","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating the submission file\ndf = pd.DataFrame([{\"imageId\": idx, \"age\": int(pred)} for idx, pred in enumerate(y_pred)])\ndf.to_csv('/kaggle/working/submission.csv', index=False)\ndf","metadata":{"papermill":{"duration":2.884945,"end_time":"2023-03-19T15:37:54.668819","exception":false,"start_time":"2023-03-19T15:37:51.783874","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}