{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom pathlib import Path\nimport os.path\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport os\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_Path = '../input/plant-pathology-2021-fgvc8/train_images'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_Path","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img_Path = '../input/plant-pathology-2021-fgvc8/test_images'\n\nimg_Path = '../input/resized-plant2021/img_sz_256'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/train.csv')\n\nsample_submission = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of pictures in the training dataset: {train.shape[0]}\\n')\nprint(f'Number of different labels: {len(train.labels.unique())}\\n')\nprint(f'Labels: {train.labels.unique()}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['labels'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14,7))\nb = sns.countplot(x='labels', data=train, order=sorted(train['labels'].unique()))\nfor item in b.get_xticklabels():\n    item.set_rotation(90)\nplt.title('Label Distribution', weight='bold')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(path/\"train.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(2021)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ở đây chúng ta sẽ đưa về kích thước ảnh là 128, cắt kích thước ảnh ngẫu nhiên từ 0.75 dến 1.00 và đưa về tỷ lệ 1:1 so\n#với ảnh cũ\nitem_tfms = [RandomResizedCrop(128, min_scale=0.75, ratio=(1., 1.))]\nbatch_tfms = [*aug_transforms(size=128, max_warp=0), Normalize.from_stats(*imagenet_stats)]\n\ndls = ImageDataLoaders.from_df(\n    df = train_df,\n    folder = \"../input/resized-plant2021/img_sz_512\",\n    item_tfms = item_tfms,\n    batch_tfms = batch_tfms,\n    splitter = RandomSplitter(valid_pct=0.1),\n    label_delim = \" \",\n    bs=256\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,40))\ni=1\nfor idx,s in train.head(9).iterrows():\n    img_path = os.path.join(img_Path,s['image'])\n    img=cv2.imread(img_path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    #gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    fig=plt.subplot(9,3,i)\n    #fig.imshow(img)\n    fig.imshow(img)\n    fig.set_title(s['labels'])\n    i+=1\n    \n  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CLASSES = train['labels'].unique().tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n\n# Preprocessing the Training set\ntrain_datagen = ImageDataGenerator(rescale=1./255,\n                                   shear_range = 0.1,\n                                   zoom_range = 0.1,\n                                   horizontal_flip = True,\n                                   validation_split=0.25)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_datagen.flow_from_dataframe(train,\n                                              directory=img_Path,\n                                              classes=CLASSES,\n                                              x_col=\"image\",\n                                              y_col=\"labels\",\n                                              target_size=(150, 150),\n                                              subset='training')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}