{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport io\nimport sys\nimport numpy as np \nimport pandas as pd \nimport bson\nimport cv2\nimport matplotlib.pyplot as plt\nfrom skimage.io import imread , imshow\nfrom PIL import Image\nimport seaborn as sns\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import RMSprop,SGD,Adam\n\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.utils import to_categorical\nimport keras\nfrom keras.preprocessing.image import load_img, img_to_array\nimport tensorflow as tf\n\nimport base64\n\nfrom io import BytesIO\n\nfrom skimage import color\n\n%matplotlib inline\nimport matplotlib.image as mpimg\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator ","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:50:24.773641Z","iopub.execute_input":"2022-06-06T14:50:24.774222Z","iopub.status.idle":"2022-06-06T14:50:32.686263Z","shell.execute_reply.started":"2022-06-06T14:50:24.774178Z","shell.execute_reply":"2022-06-06T14:50:32.685094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importation des données","metadata":{}},{"cell_type":"code","source":"category_name_df = pd.read_csv('../input/cdiscount-image-classification-challenge/category_names.csv')\ndata_file = bson.decode_file_iter(open('../input/cdiscount-image-classification-challenge/train.bson', 'rb'))\nsample_submission_file = pd.read_csv('../input/cdiscount-image-classification-challenge/sample_submission.csv')\n\ntest_bson_file =  bson.decode_file_iter(open('../input/cdiscount-image-classification-challenge/test.bson', 'rb'))\ntrain_bson_file = bson.decode_file_iter(open('../input/cdiscount-image-classification-challenge/train.bson', 'rb'))\ntrain_example_bson_file = bson.decode_file_iter(open('../input/cdiscount-image-classification-challenge/train_example.bson', 'rb'))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:50:34.058300Z","iopub.execute_input":"2022-06-06T14:50:34.059130Z","iopub.status.idle":"2022-06-06T14:50:34.712033Z","shell.execute_reply.started":"2022-06-06T14:50:34.059084Z","shell.execute_reply":"2022-06-06T14:50:34.710496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Définition des méthodes utilisées","metadata":{}},{"cell_type":"code","source":"def encode_b64(data) :\n    encoded = base64.b64encode(data)\n    return encoded\n\ndef create_dataframe(path) :\n    _ids = []\n    category_ids = []\n    imgs = []\n\n    i=0\n    for c, d in enumerate(path):\n        i+=1\n        if i>1000000:\n            break\n        product_id = d['_id']\n        category_id = d['category_id']\n        for img_dict in d['imgs']:\n            img = encode_b64(img_dict['picture'])\n            picture = img\n            decoded_pic = picture.decode('utf-8')\n            \n            _ids.append(product_id)\n            category_ids.append(category_id)\n            imgs.append(decoded_pic)\n    return pd.DataFrame({'_id':_ids, 'category_id':category_ids, 'picture':imgs})\n\ndef assure_path_exists(dir):\n    if not os.path.exists(dir):\n        os.mkdir(dir)\n    else : \n        print('Folder Exists')\n        \ndef create_category_id_folders(list_category_id, dir) :\n    assure_path_exists(dir)\n    for cat_id in list_category_id :\n        dest_path = os.path.join(dir,str(cat_id))\n        assure_path_exists(dest_path)\n        \ndef write_images_to_category_folder(data, list_category_id, dir) :\n    i=0\n    for j, d in data.iterrows() :\n        if d['category_id'] in list_category_id :\n            dest_path = os.path.join(dir,str(d['category_id']),str(d['category_id'])+'_'+str(i)+'.jpg')\n            data_ = d['picture']\n            imgdata = base64.b64decode(data_)\n            with open(dest_path, 'wb') as f:\n                f.write(imgdata)\n            i+=1\n            \ndef write_train_test_images(data, list_category_id, df_category_id, train_dir, test_dir, nbr_train, nbr_test) :\n    df_category_id['count']=0\n    dict_category_id = df_category_id.set_index('categories').T.to_dict('list')\n    nbr_total=nbr_train+nbr_test\n    for j, d in data.iterrows() :\n        if d['category_id'] in list_category_id :\n            if dict_category_id[d['category_id']][0]<=nbr_train :\n                dest_path = os.path.join(train_dir,str(d['category_id']),str(d['category_id'])+'_'+str(dict_category_id[d['category_id']][0])+'.jpg')\n                data_ = d['picture']\n                imgdata = base64.b64decode(data_)\n                with open(dest_path, 'wb') as f:\n                    f.write(imgdata)\n                dict_category_id[d['category_id']][0]+=1\n            elif dict_category_id[d['category_id']][0]>nbr_train and dict_category_id[d['category_id']][0]<=nbr_total :\n                dest_path = os.path.join(test_dir,str(d['category_id']),str(d['category_id'])+'_'+str(dict_category_id[d['category_id']][0])+'.jpg')\n                data_ = d['picture']\n                imgdata = base64.b64decode(data_)\n                with open(dest_path, 'wb') as f:\n                    f.write(imgdata)\n                dict_category_id[d['category_id']][0]+=1\n            \ndef plot_image(i, label, imgs):\n#     label, img = label[i], imgs[i]\n    img= imgs[i]\n    plt.grid(False)\n    plt.xticks([])\n    plt.yticks([])\n    \n    images, index = next(imgs)\n    \n    lbl_i = np.argwhere(index[i]>0)[0][0]\n    \n    label = label[lbl_i]\n    \n    plt.imshow(images[i])\n    \n    categ_level_1 = category_name_df[category_name_df['category_id']==label].iloc[0,1] +'\\n'\n    categ_level_2 = category_name_df[category_name_df['category_id']==label].iloc[0,2] +'\\n'\n    categ_level_3 = category_name_df[category_name_df['category_id']==label].iloc[0,3] \n    title= categ_level_1+ categ_level_2+ categ_level_3 \n\n    plt.xlabel(\"{}\".format(title), color='blue')\n    \n#     plt.xlabel(\"({})\".format(category_name_df[category_name_df['category_id']==label].iloc[0:4]),\n#                                 color='blue')\n    \n    \ndef plot_image_pred(i, predictions_array, true_label, imgs):\n    img = imgs[i]\n    plt.grid(False)\n    plt.xticks([])\n    plt.yticks([])\n    \n    images, index = next(imgs)\n    \n    lbl_i = np.argwhere(index[i]>0)[0][0]\n    \n    label = true_label[lbl_i]\n    \n    plt.imshow(images[i])\n\n#     plt.imshow(img, cmap=plt.cm.binary)\n\n    predicted_label = np.argmax(predictions_array)\n    if predicted_label == label:\n        color = 'blue'\n    else:\n        color = 'red'\n        \n    categ_level_1 = category_name_df[category_name_df['category_id']==label].iloc[0,1] +'\\n'\n    categ_level_2 = category_name_df[category_name_df['category_id']==label].iloc[0,2] +'\\n'\n    categ_level_3 = category_name_df[category_name_df['category_id']==label].iloc[0,3] \n    title= categ_level_1+ categ_level_2+ categ_level_3 \n\n    plt.xlabel(\"{} {:2.0f}% \\n ({})\".format(category_name_df.iloc[predicted_label,1:4],\n                                100*np.max(predictions_array),\n                                title),\n                                color=color, fontsize=30)\n\ndef plot_value_array(i, predictions_array, true_label, imgs):\n    images, index = next(imgs)\n    \n    lbl_i = np.argwhere(index[i]>0)[0][0]\n    \n    true_label = true_label[lbl_i]\n    \n#     true_label = true_label['category_ids'].iloc[i]\n    plt.grid(False)\n    plt.xticks(range(194))\n    plt.yticks([])\n    thisplot = plt.bar(range(194), predictions_array, color=\"#777777\")\n    plt.ylim([0, 1])\n    predicted_label = np.argmax(predictions_array)\n\n    thisplot[predicted_label].set_color('red')\n    thisplot[lbl_i].set_color('blue')\n    \n\ndef decode_images(item_imgs):\n    nx = 2 if len(item_imgs) > 1 else 1\n    ny = 2 if len(item_imgs) > 2 else 1\n    composed_img = np.zeros((ny * 180, nx * 180, 3), dtype=np.uint8)\n    for i, img_dict in enumerate(item_imgs):\n        img = decode(img_dict['picture'])\n        h, w, _ = img.shape        \n        xstart = (i % nx) * 180\n        xend = xstart + w\n        ystart = (i // nx) * 180\n        yend = ystart + h\n        composed_img[ystart:yend, xstart:xend] = img\n    return composed_img\n\n\ndef decode(data):\n    arr = np.asarray(bytearray(data), dtype=np.uint8)\n    img = cv2.imdecode(arr, cv2.IMREAD_COLOR)\n    return cv2.cvtColor(img, cv2.COLOR_BGR2RGB) ","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:14:51.701112Z","iopub.execute_input":"2022-06-06T15:14:51.701717Z","iopub.status.idle":"2022-06-06T15:14:51.742278Z","shell.execute_reply.started":"2022-06-06T15:14:51.701664Z","shell.execute_reply":"2022-06-06T15:14:51.739968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Définition des variables statiques","metadata":{}},{"cell_type":"code","source":"IMAGES_DIR = os.path.join('.','data')\nassure_path_exists(IMAGES_DIR)\nbs=50\ntrain_dir = \"./train/\"\nvalidation_dir = \"./test/\"\n\nassure_path_exists(train_dir)\nassure_path_exists(validation_dir)\n\ntrain_datagen = ImageDataGenerator( rescale = 1.0/255. )\ntest_datagen  = ImageDataGenerator( rescale = 1.0/255. )\n\ntrain_generator = train_datagen.flow_from_directory(train_dir,\n                                                    batch_size=bs,\n                                                    class_mode='categorical',\n                                                    target_size=(180,180))\n\nvalidation_generator =  test_datagen.flow_from_directory(validation_dir,\n                                                         batch_size=bs,\n                                                         class_mode  = 'categorical',\n                                                         target_size=(180,180))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:05:16.643713Z","iopub.execute_input":"2022-06-06T15:05:16.644366Z","iopub.status.idle":"2022-06-06T15:05:16.859152Z","shell.execute_reply.started":"2022-06-06T15:05:16.644178Z","shell.execute_reply":"2022-06-06T15:05:16.858274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Affichage des images et leurs catégories","metadata":{}},{"cell_type":"code","source":"max_counter = 16\ncounter = 0\nn = 4\n\nlevel_tags = category_name_df.columns[1:]\n\nfor item in bson.decode_file_iter(open('../input/cdiscount-image-classification-challenge/train.bson', 'rb')):  \n    if counter % n == 0:\n        plt.figure(figsize=(14, 12))\n    \n    mask = category_name_df['category_id'] == item['category_id']    \n    plt.subplot(1, n, counter % n + 1)\n    cat_levels = category_name_df[mask][level_tags].values.tolist()[0]\n    cat_levels = [c[:25] for c in cat_levels]\n    title = str(item['category_id']) + '\\n\\n'\n    title += '\\n'.join(cat_levels)\n    plt.title(title+'\\n')\n    plt.imshow(decode_images(item['imgs']))\n    plt.axis('off')\n    \n    counter += 1\n    if counter == max_counter:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:40:11.941067Z","iopub.execute_input":"2022-06-06T14:40:11.941919Z","iopub.status.idle":"2022-06-06T14:40:13.275745Z","shell.execute_reply.started":"2022-06-06T14:40:11.941867Z","shell.execute_reply":"2022-06-06T14:40:13.275000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Affichage du nombre des catégories et des niveaux","metadata":{}},{"cell_type":"code","source":"print(\"Unique categories: \", len(category_name_df['category_id'].unique()))\nprint(\"Unique level 1 categories: \", len(category_name_df['category_level1'].unique()))\nprint(\"Unique level 2 categories: \", len(category_name_df['category_level2'].unique()))\nprint(\"Unique level 3 categories: \", len(category_name_df['category_level3'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:40:15.608358Z","iopub.execute_input":"2022-06-06T14:40:15.609097Z","iopub.status.idle":"2022-06-06T14:40:15.623847Z","shell.execute_reply.started":"2022-06-06T14:40:15.609037Z","shell.execute_reply":"2022-06-06T14:40:15.622428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Affichage des noms des niveaux du première catégorie","metadata":{}},{"cell_type":"code","source":"for i in category_name_df['category_level1'].unique():\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:40:33.833412Z","iopub.execute_input":"2022-06-06T14:40:33.834483Z","iopub.status.idle":"2022-06-06T14:40:33.840484Z","shell.execute_reply.started":"2022-06-06T14:40:33.834435Z","shell.execute_reply":"2022-06-06T14:40:33.839719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Affichage des statistiques pour le premier niveau","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\n_ = sns.countplot(y=category_name_df['category_level1'])","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:40:36.957607Z","iopub.execute_input":"2022-06-06T14:40:36.958140Z","iopub.status.idle":"2022-06-06T14:40:37.542005Z","shell.execute_reply.started":"2022-06-06T14:40:36.958100Z","shell.execute_reply":"2022-06-06T14:40:37.540948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Création du dataFrame et enregistrement sous forme csv","metadata":{}},{"cell_type":"code","source":"df = create_dataframe(data_file)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:42:04.170080Z","iopub.execute_input":"2022-06-06T14:42:04.170516Z","iopub.status.idle":"2022-06-06T14:43:54.835569Z","shell.execute_reply.started":"2022-06-06T14:42:04.170485Z","shell.execute_reply":"2022-06-06T14:43:54.833859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assure_path_exists('./csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:43:54.838255Z","iopub.execute_input":"2022-06-06T14:43:54.838848Z","iopub.status.idle":"2022-06-06T14:43:54.846016Z","shell.execute_reply.started":"2022-06-06T14:43:54.838808Z","shell.execute_reply":"2022-06-06T14:43:54.844765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"./csv/data.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:43:54.847489Z","iopub.execute_input":"2022-06-06T14:43:54.848592Z","iopub.status.idle":"2022-06-06T14:47:48.640664Z","shell.execute_reply.started":"2022-06-06T14:43:54.848543Z","shell.execute_reply":"2022-06-06T14:47:48.638069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lecture des données d'après le fichier .csv","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('./csv/data.csv')\ndata.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:51:06.660678Z","iopub.execute_input":"2022-06-06T14:51:06.661321Z","iopub.status.idle":"2022-06-06T14:53:26.541908Z","shell.execute_reply.started":"2022-06-06T14:51:06.661274Z","shell.execute_reply":"2022-06-06T14:53:26.540778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# nombre des images par catégorie","metadata":{}},{"cell_type":"code","source":"df_count = data.groupby(['category_id'])['_id'].count()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:53:26.545299Z","iopub.execute_input":"2022-06-06T14:53:26.546236Z","iopub.status.idle":"2022-06-06T14:53:26.630946Z","shell.execute_reply.started":"2022-06-06T14:53:26.546194Z","shell.execute_reply":"2022-06-06T14:53:26.629670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_count","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:53:26.632260Z","iopub.execute_input":"2022-06-06T14:53:26.632610Z","iopub.status.idle":"2022-06-06T14:53:26.641793Z","shell.execute_reply.started":"2022-06-06T14:53:26.632582Z","shell.execute_reply":"2022-06-06T14:53:26.640614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_category_count = pd.DataFrame(df_count)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:54:52.192342Z","iopub.execute_input":"2022-06-06T14:54:52.192923Z","iopub.status.idle":"2022-06-06T14:54:52.199164Z","shell.execute_reply.started":"2022-06-06T14:54:52.192882Z","shell.execute_reply":"2022-06-06T14:54:52.198289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_category_count","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:54:56.145188Z","iopub.execute_input":"2022-06-06T14:54:56.145954Z","iopub.status.idle":"2022-06-06T14:54:56.155817Z","shell.execute_reply.started":"2022-06-06T14:54:56.145914Z","shell.execute_reply":"2022-06-06T14:54:56.155023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Catégories qui ont plus que 1500 images","metadata":{}},{"cell_type":"code","source":"df_count_1500 = df_category_count[df_category_count['_id']>1500]","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:54:58.398509Z","iopub.execute_input":"2022-06-06T14:54:58.398961Z","iopub.status.idle":"2022-06-06T14:54:58.405660Z","shell.execute_reply.started":"2022-06-06T14:54:58.398917Z","shell.execute_reply":"2022-06-06T14:54:58.404762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_count_1500","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:55:00.528062Z","iopub.execute_input":"2022-06-06T14:55:00.528625Z","iopub.status.idle":"2022-06-06T14:55:00.546119Z","shell.execute_reply.started":"2022-06-06T14:55:00.528585Z","shell.execute_reply":"2022-06-06T14:55:00.544720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_category_id_1500 = list(df_count_1500.index.values)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:55:02.313933Z","iopub.execute_input":"2022-06-06T14:55:02.314787Z","iopub.status.idle":"2022-06-06T14:55:02.319232Z","shell.execute_reply.started":"2022-06-06T14:55:02.314748Z","shell.execute_reply":"2022-06-06T14:55:02.318083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_category_id_1500","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:55:03.702051Z","iopub.execute_input":"2022-06-06T14:55:03.702510Z","iopub.status.idle":"2022-06-06T14:55:03.712931Z","shell.execute_reply.started":"2022-06-06T14:55:03.702474Z","shell.execute_reply":"2022-06-06T14:55:03.711727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(list_category_id_1500)\n# list_category_id_1500.count()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:55:05.775586Z","iopub.execute_input":"2022-06-06T14:55:05.776885Z","iopub.status.idle":"2022-06-06T14:55:05.783343Z","shell.execute_reply.started":"2022-06-06T14:55:05.776840Z","shell.execute_reply":"2022-06-06T14:55:05.782155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_category_id_1500 = pd.DataFrame(list_category_id_1500)\ndf_category_id_1500","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:17:02.736097Z","iopub.execute_input":"2022-06-06T15:17:02.736536Z","iopub.status.idle":"2022-06-06T15:17:02.749276Z","shell.execute_reply.started":"2022-06-06T15:17:02.736504Z","shell.execute_reply":"2022-06-06T15:17:02.748227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_category_id_1500 = df_category_id_1500.rename({0: 'categories'}, axis='columns')\ndf_category_id_1500","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:17:10.823438Z","iopub.execute_input":"2022-06-06T15:17:10.823920Z","iopub.status.idle":"2022-06-06T15:17:10.846108Z","shell.execute_reply.started":"2022-06-06T15:17:10.823883Z","shell.execute_reply":"2022-06-06T15:17:10.845257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Création des dossiers","metadata":{}},{"cell_type":"code","source":"# create_category_id_folders(list_category_id_1500, IMAGES_DIR)\ncreate_category_id_folders(list_category_id_1500, train_dir)\ncreate_category_id_folders(list_category_id_1500, validation_dir)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:13:32.508952Z","iopub.execute_input":"2022-06-06T15:13:32.510358Z","iopub.status.idle":"2022-06-06T15:13:32.515189Z","shell.execute_reply.started":"2022-06-06T15:13:32.510287Z","shell.execute_reply":"2022-06-06T15:13:32.513871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['category_id']","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:13:37.484627Z","iopub.execute_input":"2022-06-06T15:13:37.486929Z","iopub.status.idle":"2022-06-06T15:13:37.506858Z","shell.execute_reply.started":"2022-06-06T15:13:37.486844Z","shell.execute_reply":"2022-06-06T15:13:37.505526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ecriture des images dans les dossiers appropriés","metadata":{}},{"cell_type":"code","source":"write_train_test_images(data, list_category_id_1500, df_category_id_1500, train_dir, validation_dir, 1200, 300) # Lecture des données d'après le fichier .csv","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:17:22.442333Z","iopub.execute_input":"2022-06-06T15:17:22.443443Z","iopub.status.idle":"2022-06-06T15:18:18.922350Z","shell.execute_reply.started":"2022-06-06T15:17:22.443400Z","shell.execute_reply":"2022-06-06T15:18:18.920958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualisation des images","metadata":{}},{"cell_type":"code","source":"# i = 0\n# for i in range(0,3):\n#     plt.figure(figsize=(16, 16))\n#     plt.subplot(2,3,i+1)\n#     plot_image(i, list_category_id_1500, validation_generator)\n#     plt.show()\n#     i+=1","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:04:14.350539Z","iopub.execute_input":"2022-06-06T15:04:14.352705Z","iopub.status.idle":"2022-06-06T15:04:14.358240Z","shell.execute_reply.started":"2022-06-06T15:04:14.352607Z","shell.execute_reply":"2022-06-06T15:04:14.357049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Construction du réseau de neurons","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.Sequential([\n    tf.keras.layers.Conv2D(16,(3,3),activation = \"relu\" , input_shape = (180,180,3)) ,\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Conv2D(32,(3,3),activation = \"relu\") ,  \n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Conv2D(64,(3,3),activation = \"relu\") ,  \n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Conv2D(128,(3,3),activation = \"relu\"),  \n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Flatten(), \n    tf.keras.layers.Dense(550,activation=\"relu\"),\n    tf.keras.layers.Dropout(0.1,seed = 2019),\n    tf.keras.layers.Dense(400,activation =\"relu\"),\n    tf.keras.layers.Dropout(0.3,seed = 2019),\n    tf.keras.layers.Dense(300,activation=\"relu\"),\n    tf.keras.layers.Dropout(0.4,seed = 2019),\n    tf.keras.layers.Dense(200,activation =\"relu\"),\n    tf.keras.layers.Dropout(0.2,seed = 2019),\n    tf.keras.layers.Dense(194,activation = \"softmax\")\n])","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:04:39.532967Z","iopub.execute_input":"2022-06-06T15:04:39.533548Z","iopub.status.idle":"2022-06-06T15:04:39.702908Z","shell.execute_reply.started":"2022-06-06T15:04:39.533498Z","shell.execute_reply":"2022-06-06T15:04:39.701569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:04:42.005394Z","iopub.execute_input":"2022-06-06T15:04:42.005818Z","iopub.status.idle":"2022-06-06T15:04:42.014041Z","shell.execute_reply.started":"2022-06-06T15:04:42.005778Z","shell.execute_reply":"2022-06-06T15:04:42.012790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adam=Adam(learning_rate=0.001)\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics = ['acc'])","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:04:45.297493Z","iopub.execute_input":"2022-06-06T15:04:45.298954Z","iopub.status.idle":"2022-06-06T15:04:45.315649Z","shell.execute_reply.started":"2022-06-06T15:04:45.298876Z","shell.execute_reply":"2022-06-06T15:04:45.314155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator,\n                    validation_data=validation_generator,\n                    steps_per_epoch=1200 // bs,\n                    epochs=30,\n                    validation_steps=300 // bs,\n#                     verbose=2\n                             )","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:04:56.077824Z","iopub.execute_input":"2022-06-06T15:04:56.078292Z","iopub.status.idle":"2022-06-06T15:04:56.118602Z","shell.execute_reply.started":"2022-06-06T15:04:56.078254Z","shell.execute_reply":"2022-06-06T15:04:56.116414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation des résultats","metadata":{}},{"cell_type":"code","source":"plt.plot(history.history[\"acc\"])\nplt.plot(history.history['val_acc'])\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title(\"model accuracy\")\nplt.ylabel(\"Accuracy\")\nplt.xlabel(\"Epoch\")\nplt.legend([\"Accuracy\",\"Validation Accuracy\",\"loss\",\"Validation Loss\"])\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test et évaluation des résultats","metadata":{}},{"cell_type":"code","source":"probability_model = tf.keras.Sequential([model, \n                                         tf.keras.layers.Softmax()])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = probability_model.predict(validation_generator)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 0\nfor i in range(0,5):\n    plt.figure(figsize=(6,3))\n    plt.subplot(1,2,1)\n    plot_image_pred(i, predictions[i], list_category_id_1500, validation_generator)\n    plt.subplot(1,2,2)\n    plot_value_array(i, predictions[i],  list_category_id_1500, validation_generator)\n    plt.show()\n    i+=1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_rows = 10\nnum_cols = 1\nnum_images = num_rows*num_cols\nplt.figure(figsize=(25*2*num_cols, 4*2*2*num_rows))\nfor i in range(num_images):\n    plt.subplot(num_rows, 2*num_cols, 2*i+1)\n    plot_image_pred(i, predictions[i], list_category_id_1500, validation_generator)\n    plt.subplot(num_rows, 2*num_cols, 2*i+2)\n    plot_value_array(i, predictions[i],  list_category_id_1500, validation_generator)\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}