{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport plotly.express as px\nimport PIL","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-22T16:36:18.133528Z","iopub.execute_input":"2023-08-22T16:36:18.133875Z","iopub.status.idle":"2023-08-22T16:36:18.482635Z","shell.execute_reply.started":"2023-08-22T16:36:18.133846Z","shell.execute_reply":"2023-08-22T16:36:18.479161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:18.484491Z","iopub.execute_input":"2023-08-22T16:36:18.484921Z","iopub.status.idle":"2023-08-22T16:36:19.506424Z","shell.execute_reply.started":"2023-08-22T16:36:18.484889Z","shell.execute_reply":"2023-08-22T16:36:19.505244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/herbarium-2022-fgvc9/train_images/'\ntest_dir = '../input/herbarium-2022-fgvc9/test_images/'\n\nwith open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)\nwith open(\"../input/herbarium-2022-fgvc9/test_metadata.json\") as json_file:\n    test_meta = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:19.50855Z","iopub.execute_input":"2023-08-22T16:36:19.508944Z","iopub.status.idle":"2023-08-22T16:36:34.292243Z","shell.execute_reply.started":"2023-08-22T16:36:19.508909Z","shell.execute_reply":"2023-08-22T16:36:34.291293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_meta)\ntest_df.columns = ['file_name', 'image_id', 'license']","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:34.295522Z","iopub.execute_input":"2023-08-22T16:36:34.295905Z","iopub.status.idle":"2023-08-22T16:36:34.550199Z","shell.execute_reply.started":"2023-08-22T16:36:34.29587Z","shell.execute_reply":"2023-08-22T16:36:34.549183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_meta","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:34.551668Z","iopub.execute_input":"2023-08-22T16:36:34.552007Z","iopub.status.idle":"2023-08-22T16:36:34.558332Z","shell.execute_reply.started":"2023-08-22T16:36:34.551975Z","shell.execute_reply":"2023-08-22T16:36:34.556424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame(train_meta['annotations'])\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:34.559823Z","iopub.execute_input":"2023-08-22T16:36:34.560879Z","iopub.status.idle":"2023-08-22T16:36:36.302762Z","shell.execute_reply.started":"2023-08-22T16:36:34.5608Z","shell.execute_reply":"2023-08-22T16:36:36.301622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cat = pd.DataFrame(train_meta['categories'])\n#train_cat.columns = [ 'category_id', 'scientificName','family', 'genus']\ntrain_cat.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:36.304224Z","iopub.execute_input":"2023-08-22T16:36:36.304653Z","iopub.status.idle":"2023-08-22T16:36:36.346452Z","shell.execute_reply.started":"2023-08-22T16:36:36.30462Z","shell.execute_reply":"2023-08-22T16:36:36.345609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img = pd.DataFrame(train_meta['images'])\ntrain_img.columns = ['image_id','file_name', 'license']\ndisplay(train_img)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:36.348242Z","iopub.execute_input":"2023-08-22T16:36:36.348916Z","iopub.status.idle":"2023-08-22T16:36:37.344799Z","shell.execute_reply.started":"2023-08-22T16:36:36.348882Z","shell.execute_reply":"2023-08-22T16:36:37.343718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = pd.DataFrame(train_meta['genera'])\ntrain_gen.columns = ['genus_id', 'genus']\ndisplay(train_gen)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:37.348215Z","iopub.execute_input":"2023-08-22T16:36:37.348556Z","iopub.status.idle":"2023-08-22T16:36:37.362581Z","shell.execute_reply.started":"2023-08-22T16:36:37.34853Z","shell.execute_reply":"2023-08-22T16:36:37.361576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.merge(train_cat, on='category_id', how='outer')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:37.367875Z","iopub.execute_input":"2023-08-22T16:36:37.368506Z","iopub.status.idle":"2023-08-22T16:36:37.618149Z","shell.execute_reply.started":"2023-08-22T16:36:37.368472Z","shell.execute_reply":"2023-08-22T16:36:37.617093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.merge(train_img, on='image_id', how='outer')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:37.619745Z","iopub.execute_input":"2023-08-22T16:36:37.620134Z","iopub.status.idle":"2023-08-22T16:36:38.760377Z","shell.execute_reply.started":"2023-08-22T16:36:37.620098Z","shell.execute_reply":"2023-08-22T16:36:38.759217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:38.762036Z","iopub.execute_input":"2023-08-22T16:36:38.762438Z","iopub.status.idle":"2023-08-22T16:36:38.77033Z","shell.execute_reply.started":"2023-08-22T16:36:38.762402Z","shell.execute_reply":"2023-08-22T16:36:38.769237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:38.771997Z","iopub.execute_input":"2023-08-22T16:36:38.773849Z","iopub.status.idle":"2023-08-22T16:36:38.789648Z","shell.execute_reply.started":"2023-08-22T16:36:38.773816Z","shell.execute_reply":"2023-08-22T16:36:38.788848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"file_name\"] = train_dir + train_df[\"file_name\"]","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:38.791806Z","iopub.execute_input":"2023-08-22T16:36:38.792451Z","iopub.status.idle":"2023-08-22T16:36:38.9382Z","shell.execute_reply.started":"2023-08-22T16:36:38.79242Z","shell.execute_reply":"2023-08-22T16:36:38.937199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop([\"image_id\", \"authors\", \"license\",\"institution_id\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:38.939626Z","iopub.execute_input":"2023-08-22T16:36:38.940178Z","iopub.status.idle":"2023-08-22T16:36:38.993366Z","shell.execute_reply.started":"2023-08-22T16:36:38.940144Z","shell.execute_reply":"2023-08-22T16:36:38.992404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:38.995021Z","iopub.execute_input":"2023-08-22T16:36:38.99539Z","iopub.status.idle":"2023-08-22T16:36:40.141331Z","shell.execute_reply.started":"2023-08-22T16:36:38.995358Z","shell.execute_reply":"2023-08-22T16:36:40.140342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:40.142809Z","iopub.execute_input":"2023-08-22T16:36:40.143415Z","iopub.status.idle":"2023-08-22T16:36:40.157723Z","shell.execute_reply.started":"2023-08-22T16:36:40.14338Z","shell.execute_reply":"2023-08-22T16:36:40.156551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:40.159639Z","iopub.execute_input":"2023-08-22T16:36:40.160055Z","iopub.status.idle":"2023-08-22T16:36:40.809937Z","shell.execute_reply.started":"2023-08-22T16:36:40.160004Z","shell.execute_reply":"2023-08-22T16:36:40.809025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image, display\n\n# Assuming 'train_df' contains your DataFrame with an 'file_name' and 'label' column\n\nindex_to_view = 0  # Change this to the index of the row you want to view\n\n# Get the image file path and label from the DataFrame\nimage_path = train_df.loc[index_to_view, 'file_name']\nlabel = train_df.loc[index_to_view, 'family']  # Replace 'label' with the actual column name\n\n# Display the image and label using IPython's Image module\ndisplay(Image(filename=image_path))\nprint(\"Label:\", label)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:40.811526Z","iopub.execute_input":"2023-08-22T16:36:40.811864Z","iopub.status.idle":"2023-08-22T16:36:40.87511Z","shell.execute_reply.started":"2023-08-22T16:36:40.811832Z","shell.execute_reply":"2023-08-22T16:36:40.874227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of families to include in the filtered DataFrame\nselected_families = ['Pinaceae', 'Nyctaginaceae', 'Menispermaceae', 'Malvaceae', 'Fabaceae', 'Rosaceae',\n                     'Euphorbiaceae', 'Asteraceae', 'Cactaceae', 'Lamiaceae', 'Polygonaceae', 'Bromeliaceae',\n                     'Sapindaceae', 'Plantaginaceae', 'Berberidaceae', 'Amaranthaceae', 'Caryophyllaceae',\n                     'Orchidaceae', 'Calyceraceae', 'Arecaceae', 'Ranunculaceae', 'Acoraceae', 'Pteridaceae',\n                     'Schizaeaceae', 'Araceae', 'Bignoniaceae', 'Papaveraceae', 'Rhamnaceae', 'Viburnaceae',\n                     'Orobanchaceae', 'Ericaceae', 'Asparagaceae', 'Phytolaccaceae', 'Poaceae', 'Lauraceae',\n                     'Apiaceae', 'Nartheciaceae', 'Polemoniaceae', 'Alismataceae', 'Apocynaceae', 'Amaryllidaceae',\n                     'Betulaceae', 'Iridaceae', 'Verbenaceae', 'Gesneriaceae', 'Cyatheaceae', 'Alstroemeriaceae',\n                     'Picramniaceae', 'Melanthiaceae', 'Lythraceae', 'Vitaceae', 'Anacardiaceae']\n\n# Filter the DataFrame to include only the selected families\ntrain_df = train_df[train_df['family'].isin(selected_families)]","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:40.876108Z","iopub.execute_input":"2023-08-22T16:36:40.876441Z","iopub.status.idle":"2023-08-22T16:36:41.121456Z","shell.execute_reply.started":"2023-08-22T16:36:40.876406Z","shell.execute_reply":"2023-08-22T16:36:41.120413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:41.122766Z","iopub.execute_input":"2023-08-22T16:36:41.123594Z","iopub.status.idle":"2023-08-22T16:36:41.523169Z","shell.execute_reply.started":"2023-08-22T16:36:41.12356Z","shell.execute_reply":"2023-08-22T16:36:41.522168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:41.524602Z","iopub.execute_input":"2023-08-22T16:36:41.525194Z","iopub.status.idle":"2023-08-22T16:36:42.239223Z","shell.execute_reply.started":"2023-08-22T16:36:41.525159Z","shell.execute_reply":"2023-08-22T16:36:42.238318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:42.240729Z","iopub.execute_input":"2023-08-22T16:36:42.241083Z","iopub.status.idle":"2023-08-22T16:36:42.253233Z","shell.execute_reply.started":"2023-08-22T16:36:42.241051Z","shell.execute_reply":"2023-08-22T16:36:42.252106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"family\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:42.254734Z","iopub.execute_input":"2023-08-22T16:36:42.255414Z","iopub.status.idle":"2023-08-22T16:36:42.342238Z","shell.execute_reply.started":"2023-08-22T16:36:42.255378Z","shell.execute_reply":"2023-08-22T16:36:42.341198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the top 10 values in the 'family' column\ntop_10 = train_df['family'].value_counts().head(10)\ntop_10 = top_10.sort_values(ascending=True)\n# Create a horizontal bar chart\nplt.barh(top_10.index, top_10.values, color='green')\n\n# Add labels and title\nplt.xlabel('Count')\nplt.ylabel('Family')\nplt.title('Top 10 Families')\n\n# Display the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:42.343575Z","iopub.execute_input":"2023-08-22T16:36:42.344001Z","iopub.status.idle":"2023-08-22T16:36:42.721088Z","shell.execute_reply.started":"2023-08-22T16:36:42.343967Z","shell.execute_reply":"2023-08-22T16:36:42.720167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the top 10 values in the 'family' column\ntop_10 = train_df['genus'].value_counts().head(10)\ntop_10 = top_10.sort_values(ascending=True)\n# Create a horizontal bar chart\nplt.barh(top_10.index, top_10.values, color='green')\n\n# Add labels and title\nplt.xlabel('Count')\nplt.ylabel('Genus')\nplt.title('Top 10 Families')\n\n# Display the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:42.722696Z","iopub.execute_input":"2023-08-22T16:36:42.723345Z","iopub.status.idle":"2023-08-22T16:36:43.074155Z","shell.execute_reply.started":"2023-08-22T16:36:42.72331Z","shell.execute_reply":"2023-08-22T16:36:43.073223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the top 10 values in the 'family' column\ntop_10 = train_df['species'].value_counts().head(10)\ntop_10 = top_10.sort_values(ascending=True)\n# Create a horizontal bar chart\nplt.barh(top_10.index, top_10.values, color='green')\n\n# Add labels and title\nplt.xlabel('Count')\nplt.ylabel('species')\nplt.title('Top 10 Families')\n\n# Display the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:43.075428Z","iopub.execute_input":"2023-08-22T16:36:43.076534Z","iopub.status.idle":"2023-08-22T16:36:43.429056Z","shell.execute_reply.started":"2023-08-22T16:36:43.076498Z","shell.execute_reply":"2023-08-22T16:36:43.42816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of unique species within each genus and family\nfamily_counts = train_df.groupby('family')['species'].nunique().reset_index()\ngenus_counts = train_df.groupby(['family', 'genus'])['species'].nunique().reset_index()\n\n# Create a treemap using Plotly Express\nfig = px.treemap(genus_counts, \n                 path=['family', 'genus'],  # Define the hierarchy\n                 values='species',  # Use 'species' count as the values\n                 title='Hierarchical Treemap of Families, Genera, and Species Counts')\n\n# Show the treemap\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:43.434722Z","iopub.execute_input":"2023-08-22T16:36:43.435034Z","iopub.status.idle":"2023-08-22T16:36:45.186898Z","shell.execute_reply.started":"2023-08-22T16:36:43.435008Z","shell.execute_reply":"2023-08-22T16:36:45.185706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop = ['genus_id', 'category_id', 'scientificName']\ndf_family = train_df.drop(columns=columns_to_drop)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.188618Z","iopub.execute_input":"2023-08-22T16:36:45.188975Z","iopub.status.idle":"2023-08-22T16:36:45.218201Z","shell.execute_reply.started":"2023-08-22T16:36:45.18894Z","shell.execute_reply":"2023-08-22T16:36:45.217228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# # Data preprocessing\n# batch_size = 70\n# num_classes = 52  # Replace with the actual number of categories\n# image_size = (150, 150)  # Adjust the size as needed\n# epochs = 4  # Adjust the number of training epochs\n\n# # Create an ImageDataGenerator without shuffling\n# train_datagen = ImageDataGenerator(rescale=1.0/255.0, validation_split=0.1)\n\n# # Enable shuffling for training data generator\n# train_generator = train_datagen.flow_from_dataframe(\n#     dataframe=train_data,\n#     x_col=\"file_name\",  # Column containing image file paths\n#     y_col=\"family\",   # Column containing target labels\n#     target_size=image_size,\n#     batch_size=batch_size,\n#     subset='training',\n#     class_mode='categorical',  # Specify class mode\n#     shuffle=True , # Enable shuffling during training,\n#     seed=42\n# )\n\n# # Enable shuffling for validation data generator\n# validation_generator = train_datagen.flow_from_dataframe(\n#     dataframe=train_data,\n#     x_col=\"file_name\",\n#     y_col=\"family\",\n#     target_size=image_size,\n#     batch_size=batch_size,\n#     subset='validation',\n#     class_mode='categorical',\n#     shuffle=True,# Enable shuffling during validation\n#     seed=42\n\n# )","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.221309Z","iopub.execute_input":"2023-08-22T16:36:45.221605Z","iopub.status.idle":"2023-08-22T16:36:45.227046Z","shell.execute_reply.started":"2023-08-22T16:36:45.22158Z","shell.execute_reply":"2023-08-22T16:36:45.226044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Shape of the training set\n# train_steps_per_epoch = train_generator.samples // train_generator.batch_size\n# train_set_shape = (train_steps_per_epoch, train_generator.batch_size, *train_generator.image_shape)\n# print(\"Training set shape:\", train_set_shape)\n\n# # Shape of the validation set\n# validation_steps_per_epoch = validation_generator.samples // validation_generator.batch_size\n# validation_set_shape = (validation_steps_per_epoch, validation_generator.batch_size, *validation_generator.image_shape)\n# print(\"Validation set shape:\", validation_set_shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.228458Z","iopub.execute_input":"2023-08-22T16:36:45.229447Z","iopub.status.idle":"2023-08-22T16:36:45.23905Z","shell.execute_reply.started":"2023-08-22T16:36:45.229403Z","shell.execute_reply":"2023-08-22T16:36:45.237864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras import layers, models\n\n# # Define the model\n# model = models.Sequential()\n\n# # Convolutional Layer 1\n# model.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=(150, 150, 3)))\n# model.add(layers.MaxPooling2D((2, 2)))\n\n\n# # Convolutional Layer 3\n# model.add(layers.Conv2D(128, (3, 3), activation='relu'))\n# model.add(layers.MaxPooling2D((2, 2)))\n\n# # Flatten the output\n# model.add(layers.Flatten())\n\n# # Fully Connected Layers with Dropout\n# model.add(layers.Dense(512, activation='relu'))\n# model.add(layers.Dropout(0.6))  # Dropout layer with a 50% dropout rate\n# model.add(layers.Dense(52, activation='softmax'))  # Assuming 52 classes for classification\n\n# # Compile the model\n# model.compile(optimizer='adam',\n#               loss='categorical_crossentropy',\n#               metrics=['accuracy'])\n\n# # Print the model summary\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.240682Z","iopub.execute_input":"2023-08-22T16:36:45.241332Z","iopub.status.idle":"2023-08-22T16:36:45.249383Z","shell.execute_reply.started":"2023-08-22T16:36:45.241255Z","shell.execute_reply":"2023-08-22T16:36:45.248538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# epochs = 2\n# history = model.fit(\n#     train_generator,\n#     validation_data=validation_generator,\n#     epochs=epochs\n# )\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.251628Z","iopub.execute_input":"2023-08-22T16:36:45.253121Z","iopub.status.idle":"2023-08-22T16:36:45.262666Z","shell.execute_reply.started":"2023-08-22T16:36:45.253096Z","shell.execute_reply":"2023-08-22T16:36:45.261194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_save_path = \"/kaggle/working/my_model.h5\"  # Change the path and filename as needed\n# model.save(model_save_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.263891Z","iopub.execute_input":"2023-08-22T16:36:45.264171Z","iopub.status.idle":"2023-08-22T16:36:45.272183Z","shell.execute_reply.started":"2023-08-22T16:36:45.264148Z","shell.execute_reply":"2023-08-22T16:36:45.271295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:36:45.273769Z","iopub.execute_input":"2023-08-22T16:36:45.274148Z","iopub.status.idle":"2023-08-22T16:36:51.229188Z","shell.execute_reply.started":"2023-08-22T16:36:45.274117Z","shell.execute_reply":"2023-08-22T16:36:51.228197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into train and test sets\ntrain_data, test_data = train_test_split(df_family, test_size=0.2, random_state=42)\n\n# Split the train data into train and validation sets\ntrain_data, val_data = train_test_split(train_data, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:41:57.901876Z","iopub.execute_input":"2023-08-22T16:41:57.902245Z","iopub.status.idle":"2023-08-22T16:41:58.163276Z","shell.execute_reply.started":"2023-08-22T16:41:57.902214Z","shell.execute_reply":"2023-08-22T16:41:58.162281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size =256\nnum_classes = len(df_family['family'].unique())  # Actual number of categories\nimage_size = (150, 150)  # Adjust the size as needed\nepochs = 6  # Adjust the number of training epochs\n\n# Create an ImageDataGenerator without shuffling\ntrain_datagen = ImageDataGenerator(rescale=1.0 / 255,\n                                   validation_split=0.1,\n                                   featurewise_std_normalization=True,\n                                   samplewise_std_normalization=True,\n                                   width_shift_range=0.2,\n                                   height_shift_range=0.1)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:47:09.433102Z","iopub.execute_input":"2023-08-22T16:47:09.433482Z","iopub.status.idle":"2023-08-22T16:47:09.483577Z","shell.execute_reply.started":"2023-08-22T16:47:09.43345Z","shell.execute_reply":"2023-08-22T16:47:09.482617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Enable shuffling for training data generator\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    x_col=\"file_name\",  # Column containing image file paths\n    y_col=\"family\",  # Column containing target labels\n    target_size=image_size,\n    batch_size=batch_size,\n    subset='training',\n    class_mode='categorical',  # Specify class mode\n    shuffle=True,  # Enable shuffling during training\n    seed=42\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:47:34.904193Z","iopub.execute_input":"2023-08-22T16:47:34.904561Z","iopub.status.idle":"2023-08-22T17:04:50.860596Z","shell.execute_reply.started":"2023-08-22T16:47:34.904532Z","shell.execute_reply":"2023-08-22T17:04:50.859592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Identify the top 5 classes\n# top_classes = train_data['family'].value_counts().index[:5]\n\n# Undersample only the top 5 classes to 50% of their original size\n# undersampled_data = []\n# for family_label in train_data['family'].unique():\n#     family_data = train_data[train_data['family'] == family_label]\n#     if family_label in top_classes:\n#         # Undersample the top classes to 50% of their original size\n#         undersampled_family_data = resample(family_data, replace=False, n_samples=int(0.5 * len(family_data)), random_state=42)\n#     else:\n#         # Keep the other classes intact\n#         undersampled_family_data = family_data\n#     undersampled_data.append(undersampled_family_data)\n\n# # Concatenate the undersampled data back together\n# undersampled_train_data = pd.concat(undersampled_data)\n\n\n# Enable shuffling for validation data generator\nvalidation_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    x_col=\"file_name\",\n    y_col=\"family\",\n    target_size=image_size,\n    batch_size=batch_size,\n    subset='validation',\n    class_mode='categorical',\n    shuffle=True,  # Enable shuffling during validation\n    seed=42\n)\n\n# Now you have the correct train_data, val_data, and test_data splits.\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T17:04:50.862433Z","iopub.execute_input":"2023-08-22T17:04:50.862857Z","iopub.status.idle":"2023-08-22T17:12:25.921954Z","shell.execute_reply.started":"2023-08-22T17:04:50.862825Z","shell.execute_reply":"2023-08-22T17:12:25.92094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming your data splits are DataFrames\ntrain_data.to_csv('/kaggle/working/train_data.csv', index=False)\nval_data.to_csv('/kaggle/working/val_data.csv', index=False)\ntest_data.to_csv('/kaggle/working/test_data.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:41:45.071487Z","iopub.status.idle":"2023-08-22T16:41:45.074065Z","shell.execute_reply.started":"2023-08-22T16:41:45.073803Z","shell.execute_reply":"2023-08-22T16:41:45.073829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the CNN model\nmodel = models.Sequential()\n\n# Add a convolutional layer\nmodel.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=(image_size[0], image_size[1], 3)))\nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# Add another convolutional layer\nmodel.add(layers.Conv2D(64, (3, 3), activation='relu'))\nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# Add a fully connected layer\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(64, activation='relu'))\n\n# Add the output layer with the appropriate number of units (classes)\nmodel.add(layers.Dense(num_classes, activation='softmax'))\n\n# Compile the model\nmodel.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-08-22T17:13:46.136157Z","iopub.execute_input":"2023-08-22T17:13:46.137167Z","iopub.status.idle":"2023-08-22T17:13:48.918109Z","shell.execute_reply.started":"2023-08-22T17:13:46.137131Z","shell.execute_reply":"2023-08-22T17:13:48.917097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nepochs = 6\nhistory = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=epochs\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T17:13:53.369199Z","iopub.execute_input":"2023-08-22T17:13:53.369915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the validation data\nvalidation_loss, validation_acc = model.evaluate(validation_generator)\nprint(f\"Validation accuracy: {validation_acc}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:41:45.088711Z","iopub.status.idle":"2023-08-22T16:41:45.089474Z","shell.execute_reply.started":"2023-08-22T16:41:45.089222Z","shell.execute_reply":"2023-08-22T16:41:45.089245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test data\ntest_datagen = ImageDataGenerator(rescale=1.0 / 255.0)\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_data,\n    x_col=\"file_name\",\n    y_col=\"family\",\n    target_size=image_size,\n    batch_size=batch_size,\n    class_mode='categorical'\n)\n\ntest_loss, test_acc = model.evaluate(test_generator)\nprint(f\"Test accuracy: {test_acc}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T16:41:45.09085Z","iopub.status.idle":"2023-08-22T16:41:45.091605Z","shell.execute_reply.started":"2023-08-22T16:41:45.091373Z","shell.execute_reply":"2023-08-22T16:41:45.091395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}