{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Problem Statement\n\nPlant Pathology 2021 - FGVC8 is a [Kaggle competition](https://www.kaggle.com/c/plant-pathology-2021-fgvc8) launched on march 15 2021 and closed on mai 27 2021. \n\n## Specific Objectives\nThe main objective of the competition is to develop machine learning-based models to accurately classify a given leaf image from the test dataset to a particular disease category, and to identify an individual disease from multiple disease symptoms on a single leaf image.","metadata":{}},{"cell_type":"markdown","source":"### Librairies","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nimport matplotlib.image as img","metadata":{"execution":{"iopub.status.busy":"2021-06-01T08:26:42.617706Z","iopub.execute_input":"2021-06-01T08:26:42.618201Z","iopub.status.idle":"2021-06-01T08:26:42.624617Z","shell.execute_reply.started":"2021-06-01T08:26:42.618066Z","shell.execute_reply":"2021-06-01T08:26:42.623657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading data from directory\ntrain = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\n\ntrain.head()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:42.626446Z","iopub.execute_input":"2021-06-01T08:26:42.627051Z","iopub.status.idle":"2021-06-01T08:26:42.669610Z","shell.execute_reply.started":"2021-06-01T08:26:42.627008Z","shell.execute_reply":"2021-06-01T08:26:42.668663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# looking at basic infos\ntrain.info()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:42.671567Z","iopub.execute_input":"2021-06-01T08:26:42.671852Z","iopub.status.idle":"2021-06-01T08:26:42.688355Z","shell.execute_reply.started":"2021-06-01T08:26:42.671824Z","shell.execute_reply":"2021-06-01T08:26:42.687113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the number of images\nprint(\"Number of images:\",len(train))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:42.689801Z","iopub.execute_input":"2021-06-01T08:26:42.690249Z","iopub.status.idle":"2021-06-01T08:26:42.695289Z","shell.execute_reply.started":"2021-06-01T08:26:42.690212Z","shell.execute_reply":"2021-06-01T08:26:42.694425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample images","metadata":{}},{"cell_type":"code","source":"# preparing the path to images\ntrain[\"path\"] = \"../input/plant-pathology-2021-fgvc8/train_images/\" + train[\"image\"]\n\n# taking a sample of the dataset\ndata_sample = train.sample(16)\n\n# Showing image sample\nplt.figure(figsize=(14,9))\nn=1\nfor i in data_sample.index :\n    plt.subplot(4,4,n)\n    \n    testImage = img.imread(data_sample[\"path\"][i])\n\n    # displaying the image\n    plt.imshow(testImage)\n\n    plt.title(data_sample[\"labels\"][i])\n    plt.axis(\"off\")\n    n+=1\n_ = plt.suptitle(\"Images sample\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:42.698582Z","iopub.execute_input":"2021-06-01T08:26:42.699035Z","iopub.status.idle":"2021-06-01T08:26:57.837978Z","shell.execute_reply.started":"2021-06-01T08:26:42.698985Z","shell.execute_reply":"2021-06-01T08:26:57.836929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Labels exploration\n### Raw labels","metadata":{}},{"cell_type":"code","source":"# Preparing labels comlumns to exploration\n\ndef get_list(string):\n    list = []\n    list.append(string)\n    return list\n\ntrain[\"labels_long\"] = train[\"labels\"].apply(get_list)\n\ntrain[\"labels\"] = train[\"labels\"].astype(\"str\")\n\ntrain[\"labels\"] = train[\"labels\"].str.split(\" \")\n\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:57.839930Z","iopub.execute_input":"2021-06-01T08:26:57.840420Z","iopub.status.idle":"2021-06-01T08:26:57.964895Z","shell.execute_reply.started":"2021-06-01T08:26:57.840362Z","shell.execute_reply":"2021-06-01T08:26:57.963837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\n\ndef labels_plot(labels):\n    \"\"\" Takes the labels columns and plot a bar graph of values,\n    print number of unique labels and their respective ratio.\n    arg : label columns  \"\"\"   \n    \n    # converting each label into a column\n    mlb = MultiLabelBinarizer()\n\n    labels = mlb.fit_transform(train[labels])\n\n    labels_df =  pd.DataFrame(labels,columns=mlb.classes_, index=train.index)\n    \n    print(\"Number of labels:\",len(mlb.classes_))\n    print(mlb.classes_)\n    \n    # summing the \n    labels_count = labels_df.sum()\n\n    fig, ax = plt.subplots()\n\n    # Example data\n    labs = labels_count.index\n    y_pos = np.arange(len(labs))\n    counts = labels_count\n\n\n    ax.barh(y_pos, counts)\n    ax.set_yticks(y_pos)\n    ax.set_yticklabels(labs)\n    ax.invert_yaxis()  # labels read top-to-bottom\n    ax.set_xlabel('Nb')\n    ax.set_title('Number of labels')\n\n    plt.show()\n\n    print(\"Ratio of labels\")\n    \n    print(labels_count / len(train))\n    \nlabels_plot(\"labels_long\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-01T08:26:57.966350Z","iopub.execute_input":"2021-06-01T08:26:57.966714Z","iopub.status.idle":"2021-06-01T08:26:58.451633Z","shell.execute_reply.started":"2021-06-01T08:26:57.966680Z","shell.execute_reply":"2021-06-01T08:26:58.450758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Labels","metadata":{}},{"cell_type":"code","source":"labels_plot(\"labels\")","metadata":{"execution":{"iopub.status.busy":"2021-06-01T08:26:58.452895Z","iopub.execute_input":"2021-06-01T08:26:58.453322Z","iopub.status.idle":"2021-06-01T08:26:58.597322Z","shell.execute_reply.started":"2021-06-01T08:26:58.453291Z","shell.execute_reply":"2021-06-01T08:26:58.596493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA conclusion\n\nWe have 12 labels combinaisons made out of 6 labels.\nIn the training phase, our choice will be to use a multilabels classification approach. Therefore to predict each label independently.","metadata":{}}]}