{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-04-19T05:41:07.309736Z","iopub.execute_input":"2022-04-19T05:41:07.310665Z","iopub.status.idle":"2022-04-19T05:41:08.472199Z","shell.execute_reply.started":"2022-04-19T05:41:07.310575Z","shell.execute_reply":"2022-04-19T05:41:08.471660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/sorghum-id-fgvc-9/train_cultivar_mapping.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:05:55.784225Z","iopub.execute_input":"2022-04-19T06:05:55.784492Z","iopub.status.idle":"2022-04-19T06:05:55.824072Z","shell.execute_reply.started":"2022-04-19T06:05:55.784447Z","shell.execute_reply":"2022-04-19T06:05:55.823337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:05:57.630917Z","iopub.execute_input":"2022-04-19T06:05:57.631184Z","iopub.status.idle":"2022-04-19T06:05:57.665826Z","shell.execute_reply.started":"2022-04-19T06:05:57.631154Z","shell.execute_reply":"2022-04-19T06:05:57.664795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing null values\ndf_train = df_train.dropna(axis =0, how = 'any')\n#cheking for missing data \ndf_train.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:05:59.933108Z","iopub.execute_input":"2022-04-19T06:05:59.933511Z","iopub.status.idle":"2022-04-19T06:05:59.945611Z","shell.execute_reply.started":"2022-04-19T06:05:59.933477Z","shell.execute_reply":"2022-04-19T06:05:59.944835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:06:01.890745Z","iopub.execute_input":"2022-04-19T06:06:01.890963Z","iopub.status.idle":"2022-04-19T06:06:01.920330Z","shell.execute_reply.started":"2022-04-19T06:06:01.890937Z","shell.execute_reply":"2022-04-19T06:06:01.919452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking for duplicates\ndf_train['image'].duplicated().any()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T05:57:44.160442Z","iopub.execute_input":"2022-04-19T05:57:44.160688Z","iopub.status.idle":"2022-04-19T05:57:44.167966Z","shell.execute_reply.started":"2022-04-19T05:57:44.160665Z","shell.execute_reply":"2022-04-19T05:57:44.167030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['image_path'] = '../input/sorghum-id-fgvc-9/train_images/'+df_train['image']\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:48.833635Z","iopub.execute_input":"2022-04-19T06:14:48.834387Z","iopub.status.idle":"2022-04-19T06:14:48.851535Z","shell.execute_reply.started":"2022-04-19T06:14:48.834355Z","shell.execute_reply":"2022-04-19T06:14:48.850797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 25))\nsns.countplot(y=\"cultivar\", data=df_train,)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:51.872926Z","iopub.execute_input":"2022-04-19T06:14:51.873285Z","iopub.status.idle":"2022-04-19T06:14:53.025558Z","shell.execute_reply.started":"2022-04-19T06:14:51.873255Z","shell.execute_reply":"2022-04-19T06:14:53.024809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Maximum number of samples available for a category')\nmax(df_train['cultivar'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:53.027106Z","iopub.execute_input":"2022-04-19T06:14:53.027269Z","iopub.status.idle":"2022-04-19T06:14:53.034315Z","shell.execute_reply.started":"2022-04-19T06:14:53.027248Z","shell.execute_reply":"2022-04-19T06:14:53.033497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Minimum number of samples available for a category')\nmin(df_train['cultivar'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:53.035765Z","iopub.execute_input":"2022-04-19T06:14:53.036032Z","iopub.status.idle":"2022-04-19T06:14:53.048424Z","shell.execute_reply.started":"2022-04-19T06:14:53.035995Z","shell.execute_reply":"2022-04-19T06:14:53.048023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = os.listdir('../input/sorghum-id-fgvc-9/test')\ndf_test = pd.DataFrame(test, columns =['image'])\ndf_test['image_path'] = '../input/sorghum-id-fgvc-9/test/'+df_test['image']","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:53.049548Z","iopub.execute_input":"2022-04-19T06:14:53.049710Z","iopub.status.idle":"2022-04-19T06:14:53.068774Z","shell.execute_reply.started":"2022-04-19T06:14:53.049688Z","shell.execute_reply":"2022-04-19T06:14:53.068223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:53.300557Z","iopub.execute_input":"2022-04-19T06:14:53.300825Z","iopub.status.idle":"2022-04-19T06:14:53.350864Z","shell.execute_reply.started":"2022-04-19T06:14:53.300790Z","shell.execute_reply":"2022-04-19T06:14:53.349870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualise","metadata":{}},{"cell_type":"code","source":"# plotting image by image id for single image\ndef  visualize(image_id):\n    \n    path = df_train.loc[df_train['image'] == image_id, 'image_path'].iloc[0]\n    cultivar = df_train.loc[df_train['image'] == image_id, 'cultivar'].iloc[0]\n    \n    plt.figure(figsize=(10, 10))\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.title(f\"CULTIVAR: {cultivar} \\n Image:{image_id}\", fontsize=10)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:14:54.863526Z","iopub.execute_input":"2022-04-19T06:14:54.863831Z","iopub.status.idle":"2022-04-19T06:14:54.872273Z","shell.execute_reply.started":"2022-04-19T06:14:54.863786Z","shell.execute_reply":"2022-04-19T06:14:54.871634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize('2017-06-16__12-24-20-930.png')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:27:23.071303Z","iopub.execute_input":"2022-04-19T06:27:23.071671Z","iopub.status.idle":"2022-04-19T06:27:23.673586Z","shell.execute_reply.started":"2022-04-19T06:27:23.071637Z","shell.execute_reply":"2022-04-19T06:27:23.672941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting image by image id for multiple image. When an image id is given, function returns 20 images \n#in same class\ndef visualize_many(image_id):\n    train_image_path = \"../input/sorghum-id-fgvc-9/train_images/\"\n    \n    cultivar = df_train.loc[df_train['image'] == image_id, 'cultivar'].iloc[0]\n    df = df_train.loc[df_train['cultivar'] == cultivar]\n    \n    \n                                                           \n    if  df['image'].count()< 20:\n        x=df['image'].tolist()\n        plt.figure(figsize=(18, 18))\n        for i, j in zip(x, range(20)):       \n            plt.subplot(5, 4, j + 1)\n            path = df.loc[df['image'] == i, 'image'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            cultivar = df.loc[df['image'] == i, 'cultivar'].iloc[0]\n            imageid= df.loc[df['image'] == i, 'image'].iloc[0]\n            plt.title(f\"CULTIVAR: {cultivar} \\n Image:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n        plt.show()\n    else:\n        x = np.random.choice(df['image'], 20, replace=False).tolist()\n        plt.figure(figsize=(18, 18))\n        for i, j in zip(x, range(20)):       \n            plt.subplot(5, 4, j + 1)\n            path = df.loc[df['image'] == i, 'image'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            cultivar = df.loc[df['image'] == i, 'cultivar'].iloc[0]\n            imageid= df.loc[df['image'] == i, 'image'].iloc[0]\n            plt.title(f\"CULTIVAR: {cultivar} \\n Image:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n    \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:28:55.521936Z","iopub.execute_input":"2022-04-19T06:28:55.522198Z","iopub.status.idle":"2022-04-19T06:28:55.538270Z","shell.execute_reply.started":"2022-04-19T06:28:55.522170Z","shell.execute_reply":"2022-04-19T06:28:55.537638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_many('2017-06-16__12-24-20-930.png')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:28:56.103156Z","iopub.execute_input":"2022-04-19T06:28:56.103359Z","iopub.status.idle":"2022-04-19T06:29:01.643975Z","shell.execute_reply.started":"2022-04-19T06:28:56.103337Z","shell.execute_reply":"2022-04-19T06:29:01.643149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot image by cultivar name\ndef visualize_cultivar(name):\n    train_image_path = \"../input/sorghum-id-fgvc-9/train_images/\"\n     \n    df = df_train.loc[df_train['cultivar'] == name]\n    \n    x = np.random.choice(df['image'], 5, replace=False).tolist()\n    plt.figure(figsize=(18, 18))\n    for i, j in zip(x, range(15)):       \n            plt.subplot(1, 5, j + 1)\n            path = df.loc[df['image'] == i, 'image'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            cultivar = df.loc[df['image'] == i, 'cultivar'].iloc[0]\n            imageid= df.loc[df['image'] == i, 'image'].iloc[0]\n            plt.title(f\"CULTIVAR: {cultivar} \\n Image:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n            #plt.savefig('saved_figure.png')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:33:22.194069Z","iopub.execute_input":"2022-04-19T06:33:22.194291Z","iopub.status.idle":"2022-04-19T06:33:22.201687Z","shell.execute_reply.started":"2022-04-19T06:33:22.194267Z","shell.execute_reply":"2022-04-19T06:33:22.201091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:31:57.580630Z","iopub.execute_input":"2022-04-19T06:31:57.581205Z","iopub.status.idle":"2022-04-19T06:31:57.591748Z","shell.execute_reply.started":"2022-04-19T06:31:57.581167Z","shell.execute_reply":"2022-04-19T06:31:57.590992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_cultivar('PI_154987')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:33:26.782268Z","iopub.execute_input":"2022-04-19T06:33:26.783271Z","iopub.status.idle":"2022-04-19T06:33:28.199017Z","shell.execute_reply.started":"2022-04-19T06:33:26.783231Z","shell.execute_reply":"2022-04-19T06:33:28.198226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = df_train['cultivar'].unique()\nfor j in i:\n    visualize_cultivar(j)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T06:35:04.853290Z","iopub.execute_input":"2022-04-19T06:35:04.853628Z","iopub.status.idle":"2022-04-19T06:37:18.274307Z","shell.execute_reply.started":"2022-04-19T06:35:04.853596Z","shell.execute_reply":"2022-04-19T06:37:18.273704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}