{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os, cv2\nfrom PIL import Image\n\nimport numpy as np\nimport json\nimport pandas as pd\nfrom tqdm import tqdm\nimport random\n\nimport tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inp_fol=\"../input/cassava-leaf-disease-classification\"\nos.listdir(inp_fol)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!du -sh $inp_fol/\"train_tfrecords\"\n!du -sh $inp_fol/\"test_tfrecords\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv_path=os.path.join(inp_fol, 'train.csv')\nsample_csv_path=os.path.join(inp_fol, 'sample_submission.csv')\n\ntrain_df=pd.read_csv(train_csv_path)\nsample_df=pd.read_csv(sample_csv_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"json_path=os.path.join(inp_fol, 'label_num_to_disease_map.json')\nfd=open(json_path, \"r\")\nlabel_map=json.load(fd)\nfd.close()\n\nlabel_map","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Csv Analysis"},{"metadata":{},"cell_type":"markdown","source":"### Train csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Unique labels in train csv\nlabels=train_df.iloc[:,1]\nlab_unique=labels.unique()\nlab_unique","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(labels)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Sample csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train data Analysis"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tf_fol=os.path.join(inp_fol, 'train_tfrecords')\ntest_tf_fol=os.path.join(inp_fol, 'test_tfrecords')\n\n!ls $train_tf_fol | wc -l\n!ls $test_tf_fol | wc -l\n\n!du -sh $train_tf_fol\n!du -sh $test_tf_fol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls $train_tf_fol ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls $test_tf_fol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_im_fol=os.path.join(inp_fol, 'train_images')\n\n!ls $train_im_fol | wc -l\n!du -sh $train_im_fol\n!ls $train_im_fol | head -5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_im_fol=os.path.join(inp_fol, 'test_images')\n\n!ls $test_im_fol | wc -l\n!du -sh $test_im_fol\n!ls $test_im_fol | head -5","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train folder images"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_im_ll=os.listdir(train_im_fol)\nlen(train_im_ll)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"H,W=[],[]\n\nfor name in tqdm(train_im_ll):\n    path=os.path.join(train_im_fol, name)\n    w,h = Image.open(path).size\n    H.append(h)\n    W.append(w)\n\nnp.unique(H), np.unique(W)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Sample train images"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images\n\nrandom.seed(0)\nsample_train_ll = random.sample(train_im_ll, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_train_ll):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Images for label 0"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images for disease label 0\n\nrandom.seed(0)\nsample_lab_0=train_df[train_df.iloc[:,1]==0]['image_id'][:16].tolist()\nsample_imgs = random.sample(sample_lab_0, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_imgs):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Images for  label 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images for disease label 1\n\nrandom.seed(0)\nsample_lab=train_df[train_df.iloc[:,1]==1]['image_id'][:16].tolist()\nsample_imgs = random.sample(sample_lab, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_imgs):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Images for label 2"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images for disease label 2\n\nrandom.seed(0)\nsample_lab=train_df[train_df.iloc[:,1]==2]['image_id'][:16].tolist()\nsample_imgs = random.sample(sample_lab, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_imgs):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Images for  label 3"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images for disease label 3\n\nrandom.seed(0)\nsample_lab=train_df[train_df.iloc[:,1]==3]['image_id'][:16].tolist()\nsample_imgs = random.sample(sample_lab, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_imgs):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Images for  label 4"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sample train images for disease label 4\n\nrandom.seed(0)\nsample_lab=train_df[train_df.iloc[:,1]==4]['image_id'][:16].tolist()\nsample_imgs = random.sample(sample_lab, 16)\n\nfig = plt.figure(figsize=(14,14))\nfig.subplots_adjust(hspace=0.1, wspace=0.1)\n\nfor i,name in enumerate(sample_imgs):\n    path=os.path.join(train_im_fol, name)\n    im=Image.open(path)\n    \n    ax = fig.add_subplot(4, 4, i+1)\n    ax.axis(\"off\")\n    ax.imshow(im)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train tf-record Analysis"},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls $train_tf_fol | wc -l\n!ls $test_tf_fol | wc -l\n\n!du -sh $train_tf_fol\n!du -sh $test_tf_fol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls $train_tf_fol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import io\n\n# Create a dictionary describing the features.\nimage_feature_description = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'image_name': tf.io.FixedLenFeature([], tf.string),\n    'target': tf.io.FixedLenFeature([], tf.int64),\n}\n\ndef _parse_image_function(example_proto):\n  return tf.io.parse_single_example(example_proto, image_feature_description)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Read single image from tf-record\n\ntrain_tf1=os.path.join(train_tf_fol, \"ld_train00-1338.tfrec\")\nraw_dataset = tf.data.TFRecordDataset(train_tf1)\nparsed_image_dataset = raw_dataset.map(_parse_image_function)\n\nfor image_features in parsed_image_dataset.take(1):\n    image = image_features['image'].numpy()\n    image = Image.open(io.BytesIO(image))\n    \n    name = image_features['image_name'].numpy().decode('ascii')\n    label = image_features['target'].numpy()                    \n    \n    print(name, label)\n    plt.imshow(image); plt.show()\n    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check data mapping b/w tf-record and img folder\n\ntf_ll = []\nc=0\ntf_rec_map_count=0\n\nfor tf_name in tqdm(os.listdir(train_tf_fol)):\n    tf_path=os.path.join(train_tf_fol, tf_name)\n    raw_dataset = tf.data.TFRecordDataset(tf_path)\n    parsed_image_dataset = raw_dataset.map(_parse_image_function)\n\n    temp=0\n    for image_features in parsed_image_dataset:\n        c += 1\n        temp+=1\n        \n        name = image_features['image_name'].numpy().decode('ascii')\n        label = image_features['target'].numpy()   \n        \n        tf_label=train_df[ (train_df['image_id']==name) ]['label'].ravel()[0]\n        if tf_label==label:\n            tf_rec_map_count+=1\n\n    tf_ll.append(temp)\n    \nc, np.sum(np.array(tf_ll)), tf_rec_map_count","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}