{"cells": [{"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "af27abfbfd8f5e8816198d20e2e10ab07a38cd38", "_cell_guid": "5917d988-a42f-4761-91f8-303595dd3c4d"}, "cell_type": "markdown", "source": "# Visual Data Analysis", "execution_count": null}, {"outputs": [], "metadata": {"_uuid": "ebce62cdc6497ceb8a6462307f5599bffce7cd4b", "_execution_state": "idle", "trusted": false, "_cell_guid": "00aa24f1-3983-42f2-834b-4a5a2dcc735a"}, "cell_type": "code", "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For exaample, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pylab as plt\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.", "execution_count": 2}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "a0bcf5184d1e1e8d536ead35f6724e03e3feea2a", "_cell_guid": "11c229c4-70b1-4124-804a-50635184d5ce"}, "cell_type": "markdown", "source": "How many images we have in train and test datasets", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "dc20ead3f35eada7394078da3c0fb9c1ae217802", "trusted": false, "_cell_guid": "1a1389e6-4e5a-4eaf-ac85-42b0777d270a"}, "cell_type": "code", "source": "!ls ../input/train/ | wc -l\n!ls ../input/train_masks/ | wc -l\n!ls ../input/test/ | wc -l", "execution_count": 3}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "117327273efa5f1dab355ca5f9b3d0c601e2f73c", "_cell_guid": "ee62ab64-03ef-4d90-a9ca-06c5f9dc2089"}, "cell_type": "markdown", "source": "For example, train filenames looks like", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "07cd86b2d25c4b2e303d9e4a786b95d496144a93", "trusted": false, "_cell_guid": "a50255ce-117d-477b-a0a2-55f7e0605eb4"}, "cell_type": "code", "source": "!ls ../input/train/ | grep c_01.jpg", "execution_count": 4}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "8db16b44cb32beebebf7c54b5945274de143fd5b", "trusted": false, "_cell_guid": "0898fbf5-5525-4cb7-886c-d35088c2c2f4"}, "cell_type": "code", "source": "import os \nfrom glob import glob\n\nINPUT_PATH = '../input'\nDATA_PATH = INPUT_PATH\nTRAIN_DATA = os.path.join(DATA_PATH, \"train\")\nTRAIN_MASKS_DATA = os.path.join(DATA_PATH, \"train_masks\")\nTEST_DATA = os.path.join(DATA_PATH, \"test\")\nTRAIN_MASKS_CSV_FILEPATH = os.path.join(DATA_PATH, \"train_masks.csv\")\nMETADATA_CSV_FILEPATH = os.path.join(DATA_PATH, \"metadata.csv\")\n\nTRAIN_MASKS_CSV = pd.read_csv(TRAIN_MASKS_CSV_FILEPATH)\nMETADATA_CSV = pd.read_csv(METADATA_CSV_FILEPATH)", "execution_count": 5}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "78cb34bddedb2ce684c745df3ca5a56d68305e67", "trusted": false, "_cell_guid": "6a6b6fd3-3bb7-4ee3-9770-8f0509e69960"}, "cell_type": "code", "source": "train_files = glob(os.path.join(TRAIN_DATA, \"*.jpg\"))\ntrain_ids = [s[len(TRAIN_DATA)+1:-4] for s in train_files]\n\ntest_files = glob(os.path.join(TEST_DATA, \"*.jpg\"))\ntest_ids = [s[len(TEST_DATA)+1:-4] for s in test_files]", "execution_count": 6}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "0d7324326141c456c1433b156ff30ba513a24f01", "trusted": false, "_cell_guid": "c67d33f0-752d-43e6-a409-104dab40d415"}, "cell_type": "code", "source": "def get_filename(image_id, image_type):\n    check_dir = False\n    if \"Train\" == image_type:\n        ext = 'jpg'\n        data_path = TRAIN_DATA\n        suffix = ''\n    elif \"Train_mask\" in image_type:\n        ext = 'gif'\n        data_path = TRAIN_MASKS_DATA\n        suffix = '_mask'\n    elif \"Test\" in image_type:\n        ext = 'jpg'\n        data_path = TEST_DATA\n        suffix = ''\n    else:\n        raise Exception(\"Image type '%s' is not recognized\" % image_type)\n\n    if check_dir and not os.path.exists(data_path):\n        os.makedirs(data_path)\n\n    return os.path.join(data_path, \"{}{}.{}\".format(image_id, suffix, ext))", "execution_count": 7}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "9660dc5ffcc79be3160809e1c7db1094710ab20f", "trusted": false, "_cell_guid": "28890618-c431-45f9-ba58-c12a011947bb"}, "cell_type": "code", "source": "import cv2\nfrom PIL import Image\n\n\ndef get_image_data(image_id, image_type, **kwargs):\n    if 'mask' in image_type:\n        img = _get_image_data_pil(image_id, image_type, **kwargs)\n    else:\n        img = _get_image_data_opencv(image_id, image_type, **kwargs)\n    return img\n\ndef _get_image_data_opencv(image_id, image_type, **kwargs):\n    fname = get_filename(image_id, image_type)\n    img = cv2.imread(fname)\n    assert img is not None, \"Failed to read image : %s, %s\" % (image_id, image_type)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img\n\n\ndef _get_image_data_pil(image_id, image_type, return_exif_md=False, return_shape_only=False):\n    fname = get_filename(image_id, image_type)\n    try:\n        img_pil = Image.open(fname)\n    except Exception as e:\n        assert False, \"Failed to read image : %s, %s. Error message: %s\" % (image_id, image_type, e)\n\n    if return_shape_only:\n        return img_pil.size[::-1] + (len(img_pil.getbands()),)\n\n    img = np.asarray(img_pil)\n    assert isinstance(img, np.ndarray), \"Open image is not an ndarray. Image id/type : %s, %s\" % (image_id, image_type)\n    if not return_exif_md:\n        return img\n    else:\n        return img, img_pil._getexif()", "execution_count": 9}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "fea2eb52d372ce2e6981f70c492b03c2c5a75f03", "_cell_guid": "eee838c7-f3c6-441d-bb12-66669281e7bd"}, "cell_type": "markdown", "source": "## Display a single car with its mask", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "c8c60f3a1a2fa876528706ced825e75405a593c2", "trusted": false, "_cell_guid": "f85aa9c4-3ac8-41d5-910b-484c1bc8055c"}, "cell_type": "code", "source": "image_id = train_ids[0]\n\nplt.figure(figsize=(20, 20))\nimg = get_image_data(image_id, \"Train\")\nmask = get_image_data(image_id, \"Train_mask\")\nimg_masked = cv2.bitwise_and(img, img, mask=mask)\n\nprint(\"Image shape: {} | image type: {} | mask shape: {} | mask type: {}\".format(img.shape, img.dtype, mask.shape, mask.dtype) )\n\nplt.subplot(131)\nplt.imshow(img)\nplt.subplot(132)\nplt.imshow(mask)\nplt.subplot(133)\nplt.imshow(img_masked)", "execution_count": 10}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "0c860ae201cf6712090c28e72d5b77bc07d3d24a", "_cell_guid": "2f5726b7-db8c-4a2d-860d-b6f2593f13c2"}, "cell_type": "markdown", "source": "## Display 500 random cars from train dataset", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "569c9abc83843a081a50d20267f5b3c0a132da06", "trusted": false, "_cell_guid": "2ac5a582-5f68-4a8e-8198-7fa3a5393629"}, "cell_type": "code", "source": "_train_ids = list(train_ids)\nnp.random.shuffle(_train_ids)\n_train_ids = _train_ids[:500]\ntile_size = (256, 256)\nn = 8\n\nm = int(np.ceil(len(_train_ids) * 1.0 / n))\ncomplete_image = np.zeros((m*(tile_size[0]+2), n*(tile_size[1]+2), 3), dtype=np.uint8)\n\ncounter = 0\nfor i in range(m):\n    ys = i*(tile_size[1] + 2)\n    ye = ys + tile_size[1]\n    for j in range(n):\n        xs = j*(tile_size[0] + 2)\n        xe = xs + tile_size[0]\n        if counter == len(_train_ids):\n            break\n        image_id = _train_ids[counter]; counter+=1\n        img = get_image_data(image_id, 'Train')\n        img = cv2.resize(img, dsize=tile_size)\n        img = cv2.putText(img, image_id, (5,img.shape[0] - 5), cv2.FONT_HERSHEY_PLAIN, 1.5, (0, 255, 0), thickness=2)\n        complete_image[ys:ye, xs:xe, :] = img[:,:,:]\n    if counter == len(_train_ids):\n        break    ", "execution_count": 11}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "b2a86ac00e228498f727ea8b7fb8a69e6870a039", "trusted": false, "_cell_guid": "959b901d-0b11-4fbe-b2df-b195e8e2cd7d"}, "cell_type": "code", "source": "m = complete_image.shape[0] / (tile_size[0] + 2)\nk = 8\nn = int(np.ceil(m / k))\nfor i in range(n):\n    plt.figure(figsize=(20, 20))\n    ys = i*(tile_size[0] + 2)*k\n    ye = min((i+1)*(tile_size[0] + 2)*k, complete_image.shape[0])\n    plt.imshow(complete_image[ys:ye,:,:])\n    plt.title(\"Training dataset, part %i\" % i)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "f7f0e7082cdd0e30ff1f5ccb52354719621d2edf", "_cell_guid": "df9d583e-3dd3-4bea-8b10-ec69cda12566"}, "cell_type": "markdown", "source": "## How many different car in all datasets:", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "56c30119152db97ba6d629f4c8091ddf02a792d4", "trusted": false, "_cell_guid": "514b14fb-0b13-4087-acf2-aded1c418660"}, "cell_type": "code", "source": "len(METADATA_CSV['id'].unique()), len(METADATA_CSV['id'])", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "883eacc36056640f360ce30710cf657c7ae3e625", "_cell_guid": "3357086e-81a4-4a9e-a2a9-41e9a6d14702"}, "cell_type": "markdown", "source": "## How many different cars in train dataset:", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "3a299072a4eb752fefa37642320d559580987937", "trusted": false, "_cell_guid": "b2ea8550-881e-4f2e-b095-9449b0491397"}, "cell_type": "code", "source": "TRAIN_MASKS_CSV['id'] = TRAIN_MASKS_CSV['img'].apply(lambda x: x[:-7])\nlen(TRAIN_MASKS_CSV['id'].unique()), len(TRAIN_MASKS_CSV['id'].unique()) * 16\n", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "2bdf60a43dff35ce579618fd13d3e193ee6b4de6", "trusted": false, "_cell_guid": "13369364-1873-4050-9717-9e68691effc1"}, "cell_type": "code", "source": "all_318_car_ids = TRAIN_MASKS_CSV['id'].unique()", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "9fd191a30c714e4159295fe13865bd18df71e9a6", "_cell_guid": "9c83cf7f-d5b4-4fb3-9ec7-950780234520"}, "cell_type": "markdown", "source": "## Display all 318 cars at '03' angle from train dataset", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "b0f22c0aa7f928dc72ab46f686ee2f8db004d2d8", "trusted": false, "_cell_guid": "8ef885a3-4c82-4e2a-8638-8834111443f6"}, "cell_type": "code", "source": "all_318_cars_image_ids = [_id + '_03' for _id in all_318_car_ids]", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "915c45608b6c2d43b04eb342877e3d151d737e25", "trusted": false, "_cell_guid": "d9f52bb8-b4bf-420c-b79c-0d506def8bbc"}, "cell_type": "code", "source": "_train_ids = list(all_318_cars_image_ids)\ntile_size = (256, 256)\nn = 8\n\nm = int(np.ceil(len(_train_ids) * 1.0 / n))\ncomplete_image = np.zeros((m*(tile_size[0]+2), n*(tile_size[1]+2), 3), dtype=np.uint8)\n\ncounter = 0\nfor i in range(m):\n    ys = i*(tile_size[1] + 2)\n    ye = ys + tile_size[1]\n    for j in range(n):\n        xs = j*(tile_size[0] + 2)\n        xe = xs + tile_size[0]\n        if counter == len(_train_ids):\n            break\n        image_id = _train_ids[counter]; counter+=1\n        img = get_image_data(image_id, 'Train')\n        img = cv2.resize(img, dsize=tile_size)\n        img = cv2.putText(img, image_id, (5,img.shape[0] - 5), cv2.FONT_HERSHEY_PLAIN, 1.5, (0, 255, 0), thickness=2)\n        complete_image[ys:ye, xs:xe, :] = img[:,:,:]\n    if counter == len(_train_ids):\n        break   ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "ead1366bd420bdb4a1a05cf7e7a99e35d31e8585", "trusted": false, "_cell_guid": "03e1397e-5850-4ecf-af29-75231f186a2b"}, "cell_type": "code", "source": "m = complete_image.shape[0] / (tile_size[0] + 2)\nk = 8\nn = int(np.ceil(m / k))\nfor i in range(n):\n    plt.figure(figsize=(20, 20))\n    ys = i*(tile_size[0] + 2)*k\n    ye = min((i+1)*(tile_size[0] + 2)*k, complete_image.shape[0])\n    plt.imshow(complete_image[ys:ye,:,:])\n    plt.title(\"All 318 cars from train dataset, part %i\" % i)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "900fdca3bb080cc41d562b4b15c386d7b816300c", "_cell_guid": "017e56d4-f269-42d6-9a80-30d9a55f3b26"}, "cell_type": "markdown", "source": "## Which cars are present in the train dataset:", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "7d8bdfa3a195d66779b239d2c3756879978e5b12", "trusted": false, "_cell_guid": "73274f4d-cc4f-45fe-8d57-8e729e4eda2b"}, "cell_type": "code", "source": "METADATA_CSV.index = METADATA_CSV['id']\ntrain_metadata_csv = METADATA_CSV.loc[TRAIN_MASKS_CSV['id'].unique(),:]", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "2f6dd943cd4e767db8146b96ef7b73b216451c27", "trusted": false, "_cell_guid": "2d038e87-ff40-4a05-952e-c5acd7a2e27d"}, "cell_type": "code", "source": "import seaborn as sns\nsns.countplot(y=\"make\", data=train_metadata_csv, palette=\"Greens_d\")", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "184e57daf0c362f258d4db698912b7cffe5b0213", "_cell_guid": "7245e94f-9918-4406-996b-c99a360b52ba"}, "cell_type": "markdown", "source": "### Search for similar cars that have same year, make, model and trim1 ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "fb722cdd5062b22386c1c663228f2db370fe6cd3", "trusted": false, "_cell_guid": "6eef7c8e-1e39-4855-b35a-f7a2bce93f8e"}, "cell_type": "code", "source": "train_gb_year_make_model_trim1 = train_metadata_csv.groupby(['year', 'make', 'model', 'trim1'])\nlen(train_gb_year_make_model_trim1.groups)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "e4ce743bddbcbdf651bb7be2be0e592ef898ed06", "trusted": false, "_cell_guid": "c37d6c1a-0cb1-4a9f-a544-82bce3b525e6"}, "cell_type": "code", "source": "similar_cars = [k for k in train_gb_year_make_model_trim1.groups if len(train_gb_year_make_model_trim1.groups[k]) > 1]", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "741af68c3bcf725341e6b01ea92e9c6f8e222379", "_cell_guid": "c37dfc2b-fde9-4097-90e6-587abf2f7e0b"}, "cell_type": "markdown", "source": "#### Display similar cars", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "0ddc13c9a2e7e8bfaada9095f9688d28f206d3e5", "trusted": false, "_cell_guid": "69868ac6-bb7f-4cf6-a17d-4d846d78d258"}, "cell_type": "code", "source": "for gname in similar_cars:\n    _ids = train_gb_year_make_model_trim1.get_group(gname)['id']\n    _trim2 = train_gb_year_make_model_trim1.get_group(gname)['trim2']\n    plt.figure(figsize=(14, 6))\n    plt.suptitle(\"{}\".format(gname))    \n    n = len(_ids)\n    for i, _id in enumerate(_ids):\n        plt.subplot(1, n, i + 1)\n        plt.title('{}'.format(_trim2[i]))\n        img = get_image_data(_id + '_03', 'Train')\n        plt.imshow(img)            ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "c4b36f3c805b3760d9ce346b3775290e75be5aca", "trusted": false, "_cell_guid": "2775373d-842c-4e09-8962-af7603d3fb14"}, "cell_type": "code", "source": "", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "f116af8d9f6562864078c7b59dfcbb2b19b07bdf", "_cell_guid": "712ada0b-cd15-4614-8150-28387ccaed3f"}, "cell_type": "markdown", "source": "## Which cars are present in the test dataset:", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "3e8275948033a5340234b9de5939b757e302f36c", "trusted": false, "_cell_guid": "fcc08067-dacf-4568-8cef-be6543a09c98"}, "cell_type": "code", "source": "test_dataset_ids = list(set(METADATA_CSV['id']) - set(TRAIN_MASKS_CSV['id']))\nlen(test_dataset_ids), len(METADATA_CSV['id'])", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "6f3fab1f585eccd8cd8259c596fc3c1280a7d1e4", "trusted": false, "_cell_guid": "229f1b02-501f-45bb-882c-2c7b0baa03a8"}, "cell_type": "code", "source": "test_metadata_csv = METADATA_CSV.loc[test_dataset_ids,:]\nsns.countplot(y=\"make\", data=test_metadata_csv, palette=\"Greens_d\")", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "8909bd5e8f7c1ef2fdb0407735c13c3296004ed4", "_cell_guid": "c6f9c083-e86b-4e0a-89fe-509f5acd3b18"}, "cell_type": "markdown", "source": "### Search for similar cars that have same year, make, model and trim1 ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "29624716da67c85061b02321f016e04d792c67ed", "trusted": false, "_cell_guid": "f8536ebb-e5ef-4e44-91db-fd3a1455097d"}, "cell_type": "code", "source": "test_metadata_csv.loc[test_metadata_csv['trim1'].isnull(), 'trim1'] = '-'\ntest_gb_year_make_model_trim1 = test_metadata_csv.groupby(['year', 'make', 'model', 'trim1'])\nlen(test_gb_year_make_model_trim1.groups)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "6a4a73121d556d33c8a3a867f39b227937bc3b74", "trusted": false, "_cell_guid": "787f1cc3-ba79-49bf-b9dc-3f051bd0482c"}, "cell_type": "code", "source": "similar_cars = [k for k in test_gb_year_make_model_trim1.groups if len(test_gb_year_make_model_trim1.groups[k]) > 1]\nlen(similar_cars)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "051e8ebf2b870e434fd2fe51528b8399caceaca5", "_cell_guid": "c53ac5fa-6529-4688-bbfc-4d62c5a6ef2f"}, "cell_type": "markdown", "source": "#### Display some of similar cars", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "4094be0e763d792f5fecc163a9046c274d432994", "trusted": false, "_cell_guid": "7111dbb2-e7e7-4ea7-9ad9-b6becbfcb201"}, "cell_type": "code", "source": "k = 5 \nfor gname in similar_cars[:20]:\n    _ids = test_gb_year_make_model_trim1.get_group(gname)['id']      \n    _trim2 = test_gb_year_make_model_trim1.get_group(gname)['trim2']    \n    plt.figure(figsize=(14, 6))\n    plt.suptitle(\"{}\".format(gname))    \n    n = min(len(_ids), k)\n    m = int(np.ceil(len(_ids) * 1.0 / k))\n    for i, _id in enumerate(_ids):\n        plt.subplot(m, n, i + 1)    \n        plt.title(\"{}\".format(_trim2[i]))\n        img = get_image_data(_id + '_03', 'Test')\n        plt.imshow(img)        \n    ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "3b868dc7ca8e81a0b52a1cf5abe58ea960a744f6", "_cell_guid": "5271debf-6e4b-457a-a4c5-6159e40dbc26"}, "cell_type": "markdown", "source": "## Are there 'same' cars in train and test ?", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "6f5ddf1f410060f2fe6dbcdd77046d22cf55d95e", "trusted": false, "_cell_guid": "475d4617-e2e0-41e3-8d69-4dea5a3671a7"}, "cell_type": "code", "source": "METADATA_CSV['in_train'] = False\nMETADATA_CSV['in_test'] = False\n\nMETADATA_CSV.loc[test_dataset_ids, 'in_test'] = True\nMETADATA_CSV.loc[TRAIN_MASKS_CSV['id'].unique(), 'in_train'] = True", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "0a4c62f51a956d5aea22a8eea2f3a41719fef70f", "_cell_guid": "440935e0-460d-4016-85b2-2f130ace57bc"}, "cell_type": "markdown", "source": "No cars with the same ids", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "819f9467779ad03eb67701e74669fd5279aadc6d", "trusted": false, "_cell_guid": "db27cd45-d243-4b34-8528-6157221d9123"}, "cell_type": "code", "source": "METADATA_CSV[METADATA_CSV['in_train'] & METADATA_CSV['in_test']]", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "6e4fa2cd870ffc1456de3c2ff12f700d837c4c96", "trusted": false, "_cell_guid": "bcff7ada-4717-48ae-84e7-a4497de6345b"}, "cell_type": "code", "source": "METADATA_CSV.loc[METADATA_CSV['trim1'].isnull(), 'trim1'] = '-'\ngb_year_make_model_trim1 = METADATA_CSV.groupby(['year', 'make', 'model', 'trim1'])\nlen(gb_year_make_model_trim1.groups)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "e0a339287b42dd76a85225be28b47aaf6ec65aa8", "trusted": false, "_cell_guid": "2503f2c0-6ee3-4b5b-a2d7-10be8a47e228"}, "cell_type": "code", "source": "similar_cars = [k for k in gb_year_make_model_trim1.groups if len(gb_year_make_model_trim1.groups[k]) > 1]\nlen(similar_cars)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "99a24c919fb09a4f5e6e24e3ffedeeae354f44cf", "trusted": false, "_cell_guid": "e7614a0b-ab2a-4959-87bf-9478a0563b25"}, "cell_type": "code", "source": "gb_year_make_model_trim1.get_group(similar_cars[0])", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "2068862c7e7bda1c4ac653f7c2cdd13cde4f99e6", "_cell_guid": "2017f200-6b01-4dfc-9320-6b3ebbd2d9db"}, "cell_type": "markdown", "source": "We see that model BMW Z4 Z4 sDrive35i 2014 is in test and train dataset", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "46df0a22742fc0127edc53346985c043a4568235", "_cell_guid": "ed1ca5ed-8a08-4fe8-8299-07a153a10b82"}, "cell_type": "markdown", "source": "### Display some of similar cars", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "ffa9d38de15bff236328c383ee6b64032644153a", "trusted": false, "_cell_guid": "6cb22ee2-e2b0-4f72-a0dd-2cad1c1ba7a4"}, "cell_type": "code", "source": "k = 5 \nfor gname in similar_cars[:10]:\n    _ids = gb_year_make_model_trim1.get_group(gname)['id']      \n    _trim2 = gb_year_make_model_trim1.get_group(gname)['trim2']\n    _in_train = gb_year_make_model_trim1.get_group(gname)['in_train']\n    _in_test = gb_year_make_model_trim1.get_group(gname)['in_test']    \n    \n    plt.figure(figsize=(14, 6))\n    plt.suptitle(\"{}\".format(gname))    \n    n = min(len(_ids), k)\n    m = int(np.ceil(len(_ids) * 1.0 / k))\n    for i, _id in enumerate(_ids):\n        plt.subplot(m, n, i + 1)    \n        plt.title(\"{}\\ntrain={}, test={}\\n{}\".format(_trim2[i], _in_train[i], _in_test[i], _id))\n        image_type = \"Train\" if  _in_train[i] else \"Test\"\n        img = get_image_data(_id + '_03', image_type)\n        plt.imshow(img)        \n    ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "f8b12f98873ea2f8b376274ddd980575e30ead24", "trusted": false, "_cell_guid": "7fc6c733-9fa0-4f1c-a747-fcfdb2bc3e6b"}, "cell_type": "code", "source": "cond = lambda k: (len(gb_year_make_model_trim1.groups[k]) > 1) and gb_year_make_model_trim1.get_group(k)[['in_train', 'in_test']].any().all()\nmodels_in_train_and_test = [k for k in gb_year_make_model_trim1.groups if cond(k)]\nlen(models_in_train_and_test)", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "idle", "_uuid": "34d94173ed11bdf46e93e0c0c1c2e6f842032764", "_cell_guid": "ee4a826b-f838-47a7-9d7c-21d7c7d9c7df"}, "cell_type": "markdown", "source": "### Display only models that present in train and test\nTrain image is display with its mask and test images are blended with the train mask", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "2b6b7b5504cd29b6c3c2eafbee9414bebe6aa736", "trusted": false, "_cell_guid": "7f47d611-e61a-4da8-9c21-95b60ea86044"}, "cell_type": "code", "source": "sns.set_style(\"whitegrid\", {'axes.grid' : False})", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_execution_state": "busy", "_uuid": "7ee8a16ab0092dcb9936210031d4a401e066b3a9", "trusted": false, "_cell_guid": "4b3c07b8-ca0a-405c-9a15-1713d6454c38"}, "cell_type": "code", "source": "k = 5 \nfor gname in models_in_train_and_test[:10]:\n    _ids = gb_year_make_model_trim1.get_group(gname)['id']      \n    _trim2 = gb_year_make_model_trim1.get_group(gname)['trim2']\n    _in_train = gb_year_make_model_trim1.get_group(gname)['in_train']\n    _in_test = gb_year_make_model_trim1.get_group(gname)['in_test']    \n    \n    train_index = np.where(_in_train == True)[0][0]    \n    first_train_mask = get_image_data(_ids[train_index] + '_03', \"Train_mask\")    \n    \n    plt.figure(figsize=(14, 6))\n    plt.suptitle(\"{}\".format(gname))    \n    n = min(len(_ids), k)\n    m = int(np.ceil(len(_ids) * 1.0 / k))\n    for i, _id in enumerate(_ids):\n        plt.subplot(m, n, i + 1)    \n        plt.title(\"{}\\ntrain={}, test={}\\n{}\".format(_trim2[i], _in_train[i], _in_test[i], _id))\n        image_type = \"Train\" if  _in_train[i] else \"Test\"\n        img = get_image_data(_id + '_03', image_type)\n        if _in_train[i]:\n            img = cv2.bitwise_and(img, img, mask=first_train_mask)\n            plt.imshow(img)\n        else:\n            plt.imshow(img)\n            plt.imshow(first_train_mask, alpha=0.50)\n\n    ", "execution_count": null}, {"outputs": [], "metadata": {"collapsed": false, "_uuid": "5b24f6ef219e87d3e7ee0bb8482a179af75c2b0d", "_execution_state": "idle"}, "cell_type": "code", "source": "", "execution_count": null}], "nbformat": 4, "metadata": {"language_info": {"version": "3.6.1", "mimetype": "text/x-python", "file_extension": ".py", "nbconvert_exporter": "python", "codemirror_mode": {"version": 3, "name": "ipython"}, "pygments_lexer": "ipython3", "name": "python"}, "kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}}, "nbformat_minor": 0}