{"cells":[{"metadata":{"trusted":true,"_uuid":"e4dc59c0edde35c17bb339af505be943abf1244a"},"cell_type":"code","source":"# You can use shell commands with \"!\"\n!ls ../input\n\n# Pipe output to do basic analysis\n!ls ../input/train/ | wc -l\n!ls ../input/train/ | head","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a73dfb2850d032184b4c30c5d2e70f7988c0e5d"},"cell_type":"code","source":"# Better approach: use pathlib\nfrom pathlib import Path\n\nDATA_DIR = Path('../input')\nTRAIN_DIR = DATA_DIR/'train'\nTEST_DIR = DATA_DIR/'test'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee7d4c6e04dc4f06c2bc5abcfba3ae8a34ed5d37"},"cell_type":"code","source":"# Use the full power of Python, taking the unique ID's\ntest_ids = list(set([str(fn).split('/')[-1].split('_')[0]  for fn in TEST_DIR.iterdir()]))\nprint('Test IDs:', len(test_ids))\ntest_ids[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76d777d3d5933fbdadb61eb205a22e23be52dad9"},"cell_type":"code","source":"# You can even create directories\nSUB_DIR = Path('files/submissions')\nSUB_DIR.mkdir(parents=True, exist_ok=True)\n!ls files","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ba1b3dfdd006e39769b13646944f35a9945c51b"},"cell_type":"markdown","source":"Learn more here: https://docs.python.org/3/library/pathlib.html"},{"metadata":{"trusted":true,"_uuid":"0081d3d1f8df36304297b07b9c7374f8d10109bb"},"cell_type":"code","source":"# You could always use shell commands\nLABELS_CSV = DATA_DIR/'train.csv'\n!head {LABELS_CSV}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9a4f323097a4fc86ed964a514bdd3068d2cb782"},"cell_type":"code","source":"# Enter pandas\nimport pandas as pd\n\ntrain_df = pd.read_csv(LABELS_CSV, index_col='Id')\ntrain_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2873f7664b2bf486a7c07fb664dd3da985dc6adb"},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c745dcea68ae6b7a8b88c0533d6661c9838bb848"},"cell_type":"code","source":"# You can look at a random sample\ntrain_df.sample(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0e49d1e764532c3a1d0db753f56ac2ef7f0721d"},"cell_type":"code","source":"# Or get basic information about the data\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f35d9419d122c39ce3b0f729bda80da9f8d33cb3"},"cell_type":"code","source":"# Use Python to your advantage\ntrain_df['Target'] = train_df['Target'].str.split(' ').map(lambda x: list(map(int, x)))\ntrain_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"16b56c87fc3534a8e16e28435a1dbdcd878ba7db"},"cell_type":"markdown","source":"Exploratory Data Analysis (EDA)\n\nPandas dataframe is a great starting point for doing EDA. It provides many utilities for plotting graphs right out of the box."},{"metadata":{"trusted":true,"_uuid":"dc99ac9f4638aa9cac6bb828e46e7cc12f072f40"},"cell_type":"code","source":"label_names = [\"Nucleoplasm\", \"Nuclear membrane\", \"Nucleoli\", \"Nucleoli fibrillar center\", \n               \"Nuclear speckles\", \"Nuclear bodies\", \"Endoplasmic reticulum\", \n               \"Golgi apparatus\", \"Peroxisomes\", \"Endosomes\",\"Lysosomes\", \n               \"Intermediate filaments\", \"Actin filaments\", \"Focal adhesion sites\", \n               \"Microtubules\", \"Microtubule ends\", \"Cytokinetic bridge\", \"Mitotic spindle\", \n               \"Microtubule organizing center\", \"Centrosome\", \"Lipid droplets\", \n               \"Plasma membrane\", \"Cell junctions\", \"Mitochondria\", \"Aggresome\",   \n               \"Cytosol\", \"Cytoplasmic bodies\", \"Rods & rings\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd45281e0ddf26d4138e8e4ec51f73752784c71e"},"cell_type":"code","source":"import numpy as np\n\ndef get_label_freqs(targets, label_names, ascending=None):\n    n_classes = len(label_names)\n    freqs = np.array([0] * n_classes)\n    for lst in targets:\n        for c in range(n_classes):\n            freqs[c] += c in lst\n    data = {\n        'name': label_names, \n        'frequency': freqs, \n        'percent': (10000 * freqs / len(targets)).astype(int) / 100.,\n    }\n    cols = ['name', 'frequency', 'percent']\n    df = pd.DataFrame(data, columns=cols)\n    if ascending is not None:\n        df = df.sort_values(by='frequency', ascending=ascending)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ed3877df6bdcaa6c0f908adda9a9301fe678ac3"},"cell_type":"code","source":"# Create a frequency table\ntrain_freqs = get_label_freqs(train_df.Target, label_names, ascending=False)\ntrain_freqs","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a4061294e0563b401289bb15ac33b817b7d6087"},"cell_type":"markdown","source":"Clearly, there is a huge imbalance between the classes, and **15 of the 28 classes have less than 900 samples (~ 3% of the data)**, and 9 classes have fewer than 330 samples (~1% of the data). Any model which always predicts 0 or 'not present' for these classes is already 97% accurate.\n\nSo, it's going to be really difficult to train a model that can detect the less frequently occuring classes. This may lead to a recall of 0, which will lead to and F1 score of 0 for these classes, thus putting a ceiling of 0.465 on the evaluation metric. In fact, we might need to train a separate model for these classes."},{"metadata":{"trusted":true,"_uuid":"7c75a6693c1ea9374006c5b7688e05acb48e7e44"},"cell_type":"code","source":"# Visualize the frequency table using a chart\ntrain_freqs.plot(x='name', y='frequency', kind='bar', title='Name vs. Frequency');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fae4c9d69aa6d0747f4a060b18f58cd3fb52cb70"},"cell_type":"code","source":"# Use logarithmic axis for easier interpretation\ntrain_freqs.plot(x='name', y='frequency', kind='bar', logy=True, title='Name vs. log(Frequency)');","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4c386aa77356a6b63c6a098ae03e8bef0b949608"},"cell_type":"markdown","source":"Display an image, or show multiple images in a grid?"},{"metadata":{"trusted":true,"_uuid":"359c6e93f1990c91b1aa681ace8e9f43d837c490"},"cell_type":"code","source":"train_sample = \"ac39847a-bbb1-11e8-b2ba-ac1f6b6435d0_red.png\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"375c1432cacced9d7773bee27ba2f91c02a560cd"},"cell_type":"code","source":"from imageio import imread\nimport matplotlib.pyplot as plt\n\n# Look at one channel/filter\nimg0 = imread(str(TRAIN_DIR/train_sample))\nprint(img0.shape)\nplt.imshow(img0)\nplt.title(train_sample[0]);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5180f15b677c7f5c91177ccac2fcfa3e63d24131"},"cell_type":"code","source":"# Use a color map for grayscale images\nplt.imshow(img0, cmap=\"Reds\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c33ee7195b0a9bf7d0aaa49a6857df2e8698addc"},"cell_type":"code","source":"# For RGB images, it \"just works\"\n!curl https://www.what-dog.net/Images/faces2/scroll001.jpg -o sample.jpg\n\nimg = imread('sample.jpg')\nplt.imshow(img);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a979083a6ce077caff228c6edec7f253a01470e3"},"cell_type":"code","source":"!ls {TRAIN_DIR}/ac39847a-bbb1-11e8-b2ba-ac1f6b6435d0_*.png","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0b56f48a70e5a5c35503f68c559c3bb150d5cda"},"cell_type":"code","source":"CHANNELS = ['green', 'red', 'blue', 'yellow']\n\n# Load images for multiple channels\ndef load_image(image_id, channels=CHANNELS, img_dir=TRAIN_DIR):\n    image = np.zeros(shape=(len(channels),512,512))\n    for i, ch in enumerate(channels):\n        image[i,:,:] = imread(str(img_dir/f'{image_id}_{ch}.png'))\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da61ec49453763c16e4d25272dc88b04ecc9b00f"},"cell_type":"code","source":"# Plot multiple images in a grid\ndef show_image_filters(image, title, figsize=(16,5)):\n    fig, subax = plt.subplots(1, 4, figsize=figsize)\n    # Green channel\n    subax[0].imshow(image[0], cmap=\"Greens\")\n    subax[0].set_title(title)\n    # Red channel\n    subax[1].imshow(image[1], cmap=\"Reds\")\n    subax[1].set_title(\"Microtubules\")\n    # Blue channel\n    subax[2].imshow(image[2], cmap=\"Blues\")\n    subax[2].set_title(\"Nucleus\")\n    # Orange channel\n    subax[3].imshow(image[3], cmap=\"Oranges\")\n    subax[3].set_title(\"Endoplasmatic reticulum\")\n    return subax","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32978a49c3cde58138ec14a8a692cd600dcbad05"},"cell_type":"code","source":"# Use the traning data to show appropriate labels\ndef get_labels(image_id):\n    labels = [label_names[x] for x in train_df.loc[image_id]['Target']]\n    return ', '.join(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"004591615ad88496434a9ffbf3f9faad1038a359"},"cell_type":"code","source":"# Look at a sample grid\nimg_id = 'ac39847a-bbb1-11e8-b2ba-ac1f6b6435d0'\nimg, title = load_image(img_id), get_labels(img_id)\nshow_image_filters(img, title);\nprint(img.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7fb01ab14fec8f8753420b040f51179c9df766a"},"cell_type":"code","source":"# Combine with pandas to view a random sample\nfor img_id in train_df.sample(3).index:\n    print(img_id)\n    img, title = load_image(img_id), get_labels(img_id)\n    show_image_filters(img, title)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e3ae7763fef4bedb7ca3196162ccc179a3f8a953"},"cell_type":"markdown","source":"![](http://)Generate a submission file"},{"metadata":{"trusted":true,"_uuid":"e2661a89a89758f4ebc5a529bf413e5d5083aa05"},"cell_type":"code","source":"# Let's define a sophisticated and highly accurate model :-)\ndef model(inputs):\n    return np.random.randn(len(inputs), len(label_names))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e03d1707c07afba4135b56bdfa14e83d1f98da9"},"cell_type":"code","source":"# Generate some predictions (logits)\npreds = model(test_ids)\nprint(preds.shape)\nprint(preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49ead2966a16d35ec8dcaab990bfdcc5919ee018"},"cell_type":"code","source":"# Convert them into probabilities\ndef sigmoid(x):\n    return np.reciprocal(np.exp(-x) + 1) \n\nprobs = sigmoid(preds)\nprobs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"309fe64a82ffc3f7e4a1ba345d21dc415701b3f6"},"cell_type":"code","source":"# Convert probabilities into labels\ndef make_labels(y, thres=0.75):\n    return ' '.join(map(str, [i for i, p in enumerate(y) if p > thres]))\n\nmake_labels(probs[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9150bb35d3d02c491d5a9ecfc3d675a767470a5"},"cell_type":"code","source":"# Create a pandas dataframe\nlabels = list(map(make_labels, probs))\nsub_df = pd.DataFrame({ 'Id': test_ids, 'Predicted': labels}, columns=['Id', 'Predicted'])\nsub_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7eafd6eb6b6369ba8db7b831ec431dce9981afe7"},"cell_type":"code","source":"# Export it to a file and make sure it looks okay\nsub_fname = SUB_DIR/'basic.csv'\nsub_df.to_csv(sub_fname, index=None)\n\n!head {sub_fname}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a1776f2ccb18f5e3449e88d368fce818ae87243"},"cell_type":"code","source":"# Use FileLink to download the file\nfrom IPython.display import FileLink\n\nFileLink(sub_fname)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc6250d69d9d0cb9fd24add770f644a7576c15bf"},"cell_type":"markdown","source":"The last but **MOST IMPORTANT** step is to take all of the above code (once it works as expected), and wrap it into a function (or two)"},{"metadata":{"trusted":true,"_uuid":"d71e47c6c77129f202af1a5fcc21f136fb7528a6"},"cell_type":"code","source":"def make_sub(fname):\n    preds = model(test_ids)\n    probs = sigmoid(preds)\n    labels = list(map(make_labels, probs))\n    sub_df = pd.DataFrame({ 'Id': test_ids, 'Predicted': labels}, columns=['Id', 'Predicted'])\n    sub_df.to_csv(sub_fname, index=None)\n    fpath = SUB_DIR/fname\n    sub_df.to_csv(fpath, index=None)\n    !head {fpath}\n    return FileLink(fpath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3be77618796275d3a178f2887b242ead36dbd8aa","scrolled":true},"cell_type":"code","source":"make_sub('best_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7451df870ae288c0d9a9218b5e0a1c28c3d7bf4b"},"cell_type":"markdown","source":"Now you can generate test predictions with a single line of code!"},{"metadata":{"trusted":true,"_uuid":"de1a5a3827c327b8e6476fe5aa90593a37c7a0a6"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}