{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\n\nimport glob\nimport csv\n\nimport numpy as np\nimport pandas as pd\n\nimport tifffile as tiff","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_path='/kaggle/input/hubmap-kidney-segmentation/'\ntrain_path=main_path+'train/'\ntest_path=main_path+'test/'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_shape(tiffile):# this function call fails on the first test image\n    tif=tiff.TiffFile(tiffile)\n    shape=tif.series[0].shape\n    print('get shape', shape)\n\n    #if shape[0]==1:#(1,1,3,x,y):\n    #    shape=shape[3:]\n    #elif shape[0]==3:#(3,x,y)\n    #    shape=shape[1:]\n    #elif shape[2]==3:#(x,y,3)\n    #    shape=shape[0:2]\n    return shape\n\ndef read_tiff(file_path):\n    img = tiff.imread(file_path)\n    print(file_path, img.shape)\n    if img.shape[0]==1:#(1,1,3,x,y):\n        img=np.moveaxis(img[0][0],0,-1)\n    elif img.shape[0]==3:\n        img=np.moveaxis(img,0,-1)\n\n    print(file_path, img.shape)\n    return img\n\ndef rle_encode_less_memory(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    This simplified method requires first and last pixel to be zero\n    '''\n    pixels = img.flatten()\n    \n    # This simplified method requires first and last pixel to be zero\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#You can use either of the two lines below. Both options work fine. It was a test to see of sample_submission.csv is a valid source of test_id\ntest_ids=[os.path.splitext(os.path.split(f)[1])[0] for f in glob.glob(test_path+'*.tiff')]\ntest_ids=list(pd.read_csv('../input/hubmap-kidney-segmentation/sample_submission.csv').id.values)\ntest_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#generate random masks to have score (imitates the CNN inference)\n\ntry:\n    np.random.seed(42)\n\n    ids, preds = [], []\n    for t_id in test_ids:\n        test_tiff=test_path+t_id+'.tiff'\n        shape=get_shape(test_tiff)#if I comment this like out the submission works fine. If the line is present the submission fails.\n        shape=(10000,10000)\n        print(test_tiff, shape)\n        mask=np.zeros(shape, dtype=np.uint8)\n        t=1/0\n        for i in range(100):\n            y=np.random.randint(0,shape[0]-1000)\n            x=np.random.randint(0,shape[1]-1000)\n            mask[y:y+1000, x:x+1000]=1\n\n        encoded=rle_encode_less_memory(mask.T)\n\n        ids.append(t_id)\n        preds.append(encoded)   \nexcept:\n    print(\"Unexpected error:\", sys.exc_info()[0])\n    raise\n\nelse:\n    print('We are cool')\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#If the previous cell fails this cell should never be exectuted and no submission.csv appear\n#But this cell is executed even if the previous cell thows an exception\n\ndf = pd.DataFrame({'id':ids, 'predicted':preds})\ndf.to_csv('sub.csv', index=False)\ndf.to_csv('submission.csv', index=False)\ndf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this is the \"hack\" that I used to make the submission successful\n#my assumption was that I must be forming the submission.csv incorrectly as kaggle was giving me \"Error Scoring Submission\"\n#It turned out that Kaggle kept executing notebook cells after the exeption was thrown thus resulting in a malformed CSV with no private test submission\n#This, in turn, resulted in perfectly valid puublic test score and 0 private test score.\n\nif True:\n    sample_sub=pd.read_csv('../input/hubmap-kidney-segmentation/sample_submission.csv')\n    submission = pd.read_csv('sub.csv', index_col='id')\n    sample_sub = pd.read_csv('/kaggle/input/hubmap-kidney-segmentation/test/sample_submission.csv', index_col='id')\n    print(sample_sub.index.values)\n    pub_ids = submission.index.values\n\n    predictions = submission.values\n    sample_sub.loc[pub_ids] = predictions\n    sample_sub.to_csv('submission.csv')\n    print('Done')\n    # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n    # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}