{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport glob\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom ipywidgets import interact\nfrom PIL import Image\nfrom math import ceil\nfrom bs4 import BeautifulSoup\nimport requests\nfrom io import StringIO, BytesIO\nfrom tqdm.auto import tqdm\n\nimport torch\nimport matplotlib.pyplot as plt\nimport zipfile","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:23.707116Z","iopub.execute_input":"2022-08-12T01:03:23.707798Z","iopub.status.idle":"2022-08-12T01:03:26.476514Z","shell.execute_reply.started":"2022-08-12T01:03:23.707637Z","shell.execute_reply":"2022-08-12T01:03:26.474795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hubmap-data-csv/GTEx Portal.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.480407Z","iopub.execute_input":"2022-08-12T01:03:26.481478Z","iopub.status.idle":"2022-08-12T01:03:26.673384Z","shell.execute_reply.started":"2022-08-12T01:03:26.481424Z","shell.execute_reply":"2022-08-12T01:03:26.672167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# organ_names = ['Kidney','Prostate','Colon', 'Spleen','Lung']\n# df_sub = df.loc[df['Tissue'].str.contains('Kidney|Prostate|Colon|Spleen|Lung')]\ndebug = False\norgan_name = 'Prostate'\ndf_organ = df.loc[df['Tissue'].str.contains(organ_name)]\nprint(df_organ.shape)\ndf_organ.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.675306Z","iopub.execute_input":"2022-08-12T01:03:26.675690Z","iopub.status.idle":"2022-08-12T01:03:26.717796Z","shell.execute_reply.started":"2022-08-12T01:03:26.675657Z","shell.execute_reply":"2022-08-12T01:03:26.716793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = 14\ncount = 40\nstart_index = count*index\nend_index = count*(index+1)\ndf_organ_sub = df_organ[start_index:end_index]\ndf_organ_sub.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.719813Z","iopub.execute_input":"2022-08-12T01:03:26.720462Z","iopub.status.idle":"2022-08-12T01:03:26.734240Z","shell.execute_reply.started":"2022-08-12T01:03:26.720420Z","shell.execute_reply":"2022-08-12T01:03:26.731390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def download_tiff(tissue_sample_id, img_out):\n    url = f\"https://brd.nci.nih.gov/brd/imagedownload/{tissue_sample_id}\"\n    r = requests.get(url)\n    img_out.writestr(f'{tissue_sample_id}.tiff', r.content)\n    \ndef map_organ(tissue):\n    organ_name = ''\n    if 'Kidney' in tissue:\n         organ_name = 'kidney'\n    elif 'Colon' in tissue:\n        organ_name = 'largeintestine'\n    elif 'Spleen' in tissue:\n        organ_name = 'spleen'\n    elif 'Prostate' in tissue:\n        organ_name = 'prostate'\n    elif 'Lung' in tissue:\n        organ_name = 'lung'\n    else:\n        print(f'tissue is {tissue}')\n        \n    return organ_name","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.736259Z","iopub.execute_input":"2022-08-12T01:03:26.737406Z","iopub.status.idle":"2022-08-12T01:03:26.751268Z","shell.execute_reply.started":"2022-08-12T01:03:26.737330Z","shell.execute_reply":"2022-08-12T01:03:26.750164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef show_image(img,mask=None):\n    plt.figure(figsize=(10,10))\n    plt.imshow(img)\n    if mask is not None:\n        plt.imshow(mask, cmap='coolwarm', alpha=0.5)\n    plt.axis(\"off\")\n    \n\ndef show_images(img_list, rows=5, cols=3):\n    # create the figure\n    fig, axs = plt.subplots(nrows=rows, ncols=cols, figsize=(20, 20))\n    # flatten the axis into a 1-d array to make it easier to access each axes\n    axs = axs.flatten()\n    # iterate through and enumerate the files, use i to index the axes\n    for i, img in enumerate(img_list):\n        # add the image to the axes\n        axs[i].imshow(img)\n        # add an axes title; .stem is a pathlib method to get the filename\n        axs[i].set(title=f'{i:04d}')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.754902Z","iopub.execute_input":"2022-08-12T01:03:26.757333Z","iopub.status.idle":"2022-08-12T01:03:26.769725Z","shell.execute_reply.started":"2022-08-12T01:03:26.757268Z","shell.execute_reply":"2022-08-12T01:03:26.767831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_out = f'/kaggle/working/hubmap_organ_{organ_name}_{start_index}_{end_index}.zip'\nwith zipfile.ZipFile(images_out, 'w') as img_out:\n    for index, (t, d) in tqdm(enumerate(df_organ_sub.iterrows()),total=df_organ_sub.shape[0]):\n        tissue_sample_id = d['Tissue Sample ID']\n        tissue = d['Tissue']\n        sex = d['Sex']\n        age_bracket = d['Age Bracket']\n        organ_name = map_organ(tissue)   \n        download_tiff(tissue_sample_id, img_out)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T01:03:26.771525Z","iopub.execute_input":"2022-08-12T01:03:26.773114Z","iopub.status.idle":"2022-08-12T01:12:53.233695Z","shell.execute_reply.started":"2022-08-12T01:03:26.773052Z","shell.execute_reply":"2022-08-12T01:12:53.231650Z"},"trusted":true},"execution_count":null,"outputs":[]}]}