{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom os import listdir\nimport pandas as pd\nimport numpy as np\nimport glob\nimport tqdm\nfrom typing import Dict\nimport matplotlib.pyplot as plt\n#import pandas_profiling as pdp\nimport json\n%matplotlib inline\nimport shapely.geometry as sg\nimport shapely.ops as so\nimport zipfile\nimport cv2\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport plotly.offline as pyo\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\n#seaborn\nimport seaborn as sns\n\n#color\nfrom colorama import Fore, Back, Style\n\n#networkx\nimport networkx as nx\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")\n\n#tifffile\nfrom PIL import Image\nimport tifffile as tiff\nimport cv2\nfrom tqdm.notebook import tqdm\nimport zipfile\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:01.544647Z","iopub.execute_input":"2023-07-26T23:50:01.545408Z","iopub.status.idle":"2023-07-26T23:50:18.801425Z","shell.execute_reply.started":"2023-07-26T23:50:01.545368Z","shell.execute_reply":"2023-07-26T23:50:18.800398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List files available\nlist(os.listdir(\"../input/hubmap-kidney-segmentation\"))","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:18.803288Z","iopub.execute_input":"2023-07-26T23:50:18.803706Z","iopub.status.idle":"2023-07-26T23:50:18.813204Z","shell.execute_reply.started":"2023-07-26T23:50:18.803663Z","shell.execute_reply":"2023-07-26T23:50:18.812175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = \"../input/hubmap-kidney-segmentation/\"\ntrain_df=pd.read_csv('../input/hubmap-kidney-segmentation/train.csv')\nhubmap_df=pd.read_csv('../input/hubmap-kidney-segmentation/HuBMAP-20-dataset_information.csv')\ntest_df=pd.read_csv('../input/hubmap-kidney-segmentation/sample_submission.csv')\nprint(Fore.YELLOW+'Training data shape:',Style.RESET_ALL,train_df.shape)\nprint(Fore.YELLOW + 'HubMap data shape: ',Style.RESET_ALL,hubmap_df.shape)\nprint(Fore.YELLOW + 'Test data shape: ',Style.RESET_ALL,test_df.shape)\n\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:18.814687Z","iopub.execute_input":"2023-07-26T23:50:18.815262Z","iopub.status.idle":"2023-07-26T23:50:19.193231Z","shell.execute_reply.started":"2023-07-26T23:50:18.815227Z","shell.execute_reply":"2023-07-26T23:50:19.192243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train masks**","metadata":{}},{"cell_type":"code","source":"hubmap_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.196175Z","iopub.execute_input":"2023-07-26T23:50:19.196868Z","iopub.status.idle":"2023-07-26T23:50:19.215203Z","shell.execute_reply.started":"2023-07-26T23:50:19.196831Z","shell.execute_reply":"2023-07-26T23:50:19.213999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df.groupby(['race']).count()['sex'].to_frame()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.216720Z","iopub.execute_input":"2023-07-26T23:50:19.217074Z","iopub.status.idle":"2023-07-26T23:50:19.237271Z","shell.execute_reply.started":"2023-07-26T23:50:19.217041Z","shell.execute_reply":"2023-07-26T23:50:19.236395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EDA**","metadata":{}},{"cell_type":"code","source":"# Null values and Data types\nprint(Fore.YELLOW + 'Train Set !!',Style.RESET_ALL)\nprint(train_df.info())\nprint('-------------')\nprint(Fore.BLUE+'Test Set !!',Style.RESET_ALL)\nprint(test_df.info())\nprint('-------------')\nprint(Fore.GREEN+'HuBMAP Set !!',Style.RESET_ALL)\nprint(hubmap_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.239306Z","iopub.execute_input":"2023-07-26T23:50:19.240272Z","iopub.status.idle":"2023-07-26T23:50:19.266828Z","shell.execute_reply.started":"2023-07-26T23:50:19.240239Z","shell.execute_reply":"2023-07-26T23:50:19.265773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Missing values**","metadata":{}},{"cell_type":"code","source":"hubmap_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.268409Z","iopub.execute_input":"2023-07-26T23:50:19.268807Z","iopub.status.idle":"2023-07-26T23:50:19.284404Z","shell.execute_reply.started":"2023-07-26T23:50:19.268774Z","shell.execute_reply":"2023-07-26T23:50:19.283171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.285611Z","iopub.execute_input":"2023-07-26T23:50:19.286226Z","iopub.status.idle":"2023-07-26T23:50:19.299536Z","shell.execute_reply.started":"2023-07-26T23:50:19.286192Z","shell.execute_reply":"2023-07-26T23:50:19.298623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! ls ../input/hubmap-kidney-segmentation/train/","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:19.301111Z","iopub.execute_input":"2023-07-26T23:50:19.301572Z","iopub.status.idle":"2023-07-26T23:50:20.285715Z","shell.execute_reply.started":"2023-07-26T23:50:19.301539Z","shell.execute_reply":"2023-07-26T23:50:20.284493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Total number of Patient in the dataset(train+test)\n\nprint(Fore.YELLOW +\"Total Patients in Train set: \",Style.RESET_ALL,train_df['id'].count())\nprint(Fore.BLUE +\"Total Patients in Test set: \",Style.RESET_ALL,test_df['id'].count())","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.291085Z","iopub.execute_input":"2023-07-26T23:50:20.291418Z","iopub.status.idle":"2023-07-26T23:50:20.298656Z","shell.execute_reply.started":"2023-07-26T23:50:20.291387Z","shell.execute_reply":"2023-07-26T23:50:20.297409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Unique Patients(Ids)**","metadata":{}},{"cell_type":"code","source":"print(Fore.YELLOW+\"The total patient ids are\",Style.RESET_ALL,f\"{train_df['id'].count()},\",Fore.BLUE+\"form those the uinque ids are\",Style.RESET_ALL,f\"{train_df['id'].value_counts().shape[0]}.\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.299930Z","iopub.execute_input":"2023-07-26T23:50:20.300220Z","iopub.status.idle":"2023-07-26T23:50:20.311831Z","shell.execute_reply.started":"2023-07-26T23:50:20.300196Z","shell.execute_reply":"2023-07-26T23:50:20.310941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['id'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.313330Z","iopub.execute_input":"2023-07-26T23:50:20.313998Z","iopub.status.idle":"2023-07-26T23:50:20.322325Z","shell.execute_reply.started":"2023-07-26T23:50:20.313961Z","shell.execute_reply":"2023-07-26T23:50:20.321231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_patient_ids=set(train_df['id'].unique())\ntest_patient_ids=set(test_df['id'].unique())\n\ntrain_patient_ids.intersection(test_patient_ids)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.323983Z","iopub.execute_input":"2023-07-26T23:50:20.324443Z","iopub.status.idle":"2023-07-26T23:50:20.334625Z","shell.execute_reply.started":"2023-07-26T23:50:20.324412Z","shell.execute_reply":"2023-07-26T23:50:20.333370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns=train_df.keys()\ncolumns=list(columns)\nprint(columns)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.336183Z","iopub.execute_input":"2023-07-26T23:50:20.336876Z","iopub.status.idle":"2023-07-26T23:50:20.342389Z","shell.execute_reply.started":"2023-07-26T23:50:20.336844Z","shell.execute_reply":"2023-07-26T23:50:20.341508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Patient Counts**","metadata":{}},{"cell_type":"code","source":"train_df['id'].value_counts().max()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.344149Z","iopub.execute_input":"2023-07-26T23:50:20.344958Z","iopub.status.idle":"2023-07-26T23:50:20.354639Z","shell.execute_reply.started":"2023-07-26T23:50:20.344926Z","shell.execute_reply":"2023-07-26T23:50:20.353508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['id'].value_counts().max()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.356165Z","iopub.execute_input":"2023-07-26T23:50:20.356562Z","iopub.status.idle":"2023-07-26T23:50:20.366876Z","shell.execute_reply.started":"2023-07-26T23:50:20.356532Z","shell.execute_reply":"2023-07-26T23:50:20.365906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of Patients in Training**","metadata":{}},{"cell_type":"code","source":"files=folders=0 \nfolder_names={'a'}\nfolder_names.remove('a')\n\npath=\"../input/hubmap-kidney-segmentation/train\"\nfor _,dirnames,filesnames in os.walk(path):\n    \n    files +=len(filesnames)\n    for j in range(files):\n        folder_names.add(filesnames[j][:9])\n        \n    folders=len(folder_names)\nprint(Fore.YELLOW +f'{files:,}',Style.RESET_ALL,\"files, \" + Fore.BLUE + f'{folders:,}',Style.RESET_ALL ,'patients/ids')    ","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.368294Z","iopub.execute_input":"2023-07-26T23:50:20.368933Z","iopub.status.idle":"2023-07-26T23:50:20.377922Z","shell.execute_reply.started":"2023-07-26T23:50:20.368902Z","shell.execute_reply":"2023-07-26T23:50:20.377234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = folders = 0\nfolder_names = {'a'}\nfolder_names.remove('a')\n\npath = \"../input/hubmap-kidney-segmentation/test\"\n\nfor _,dirnames,filenames in os.walk(path):\n    \n    files +=len(filenames)\n    for j in range(files):\n        folder_names.add(filenames[j][:9])\n    folders = len(folder_names) \nprint(Fore.YELLOW +f'{files:,}',Style.RESET_ALL,\"files, \" + Fore.BLUE + f'{folders:,}',Style.RESET_ALL ,'patients/ids')    ","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.379383Z","iopub.execute_input":"2023-07-26T23:50:20.380003Z","iopub.status.idle":"2023-07-26T23:50:20.395297Z","shell.execute_reply.started":"2023-07-26T23:50:20.379971Z","shell.execute_reply":"2023-07-26T23:50:20.394568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = folders = 0\nfolder_names = {'a'}\nfolder_names.remove('a')\n\npath = \"../input/hubmap-kidney-segmentation/train\"\n\nfor _, dirnames, filenames in os.walk(path):\n    \n    files +=len(filenames)\n    for j in range(files):\n        if filenames[j][-4:]=='json':\n            folder_names.add(filenames[j][:9])\n    folders=len(folder_names)\nprint(Fore.YELLOW+f'{folders:,}',Style.RESET_ALL,\"json-files\")    ","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.398497Z","iopub.execute_input":"2023-07-26T23:50:20.398796Z","iopub.status.idle":"2023-07-26T23:50:20.406398Z","shell.execute_reply.started":"2023-07-26T23:50:20.398771Z","shell.execute_reply":"2023-07-26T23:50:20.405410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = folders = 0\nfolder_names = {'a'}\nfolder_names.remove('a')\n\npath = \"../input/hubmap-kidney-segmentation/test\"\n\nfor _, dirnames, filenames in os.walk(path):\n  # ^ this idiom means \"we won't be using this value\"\n    files += len(filenames)\n    for j in range(files):\n        if filenames[j][-4:]=='json':\n            folder_names.add(filenames[j][:9])\n    folders = len(folder_names)\n#print(Fore.YELLOW +\"Total Patients in Train set: \",Style.RESET_ALL,train_df['Patient'].count())\nprint(Fore.YELLOW +f'{folders:,}',Style.RESET_ALL,\" json-files\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.408065Z","iopub.execute_input":"2023-07-26T23:50:20.408797Z","iopub.status.idle":"2023-07-26T23:50:20.423013Z","shell.execute_reply.started":"2023-07-26T23:50:20.408761Z","shell.execute_reply":"2023-07-26T23:50:20.422329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Normal Json**","metadata":{}},{"cell_type":"code","source":"files=0 \npath = \"../input/hubmap-kidney-segmentation/train\"\nfor _,dirnmes,filenames in os.walk(path):\n    \n    files +=len(filesnames)\n    for j in range(files):\n        if filenames[j][-4:]=='json'and len(filenames[j])<15:\n            df=pd.read_json(f'../input/hubmap-kidney-segmentation/train/{filenames[j]}')\n            print(Fore.RED + f'{df.shape[0]}',Style.RESET_ALL, \"rows and \"+ Fore.GREEN + f'{df.shape[1]}',Style.RESET_ALL,f\" columns in the {filenames[j][:-5]} json file\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:20.424929Z","iopub.execute_input":"2023-07-26T23:50:20.425757Z","iopub.status.idle":"2023-07-26T23:50:21.471370Z","shell.execute_reply.started":"2023-07-26T23:50:20.425721Z","shell.execute_reply":"2023-07-26T23:50:21.470324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = 0\npath = \"../input/hubmap-kidney-segmentation\"\ni = 0\n\nfor _, dirnames, filenames in os.walk(path):\n  \n    if i==0:\n        i += 1\n        files += len(filenames)\n        for j in range(files):\n            if filenames[j][-4:]=='json' and len(filenames[j])<15:\n                df = pd.read_json(f'../input/hubmap-kidney-segmentation/{filenames[j]}')\n                print(Fore.RED + f'{df.shape[0]}',Style.RESET_ALL, \"rows and \"+ Fore.GREEN + f'{df.shape[1]}',Style.RESET_ALL,f\" columns in the {filenames[j][:-5]} json file\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.472933Z","iopub.execute_input":"2023-07-26T23:50:21.473306Z","iopub.status.idle":"2023-07-26T23:50:21.483557Z","shell.execute_reply.started":"2023-07-26T23:50:21.473261Z","shell.execute_reply":"2023-07-26T23:50:21.482554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train**","metadata":{}},{"cell_type":"code","source":"files = 0\npath = \"../input/hubmap-kidney-segmentation/train\"\n\nfor _, dirnames, filenames in os.walk(path):\n  # ^ this idiom means \"we won't be using this value\"\n    files += len(filenames)\n    for j in range(files):\n        if filenames[j][-4:]=='json' and len(filenames[j])>15:\n            df = pd.read_json(f'../input/hubmap-kidney-segmentation/train/{filenames[j]}')\n            print(Fore.RED + f'{df.shape[0]}',Style.RESET_ALL, \"rows and \"+ Fore.GREEN + f'{df.shape[1]}',Style.RESET_ALL,f\" columns in the {filenames[j][:-5]} json file\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.485408Z","iopub.execute_input":"2023-07-26T23:50:21.487280Z","iopub.status.idle":"2023-07-26T23:50:21.601686Z","shell.execute_reply.started":"2023-07-26T23:50:21.487245Z","shell.execute_reply":"2023-07-26T23:50:21.600691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = 0\npath = \"../input/hubmap-kidney-segmentation/test\"\n\nfor _, dirnames, filenames in os.walk(path):\n  # ^ this idiom means \"we won't be using this value\"\n    files += len(filenames)\n    for j in range(files):\n        if filenames[j][-4:]=='json' and len(filenames[j])>15:\n            df = pd.read_json(f'../input/hubmap-kidney-segmentation/test/{filenames[j]}')\n            print(Fore.RED + f'{df.shape[0]}',Style.RESET_ALL, \"rows and \"+ Fore.GREEN + f'{df.shape[1]}',Style.RESET_ALL,f\" columns in the {filenames[j][:-5]} json file\")","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.603191Z","iopub.execute_input":"2023-07-26T23:50:21.603956Z","iopub.status.idle":"2023-07-26T23:50:21.656583Z","shell.execute_reply.started":"2023-07-26T23:50:21.603915Z","shell.execute_reply":"2023-07-26T23:50:21.655556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.658085Z","iopub.execute_input":"2023-07-26T23:50:21.658714Z","iopub.status.idle":"2023-07-26T23:50:21.697441Z","shell.execute_reply.started":"2023-07-26T23:50:21.658679Z","shell.execute_reply":"2023-07-26T23:50:21.696390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**RLE Encoding**","metadata":{}},{"cell_type":"code","source":"def encode(s):\n\n    encoding = \"\" \n    i = 0\n    while i < len(s):\n        # count occurrences of character at index i\n        count = 1\n\n        while i + 1 < len(s) and s[i] == s[i + 1]:\n            count = count + 1\n            i = i + 1\n        encoding += str(count) + s[i]\n        i = i + 1\n\n    return encoding\ns = \"ABBCCCD\"\nprint(encode(s))","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.699266Z","iopub.execute_input":"2023-07-26T23:50:21.699796Z","iopub.status.idle":"2023-07-26T23:50:21.707233Z","shell.execute_reply.started":"2023-07-26T23:50:21.699757Z","shell.execute_reply":"2023-07-26T23:50:21.706123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EDA**","metadata":{}},{"cell_type":"code","source":"hubmap_df[\"split\"]=\"test\"\nhubmap_df.loc[hubmap_df[\"image_file\"].isin(os.listdir(os.path.join(IMAGE_PATH,\"train\"))),\"split\"]=\"train\"","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.709015Z","iopub.execute_input":"2023-07-26T23:50:21.709504Z","iopub.status.idle":"2023-07-26T23:50:21.720477Z","shell.execute_reply.started":"2023-07-26T23:50:21.709470Z","shell.execute_reply":"2023-07-26T23:50:21.719544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.730458Z","iopub.execute_input":"2023-07-26T23:50:21.731098Z","iopub.status.idle":"2023-07-26T23:50:21.737973Z","shell.execute_reply.started":"2023-07-26T23:50:21.731072Z","shell.execute_reply":"2023-07-26T23:50:21.736766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Parallel Diagram**","metadata":{}},{"cell_type":"code","source":"parallel_diagram=hubmap_df[['patient_number','age','sex','race','percent_cortex']]\nfig=px.parallel_categories(parallel_diagram,color_continuous_scale=px.colors.sequential.Inferno)\nfig.update_layout(title='Parallel category diagram 1 on hubmap set')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:21.739822Z","iopub.execute_input":"2023-07-26T23:50:21.740403Z","iopub.status.idle":"2023-07-26T23:50:22.820155Z","shell.execute_reply.started":"2023-07-26T23:50:21.740369Z","shell.execute_reply":"2023-07-26T23:50:22.819261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parallel_diagram = hubmap_df[['patient_number','age','sex','race','percent_medulla']]\nfig = px.parallel_categories(parallel_diagram, color_continuous_scale=px.colors.sequential.Inferno)\nfig.update_layout(title='Parallel category diagram 2 on hubmap set')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:22.821629Z","iopub.execute_input":"2023-07-26T23:50:22.822652Z","iopub.status.idle":"2023-07-26T23:50:22.887010Z","shell.execute_reply.started":"2023-07-26T23:50:22.822616Z","shell.execute_reply":"2023-07-26T23:50:22.886029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dist(df,column,color):\n    sns.distplot(df[column],label=column,color=color)\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:22.888455Z","iopub.execute_input":"2023-07-26T23:50:22.888888Z","iopub.status.idle":"2023-07-26T23:50:22.895419Z","shell.execute_reply.started":"2023-07-26T23:50:22.888854Z","shell.execute_reply":"2023-07-26T23:50:22.893319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['race','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:22.896650Z","iopub.execute_input":"2023-07-26T23:50:22.897687Z","iopub.status.idle":"2023-07-26T23:50:22.912456Z","shell.execute_reply.started":"2023-07-26T23:50:22.897654Z","shell.execute_reply":"2023-07-26T23:50:22.911371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['race','split']].value_counts().iplot(kind='bar',yTitle='Counts',linecolor='black', opacity=0.7,\n                                              color='blue',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the race column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:22.913848Z","iopub.execute_input":"2023-07-26T23:50:22.914481Z","iopub.status.idle":"2023-07-26T23:50:23.013897Z","shell.execute_reply.started":"2023-07-26T23:50:22.914441Z","shell.execute_reply":"2023-07-26T23:50:23.012882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['race','split'])['race'].count().reset_index(name = 'counts')\ndf['race_split']=df['race']+'_'+df['split']\nfig = px.pie(df, values='counts', names='race_split', title='Patient Race Count')\nfig.update_layout(\n\n  autosize=False,\n  width=800,\n  height=650,\n  margin=dict(\n  \n  l=50,\n  r=50,\n  b=50,\n  t=50,\n  pad=4    \n  \n  \n  \n  ),\n  paper_bgcolor=\"LightSteelBlue\",  \n\n\n\n)\ndf = hubmap_df.groupby(['race','split'])['race'].count().reset_index(name = 'counts')\ndf['race_split'] = df['race']+'_'+df['split']\nfig = px.pie(df, values='counts', names='race_split', title='Patient Race Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:23.015463Z","iopub.execute_input":"2023-07-26T23:50:23.016038Z","iopub.status.idle":"2023-07-26T23:50:23.158682Z","shell.execute_reply.started":"2023-07-26T23:50:23.016003Z","shell.execute_reply":"2023-07-26T23:50:23.157762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['race','split'])['race'].count().reset_index(name = 'counts')\nplt.figure(figsize=(10,10))\ndist(df,\"counts\",\"red\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:23.160156Z","iopub.execute_input":"2023-07-26T23:50:23.160752Z","iopub.status.idle":"2023-07-26T23:50:23.714252Z","shell.execute_reply.started":"2023-07-26T23:50:23.160715Z","shell.execute_reply":"2023-07-26T23:50:23.713258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"white=hubmap_df[hubmap_df['race']=='White']\nblack=hubmap_df[hubmap_df['race']=='Black or African American']\ncount_white=white['split'].value_counts().reset_index()\ncount_black=black['split'].value_counts().reset_index()\npie_white=go.Pie(labels=count_white['index'],values=count_white['split'],name=\"White\",hole=0.4,domain={'x':[0,0.46]})\npie_black = go.Pie(labels=count_black['index'],values=count_black['split'],name=\"Black or African American\",hole=0.5,domain={'x': [0.52,1]})\nlayout = dict(title = 'Race', font=dict(size=10), legend=dict(orientation=\"h\"),\n              annotations = [dict(x=0.2, y=0.5, text='White', showarrow=False, font=dict(size=20)),\n                             dict(x=0.8, y=0.5, text='Black or African American', showarrow=False, font=dict(size=20)) ])\nfig=dict(data=[pie_white,pie_black],layout=layout)\npyo.iplot(fig)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:23.715863Z","iopub.execute_input":"2023-07-26T23:50:23.716503Z","iopub.status.idle":"2023-07-26T23:50:23.776124Z","shell.execute_reply.started":"2023-07-26T23:50:23.716468Z","shell.execute_reply":"2023-07-26T23:50:23.775263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(20,12))\nfor i,t in enumerate(hubmap_df['race'].unique()):\n    hubmap_df_type=hubmap_df.loc[hubmap_df['race']==t]\n    hubmap_df_type = hubmap_df.loc[hubmap_df['race'] == t]\n    bad_weight = list(hubmap_df_type['weight_kilograms'].value_counts(normalize=True)[hubmap_df_type['weight_kilograms'].value_counts(normalize=True) < 0.01].index)\n    bad_height = list(hubmap_df_type['height_centimeters'].value_counts(normalize=True)[hubmap_df_type['height_centimeters'].value_counts(normalize=True) < 0.01].index)\n    bad_mes=list(set(bad_weight+bad_height))\n    hubmap_df_type = hubmap_df_type.loc[(hubmap_df_type['weight_kilograms'].isin(bad_weight) == False) & (hubmap_df_type['height_centimeters'].isin(bad_height) == False)]\n    G = nx.from_pandas_edgelist(hubmap_df_type, 'weight_kilograms', 'height_centimeters', ['bmi_kg/m^2'])\n    plt.subplot(2, 1, i + 1)\n    nx.draw(G, with_labels=True)\n    plt.title(f'Network Graph for race {t}')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:23.777488Z","iopub.execute_input":"2023-07-26T23:50:23.778453Z","iopub.status.idle":"2023-07-26T23:50:24.377941Z","shell.execute_reply.started":"2023-07-26T23:50:23.778418Z","shell.execute_reply":"2023-07-26T23:50:24.377018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['patient_number','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.378948Z","iopub.execute_input":"2023-07-26T23:50:24.379267Z","iopub.status.idle":"2023-07-26T23:50:24.393465Z","shell.execute_reply.started":"2023-07-26T23:50:24.379236Z","shell.execute_reply":"2023-07-26T23:50:24.392410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['patient_number','split']].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='red',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the patient_number column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.395277Z","iopub.execute_input":"2023-07-26T23:50:24.395740Z","iopub.status.idle":"2023-07-26T23:50:24.447440Z","shell.execute_reply.started":"2023-07-26T23:50:24.395703Z","shell.execute_reply":"2023-07-26T23:50:24.446405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=hubmap_df.groupby(['patient_number','split'])['patient_number'].count().reset_index(name='counts')\ndf['patient_number_split']=df['patient_number'].astype('str')+'_'+df['split']\nfig=px.pie(df,values='counts',names='patient_number_split',title='Patient Number Count')\nfig.update_layout(\n\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n    \n       l=50,\n       r=50, \n       b=50,\n       t=50,\n       pad=4\n    \n    \n    \n    \n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.448954Z","iopub.execute_input":"2023-07-26T23:50:24.449424Z","iopub.status.idle":"2023-07-26T23:50:24.514726Z","shell.execute_reply.started":"2023-07-26T23:50:24.449387Z","shell.execute_reply":"2023-07-26T23:50:24.513581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['ethnicity','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.515975Z","iopub.execute_input":"2023-07-26T23:50:24.519598Z","iopub.status.idle":"2023-07-26T23:50:24.530547Z","shell.execute_reply.started":"2023-07-26T23:50:24.519561Z","shell.execute_reply":"2023-07-26T23:50:24.529498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['ethnicity','split']].value_counts().iplot(kind='bar',yTitle='Counts',linecolor='black',opacity=0.7,color='green',theme='pearl',bargap=0.5,gridcolor='white',title='Distribution of the Ethnicity column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.532172Z","iopub.execute_input":"2023-07-26T23:50:24.532530Z","iopub.status.idle":"2023-07-26T23:50:24.587711Z","shell.execute_reply.started":"2023-07-26T23:50:24.532495Z","shell.execute_reply":"2023-07-26T23:50:24.586638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['ethnicity','split'])['ethnicity'].count().reset_index(name = 'counts')\ndf['ethnicity_split']=df['ethnicity']+'_'+df['split']\nfig = px.pie(df, values='counts', names='ethnicity_split', title='Patient Ethnicity Count')\nfig.update_layout(\n    autosize=False,\n    width=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    \n    \n    \n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n   \n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.589182Z","iopub.execute_input":"2023-07-26T23:50:24.590093Z","iopub.status.idle":"2023-07-26T23:50:24.654134Z","shell.execute_reply.started":"2023-07-26T23:50:24.590058Z","shell.execute_reply":"2023-07-26T23:50:24.653133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['sex','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.655641Z","iopub.execute_input":"2023-07-26T23:50:24.656216Z","iopub.status.idle":"2023-07-26T23:50:24.667593Z","shell.execute_reply.started":"2023-07-26T23:50:24.656179Z","shell.execute_reply":"2023-07-26T23:50:24.666528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['sex','split']].value_counts().iplot(kind='bar',\n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='yellow',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the Sex column in the HuBMAP-20 Set')\n                                                \n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.669417Z","iopub.execute_input":"2023-07-26T23:50:24.669804Z","iopub.status.idle":"2023-07-26T23:50:24.724626Z","shell.execute_reply.started":"2023-07-26T23:50:24.669772Z","shell.execute_reply":"2023-07-26T23:50:24.723351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['sex','split'])['sex'].count().reset_index(name = 'counts')\ndf['sex_split']=df['sex']+'_'+df['split']\nfig = px.pie(df, values='counts', names='sex_split', title='Patient Gender Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n\n\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.726091Z","iopub.execute_input":"2023-07-26T23:50:24.726676Z","iopub.status.idle":"2023-07-26T23:50:24.789730Z","shell.execute_reply.started":"2023-07-26T23:50:24.726640Z","shell.execute_reply":"2023-07-26T23:50:24.788762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"male=hubmap_df[hubmap_df['sex']=='Male']\nfemale=hubmap_df[hubmap_df['sex']== 'Female']\ncount_male=male['split'].value_counts().reset_index()\ncount_female=female['split'].value_counts().reset_index()\npie_male=go.Pie(labels=count_male['index'],values=count_male['split'],name=\"Male\",hole=0.4,domain={'x': [0,0.46]})\npie_female = go.Pie(labels=count_female['index'],values=count_female['split'],name=\"Female\",hole=0.5,domain={'x': [0.52,1]})\nlayout=dict(title='sex',font=dict(size=10),legend=dict(orientation=\"h\"),\n           annotations = [dict(x=0.2, y=0.5, text='Male', showarrow=False, font=dict(size=20)),\n            dict(x=0.8, y=0.5, text='Female', showarrow=False, font=dict(size=20)) ])\n\nfig=dict(data=[pie_male,pie_female],layout=layout)\npyo.iplot(fig)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.791027Z","iopub.execute_input":"2023-07-26T23:50:24.791931Z","iopub.status.idle":"2023-07-26T23:50:24.837328Z","shell.execute_reply.started":"2023-07-26T23:50:24.791894Z","shell.execute_reply":"2023-07-26T23:50:24.836228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['age','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.839152Z","iopub.execute_input":"2023-07-26T23:50:24.839518Z","iopub.status.idle":"2023-07-26T23:50:24.854107Z","shell.execute_reply.started":"2023-07-26T23:50:24.839486Z","shell.execute_reply":"2023-07-26T23:50:24.852812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['age','split']].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='orange',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the Age column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.856008Z","iopub.execute_input":"2023-07-26T23:50:24.856628Z","iopub.status.idle":"2023-07-26T23:50:24.908032Z","shell.execute_reply.started":"2023-07-26T23:50:24.856594Z","shell.execute_reply":"2023-07-26T23:50:24.907095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=hubmap_df.groupby(['age','split'])['age'].count().reset_index(name='counts')\ndf['age_split']=df['age'].astype('str')+'_'+df['split']\nfig=px.pie(df,values='counts',names='age_split', title='Patient Ages Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n\n    ),\n    paper_bgcolor=\"LightSteelBlue\"\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.909460Z","iopub.execute_input":"2023-07-26T23:50:24.910032Z","iopub.status.idle":"2023-07-26T23:50:24.975000Z","shell.execute_reply.started":"2023-07-26T23:50:24.909997Z","shell.execute_reply":"2023-07-26T23:50:24.973992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['laterality','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.976905Z","iopub.execute_input":"2023-07-26T23:50:24.977837Z","iopub.status.idle":"2023-07-26T23:50:24.988512Z","shell.execute_reply.started":"2023-07-26T23:50:24.977801Z","shell.execute_reply":"2023-07-26T23:50:24.987390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['laterality','split']].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='purple',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the Laterality column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:24.990425Z","iopub.execute_input":"2023-07-26T23:50:24.990778Z","iopub.status.idle":"2023-07-26T23:50:25.046614Z","shell.execute_reply.started":"2023-07-26T23:50:24.990746Z","shell.execute_reply":"2023-07-26T23:50:25.045708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['laterality','split'])['laterality'].count().reset_index(name = 'counts')\ndf['laterality_split'] = df['laterality']+'_'+df['split']\nfig = px.pie(df, values='counts', names='laterality_split', title='Patient Laterality Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.047873Z","iopub.execute_input":"2023-07-26T23:50:25.048934Z","iopub.status.idle":"2023-07-26T23:50:25.116371Z","shell.execute_reply.started":"2023-07-26T23:50:25.048897Z","shell.execute_reply":"2023-07-26T23:50:25.115466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['percent_medulla','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.117638Z","iopub.execute_input":"2023-07-26T23:50:25.118067Z","iopub.status.idle":"2023-07-26T23:50:25.129841Z","shell.execute_reply.started":"2023-07-26T23:50:25.118031Z","shell.execute_reply":"2023-07-26T23:50:25.128688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['percent_medulla','split']].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='pink',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the Medulla column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.131821Z","iopub.execute_input":"2023-07-26T23:50:25.132177Z","iopub.status.idle":"2023-07-26T23:50:25.187905Z","shell.execute_reply.started":"2023-07-26T23:50:25.132145Z","shell.execute_reply":"2023-07-26T23:50:25.186986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=hubmap_df.groupby(['percent_medulla','split'])['percent_medulla'].count().reset_index(name='counts')\ndf['percent_medulla_split']=df['percent_medulla'].astype('str')+'_'+df['split']\nfg=px.pie(df, values='counts', names='percent_medulla_split', title='Patient Percent_Medulla Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.189152Z","iopub.execute_input":"2023-07-26T23:50:25.190190Z","iopub.status.idle":"2023-07-26T23:50:25.253045Z","shell.execute_reply.started":"2023-07-26T23:50:25.190151Z","shell.execute_reply":"2023-07-26T23:50:25.252135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['percent_cortex','split']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.254696Z","iopub.execute_input":"2023-07-26T23:50:25.255324Z","iopub.status.idle":"2023-07-26T23:50:25.267721Z","shell.execute_reply.started":"2023-07-26T23:50:25.255290Z","shell.execute_reply":"2023-07-26T23:50:25.266528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hubmap_df[['percent_cortex','split']].value_counts().iplot(kind='bar',\n                                              yTitle='Counts', \n                                              linecolor='black', \n                                              opacity=0.7,\n                                              color='cyan',\n                                              theme='pearl',\n                                              bargap=0.5,\n                                              gridcolor='white',\n                                              title='Distribution of the Medulla column in the HuBMAP-20 Set')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.269051Z","iopub.execute_input":"2023-07-26T23:50:25.269588Z","iopub.status.idle":"2023-07-26T23:50:25.325267Z","shell.execute_reply.started":"2023-07-26T23:50:25.269552Z","shell.execute_reply":"2023-07-26T23:50:25.324285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['percent_cortex','split'])['percent_cortex'].count().reset_index(name = 'counts')\ndf['percent_cortex_split'] = df['percent_cortex'].astype('str')+'_'+df['split']\nfig = px.pie(df, values='counts', names='percent_cortex_split', title='Patient Percent_Cortex Count')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.326710Z","iopub.execute_input":"2023-07-26T23:50:25.327269Z","iopub.status.idle":"2023-07-26T23:50:25.391650Z","shell.execute_reply.started":"2023-07-26T23:50:25.327230Z","shell.execute_reply":"2023-07-26T23:50:25.390744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution Of Age**","metadata":{}},{"cell_type":"code","source":"df=hubmap_df\nfig=px.violin(df,y='age',x='race',box=True,color='sex',points=\"all\",hover_data=hubmap_df.columns)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.392879Z","iopub.execute_input":"2023-07-26T23:50:25.393996Z","iopub.status.idle":"2023-07-26T23:50:25.516272Z","shell.execute_reply.started":"2023-07-26T23:50:25.393956Z","shell.execute_reply":"2023-07-26T23:50:25.515292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\nsns.kdeplot(df.loc[df['race']=='White','age'],label='White',shade=True)\nsns.kdeplot(df.loc[df['race'] == 'Black or African American', 'age'], label = 'Black or African American',shade=True)\n\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.517756Z","iopub.execute_input":"2023-07-26T23:50:25.518725Z","iopub.status.idle":"2023-07-26T23:50:25.988276Z","shell.execute_reply.started":"2023-07-26T23:50:25.518690Z","shell.execute_reply":"2023-07-26T23:50:25.987277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\nax=sns.violinplot(x=df['age'],y=df['race'],palette='Reds')\nax.set_xlabel(xlabel='age',fontsize=15)\nax.set_ylabel(ylabel = 'race', fontsize = 15)\nax.set_title(label = 'Distribution of age over race', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:25.989715Z","iopub.execute_input":"2023-07-26T23:50:25.990176Z","iopub.status.idle":"2023-07-26T23:50:26.384784Z","shell.execute_reply.started":"2023-07-26T23:50:25.990139Z","shell.execute_reply":"2023-07-26T23:50:26.383732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=hubmap_df.groupby(['age','race'])['age'].count().reset_index(name='counts')\ndf['age_race']=df['age'].astype('str')+'_'+df['race']\nfig = px.pie(df, values='counts', names='age_race', title='Distribution Of Age Over Race')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:26.386290Z","iopub.execute_input":"2023-07-26T23:50:26.387269Z","iopub.status.idle":"2023-07-26T23:50:26.452150Z","shell.execute_reply.started":"2023-07-26T23:50:26.387232Z","shell.execute_reply":"2023-07-26T23:50:26.451261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution of race**","metadata":{}},{"cell_type":"code","source":"df = hubmap_df\nfig=px.violin(df,x='race',y='percent_medulla',box=True,color='sex',points='all',hover_data=hubmap_df.columns)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:26.453612Z","iopub.execute_input":"2023-07-26T23:50:26.454267Z","iopub.status.idle":"2023-07-26T23:50:26.534262Z","shell.execute_reply.started":"2023-07-26T23:50:26.454227Z","shell.execute_reply":"2023-07-26T23:50:26.533261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.violinplot(x = df['race'], y = df['percent_medulla'], palette = 'Reds')\nax.set_xlabel(xlabel = 'race', fontsize = 15)\nax.set_ylabel(ylabel = 'percent_medulla', fontsize = 15)\nax.set_title(label = 'Distribution of race over medulla', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:26.535845Z","iopub.execute_input":"2023-07-26T23:50:26.536504Z","iopub.status.idle":"2023-07-26T23:50:26.881739Z","shell.execute_reply.started":"2023-07-26T23:50:26.536470Z","shell.execute_reply":"2023-07-26T23:50:26.880818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['race','percent_medulla'])['race'].count().reset_index(name = 'counts')\ndf['race_percent_medulla'] = df['race']+'_'+df['percent_medulla'].astype('str')\nfig = px.pie(df, values='counts', names='race_percent_medulla', title='Distribution Of Race Over Percent Medulla')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:26.882979Z","iopub.execute_input":"2023-07-26T23:50:26.884069Z","iopub.status.idle":"2023-07-26T23:50:26.948894Z","shell.execute_reply.started":"2023-07-26T23:50:26.884032Z","shell.execute_reply":"2023-07-26T23:50:26.947814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distiribution of age over percent medulla**","metadata":{}},{"cell_type":"code","source":"df=hubmap_df\nplt.figure(figsize=(16, 6))\nsns.kdeplot(df.loc[df['percent_medulla']==20,'age'],label=20,shade=True)\nsns.kdeplot(df.loc[df['percent_medulla'] == 25, 'age'], label = 25,shade=True)\nsns.kdeplot(df.loc[df['percent_medulla'] == 35, 'age'], label = 35,shade=True)\nsns.kdeplot(df.loc[df['percent_medulla'] == 45, 'age'], label = 45,shade=True)\n\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:26.950407Z","iopub.execute_input":"2023-07-26T23:50:26.950837Z","iopub.status.idle":"2023-07-26T23:50:27.471089Z","shell.execute_reply.started":"2023-07-26T23:50:26.950795Z","shell.execute_reply":"2023-07-26T23:50:27.470129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['age','percent_medulla'])['age'].count().reset_index(name = 'counts')\ndf['age_percent_medulla'] = df['age'].astype('str')+'_'+df['percent_medulla'].astype('str')\nfig = px.pie(df, values='counts', names='age_percent_medulla', title='Distribution Of Age Over Percent Medulla')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:27.472519Z","iopub.execute_input":"2023-07-26T23:50:27.472948Z","iopub.status.idle":"2023-07-26T23:50:27.538057Z","shell.execute_reply.started":"2023-07-26T23:50:27.472914Z","shell.execute_reply":"2023-07-26T23:50:27.537145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df\nplt.figure(figsize=(16, 6))\nsns.kdeplot(df.loc[df['percent_cortex'] == 80, 'age'], label = 20,shade=True)\nsns.kdeplot(df.loc[df['percent_cortex'] == 75, 'age'], label = 25,shade=True)\nsns.kdeplot(df.loc[df['percent_cortex'] == 65, 'age'], label = 35,shade=True)\nsns.kdeplot(df.loc[df['percent_cortex'] == 55, 'age'], label = 45,shade=True)\n\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:27.539510Z","iopub.execute_input":"2023-07-26T23:50:27.540069Z","iopub.status.idle":"2023-07-26T23:50:28.047680Z","shell.execute_reply.started":"2023-07-26T23:50:27.540033Z","shell.execute_reply":"2023-07-26T23:50:28.046730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=hubmap_df.groupby(['age','percent_cortex'])['age'].count().reset_index(name='counts')\ndf['age_percent_cortex'] = df['age'].astype('str')+'_'+df['percent_cortex'].astype('str')\nfig = px.pie(df, values='counts', names='age_percent_cortex', title='Distribution Of Age Over Percent Cortex')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.049042Z","iopub.execute_input":"2023-07-26T23:50:28.049597Z","iopub.status.idle":"2023-07-26T23:50:28.113839Z","shell.execute_reply.started":"2023-07-26T23:50:28.049563Z","shell.execute_reply":"2023-07-26T23:50:28.112780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(hubmap_df['percent_medulla']+hubmap_df['percent_cortex']).unique()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.115741Z","iopub.execute_input":"2023-07-26T23:50:28.116086Z","iopub.status.idle":"2023-07-26T23:50:28.123209Z","shell.execute_reply.started":"2023-07-26T23:50:28.116054Z","shell.execute_reply":"2023-07-26T23:50:28.122276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sex and Race**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na=sns.countplot(data=hubmap_df,x='race',hue='sex')\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Sex split by Race', fontsize=16)\nsns.despine(left=True, bottom=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.125032Z","iopub.execute_input":"2023-07-26T23:50:28.125727Z","iopub.status.idle":"2023-07-26T23:50:28.502825Z","shell.execute_reply.started":"2023-07-26T23:50:28.125692Z","shell.execute_reply":"2023-07-26T23:50:28.501896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(hubmap_df, x=\"sex\", y=\"age\", points=\"all\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.504217Z","iopub.execute_input":"2023-07-26T23:50:28.504885Z","iopub.status.idle":"2023-07-26T23:50:28.589429Z","shell.execute_reply.started":"2023-07-26T23:50:28.504846Z","shell.execute_reply":"2023-07-26T23:50:28.588509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = hubmap_df.groupby(['sex','race'])['sex'].count().reset_index(name = 'counts')\ndf['sex_race'] = df['sex'].astype('str')+'_'+df['race'].astype('str')\nfig = px.pie(df, values='counts', names='sex_race', title='Sex Versus Race')\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=650,\n    margin=dict(\n        l=50,\n        r=50,\n        b=50,\n        t=50,\n        pad=4\n    ),\n    paper_bgcolor=\"LightSteelBlue\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.590837Z","iopub.execute_input":"2023-07-26T23:50:28.591162Z","iopub.status.idle":"2023-07-26T23:50:28.657188Z","shell.execute_reply.started":"2023-07-26T23:50:28.591130Z","shell.execute_reply":"2023-07-26T23:50:28.656299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"male = hubmap_df[hubmap_df['sex']=='Male']\nfemale = hubmap_df[hubmap_df['sex']== 'Female']\ncount_male = male[['race','split']].value_counts().reset_index()\ncount_female = female[['race','split']].value_counts().reset_index()\npie_male = go.Pie(labels=count_male[['race','split']],values=count_male[0],name=\"Male\",hole=0.4,domain={'x': [0,0.46]})\npie_female = go.Pie(labels=count_female[['race','split']],values=count_female[0],name=\"Female\",hole=0.5,domain={'x': [0.52,1]})\nlayout = dict(title = 'Sex_Race', font=dict(size=10), legend=dict(orientation=\"h\"),\n              annotations = [dict(x=0.2, y=0.5, text='Male', showarrow=False, font=dict(size=20)),\n                             dict(x=0.8, y=0.5, text='Female', showarrow=False, font=dict(size=20)) ])\n\nfig = dict(data=[pie_male, pie_female], layout=layout)\npyo.iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.658556Z","iopub.execute_input":"2023-07-26T23:50:28.659097Z","iopub.status.idle":"2023-07-26T23:50:28.712022Z","shell.execute_reply.started":"2023-07-26T23:50:28.659057Z","shell.execute_reply":"2023-07-26T23:50:28.711107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"male = hubmap_df[hubmap_df['sex']=='Male']\nfemale = hubmap_df[hubmap_df['sex']== 'Female']\ncount_male = male['race'].value_counts().reset_index()\ncount_female = female['race'].value_counts().reset_index()\npie_male = go.Pie(labels=count_male['index'],values=count_male['race'],name=\"Male\",hole=0.4,domain={'x': [0,0.46]})\npie_female = go.Pie(labels=count_female['index'],values=count_female['race'],name=\"Female\",hole=0.5,domain={'x': [0.52,1]})\nlayout = dict(title = 'Sex_Race', font=dict(size=10), legend=dict(orientation=\"h\"),\n              annotations = [dict(x=0.2, y=0.5, text='Male', showarrow=False, font=dict(size=20)),\n                             dict(x=0.8, y=0.5, text='Female', showarrow=False, font=dict(size=20)) ])\n\nfig = dict(data=[pie_male, pie_female], layout=layout)\npyo.iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.713462Z","iopub.execute_input":"2023-07-26T23:50:28.714100Z","iopub.status.idle":"2023-07-26T23:50:28.762252Z","shell.execute_reply.started":"2023-07-26T23:50:28.714058Z","shell.execute_reply":"2023-07-26T23:50:28.761358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**HeatMap**","metadata":{}},{"cell_type":"code","source":"corrmat=hubmap_df.corr()\nf,ax=plt.subplots(figsize=(9,8))\nsns.heatmap(corrmat,ax=ax,cmap='RdYlBu_r',linewidths = 0.5)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:28.763767Z","iopub.execute_input":"2023-07-26T23:50:28.764195Z","iopub.status.idle":"2023-07-26T23:50:29.525330Z","shell.execute_reply.started":"2023-07-26T23:50:28.764155Z","shell.execute_reply":"2023-07-26T23:50:29.524278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Fore.YELLOW + 'Train .tiff number of images:',Style.RESET_ALL, len(list(os.listdir('../input/hubmap-kidney-segmentation/train')))/3, '\\n' +\n      Fore.BLUE + 'Test .tiff number of images:',Style.RESET_ALL, len(list(os.listdir('../input/hubmap-kidney-segmentation/test')))/2)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:29.526997Z","iopub.execute_input":"2023-07-26T23:50:29.527688Z","iopub.status.idle":"2023-07-26T23:50:29.536813Z","shell.execute_reply.started":"2023-07-26T23:50:29.527652Z","shell.execute_reply":"2023-07-26T23:50:29.535825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = '../input/hubmap-kidney-segmentation'\nos.path.join(IMAGE_PATH, 'train/095bf7a1f.tiff')","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:29.538314Z","iopub.execute_input":"2023-07-26T23:50:29.538675Z","iopub.status.idle":"2023-07-26T23:50:29.547679Z","shell.execute_reply.started":"2023-07-26T23:50:29.538643Z","shell.execute_reply":"2023-07-26T23:50:29.546377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im=tiff.imread(\n\n    os.path.join(IMAGE_PATH,\"train/0486052bb.tiff\")\n)\nplt.figure(figsize=(10,10))\nplt.imshow(im)\nplt.axis(\"off\")\ndel im","metadata":{"execution":{"iopub.status.busy":"2023-07-26T23:50:29.548745Z","iopub.execute_input":"2023-07-26T23:50:29.549030Z","iopub.status.idle":"2023-07-26T23:52:28.912196Z","shell.execute_reply.started":"2023-07-26T23:50:29.548998Z","shell.execute_reply":"2023-07-26T23:52:28.911293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}