{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['Sex'] = df['Sex'].apply(lambda x : 1 if x=='Male' else 0)\ndf['SmokingStatus'] = df['SmokingStatus'].apply(lambda x : 0 if x=='Never smoked' else (1 if x=='Currently smokes'  else 2 ))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"len(df.Patient.unique()),df.shape,df.drop_duplicates().shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df.groupby(['Patient']).Weeks.nunique().describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df.groupby(['SmokingStatus']).Patient.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df.groupby(['Age']).Patient.nunique(()).plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df[df.SmokingStatus==0].groupby(['Age']).Patient.nunique(()).plot(label='ex-smoker')\ndf[df.SmokingStatus==1].groupby(['Age']).Patient.nunique(()).plot(label='currently smokes')\ndf[df.SmokingStatus==2].groupby(['Age']).Patient.nunique(()).plot(label='never smoked')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df.groupby(['Patient'],as_index=False).Percent.min().plot()\ndf.groupby(['Patient'],as_index=False).Percent.max().plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pydicom\nimport matplotlib.pyplot as plt\nimport glob\n%matplotlib inline\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"patient_ls = df.Patient.unique()\nfor i in range(10):\n    data_sub = df[df.Patient==patient_ls[i]]\n    min_week = data_sub[data_sub.Percent == data_sub.Percent.min()].Weeks.iloc[0]\n    max_week = data_sub[data_sub.Percent == data_sub.Percent.max()].Weeks.iloc[0]\n    if min_week<=0:    min_week=1\n    elif max_week<=0:  max_week=1\n    images = glob.glob(f\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/{patient_ls[i]}/*.dcm\")\n    week_num = [file for file in images if int(file.split('/')[-1].split(\".\")[0]) in [min_week,max_week]]\n    try :\n        plt.figure(figsize=(15,5))\n        for idx,file in enumerate(week_num):\n            plt.subplot(1,3,idx+1)\n            img = pydicom.dcmread(file)\n            plt.title(file.split('/')[-1].split(\".\")[0])\n            plt.imshow(img.pixel_array)\n\n\n        plt.subplot(1,3,3)\n        plt.plot('Weeks','Percent',marker='o',data=data_sub)\n        plt.title(patient_ls[i])\n    except :\n        pass\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df.head(9)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = glob.glob(f\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/*.dcm\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"set([int(file.split('/')[-1].split(\".\")[0]) for file in images])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = pydicom.dcmread(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/1.dcm\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test_data = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv\")\nsub_data = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_data['Patient'] = sub_data.Patient_Week.apply(lambda x: x.split(\"_\")[0])\nsub_data['Weeks'] = sub_data.Patient_Week.apply(lambda x: x.split(\"_\")[1]).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"list(sub_data[sub_data.Patient == 'ID00419637202311204720264'].Weeks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"set([int(file.split('/')[-1].split(\".\")[0]) for file in images])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport glob\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pydicom\n%matplotlib inline\n\nimport torch\nimport torch.nn as nn\nfrom torch.nn import Sequential\n\nfrom skimage.segmentation import clear_border\nfrom skimage.measure import regionprops,label\nfrom skimage import segmentation,measure\nimport scipy.ndimage as ndimage","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv\")\ntrain_images = glob.glob(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/*/*\")\ntrain_images[:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_scan(images_ls):\n    slices = [pydicom.read_file(file) for file in images_ls]\n    slices.sort(key = lambda x: x.InstanceNumber)\n    \n    try :\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2]-slices[1].ImagePositionPatient[2])\n    except :\n        slice_thickness = np.abs(slices[0].SliceLocation-slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceWidth = slice_thickness\n    \n    return slices\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def image_hu(slices):\n    img_array = np.stack([s.pixel_array for s in slices]).astype(np.int16)\n    slope = slices[0].RescaleSlope\n    intercept = slices[0].RescaleIntercept\n    img_array[img_array==-2000] = 0\n    \n    img_array  = slope*img_array + intercept\n    return img_array.astype(np.int16)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def creater_markers(img):\n    internal1 = img<-400\n    internal2 = segmentation.clear_border(internal1)\n    internal_labels = measure.label(internal2)\n    internal_labels1 = internal_labels.copy()\n    \n    areas =[r.area for r in measure.regionprops(internal_labels)]\n    areas.sort()\n    if len(areas)>2:\n        for region in measure.regionprops(internal_labels):\n            if region.area<areas[-2]:\n                for coord in region.coords:\n                    internal_labels[coord[0],coord[1]] = 0\n    \n    internal_labels = internal_labels>0\n    \n    external_a = ndimage.binary_dilation(internal_labels,iterations=10)\n    external_b = ndimage.binary_dilation(internal_labels,iterations=55)\n    external = external_b^external_a\n    \n    watershed = np.zeros((512,512),dtype=np.int)\n    watershed1 = watershed + internal_labels*255\n    watershed2 = watershed1 + external*128\n        \n    return internal_labels,external,watershed2\n    \n    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oneimage = load_scan(train_images[:2])\nplt.figure(figsize=(15,5))\nout_ls = creater_markers(image_hu(oneimage)[0])\n\nplt.subplot(1,len(out_ls)+1,1)\nplt.imshow(image_hu(oneimage)[0],cmap='gray')\nfor i in range(len(out_ls)):\n    try :\n        plt.subplot(1,len(out_ls)+1,i+2)\n        plt.imshow(out_ls[i],cmap='gray')\n    except : pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}