{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom\nimport scipy.ndimage\n\nfrom skimage import measure \nfrom mpl_toolkits.mplot3d.art3d import Poly3DCollection\n\nfrom IPython.display import HTML\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nfrom os import listdir","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"basepath = \"../input/osic-pulmonary-fibrosis-progression/\"\nlistdir(basepath)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(basepath + \"train.csv\")\ntest = pd.read_csv(basepath + \"test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef load_scans(dcm_path):\n    slices = [pydicom.dcmread(dcm_path + \"/\" + file) for file in listdir(dcm_path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    return slices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"example = basepath + \"train/\" + train.Patient.values[0]\nscans = load_scans(example)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.isnull().values.any() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scans[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\nfor n in range(10):\n    image = scans[n].pixel_array.flatten()\n    rescaled_image = image * scans[n].RescaleSlope + scans[n].RescaleIntercept\n    sns.distplot(image.flatten(), ax=ax[0]);\n    sns.distplot(rescaled_image.flatten(), ax=ax[1])\nax[0].set_title(\"Raw pixel array distributions for 10 examples\")\nax[1].set_title(\"HU unit distributions for 10 examples\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def transform_to_hu(slices):\n    images = np.stack([file.pixel_array for file in slices])\n    images = images.astype(np.int16)\n\n    # convert ouside pixel-values to air:\n    # I'm using <= -1000 to be sure that other defaults are captured as well\n    images[images <= -1000] = 0\n    \n    # convert to HU\n    for n in range(len(slices)):\n        \n        intercept = slices[n].RescaleIntercept\n        slope = slices[n].RescaleSlope\n        \n        if slope != 1:\n            images[n] = slope * images[n].astype(np.float64)\n            images[n] = images[n].astype(np.int16)\n            \n        images[n] += np.int16(intercept)\n    \n    return np.array(images, dtype=np.int16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"hu_scans = transform_to_hu(scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfig, ax = plt.subplots(1,4,figsize=(20,3))\nax[0].set_title(\"Original CT-scan\")\nax[0].imshow(scans[0].pixel_array, cmap=\"bone\")\nax[1].set_title(\"Pixelarray distribution\");\nsns.distplot(scans[0].pixel_array.flatten(), ax=ax[1]);\n\nax[2].set_title(\"CT-scan in HU\")\nax[2].imshow(hu_scans[0], cmap=\"bone\")\nax[3].set_title(\"HU values distribution\");\nsns.distplot(hu_scans[0].flatten(), ax=ax[3]);\n\nfor m in [0,2]:\n    ax[m].grid(False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pixelspacing_w = []\npixelspacing_h = []\nslice_thicknesses = []\npatient_id = []\npatient_pth = []\n\nfor patient in train.Patient.values:\n    patient_id.append(patient)\n    example_dcm = listdir(basepath + \"train/\" + patient + \"/\")[0]\n    patient_pth.append(basepath + \"train/\" + patient)\n    dataset = pydicom.dcmread(basepath + \"train/\" + patient + \"/\" + example_dcm)\n    spacing = dataset.PixelSpacing\n    slice_thicknesses.append(dataset.SliceThickness)\n    pixelspacing_w.append(spacing[0])\n    pixelspacing_h.append(spacing[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\nsns.distplot(pixelspacing_w, ax=ax[0], color=\"Limegreen\", kde=False)\nax[0].set_title(\"Pixel spacing width \\n distribution\")\nax[0].set_ylabel(\"Counts in train\")\nax[0].set_xlabel(\"width in mm\")\nsns.distplot(pixelspacing_h, ax=ax[1], color=\"Mediumseagreen\", kde=False)\nax[1].set_title(\"Pixel spacing height \\n distribution\");\nax[1].set_ylabel(\"Counts in train\");\nax[1].set_xlabel(\"height in mm\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,5))\nsns.distplot(slice_thicknesses, color=\"orangered\", kde=False)\nplt.title(\"Slice thicknesses of all patients\");\nplt.xlabel(\"Slice thickness in mm\")\nplt.ylabel(\"Counts in train\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_attribute = pixelspacing_w","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"min_idx = np.argsort(my_attribute)[0]\nmax_idx = np.argsort(my_attribute)[-1]\n\npatient_min = patient_pth[min_idx]\npatient_max = patient_pth[max_idx]\n\nmin_scans = load_scans(patient_min)\nmin_hu_scans = transform_to_hu(min_scans)\n\nmax_scans = load_scans(patient_max)\nmax_hu_scans = transform_to_hu(max_scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_3d(image, threshold=700, color=\"navy\"):\n    \n    # Position the scan upright, \n    # so the head of the patient would be at the top facing the camera\n    p = image.transpose(2,1,0)\n    \n    verts, faces,_,_ = measure.marching_cubes_lewiner(p, threshold)\n\n    fig = plt.figure(figsize=(10, 10))\n    ax = fig.add_subplot(111, projection='3d')\n\n    # Fancy indexing: `verts[faces]` to generate a collection of triangles\n    mesh = Poly3DCollection(verts[faces], alpha=0.5)\n    mesh.set_facecolor(color)\n    ax.add_collection3d(mesh)\n\n    ax.set_xlim(0, p.shape[0])\n    ax.set_ylim(0, p.shape[1])\n    ax.set_zlim(0, p.shape[2])\n\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_3d(max_hu_scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def resample(image, scan, new_spacing=[1,1,1]):\n    # Determine current pixel spacing\n    spacing = np.array([scan[0].SliceThickness] + list(scan[0].PixelSpacing), dtype=np.float32)\n\n    resize_factor = spacing / new_spacing\n    new_real_shape = image.shape * resize_factor\n    new_shape = np.round(new_real_shape)\n    real_resize_factor = new_shape / image.shape\n    new_spacing = spacing / real_resize_factor\n    \n    image = scipy.ndimage.interpolation.zoom(image, real_resize_factor, mode='nearest')\n    \n    return image, new_spacing","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pix_resampled, spacing = resample(max_hu_scans, scans, [1,1,1])\nprint(\"Shape before resampling\\t\", hu_scans.shape)\nprint(\"Shape after resampling\\t\", pix_resampled.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def segment_tissue(image, threshold=-300, fill_lung_structures=True):\n    \n    labelled_image = np.array(image > threshold, dtype=np.int8)+1\n    labels = measure.label(labelled_image)\n    \n    \n    #Fill the air around the person\n    background_label = labels[0,0]\n    labelled_image[background_label == labels] = 2\n    \n    labelled_image -= 1 #Make the image actual binary\n    labelled_image = 1-labelled_image # Invert it, lungs are now 1\n    \n    if fill_lung_structures:\n        # For every slice we determine the largest solid structure\n        for i, axial_slice in enumerate(labelled_image):\n            axial_slice = axial_slice - 1\n            labeling = measure.label(axial_slice)\n            l_max = largest_label_volume(labeling, bg=0)\n            \n            if l_max is not None: #This slice contains some lung\n                labelled_image[i][labeling != l_max] = 1\n    \n    # Remove other air pockets insided body\n    labels = measure.label(labelled_image, background=0)\n    l_max = largest_label_volume(labels, bg=0)\n    if l_max is not None: # There are air pockets\n        labelled_image[labels != l_max] = 0\n    \n    return labelled_image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"segmented_lungs = segment_tissue(max_hu_scans, fill_lung_structures=False)\nsegmented_lungs_fill = segment_tissue(max_hu_scans, fill_lung_structures=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,10))\nax[0].imshow(segmented_lungs[20], cmap=\"bone_r\")\nax[1].imshow(segmented_lungs_fill[20], cmap=\"bone_r\")\nfor n in range(2):\n    ax[n].grid(False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_3d(segmented_lungs_fill, threshold=0, color=\"crimson\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_3d(segmented_lungs_fill-segmented_lungs, threshold=0, color=\"crimson\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!conda install -c conda-forge gdcm -y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport os\nfrom sympy import symbols\nfrom sympy import expand\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.ensemble import RandomForestRegressor\nfrom skimage import morphology\nfrom skimage import measure\nfrom skimage.transform import resize\nimport tensorflow as tf\nfrom sklearn.cluster import KMeans\nimport matplotlib.patches as patches\nimport PIL","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_ids=train_csv['Patient'].unique()\nweek_x=[]\nfvc_y=[]\npercent=[]\nfor id in unique_ids:\n    week=np.array(train_csv[train_csv['Patient']==id]['Weeks'])\n    fvc=np.array(train_csv[train_csv['Patient']==id]['FVC'])\n    per=np.array(train_csv[train_csv['Patient']==id]['Percent'])\n    week_x.append(week)\n    fvc_y.append(fvc)\n    percent.append(per)\n    \nunique_train=pd.DataFrame(train_csv['Patient'].unique(),columns=['Patient'])\nunique_train['week_x']=week_x\nunique_train['fvc_y']=fvc_y\nunique_train['percent']=percent\nunique_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=unique_train['week_x'][0]\nY=unique_train['fvc_y'][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def using_poly_reg(X,Y,degree=3):\n    poly_features=PolynomialFeatures(degree=degree,include_bias=False)\n    x_poly=poly_features.fit_transform(X[:,np.newaxis])\n\n    lin_reg=LinearRegression()\n    lin_reg.fit(x_poly,Y)\n\n    x_test=np.arange(-12,133)[:,np.newaxis]\n    x_test_poly=poly_features.fit_transform(x_test)\n    plt.plot(x_test,lin_reg.predict(x_test_poly))\n    plt.plot(X,Y)\n    #plt.ylim(0,6400)\n    #plt.xlim(-12,133)\n    plt.grid(True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"using_poly_reg(X,Y,degree=3)  #here we can customize the degree of the polynomial so it is better\n                              #by the way if degree=len(X)-1 then it is same as interpolation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv['healthy_person_FVC']=(train_csv['FVC']/(train_csv['Percent']/100)).round()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"healthy_fvc_info=train_csv.groupby(['Age','Sex','SmokingStatus'])['healthy_person_FVC'].mean().round()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(healthy_fvc_info[:,'Male','Ex-smoker'],label='male ex smoker')\nplt.plot(healthy_fvc_info[:,'Male','Never smoked'],label='male never smoked')\nplt.plot(healthy_fvc_info[:,'Male','Currently smokes'],label='male currently smokes')\n\nplt.plot(healthy_fvc_info[:,'Female','Ex-smoker'],label='female ex smoker')\nplt.plot(healthy_fvc_info[:,'Female','Never smoked'],label='female never smoked')\nplt.plot(healthy_fvc_info[:,'Female','Currently smokes'],label='female currently smokes')\n\nplt.title('healthy fvc related to age,sex and smoking status')\nplt.legend()\nplt.grid(True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def RForestRegressor(x,y):\n    reg=RandomForestRegressor(n_estimators=50)\n    reg.fit(x[:,np.newaxis],y)\n    x_test=np.arange(0,100)\n    y_test=reg.predict(x_test[:,np.newaxis])\n    plt.plot(x_test,y_test,label='predicted')\n    plt.plot(x,y,label='real')\n    plt.grid(True)\n    plt.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#x=np.array(healthy_fvc_info[:,'Male','Ex-smoker'].index)\n#y=np.array(healthy_fvc_info[:,'Male','Ex-smoker'].values)\nx=np.array(healthy_fvc_info[:,'Male','Never smoked'].index)\ny=np.array(healthy_fvc_info[:,'Male','Never smoked'].values)\n\nRForestRegressor(x,y)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=unique_train['week_x'][0]\nY=unique_train['fvc_y'][0]\nRForestRegressor(X,Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"age=train_csv.groupby('Patient')['Age'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for item in age:\n    if len(item)==1:\n        continue\n    else:\n        print(item.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SS=train_csv.groupby('Patient')['SmokingStatus'].unique()\nfor item in SS:\n    if len(item)==1:\n        continue\n    else:\n        print(item)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sex=[]\nfor id in unique_train['Patient']:\n    sex.append(train_csv[train_csv['Patient']==id]['Sex'].unique()[0])\n    \nunique_train['sex']=sex","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"age=[]\nfor id in unique_train['Patient']:\n    age.append(train_csv[train_csv['Patient']==id]['Age'].unique()[0])\n    \nunique_train['age']=age","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss=[]\nfor id in unique_train['Patient']:\n    ss.append(train_csv[train_csv['Patient']==id]['SmokingStatus'].unique()[0])\n    \nunique_train['smoking-status']=ss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''from sklearn.cluster import KMeans\n\nlung=pydicom.dcmread('../input/osic-pulmonary-fibrosis-progression/train/ID00012637202177665765362/26.dcm')\nimage=lung.pixel_array\nX = image.reshape(-1,1)\n#X=image\n\n#good_init=np.array([[-2048],[-1000],[892],[-177],[190]])\n#kmeans = KMeans(n_clusters=8,init=good_init,n_init=1).fit(X)\n\nkmeans = KMeans(n_clusters=6).fit(X)\n\nsegmented_img = kmeans.cluster_centers_[kmeans.labels_]\nsegmented_img = segmented_img.reshape(image.shape)\nplt.imshow(segmented_img)'''\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def fitter(img):\n    row_size= img.shape[0]\n    col_size = img.shape[1]\n    \n    mean = np.mean(img)\n    std = np.std(img)\n    img = img-mean\n    img = img/std\n    # Find the average pixel value near the lungs\n    # to renormalize washed out images\n    middle = img[int(col_size/4):int(col_size/4*3),int(row_size/4):int(row_size/4*3)] \n    mean = np.mean(middle)  \n    max = np.max(img)\n    min = np.min(img)\n    # To improve threshold finding, I'm moving the \n    # underflow and overflow on the pixel spectrum\n    img[img==max]=mean\n    img[img==min]=mean\n    \n    #\n    # Using Kmeans to separate foreground (soft tissue / bone) and background (lung/air)\n    #\n    kmeans = KMeans(n_clusters=2).fit(np.reshape(middle,[np.prod(middle.shape),1]))\n    return kmeans\n\nlung=pydicom.dcmread('../input/osic-pulmonary-fibrosis-progression/train/ID00012637202177665765362/26.dcm')\nimage=lung.pixel_array\n\nkmeans=fitter(image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def make_lungmask(img,kmeans,display=False):\n    image=img\n    row_size= img.shape[0]\n    col_size = img.shape[1]\n    mean = np.mean(img)\n    std = np.std(img)\n    img = img-mean\n    img = img/std\n    # Find the average pixel value near the lungs\n    # to renormalize washed out images\n    middle = img[int(col_size/4):int(col_size/4*3),int(row_size/4):int(row_size/4*3)] \n    mean = np.mean(middle)  \n    max = np.max(img)\n    min = np.min(img)\n    # To improve threshold finding, I'm moving the \n    # underflow and overflow on the pixel spectrum\n    img[img==max]=mean\n    img[img==min]=mean\n    #\n    # Using Kmeans to separate foreground (soft tissue / bone) and background (lung/air)\n\n    \n    #\n    centers = sorted(kmeans.cluster_centers_.flatten())\n    threshold = np.mean(centers)\n    thresh_img = np.where(img<threshold,1.0,0.0)  # threshold the image\n\n    # First erode away the finer elements, then dilate to include some of the pixels surrounding the lung.  \n    # We don't want to accidentally clip the lung.\n\n    eroded = morphology.erosion(thresh_img,np.ones([5,5]))\n    dilation = morphology.dilation(eroded,np.ones([8,8]))\n\n    labels = measure.label(dilation) # Different labels are displayed in different colors\n    label_vals = np.unique(labels)\n    regions = measure.regionprops(labels)\n    good_labels = []\n    #for prop in regions:\n    #    b = prop.bbox\n    #    if (abs((b[2]+b[0])/2-(row_size/2))<100) and ( (abs((b[3]+b[1])/2-(col_size/4))<110) or (abs((b[3]+b[1])/2-(col_size/4)*3)<110) ):\n    #        good_labels.append(prop.label)\n    \n\n    for prop in regions:\n        b = prop.bbox\n        lung_row=abs((b[2]+b[0])/2-(row_size/2))\n        left_lung_col=abs((b[3]+b[1])/2-(col_size/4))\n        right_lung_col=abs((b[3]+b[1])/2-(col_size/4)*3)\n        \n        if lung_row<100 and (left_lung_col<110 or right_lung_col<110):\n            good_labels.append(prop.label)\n            \n    mask = np.ndarray([row_size,col_size],dtype=np.int8)\n    mask[:] = 0\n\n    #\n    #  After just the lungs are left, we do another large dilation\n    #  in order to fill in and out the lung mask \n    #\n    for N in good_labels:\n        mask = mask + np.where(labels==N,1,0)\n    mask = morphology.dilation(mask,np.ones([8,8])) # one last dilation\n    \n    \n    if (display):\n        fig, ax = plt.subplots(3, 2, figsize=[12, 12])\n        ax[0, 0].set_title(\"Original\")\n        ax[0, 0].imshow(img, cmap='gray')\n        ax[0, 0].axis('off')\n        ax[0, 1].set_title(\"Threshold\")\n        ax[0, 1].imshow(thresh_img, cmap='gray')\n        ax[0, 1].axis('off')\n        ax[1, 0].set_title(\"After Erosion and Dilation\")\n        ax[1, 0].imshow(dilation, cmap='gray')\n        ax[1, 0].axis('off')\n        ax[1, 1].set_title(\"Color Labels\")\n        ax[1, 1].imshow(labels)\n        ax[1, 1].axis('off')\n        ax[2, 0].set_title(\"Final Mask\")\n        ax[2, 0].imshow(mask, cmap='gray')\n        ax[2, 0].axis('off')\n        ax[2, 1].set_title(\"Apply Mask on Original\")\n        ax[2, 1].imshow(mask*img, cmap='gray')\n        ax[2, 1].axis('off')\n        \n        plt.show()\n        \n        \n    air=[]\n    for i in range(image.shape[0]):\n        for j in range(image.shape[1]):\n            if mask[i][j]==1:\n                air.append(image[i][j])\n    if len(air)==0:\n        air_percent=0.0\n    else:\n        air_percent=abs((sum(air)/len(air))/10).round(4)\n    return mask,air_percent","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"id='ID00011637202177653955184'\npath='../input/osic-pulmonary-fibrosis-progression/train/'+id+'/' \nfilenames=os.listdir(path) \nfileno=int(len(filenames)/2)\nimg=pydicom.dcmread(path+str(fileno)+'.dcm')\n#make_lungmask(img,kmeans,display=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=plt.figure(figsize=(20,20)) \ncol=14\nrow=14 \ni=1 \nair_percent_dict={}\nfor id in train_csv['Patient'].unique(): \n    path='../input/osic-pulmonary-fibrosis-progression/train/'+id+'/' \n    filenames=os.listdir(path) \n    fileno=int(len(filenames)/2)\n    try:\n        lung=pydicom.dcmread(path+str(fileno)+'.dcm') \n        image=lung.pixel_array \n        fig.add_subplot(row,col,i) \n        mask,air_percent=make_lungmask(image,kmeans,display=False)\n        air_percent_dict[id]=air_percent\n        plt.title(air_percent)\n        plt.imshow(mask,cmap='gray')\n        plt.grid(False)\n        plt.axis(False)\n    except: \n        print(id) \n    i=i+1 \n    if i==177: \n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train=train_csv\nfrom sklearn.preprocessing import LabelEncoder\nlb=LabelEncoder()#sex\ntrain.iloc[:,5]=lb.fit_transform(train.iloc[:,5])\nlb2=LabelEncoder()#ss\ntrain.iloc[:,6]=lb2.fit_transform(train.iloc[:,6])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lung_percent=[]\nfor id in train['Patient']:\n    lung_percent.append(float(air_percent_dict[id]))\ntrain['lung percent']=lung_percent\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\ntest.iloc[:,5]=lb.transform(test.iloc[:,5])\ntest.iloc[:,6]=lb2.transform(test.iloc[:,6])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"air_percent_dict={}\nfor id in test['Patient'].unique(): \n    path='../input/osic-pulmonary-fibrosis-progression/test/'+id+'/' \n    filenames=os.listdir(path) \n    fileno=int(len(filenames)/2)\n    try:\n        lung=pydicom.dcmread(path+str(fileno)+'.dcm') \n        image=lung.pixel_array  \n        mask,air_percent=make_lungmask(image,kmeans,display=False)\n        air_percent_dict[id]=air_percent\n    except: \n        print(id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lung_percent_test=[]\nfor id in test['Patient']:\n    lung_percent_test.append(float(air_percent_dict[id]))\ntest['lung percent']=lung_percent_test\ntest","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.to_csv('changed_train.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def healthy_fvc_predictor(age,sex,smoking_status):\n    x=np.array(healthy_fvc_info[:,sex,smoking_status].index)\n    y=np.array(healthy_fvc_info[:,sex,smoking_status].values)\n    reg=RandomForestRegressor(n_estimators=50)\n    reg.fit(x[:,np.newaxis],y)\n    return reg.predict([[age]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def RForestRegressor(x,y):\n    reg=RandomForestRegressor(n_estimators=400)\n    reg.fit(np.array(x),np.array(y))\n    return reg\n\nx=train[['Weeks','Percent','SmokingStatus','Sex','Age']]\n#x=train[['Percent','lung percent','Weeks','Sex']]\ny=train['FVC']\npercent_reg=RForestRegressor(x,y)\n\n\n\ntest_csv=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_fi(forest,X):\n    importances = forest.feature_importances_\n    std = np.std([tree.feature_importances_ for tree in forest.estimators_],axis=0)\n    indices = np.argsort(importances)[::-1]\n\n    # Print the feature ranking\n    print(\"Feature ranking:\")\n\n    for f in range(X.shape[1]):\n        print(\"%d. feature : %s (%f)\" % (f + 1, np.array(X.columns)[indices[f]], importances[indices[f]]))\n\n    # Plot the impurity-based feature importances of the forest\n    plt.figure()\n    plt.title(\"Feature importances\")\n    plt.bar(range(X.shape[1]), importances[indices],color=\"g\", yerr=std[indices])\n    plt.xticks(range(X.shape[1]),np.array(X.columns)[indices])\n    plt.xlim([-1, X.shape[1]])\n    plt.show()\n    \nplot_fi(percent_reg,x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_csv=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\ntest_csv.iloc[:,5]=lb.transform(test_csv.iloc[:,5])\ntest_csv.iloc[:,6]=lb2.transform(test_csv.iloc[:,6])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weeks=np.arange(-12,134)\nresult={}\nfor id in test_csv['Patient'].unique():\n    percent=np.array(test_csv[test_csv['Patient']==id]['Percent'])\n    sex=np.array(test_csv[test_csv['Patient']==id]['Sex'])\n    age=np.array(test_csv[test_csv['Patient']==id]['Age'])\n    ss=np.array(test_csv[test_csv['Patient']==id]['SmokingStatus'])\n    percent=np.repeat(percent,len(weeks))\n    sex=np.repeat(sex,len(weeks))\n    age=np.repeat(age,len(weeks))\n    ss=np.repeat(ss,len(weeks))\n    x=np.concatenate([weeks[:,np.newaxis],percent[:,np.newaxis],ss[:,np.newaxis],sex[:,np.newaxis],age[:,np.newaxis]],axis=1)\n    outcome=percent_reg.predict(x)\n    result[id]=outcome","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ans_df_list=[]\nfor i,id in enumerate(result):\n    ID=np.repeat(id,len(weeks))\n    ans=np.concatenate([ID[:,np.newaxis],weeks[:,np.newaxis],result[id][:,np.newaxis]],axis=1)\n    ans=pd.DataFrame(ans)\n    ans_df_list.append(ans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit=pd.concat(ans_df_list,ignore_index=True)\nsubmit.columns=['Patient','Weeks','FVC']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_csv=pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"confidence_dict={}\nfor id in submit['Patient'].unique():\n    real=float(test_csv[test_csv['Patient']==id]['FVC'])\n    predicted=float(submit[(submit['Patient']==id) & (submit['Weeks'].astype(int)==int(test_csv[test_csv['Patient']==id]['Weeks']))]['FVC'])\n    confidence_dict[id]=abs(real-predicted)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"confidence=[]\nfor i in range(len(submit)):\n    confidence.append(confidence_dict[submit.iloc[i,0]])\nsubmit['Confidence']=confidence","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit['Patient']=submit['Patient']+'_'+(submit['Weeks'].astype(str))\nsubmit.drop(['Weeks'],axis=1,inplace=True)\nsubmit.columns=['Patient_Week','FVC','Confidence']\nsubmit['FVC']=submit['FVC'].astype(float)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}