{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# import the packages and H2ODeepLearningEstimator object\nimport h2o\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pydicom as dcm\n\nfrom pathlib import Path\nfrom os.path import join, isfile\nimport os\nimport fnmatch\nimport math\nfrom random import *\n\nfrom h2o.estimators.deeplearning import H2ODeepLearningEstimator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Starting H2O\nh2o.init()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import the OSIC Pulmonary Fibrosis  data\ntrain = h2o.import_file(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest = h2o.import_file(\"/kaggle/input/osic-pulmonary-fibrosis-progression/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get a brief summary of the data\nprint(\"dimension of train data: \",train.dim)\nprint(\"dimension of test data: \",test.dim)\n\n\nprint(\"description of train data: \\n\" )\ntrain.describe()\nprint(\"description of test data: \\n\" )\ntest.describe()\n\n# Group number of pateints by sex\nSex_patients = train.group_by(\"Sex\")\nSex_patients.count()\nSex_patients.get_frame()\n\n# Group number of pateints by age\nAge_patients = train.group_by(\"Age\")\nAge_patients.count()\nAge_patients.get_frame()\n\n# Group number of pateints by SmokingStatus\nSmokingStatus_patients = train.group_by(\"SmokingStatus\")\nSmokingStatus_patients.count()\nSmokingStatus_patients.get_frame()\n\n# Find number of patients per Sex based on the Age\ncols = [\"Sex\",\"Age\"]\npatients_by_Sex_Age = train.group_by(by=cols).count()\npatients_by_Sex_Age.get_frame()\n\n# Mean impute the FVC column based on the Sex and Age columns\nFVC_imputeSA = train.impute(\"FVC\", method = \"mean\", by = [\"Sex\", \"Age\"])\nFVC_imputeSA\n\n# Mean impute the FVC column based on the Sex and SmokingStatus columns\nFVC_imputeSSS = train.impute(\"FVC\", method = \"mean\", by = [\"Sex\", \"SmokingStatus\"])\nFVC_imputeSSS\n\n# Mean impute the FVC column based on the Age and SmokingStatus columns\nFVC_imputeASS = train.impute(\"FVC\", method = \"mean\", by = [\"Age\", \"SmokingStatus\"])\nFVC_imputeASS\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Show dcm images\nworkingFolder = os.getcwd()\nlistFolderDCM=os.listdir(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train\")\ndirpath=Path(f'/kaggle/input/osic-pulmonary-fibrosis-progression/train')\n\nlistFolderDCM100=[]\nfor varFolder in listFolderDCM :\n  \n   varFolder1=Path(f'{dirpath}/{varFolder}')\n   lenFolder=len([name for name in os.listdir(varFolder1) if os.path.isfile(os.path.join(varFolder1, name))])\n   if lenFolder <=100 :\n      listFolderDCM100.append(varFolder)\n      \nfolderDCMSample=choice(listFolderDCM100)\npathDCM= Path(f'/kaggle/input/osic-pulmonary-fibrosis-progression/train/{folderDCMSample}/')\n\nnumberDCM=len(fnmatch.filter(os.listdir(pathDCM), \"*.dcm\"))\nquotient = numberDCM / (int(math.sqrt(numberDCM))*int(math.sqrt(numberDCM)))\nquotient=int(quotient)\n\nnumberRows   =int(math.sqrt(numberDCM))+quotient\nnumberColumns=int(math.sqrt(numberDCM))\n\nprint(\"number of DCM: \", numberDCM ,\"number of rows: \",numberRows, \"number of columns: \", numberColumns )\n\nfig=plt.figure(figsize=(16, 16))\n\nfor i in range(1, numberDCM+1):\n    dcmr = dcm.dcmread(pathDCM / f\"{i}.dcm\")\n    img = dcmr.pixel_array\n    img[img == -2000] = 0\n    fig.add_subplot(numberRows, numberColumns, i)\n    plt.imshow(img)\nplt.show()\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}