{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"# Let's get started!!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport os\nimport seaborn as sns\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reading data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"list(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Data Description\n\n* train.csv - the training set, contains full history of clinical information\n* test.csv - the test set, contains only the baseline measurement\n* train/ - contains the training patients' baseline CT scan in DICOM format\n* test/ - contains the test patients' baseline CT scan in DICOM format\n* sample_submission.csv - demonstrates the submission format","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest_df  = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/test.csv\")\n\nprint(\"Train data shape: \",train_df.shape)\nprint(\"Test data shape: \",test_df.shape)\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Patients present in train data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Patients in train data : \",train_df.Patient.value_counts().shape[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's check target column distribution : FVC distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(train_df.FVC,color=\"green\")\nprint('Mean: ',np.mean(train_df.FVC))\nprint('Standard deviation: ',np.std(train_df.FVC))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Number of patient samples distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame()\ndf['Sample_count'] = train_df['Patient'].value_counts()\nsns.countplot(x=\"Sample_count\",data=df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Grouping data by patient","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame()\ndf_m = train_df[[\"Patient\", \"Age\", \"Sex\", \"SmokingStatus\"]].drop_duplicates()\ndf_1 = train_df.groupby(\"Patient\")['Weeks'].apply(lambda x: x.tolist()).reset_index()\ndf_2 = train_df.groupby(\"Patient\")['FVC'].apply(lambda x: x.tolist()).reset_index()\ndf_3 = train_df.groupby(\"Patient\")['Percent'].apply(lambda x: x.tolist()).reset_index()\ndf_12 = pd.merge(df_m,df_1,on='Patient', how='inner')\ndf_23 = pd.merge(df_2,df_3,on='Patient', how='inner')\ndf = pd.merge(df_12,df_23,on='Patient', how='inner')\n\nprint(\"Train data shape: \",df.shape)\ndf.head()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Age distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(df['Age'],color='magenta')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FVC range distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def fvc_min_max_range(x):\n    return x[0]-x[-1]\n\ndf['FVC-range'] = df['FVC'].map(lambda x:fvc_min_max_range(x))\nsns.distplot(df['FVC-range'],color='#37AA9C')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df[df['FVC-range']<0].shape[0],\" patients are seen with increase in FVC values\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Pair plot b/w Age,FVC-range and Smoking status","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"g = sns.pairplot(df[[\"Age\", \"FVC-range\", \"SmokingStatus\"]], \\\n                 hue=\"SmokingStatus\", corner=True)\ng.fig.set_figwidth(10)\ng.fig.set_figheight(10)\nprint(np.corrcoef(df[\"FVC-range\"], df[\"Age\"]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Weeks range distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def weeks_max_min_range(x):\n    return x[-1]-x[0]\n\ndf['Weeks-range'] = df['Weeks'].map(lambda x:weeks_max_min_range(x))\nsns.distplot(df['Weeks-range'],color='blue')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Pair plot b/w FVC-range,Weeks-range and Smoking status","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"g = sns.pairplot(df[[\"Weeks-range\", \"FVC-range\", \"SmokingStatus\"]], \\\n                 hue=\"SmokingStatus\", corner=True)\ng.fig.set_figwidth(10)\ng.fig.set_figheight(10)\nprint(np.corrcoef(df[\"FVC-range\"], df[\"Weeks-range\"]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# LONG WAY TO GOOOOOOOOO....SEE U SOON....Thanks!!","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}