{"cells":[{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"!pip install git+https://github.com/vik228/deeplearning_ai","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"!pip install nb_black","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%load_ext nb_black\nimport numpy as np\nimport pandas as pd\nimport math\nfrom data_visualisation.plot import Plot\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest_data = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/test.csv\")\nimage_dir = \"../input/osic-pulmonary-fibrosis-progression\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def count_plot(x_key, y_key, df, **kwargs):\n    xlabel = kwargs.get(\"xlabel\")\n    ylabel = kwargs.get(\"ylabel\")\n    title = kwargs.get(\"title\")\n    p1 = Plot(df)\n    plots = [\n        {\n            \"type\": \"bar\",\n            \"params\": (\n                x_key,\n                y_key,\n                {\n                    \"axes_settings\": {\n                        \"xlabel\": xlabel,\n                        \"ylabel\": ylabel,\n                        \"title\": title,\n                    },\n                    \"ci\": None,\n                },\n            ),\n        }\n    ]\n    p1.plot_bulk(plots, single_figure=True, single_axes=True, figsize=(20, 10))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**_Now, lets analyse the information given in train and test set. Mainly we are going to be looking at the following_**\n* The unique number of patients in train and test set\n* The number of distinct week values\n* Number of results per week value\n* Age variablity\n* The number of unique male and female in the dataset\n* The number of unique Ex-smokers, never smoked and currently smokes entries we have in the dataset.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"1. **The unique number of patients in train and test set**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# The unique number of patients in train and test set\nall_patients_train = train_data[\"Patient\"].unique()\nprint(\"Number of Unique Patients in train dataset are \", len(all_patients_train))\n\nall_patients_test = test_data[\"Patient\"].unique()\nprint(\"Number of Unique Patients in test dataset are \", len(all_patients_test))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"2. **The number of distinct week values and number of results per week value**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# The number of distinct week values\ndef explore_weeks_attribute(df):\n    distinct_values_for_week = df[\"Weeks\"].unique()\n    print(\n        \"Total distinct value for weeks \",\n        len(distinct_values_for_week),\n    )\n    df[\"count\"] = 1\n    weeks_df = (\n        df.groupby(\"Weeks\")[\"count\"]\n        .sum()\n        .reset_index()\n        .sort_values(by=[\"count\"], ascending=False)\n    )\n    print(\"Top 10 week numbers with most entries in data\\n\", weeks_df.head(10))\n    count_plot(\"Weeks\", \n               \"count\", \n               weeks_df, xlabel=\"Week numbers\", \n               ylabel=\"count\", \n               title=\"Distribution of week numbers\")\nprint(\"\\nWeek exploration for train data are\\n\")\nexplore_weeks_attribute(train_data)\nprint(\"\\nWeek exploration for test data are\\n\")\nexplore_weeks_attribute(test_data)\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"4. **Age Variablity**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Age Variablity\ndef explore_age_attribute(df):\n    df[\"count\"] = 1\n    age_values = df[\"Age\"].unique()\n    print(\"Number of unique age values are \", len(age_values))\n    groupped_by_age = (\n        df.groupby(\"Age\")[\"count\"]\n        .sum()\n        .reset_index()\n        .sort_values(by=[\"count\"], ascending=False)\n    )\n    print(\"Top 10 Age numbers with most entries in data\\n\", groupped_by_age.head(10))\n    count_plot(\n        \"Age\",\n        \"count\",\n        groupped_by_age,\n        xlabel=\"Age\",\n        ylabel=\"count\",\n        title=\"Distribution of Age\",\n    )\n\n\nprint(\"Age exploration for train data\\n\")\nexplore_age_attribute(train_data)\nprint(\"Min age in train data is \\n\", train_data[\"Age\"].min())\nprint(\"Max age in train data is \\n\", train_data[\"Age\"].max())\nprint(\"Median age in train data is \\n\", train_data[\"Age\"].median())\nprint(\"Mean age in train data is \\n\", train_data[\"Age\"].mean())\nprint(\"\\nAge exploration for test data\\n\")\nexplore_age_attribute(test_data)\nprint(\"Min age in test data is \\n\", test_data[\"Age\"].min())\nprint(\"Max age in test data is \\n\", test_data[\"Age\"].max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Explore sex attributes\n\n\ndef explore_sex_attribute(df):\n    df[\"count\"] = 1\n    age_values = df[\"Sex\"].unique()\n    print(\"\\nNumber of unique sex values are \", len(age_values))\n    groupped_by_sex = (\n        df.groupby(\"Sex\")[\"count\"]\n        .sum()\n        .reset_index()\n        .sort_values(by=[\"count\"], ascending=False)\n    )\n    print(\n        \"\\nNumber of entries in different gender values are \\n\",\n        groupped_by_sex.head(10),\n    )\n    count_plot(\n        \"Sex\",\n        \"count\",\n        groupped_by_sex,\n        xlabel=\"Sex\",\n        ylabel=\"Count\",\n        title=\"Distribution of Sex\",\n    )\n\n\nexplore_sex_attribute(train_data)\nexplore_sex_attribute(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Explore SmokingStatus Attribute\ndef explore_sex_attribute(df):\n    df[\"count\"] = 1\n    smoking_status_df = df[\"SmokingStatus\"].unique()\n    print(\"\\nNumber of unique sex values are \", len(smoking_status_df))\n    groupped_by_smoking_status = (\n        df.groupby(\"SmokingStatus\")[\"count\"]\n        .sum()\n        .reset_index()\n        .sort_values(by=[\"count\"], ascending=False)\n    )\n    print(\n        \"\\nNumber of entries in different smoking statuses are \\n\",\n        groupped_by_smoking_status.head(10),\n    )\n    count_plot(\n        \"SmokingStatus\",\n        \"count\",\n        groupped_by_smoking_status,\n        xlabel=\"SmokingStatus\",\n        ylabel=\"Count\",\n        title=\"Distribution of Smoking Status\",\n    )\n\n\nexplore_sex_attribute(train_data)\nexplore_sex_attribute(test_data)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Gender vs Age","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"1. Distribution of age in Females","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"female_df = train_data[train_data[\"Sex\"] == \"Female\"]\nexplore_age_attribute(train_data)\nexplore_age_attribute(female_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image data visualisation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\n\nuser_images = os.listdir(image_dir + \"/train/ID00007637202177411956430\")\nnrows = 4\nncols = 4\npic_index = 10\npic_index += 8\nuser_images_paths = [\n    image_dir + \"/train/ID00007637202177411956430/\" + fname\n    for fname in user_images[pic_index - 8 : pic_index]\n]\nfig = plt.gcf()\nfig.set_size_inches(ncols * 4, nrows * 4)\nfor i, file in enumerate(user_images_paths):\n    sp = plt.subplot(nrows, ncols, i + 1)\n    sp.axis(\"Off\")  # Don't show axes (or gridlines)\n    img = pydicom.dcmread(file)\n    plt.imshow(img.pixel_array, cmap=plt.cm.bone)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FVC vs Percent","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"p1 = Plot(train_data)\np1.distplot(\n    \"FVC\", axes_settings={\"title\": \"Distribution of Percent\"}, figure_size=(20, 10)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"p1.distplot(\n    \"Percent\", axes_settings={\"title\": \"Distribution of Percent\"}, figure_size=(20, 10)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"group_by_week = train_data.groupby(\"Weeks\")[\"Percent\", \"FVC\",].mean()\ngroup_by_week[\"Weeks\"] = group_by_week.index\ngroup_by_week.head()\n\np4 = Plot(group_by_week)\nregplots_weeks = [\n    [\n        {\n            \"type\": \"regplot\",\n            \"params\": (\n                \"Weeks\",\n                \"FVC\",\n                {\n                    \"axes_settings\": {\"title\": \"Distribution of FVC by week\"},\n                    \"truncate\": False,\n                    \"color\": \"c\",\n                    \"order\": 3,\n                },\n            ),\n        },\n        {\n            \"type\": \"regplot\",\n            \"params\": (\n                \"Weeks\",\n                \"Percent\",\n                {\n                    \"axes_settings\": {\"title\": \"Distribution of Percent by week\"},\n                    \"truncate\": False,\n                    \"color\": \"m\",\n                    \"order\": 3,\n                },\n            ),\n        },\n    ]\n]\np4.plot_bulk(regplots_weeks, figsize=(25, 10))\ngroup_by_week.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"group_by_age = train_data.groupby(\"Age\")[\"Percent\", \"FVC\"].mean()\ngroup_by_age[\"Age\"] = group_by_age.index\ngroup_by_age.head()\n\np5 = Plot(group_by_age)\nregplots_age = [\n    [\n        {\n            \"type\": \"regplot\",\n            \"params\": (\n                \"Age\",\n                \"FVC\",\n                {\n                    \"axes_settings\": {\"title\": \"Distribution of FVC by Age\"},\n                    \"truncate\": False,\n                    \"order\": 2,\n                },\n            ),\n        },\n        {\n            \"type\": \"regplot\",\n            \"params\": (\n                \"Age\",\n                \"Percent\",\n                {\n                    \"axes_settings\": {\"title\": \"Distribution of Percent by Age\"},\n                    \"truncate\": False,\n                    \"order\": 2,\n                },\n            ),\n        },\n    ]\n]\np5.plot_bulk(regplots_age, figsize=(25, 10))\ngroup_by_age.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* In this case we can see that FVC grows with age and then falls ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"group_by_smokingstatus = train_data.groupby(\"SmokingStatus\")[\"Percent\", \"FVC\"].mean()\ngroup_by_smokingstatus[\"SmokingStatus\"] = group_by_smokingstatus.index\np6 = Plot(group_by_smokingstatus)\nregplots_smokingstatus = [\n    [\n        {\n            \"type\": \"bar\",\n            \"params\": (\n                \"SmokingStatus\",\n                \"FVC\",\n                {\"axes_settings\": {\"title\": \"Distribution of FVC by SmokingStatus\"}},\n            ),\n        },\n        {\n            \"type\": \"bar\",\n            \"params\": (\n                \"SmokingStatus\",\n                \"Percent\",\n                {\n                    \"axes_settings\": {\n                        \"title\": \"Distribution of Percent by SmokingStatus\"\n                    }\n                },\n            ),\n        },\n    ]\n]\np6.plot_bulk(regplots_smokingstatus, figsize=(25, 10))\ngroup_by_smokingstatus.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Smokers have better FVC, but this is strange. Need to investigate reason for this.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}