{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20604,"databundleVersionId":1357052,"sourceType":"competition"},{"sourceId":1426653,"sourceType":"datasetVersion","datasetId":804202},{"sourceId":2378330,"sourceType":"datasetVersion","datasetId":492658},{"sourceId":38923375,"sourceType":"kernelVersion"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mpimg\nfrom tabulate import tabulate\nimport missingno as msno\nfrom IPython.display import display_html\nfrom PIL import Image\nimport gc\nimport cv2\nfrom scipy.stats import pearsonr","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:30.778312Z","iopub.execute_input":"2024-04-23T05:26:30.77865Z","iopub.status.idle":"2024-04-23T05:26:33.114841Z","shell.execute_reply.started":"2024-04-23T05:26:30.778621Z","shell.execute_reply":"2024-04-23T05:26:33.114063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we are importing necessary libraries like pandas,numpy, saborn etc.\n* **pandas:** Data manipulation and analysis tool, particularly suited for handling structured data (e.g., tables, spreadsheets).\n* **numpy:** Numerical computing library providing support for large, multi-dimensional arrays and matrices, along with a collection of mathematical functions to operate on these arrays.\n* **seaborn:** Statistical data visualization library based on matplotlib, offering high-level abstractions to create attractive and informative statistical graphics.\n* **matplotlib.pyplot:** Plotting library for creating static, animated, and interactive visualizations in Python.\n* **%matplotlib inline:** IPython magic command to display matplotlib plots inline within Jupyter notebooks.\n* **matplotlib.image:** Module within matplotlib for loading, displaying, and manipulating image data.","metadata":{}},{"cell_type":"markdown","source":"* **cv2:** OpenCV (Open Source Computer Vision Library) for computer vision and image processing tasks like object detection, recognition, and tracking.\n* **scipy.stats.pearsonr:** Function from SciPy's statistics module for calculating the Pearson correlation coefficient and its associated p-value for two arrays.","metadata":{}},{"cell_type":"code","source":"from skimage.transform import resize\nimport copy\nimport re\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:33.116304Z","iopub.execute_input":"2024-04-23T05:26:33.116806Z","iopub.status.idle":"2024-04-23T05:26:33.204435Z","shell.execute_reply.started":"2024-04-23T05:26:33.11678Z","shell.execute_reply":"2024-04-23T05:26:33.203516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Segmentation\nimport scipy.ndimage\nfrom skimage import morphology\nfrom skimage import measure\nfrom skimage.transform import resize\nfrom sklearn.cluster import KMeans\nfrom plotly import __version__\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\nfrom plotly.tools import FigureFactory as FF\nfrom plotly.graph_objs import *\ninit_notebook_mode(connected=True)\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\n# Set Color Palettes for the notebook\ncustom_colors = ['#74a09e','#86c1b2','#98e2c6','#f3c969','#f2a553', '#d96548', '#c14953']\nsns.palplot(sns.color_palette(custom_colors))\n\n# The despine() is a function that removes the spines from the right and upper portion of the plot by default.\nsns.set_style(\"whitegrid\")\nsns.despine(left=True, bottom=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:33.205629Z","iopub.execute_input":"2024-04-23T05:26:33.206024Z","iopub.status.idle":"2024-04-23T05:26:34.533705Z","shell.execute_reply.started":"2024-04-23T05:26:33.205999Z","shell.execute_reply":"2024-04-23T05:26:34.532396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> *The above code customizes the color palette for visualizations and adjusts the plot style by setting it to \"whitegrid\" while removing the spines from the left and bottom sides of the plots for a cleaner and more visually appealing presentation of data.*","metadata":{}},{"cell_type":"markdown","source":"> The code below infers reading train and test data getting total readings in train data i.e., **1549 ** patient readings","metadata":{}},{"cell_type":"code","source":"# Import train + test data\ntrain = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\ntest = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/test.csv\")\n\n# Train len\nprint(\"Total Recordings in Train Data: {:,}\".format(len(train)))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:34.536568Z","iopub.execute_input":"2024-04-23T05:26:34.536919Z","iopub.status.idle":"2024-04-23T05:26:34.577943Z","shell.execute_reply.started":"2024-04-23T05:26:34.536887Z","shell.execute_reply":"2024-04-23T05:26:34.576969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* In the code below we are getting head and tail of train data.","metadata":{}},{"cell_type":"code","source":"\ndf1_styler = train.head().style.set_table_attributes(\"style='display:inline'\").set_caption('Head Train Data')\ndf2_styler = test.style.set_table_attributes(\"style='display:inline'\").set_caption('Test Data (rest Hidden)')\n\ndisplay_html(df1_styler._repr_html_() + df2_styler._repr_html_(), raw=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:34.579385Z","iopub.execute_input":"2024-04-23T05:26:34.579906Z","iopub.status.idle":"2024-04-23T05:26:34.779841Z","shell.execute_reply.started":"2024-04-23T05:26:34.579875Z","shell.execute_reply":"2024-04-23T05:26:34.778807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* From this we get to know what are the attributes present in this dataset.\n* The attributes are given by patient, weeks, FVC, Percent, Age, Sex, Smoking status.\n* From this the most useful attrbutes for this project are Age,Sex,Smoking status,Patient.","metadata":{}},{"cell_type":"code","source":"#checking missing values\nprint(format(train.isnull().values.any()))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:34.781143Z","iopub.execute_input":"2024-04-23T05:26:34.781712Z","iopub.status.idle":"2024-04-23T05:26:34.788317Z","shell.execute_reply.started":"2024-04-23T05:26:34.781664Z","shell.execute_reply":"2024-04-23T05:26:34.787273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> This code snippet allows us to check for any missing values in the dataset. Intially there are no missing values found.","metadata":{}},{"cell_type":"markdown","source":"# EXPLORATORY DATA ANALYSIS","metadata":{}},{"cell_type":"markdown","source":"The provided code is used to perform data analysis and visualization on a dataset. It starts by grouping the data by patient and selecting the unique values for age, sex, and smoking status. The resulting DataFrame is then used to create subplots for different types of plots.\n\n* Patient Age Distribution: A box plot is created to show the distribution of patient ages. The box plot displays the median, quartiles, and outliers of the age data, providing insights into the spread and skewness of the data.\n\n* Sex Frequency: A pie chart is used to visualize the frequency of different sexes in the dataset. The pie chart shows the proportion of each sex, allowing for a quick comparison of the distribution of sexes.\n\n* Smoking Status: A bar plot is created to show the distribution of smoking statuses. The bar plot displays the count of patients with each smoking status, helping to identify the most common and least common smoking statuses in the dataset.\n\nThe code uses several libraries, including pandas, seaborn, and matplotlib, to perform the data analysis and visualization tasks. The seaborn library is used to create the box plot, pie chart, and bar plot, while pandas is used for data manipulation and grouping. Matplotlib is used to create the subplots and customize the appearance of the plots.","metadata":{}},{"cell_type":"code","source":"# Select unique bio info for the patients like age, sex, smoking status\ndata = train.groupby(by=\"Patient\")[[\"Patient\", \"Age\", \"Sex\", \"SmokingStatus\"]].first().reset_index(drop=True)\n\n# Create subplots for different plot types\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(16, 6))\n\n# Plot for Patient Age Distribution using a Box Plot\nsns.boxplot(y=data[\"Age\"], ax=ax1, color=custom_colors[1])\nax1.set_title(\"Patient Age Distribution\", fontsize=16)\nax1.set_ylabel(\"Age\", fontsize=14)\n\n# Plot for Sex Frequency using a Pie Chart\nsex_counts = data[\"Sex\"].value_counts()\nax2.pie(sex_counts, labels=sex_counts.index, colors=custom_colors[2:4], autopct='%1.1f%%', startangle=90)\nax2.set_title(\"Sex Frequency\", fontsize=16)\n\n# Plot for Smoking Status using a Bar Plot\nsns.countplot(y=data[\"SmokingStatus\"], ax=ax3, palette=custom_colors[4:7])\nax3.set_title(\"Smoking Status\", fontsize=16)\nax3.set_ylabel(\"Smoking Status\", fontsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:34.789962Z","iopub.execute_input":"2024-04-23T05:26:34.790294Z","iopub.status.idle":"2024-04-23T05:26:35.58319Z","shell.execute_reply.started":"2024-04-23T05:26:34.790251Z","shell.execute_reply":"2024-04-23T05:26:35.582074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The above code we infer that our attributes age, sex and smoking status into various graphs to understand the data.\n* The patient age is given by a box plot which clearly says that ost patients are in between 60-75 age group.\n* And teh pie chart for sex infers that there are more male patients than female patients with ratio od arund 7.9:2.1 i.e., for every 100 patient 7 patients are male and 21 are female.\n* Fror smoking status we used bar chart to visualize data. Here we get to knw that people who never smoked and pelple wh stopped smoking are more likely to get this disease.\n* Current smokers are less likely to get pulmonary fibrosis.","metadata":{}},{"cell_type":"markdown","source":"> **FVC rate**\nFVC stands for Forced Vital Capacity. It's a common measure used in pulmonary function tests (PFTs) to assess lung function.","metadata":{}},{"cell_type":"markdown","source":"The code provided is used to analyze and visualize the distribution of two variables, \"FVC\" and \"Percent,\" in a dataset. It consists of two main parts:\n\n* Printing the minimum and maximum values: The first part of the code prints the minimum and maximum values for both variables using the min() and max() functions on the \"FVC\" and \"Percent\" columns of the dataset.\n\n* Creating swarm plots: The second part of the code uses Seaborn's swarmplot() function to create swarm plots for the \"FVC\" and \"Percent\" variables. Swarm plots are a type of scatter plot where one variable is categorical, and the points are adjusted to not overlap. This provides a better representation of the distribution of values, especially when the number of observations is not too large","metadata":{}},{"cell_type":"code","source":"print(\"Min FVC value: {:,}\".format(train[\"FVC\"].min()), \"\\n\" +\n      \"Max FVC value: {:,}\".format(train[\"FVC\"].max()), \"\\n\" +\n      \"\\n\" +\n      \"Min Percent value: {:.4}%\".format(train[\"Percent\"].min()), \"\\n\" +\n      \"Max Percent value: {:.4}%\".format(train[\"Percent\"].max()))\n\n\n\n# Create subplots for Swarm Plot\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Swarm Plot for FVC\nsns.swarmplot(x=train[\"FVC\"], ax=ax1, color=custom_colors[6])\nax1.set_title(\"FVC Distribution\", fontsize=16)\nax1.set_xlabel(\"FVC\", fontsize=14)\n\n# Swarm Plot for Percent\nsns.swarmplot(x=train[\"Percent\"], ax=ax2, color=custom_colors[4])\nax2.set_title(\"Percent Distribution\", fontsize=16)\nax2.set_xlabel(\"Percent\", fontsize=14)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:26:35.58461Z","iopub.execute_input":"2024-04-23T05:26:35.584993Z","iopub.status.idle":"2024-04-23T05:27:00.816002Z","shell.execute_reply.started":"2024-04-23T05:26:35.58496Z","shell.execute_reply":"2024-04-23T05:27:00.814776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> * FVC represents the total amount of air a person can forcibly exhale after taking a deep breath. \n> * During the test, the individual inhales deeply and then exhales as forcefully and completely as possible into a spirometer, a device that measures lung volume and airflow.\n> \n> * FVC is a key indicator of lung health and is often used to diagnose and monitor various respiratory conditions such as asthma, chronic obstructive pulmonary disease (COPD), and restrictive lung diseases.\n> * Abnormalities in FVC values can indicate lung disorders, impaired lung function, or the presence of respiratory issues. \n> * FVC results are often compared to predicted values based on factors such as age, height, sex, and ethnicity to evaluate lung function relative to expected norms.","metadata":{}},{"cell_type":"markdown","source":"* From the above code displays the minimum and maximum values of the 'FVC' (Forced Vital Capacity) and 'Percent' columns from the 'train' dataset.\n* The minium and maximum values got by the code are as follows Minimum FVC value: 827 \n  Maximum FVC value: 6,399.\n* The percentage of min and max value is given by Min Percent value: 28.88% \n   Max Percent value: 153.1%\n\n","metadata":{}},{"cell_type":"markdown","source":"* This code is used to analyze the distribution of patient entries across different weeks. It starts by grouping the data by patient and counting the number of entries for each patient using the groupby() and count() functions. The resulting DataFrame is then sorted by the number of entries in descending order.\n* The code then calculates the minimum and maximum number of entries for all patients using the min() and max() functions on the \"Weeks\" column.\n* Next, a line plot is created using Seaborn's lineplot() function. The plot shows the number of entries for each patient, with the x-axis representing the patient number and the y-axis representing the number of entries. The plot is customized with a title and labels for the x- and y-axes.\n* Finally, the x-axis of the plot is hidden using p.axes.get_xaxis().set_visible(False) to focus on the distribution of entries across patients.\n* In summary, this code is used to visualize the distribution of patient entries across different weeks, providing insights into the frequency of patient entries.\n* ","metadata":{}},{"cell_type":"code","source":"print(\"There are {} unique patients in Train Data.\".format(len(train[\"Patient\"].unique())), \"\\n\")\n\n# Recordings per Patient\ndata = train.groupby(by=\"Patient\")[\"Weeks\"].count().reset_index(drop=False)\n# Sort by Weeks\ndata = data.sort_values(['Weeks']).reset_index(drop=True)\nprint(\"Minimum number of entries are: {}\".format(data[\"Weeks\"].min()), \"\\n\" +\n      \"Maximum number of entries are: {}\".format(data[\"Weeks\"].max()))\n\n\n\n\nplt.figure(figsize=(10, 6))\np = sns.lineplot(x=data[\"Patient\"], y=data[\"Weeks\"], color=custom_colors[2 % len(custom_colors)])\n\nplt.title(\"Number of Entries per Patient\", fontsize=17)\nplt.xlabel('Patient', fontsize=14)\nplt.ylabel('Frequency', fontsize=14)\n\np.axes.get_xaxis().set_visible(False)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:00.817151Z","iopub.execute_input":"2024-04-23T05:27:00.817454Z","iopub.status.idle":"2024-04-23T05:27:01.345598Z","shell.execute_reply.started":"2024-04-23T05:27:00.817428Z","shell.execute_reply":"2024-04-23T05:27:01.344663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The code provided is used for data analysis, specifically for exploring the distribution of patient information, including age, sex, and smoking status. The code can be broken down into the following steps:\n* Data selection: The code selects unique patient information, such as age, sex, and smoking status, from the dataset. This is done using the groupby() function in pandas, which groups the data by patient and then selects the first row of each group using the first() function. The resulting DataFrame is stored in the variable data.\n* Subplots creation: The code creates subplots for different plot types using Seaborn's subplots() function. The subplots are arranged in a single row with three columns, each representing a different plot type.\n* Violin plot for patient age distribution: A violin plot is created to visualize the distribution of patient ages. The violin plot shows the distribution of ages across different patients, with the x-axis representing the age and the y-axis representing the number of patients. The plot is customized with a title and labels for the x- and y-axes.\n* Pie chart for sex frequency: A pie chart is used to visualize the frequency of different sexes in the dataset. The pie chart shows the proportion of each sex, allowing for a quick comparison of the distribution of sexes.\n* Swarm plot for smoking status: A swarm plot is created to show the distribution of smoking statuses. The swarm plot displays the count of patients with each smoking status, helping to identify the most common and least common smoking statuses in the dataset.\n* The code uses several libraries, including pandas, Seaborn, and Matplotlib, to perform the data analysis and visualization tasks. The Seaborn library is used to create the violin plot, pie chart, and swarm plot, while pandas is used for data manipulation and grouping. Matplotlib is used to create the subplots and customize the appearance of the plots.\n* In summary, the code is used to analyze and visualize the distribution of patient information, such as age, sex, and smoking status, in a dataset. The subplots are used to show different aspects of the data distribution, providing insights into the data.","metadata":{}},{"cell_type":"code","source":"# Select unique bio info for the patients\ndata = train.groupby(by=\"Patient\")[[\"Patient\", \"Age\", \"Sex\", \"SmokingStatus\"]].first().reset_index(drop=True)\n\n# Create subplots with different plot types for enhanced representation\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(16, 6))\n\n# Plot for Patient Age Distribution using a Violin Plot\nsns.violinplot(y=data[\"Age\"], ax=ax1, color=custom_colors[1])\nax1.set_title(\"Patient Age Distribution\", fontsize=16)\nax1.set_ylabel(\"Age\", fontsize=14)\n\n# Plot for Sex Frequency using a Pie Chart\nsex_counts = data[\"Sex\"].value_counts()\nax2.pie(sex_counts, labels=sex_counts.index, colors=custom_colors[2:4], autopct='%1.1f%%', startangle=90)\nax2.set_title(\"Sex Frequency\", fontsize=16)\n\n# Plot for Smoking Status using a Swarm Plot\nsns.swarmplot(y=data[\"SmokingStatus\"], ax=ax3, palette=custom_colors[4:7])\nax3.set_title(\"Smoking Status\", fontsize=16)\nax3.set_ylabel(\"Smoking Status\", fontsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:01.349584Z","iopub.execute_input":"2024-04-23T05:27:01.349871Z","iopub.status.idle":"2024-04-23T05:27:03.823224Z","shell.execute_reply.started":"2024-04-23T05:27:01.349846Z","shell.execute_reply":"2024-04-23T05:27:03.822153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Printing Min and Max Values:\n    * The code prints the minimum and maximum values for FVC and Percent columns in the dataset using the min() and max() functions. This provides an overview of the range of values for these variables.\n* Creating Subplots:\n    * Two subplots are created side by side using plt.subplots() to accommodate different plot types for FVC and Percent distributions.\n* Box Plot for FVC Distribution:\n    * A box plot is used to visualize the distribution of FVC values. The box plot displays the median, quartiles, and potential outliers in the FVC data, offering insights into the spread and central tendency of the FVC values.\n* Violin Plot for Percent Distribution:\n    * A violin plot is employed to show the distribution of Percent values. The violin plot provides a visual representation of the distribution's shape, spread, and central tendency for the Percent variable.\n* Plot Customization:\n    * Each plot is customized with titles, y-axis labels, and appropriate colors to enhance readability and interpretation.\n    * plt.tight_layout() is used to adjust the layout for better spacing between the subplots.","metadata":{}},{"cell_type":"code","source":"\n# Print Min and Max values for FVC and Percent\nprint(\"Min FVC value: {:,}\".format(train[\"FVC\"].min()), \"\\n\" +\n      \"Max FVC value: {:,}\".format(train[\"FVC\"].max()), \"\\n\" +\n      \"\\n\" +\n      \"Min Percent value: {:.4}%\".format(train[\"Percent\"].min()), \"\\n\" +\n      \"Max Percent value: {:.4}%\".format(train[\"Percent\"].max()))\n\n# Create subplots with different plot types for enhanced representation\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Plot for FVC Distribution using a Box Plot\nsns.boxplot(y=train[\"FVC\"], ax=ax1, color=custom_colors[6])\nax1.set_title(\"FVC Distribution\", fontsize=16)\nax1.set_ylabel(\"FVC\", fontsize=14)\n\n# Plot for Percent Distribution using a Violin Plot\nsns.violinplot(y=train[\"Percent\"], ax=ax2, color=custom_colors[4])\nax2.set_title(\"Percent Distribution\", fontsize=16)\nax2.set_ylabel(\"Percent\", fontsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:03.824458Z","iopub.execute_input":"2024-04-23T05:27:03.824805Z","iopub.status.idle":"2024-04-23T05:27:04.344389Z","shell.execute_reply.started":"2024-04-23T05:27:03.824776Z","shell.execute_reply":"2024-04-23T05:27:04.34345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The provided sources discuss topics related to coding, specifically seaborn.histplot, and the usage of histograms in data visualization. Source and detail the functionality of seaborn.histplot, which allows for the creation of univariate or bivariate histograms to represent the distribution of variables. It provides options for customizing the appearance of histograms, such as log scaling, different bar styles, and cumulative histograms. Additionally, it can handle categorical variables and hue semantics for more complex visualizations. On the other hand, source explains the difference between coding and programming, highlighting that coding involves writing code to fulfill specific user stories, while programming encompasses broader considerations like scalability and security.\nIn summary, seaborn.histplot is a versatile tool for creating histograms in Python, offering various customization options for data visualization, while coding involves the practical implementation of code to achieve specific functionalities within a program or system","metadata":{}},{"cell_type":"code","source":"# Print Min and Max values for Weeks\nprint(\"Minimum no. weeks before CT: {}\".format(train['Weeks'].min()), \"\\n\" +\n      \"Maximum no. weeks after CT: {}\".format(train['Weeks'].max()))\n\n# Create a figure for the plot\nplt.figure(figsize=(16, 6))\n\n# Plot for Number of Weeks using a Histogram\nsns.histplot(train['Weeks'], color=custom_colors[3], kde=True, linewidth=8, linestyle=\"--\")\nplt.title(\"Number of weeks before/after the CT scan\", fontsize=16)\nplt.xlabel(\"Weeks\", fontsize=14)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:04.345563Z","iopub.execute_input":"2024-04-23T05:27:04.345865Z","iopub.status.idle":"2024-04-23T05:27:04.80759Z","shell.execute_reply.started":"2024-04-23T05:27:04.345838Z","shell.execute_reply":"2024-04-23T05:27:04.806697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Correlation Computation: The code calculates Pearson correlation coefficients between pairs of variables (FVC and Percent, FVC and Age, Percent and Age) using the pearsonr function. The Pearson correlation coefficient measures the linear relationship between two variables, ranging from -1 to 1, where 1 indicates a perfect positive correlation, -1 a perfect negative correlation, and 0 no correlation. The results are then printed to show the strength and direction of the relationships.\n* Figure Creation: The code creates a figure with three subplots arranged horizontally using plt.subplots(). Each subplot represents a scatter plot showing the relationship between two variables. Different colors and styles are used to differentiate data points based on the \"Sex\" variable.\n* Scatter Plots: Three scatter plots are generated to visualize the correlations:\n    * Scatter plot between FVC and Percent\n    * Scatter plot between FVC and Age\n    * Scatter plot between Percent and Age\n* Plot Customization: Each scatter plot is customized with titles, x-axis labels, and y-axis labels to indicate the variables being compared and the correlation being visualized.\n\n\nthis code is used to create visualizations that compare the mean FVC and Percent values across different smoking statuses. It helps in understanding how these lung function metrics vary based on smoking status categories.","metadata":{}},{"cell_type":"code","source":"# Compute Correlation\ncorr1, _ = pearsonr(train[\"FVC\"], train[\"Percent\"])\ncorr2, _ = pearsonr(train[\"FVC\"], train[\"Age\"])\ncorr3, _ = pearsonr(train[\"Percent\"], train[\"Age\"])\nprint(\"Pearson Corr FVC x Percent: {:.4}\".format(corr1), \"\\n\" +\n      \"Pearson Corr FVC x Age: {:.0}\".format(corr2), \"\\n\" +\n      \"Pearson Corr Percent x Age: {:.2}\".format(corr3))\n\n# Figure\nf, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (16, 6))\n\na = sns.scatterplot(x = train[\"FVC\"], y = train[\"Percent\"], palette=[custom_colors[2], custom_colors[6]],\n                    hue = train[\"Sex\"], style = train[\"Sex\"], s=100, ax=ax1)\n\nb = sns.scatterplot(x = train[\"FVC\"], y = train[\"Age\"], palette=[custom_colors[2], custom_colors[6]],\n                    hue = train[\"Sex\"], style = train[\"Sex\"], s=100, ax=ax2)\n\nc = sns.scatterplot(x = train[\"Percent\"], y = train[\"Age\"], palette=[custom_colors[2], custom_colors[6]],\n                    hue = train[\"Sex\"], style = train[\"Sex\"], s=100, ax=ax3)\n\na.set_title(\"Correlation between FVC and Percent\", fontsize = 16)\na.set_xlabel(\"FVC\", fontsize = 14)\na.set_ylabel(\"Percent\", fontsize = 14)\n\nb.set_title(\"Correlation between FVC and Age\", fontsize = 16)\nb.set_xlabel(\"FVC\", fontsize = 14)\nb.set_ylabel(\"Age\", fontsize = 14)\n\nc.set_title(\"Correlation between Percent and Age\", fontsize = 16)\nc.set_xlabel(\"Percent\", fontsize = 14)\nc.set_ylabel(\"Age\", fontsize = 14);","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:04.808881Z","iopub.execute_input":"2024-04-23T05:27:04.809173Z","iopub.status.idle":"2024-04-23T05:27:06.463352Z","shell.execute_reply.started":"2024-04-23T05:27:04.809148Z","shell.execute_reply":"2024-04-23T05:27:06.462461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Subplots Creation: The code creates a figure with two subplots arranged horizontally using plt.subplots(). Each subplot will display a point plot showing the mean values of FVC and Percent for different smoking statuses.\n* Point Plots for Mean Values:\n     * Mean FVC per Smoking Status: A point plot is created to show the mean FVC values for each smoking status. The x-axis represents the smoking status categories, and the y-axis represents the mean FVC values. Different colors are used to distinguish between the smoking statuses. The plot helps visualize how FVC values vary across different smoking statuses.\n     * Mean Percent per Smoking Status: Another point plot is generated to display the mean Percent values for each smoking status. Similar to the first plot, this plot shows the relationship between smoking status and mean Percent values. It allows for a comparison of Percent values based on smoking status.\n* Plot Customization:\n    * Titles, x-axis labels, and y-axis labels are added to each subplot to provide context and clarity.\n    * The plot layout is adjusted using plt.tight_layout() to ensure proper spacing between subplots.\n    \n    this code identifies and visualizes the decrease in FVC measurements over time for patients with significant FVC differences between their initial and final check-ins. It helps in understanding the progression of FVC values for selected patients and can provide insights into potential patterns or trends in lung function decline.","metadata":{}},{"cell_type":"code","source":"# Create subplots with adjusted colors and plot types\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Plot for Mean FVC per Smoking Status using a Point Plot\nsns.pointplot(x=train[\"SmokingStatus\"], y=train[\"FVC\"], ax=ax1, color=custom_colors[0])\nax1.set_title(\"Mean FVC per Smoking Status\", fontsize=16)\nax1.set_xlabel(\"Smoking Status\", fontsize=14)\nax1.set_ylabel(\"Mean FVC\", fontsize=14)\n\n# Plot for Mean Percent per Smoking Status using a Point Plot\nsns.pointplot(x=train[\"SmokingStatus\"], y=train[\"Percent\"], ax=ax2, color=custom_colors[4])\nax2.set_title(\"Mean Percent per Smoking Status\", fontsize=16)\nax2.set_xlabel(\"Smoking Status\", fontsize=14)\nax2.set_ylabel(\"Mean Percent\", fontsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:06.464424Z","iopub.execute_input":"2024-04-23T05:27:06.464678Z","iopub.status.idle":"2024-04-23T05:27:07.101338Z","shell.execute_reply.started":"2024-04-23T05:27:06.464656Z","shell.execute_reply":"2024-04-23T05:27:07.100438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Creating a Time Variable:\n    * The code first counts the number of FVC check-ins in ascending order for each patient and stores this information in the data_time DataFrame.\n    * It then creates a new variable \"Time\" in the train DataFrame, which ranks the weeks for each patient to track the progression of FVC measurements over time.\n* Filtering Patients with Significant FVC Differences:\n    * Initial and final FVC values are extracted for each patient to calculate the FVC differences.\n    * Patients with the largest FVC differences are selected based on the calculated differences.\n* Plotting FVC Decrease Over Weeks:\n    * The code plots the FVC decrease over weeks for the selected top patients with the largest FVC differences.\n    * A line plot is used to show the trend of FVC measurements over time for each patient, with different patients represented by different colors.\n    * The plot is customized with a title, x-axis label (Weeks), and y-axis label (FVC) to provide context and interpretation.\n    \n    \n    this code prepares the data by creating a path column to locate the CT scan files for each patient and then counts the number of CT scans available for each patient. This information can be valuable for further analysis and understanding the dataset in the context of pulmonary fibrosis progression.","metadata":{}},{"cell_type":"code","source":"# Create a time variable to count the number of FVC check-ins in ascending order for each patient\ndata_time = train.groupby(by=\"Patient\")[\"Weeks\"].count().reset_index()\ntrain[\"Time\"] = train.groupby(\"Patient\")[\"Weeks\"].rank(method='dense')\n\n# Filter patients with significant FVC differences between initial and final check-ins\nmin_fvc = train.loc[train[\"Time\"] == 1, [\"Patient\", \"FVC\"]].reset_index(drop=True)\nmax_fvc = train.loc[train.groupby(\"Patient\")[\"Weeks\"].transform(max) == train[\"Weeks\"], [\"Patient\", \"FVC\"]].reset_index(drop=True)\n\n# Calculate FVC differences and select top patients with the largest differences\ndata = pd.merge(min_fvc, max_fvc, how=\"inner\", on=\"Patient\")\ndata[\"FVC_Difference\"] = data[\"FVC_x\"] - data[\"FVC_y\"]\n\n# Select top patients with the largest FVC differences\ntop_patients = data.sort_values(\"FVC_Difference\", ascending=False).head(100)[\"Patient\"]\nfiltered_data = train[train[\"Patient\"].isin(top_patients)]\n\n# Plotting the FVC decrease over weeks for selected patients\nplt.figure(figsize=(16, 6))\nsns.lineplot(x=\"Time\", y=\"FVC\", hue=\"Patient\", data=filtered_data, legend=False,\n             palette=sns.color_palette(\"GnBu_d\", n_colors=100), size=1)\n\nplt.title(\"Patient FVC Decrease Over Weeks\", fontsize=16)\nplt.xlabel(\"Weeks\", fontsize=14)\nplt.ylabel(\"FVC\", fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:07.102608Z","iopub.execute_input":"2024-04-23T05:27:07.10296Z","iopub.status.idle":"2024-04-23T05:27:08.000267Z","shell.execute_reply.started":"2024-04-23T05:27:07.102928Z","shell.execute_reply":"2024-04-23T05:27:07.999343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Setting the Base Directory:\n    * The code sets the base directory where the CT scan files for the patients are located. This directory is specified as \"../input/osic-pulmonary-fibrosis-progression/train\".\n* Creating Path Column:\n    * A new column \"Path\" is created in the train DataFrame by concatenating the base directory with the patient IDs. This column contains the path to each patient's CT scan files.\n* Counting the Number of CT Scans:\n    * A new column \"CT_number\" is initialized with zeros to store the number of CT scans each patient has.\n    * A loop iterates over the paths in the \"Path\" column, and for each path, it counts the number of files (CT scans) in that directory using os.listdir(path) and assigns the count to the corresponding row in the \"CT_number\" column.\n    \n    \n    this code provides insights into the distribution of CT scans per patient, highlighting the range of scans each patient has undergone. The point plot visualization helps in understanding the frequency of CT scans across patients and identifies any potential patterns or variations in the data.","metadata":{}},{"cell_type":"code","source":"# Create base director for Train .dcm files\ndirector = \"../input/osic-pulmonary-fibrosis-progression/train\"\n\n# Create path column with the path to each patient's CT\ntrain[\"Path\"] = director + \"/\" + train[\"Patient\"]\n\n# Create variable that shows how many CT scans each patient has\ntrain[\"CT_number\"] = 0\n\nfor k, path in enumerate(train[\"Path\"]):\n    train[\"CT_number\"][k] = len(os.listdir(path))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:08.001437Z","iopub.execute_input":"2024-04-23T05:27:08.001719Z","iopub.status.idle":"2024-04-23T05:27:24.859055Z","shell.execute_reply.started":"2024-04-23T05:27:08.001693Z","shell.execute_reply":"2024-04-23T05:27:24.858052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Printing Min and Max Number of CT Scans:\n    * The code prints the minimum and maximum number of CT scans per patient in the dataset using the min() and max() functions on the \"CT_number\" column of the train DataFrame.\n* Calculating Scans per Patient:\n    * The code groups the data by patient and counts the number of CT scans for each patient, storing this information in the data DataFrame.\n    * The data is then sorted by the number of CT scans in ascending order for visualization.\n* Plotting Number of CT Scans per Patient:\n    * A point plot is created to show the distribution of the number of CT scans per patient.\n    * Each point represents a patient, and the y-axis shows the count of CT scans.\n    * A horizontal dashed line is added at the median value of 94 for reference.\n    * The plot is customized with a title, x-axis label (Patient), y-axis label (Frequency), and a text annotation for the median value\n    \n    \n    this code provides insights into the distribution of CT scans per patient, highlighting the range of scans each patient has undergone. The point plot visualization helps in understanding the frequency of CT scans across patients and identifies any potential patterns or variations in the data.","metadata":{}},{"cell_type":"code","source":"# Print Min and Max number of CT scans\nprint(\"Minimum number of CT scans: {}\".format(train[\"CT_number\"].min()), \"\\n\" +\n      \"Maximum number of CT scans: {:,}\".format(train[\"CT_number\"].max()))\n\n# Scans per Patient\ndata = train.groupby(by=\"Patient\")[\"CT_number\"].count().reset_index()\n\n# Sort by CT_number\ndata = data.sort_values('CT_number').reset_index(drop=True)\n\n# Plot with a different plot type (Point Plot)\nplt.figure(figsize=(16, 6))\np = sns.pointplot(x=data.index, y=data[\"CT_number\"], color=custom_colors[5])\nplt.axhline(y=94, color=custom_colors[2], linestyle='--', lw=3)\n\nplt.title(\"Number of CT Scans per Patient\", fontsize=17)\nplt.xlabel('Patient', fontsize=14)\nplt.ylabel('Frequency', fontsize=14)\n\nplt.text(0, 100, \"Median=94\", fontsize=13)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:24.860433Z","iopub.execute_input":"2024-04-23T05:27:24.860758Z","iopub.status.idle":"2024-04-23T05:27:27.175895Z","shell.execute_reply.started":"2024-04-23T05:27:24.86073Z","shell.execute_reply":"2024-04-23T05:27:27.174868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Calculating Scans per Patient:\n    * The code groups the data by patient and counts the number of CT scans for each patient, storing this information in the data DataFrame.\n* Sorting by CT Number:\n    * The data is then sorted by the number of CT scans in ascending order for visualization.\n* Plotting Number of CT Scans per Patient:\n    * A violin plot is created to display the distribution of the number of CT scans per patient.\n    * The y-axis represents the count of CT scans, and the width of the plot at a given point indicates the frequency of patients with that number of scans.\n    * The plot is customized with a title (\"Number of CT Scans per Patient\"), y-axis label (\"Frequency\"), and appropriate styling.\nIn summary, this code efficiently visualizes the distribution of CT scans per patient using a violin plot, providing insights into the variability and frequency of the number of scans across patients in the dataset.","metadata":{}},{"cell_type":"code","source":"# Scans per Patient\ndata = train.groupby(by=\"Patient\")[\"CT_number\"].count().reset_index()\n\n# Sort by CT_number\ndata = data.sort_values('CT_number').reset_index(drop=True)\n\n# Plot with a Violin Plot\nplt.figure(figsize=(16, 6))\nsns.violinplot(y=data[\"CT_number\"], color=custom_colors[5])\n\nplt.title(\"Number of CT Scans per Patient\", fontsize=17)\nplt.ylabel('Frequency', fontsize=14)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:27.177437Z","iopub.execute_input":"2024-04-23T05:27:27.178132Z","iopub.status.idle":"2024-04-23T05:27:27.500421Z","shell.execute_reply.started":"2024-04-23T05:27:27.178091Z","shell.execute_reply":"2024-04-23T05:27:27.499354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Setting the Base Directory:\n    * The code sets the base directory where the CT scan files for the patients are located. This directory is specified as \"../input/osic-pulmonary-fibrosis-progression/train\".\n* Creating Path Column:\n    * A new column \"Path\" is created in the train DataFrame by combining the base directory with each patient's ID. This column contains the path to each patient's CT scan files.\n* Counting the Number of CT Scans:\n    * A new column \"CT_number\" is initialized with zeros to store the number of CT scans each patient has.\n    * A loop iterates over the paths in the \"Path\" column, and for each path, it counts the number of files (CT scans) in that directory using os.listdir(path) and assigns the count to the corresponding row in the \"CT_number\" column.\n    \nIn summary, this code prepares the data by creating a path column to locate the CT scan files for each patient and then counts the number of CT scans available for each patient. This information can be valuable for further analysis and understanding the dataset in the context of pulmonary fibrosis progression.","metadata":{}},{"cell_type":"code","source":"# Create base director for Train .dcm files\ndirector = \"../input/osic-pulmonary-fibrosis-progression/train\"\n\n# Create path column with the path to each patient's CT\ntrain[\"Path\"] = director + \"/\" + train[\"Patient\"]\n\n# Create variable that shows how many CT scans each patient has\ntrain[\"CT_number\"] = 0\n\nfor k, path in enumerate(train[\"Path\"]):\n    train[\"CT_number\"][k] = len(os.listdir(path))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:27.501876Z","iopub.execute_input":"2024-04-23T05:27:27.502204Z","iopub.status.idle":"2024-04-23T05:27:28.842276Z","shell.execute_reply.started":"2024-04-23T05:27:27.502174Z","shell.execute_reply":"2024-04-23T05:27:28.841464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first source from Reddit discusses how to plot the 10 most frequent values in a barplot using Seaborn. The user was trying to plot the 10 most frequent values from the value_count() output but faced issues with the barplot representation. They mentioned trying different methods like sns.barplot(df['var'].value_counts().head(10)) and sns.barplot(df['var'].value_counts()[:10]), which did not provide the desired output. Sorting by count versus index was suggested as a potential solution.\n\nThe second source from GeeksforGeeks explains how to show values on a Seaborn barplot in Python. It details the process of plotting values on a Seaborn barplot using sns.barplot() function and the sub-method containers returned by sns.barplot(). The article provides syntax, parameters, and code examples to demonstrate how to display values on top of a Seaborn barplot. It involves importing necessary packages like pandas, numpy, and seaborn, reading datasets, creating barplots, and displaying bar values on the plot.\n\n\nIn summary, the Reddit post addresses challenges in plotting the 10 most frequent values in a barplot using Seaborn, while the GeeksforGeeks article offers a detailed guide on how to show values on Seaborn barplots in Python.\n","metadata":{}},{"cell_type":"code","source":"# Print Min and Max number of CT scans\nprint(\"Minimum number of CT scans: {}\".format(train[\"CT_number\"].min()))\nprint(\"Maximum number of CT scans: {:,}\".format(train[\"CT_number\"].max()))\n\n# Scans per Patient\ndata = train.groupby(by=\"Patient\")[\"CT_number\"].first().reset_index()\n# Sort by Weeks\ndata = data.sort_values('CT_number').reset_index(drop=True)\n\n# Plot\nplt.figure(figsize=(16, 6))\np = sns.barplot(x=data[\"Patient\"], y=data[\"CT_number\"], color=custom_colors[5])\nplt.axhline(y=85, color=custom_colors[2], linestyle='--', lw=3)\n\nplt.title(\"Number of CT Scans per Patient\", fontsize=17)\nplt.xlabel('Patient', fontsize=14)\nplt.ylabel('Frequency', fontsize=14)\n\nplt.text(86, 850, \"Median=94\", fontsize=13)\n\nplt.xticks(rotation=45)  # Rotate x-axis labels for better readability","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-23T05:27:28.843398Z","iopub.execute_input":"2024-04-23T05:27:28.843675Z","iopub.status.idle":"2024-04-23T05:27:32.376375Z","shell.execute_reply.started":"2024-04-23T05:27:28.84365Z","shell.execute_reply":"2024-04-23T05:27:32.375493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Importing Libraries:\n    * The code imports necessary libraries: pydicom for working with DICOM files and matplotlib.pyplot for plotting images.\n* Reading and Displaying DICOM Image:\n    * It specifies the file path of a DICOM image to be read and stored in the dataset variable using pydicom.dcmread(path).\n    * Information about the DICOM image is then printed, including Patient ID, Modality, Rows, and Columns.\n    * A matplotlib figure is created with a size of 7x7 inches to display the image.\n    * The DICOM image's pixel array is visualized using plt.imshow(dataset.pixel_array, cmap=\"plasma\"), where cmap=\"plasma\" sets the color map for the image.\n    * plt.axis('off') is used to turn off the axis labels for a cleaner image display.\n    * Finally, plt.show() is called to show the DICOM image.\n    \n In summary, this code snippet reads a DICOM image, extracts relevant information from the metadata, and displays the image using matplotlib. It provides a simple and effective way to view medical imaging data stored in DICOM format.","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\npath = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/19.dcm\"\ndataset = pydicom.dcmread(path)\n\nprint(\"Patient id.......:\", dataset.PatientID, \"\\n\" +\n      \"Modality.........:\", dataset.Modality, \"\\n\" +\n      \"Rows.............:\", dataset.Rows, \"\\n\" +\n      \"Columns..........:\", dataset.Columns)\n\nplt.figure(figsize=(7, 7))\nplt.imshow(dataset.pixel_array, cmap=\"plasma\")\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:32.377783Z","iopub.execute_input":"2024-04-23T05:27:32.37836Z","iopub.status.idle":"2024-04-23T05:27:32.837421Z","shell.execute_reply.started":"2024-04-23T05:27:32.378327Z","shell.execute_reply":"2024-04-23T05:27:32.836487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The code beow reads DICOM images from a specified directory related to a patient's data.\n* It orders the DICOM files based on their numeric values in the filenames.\n* It reads the DICOM files using Pydicom library and stores them in a list called datasets.\n* It then plots the pixel arrays of the DICOM images in a grid layout using Matplotlib.\n* The code visualizes a subset of the DICOM images (up to 30 images) in a 10x3 grid with image numbers as titles.\n","metadata":{}},{"cell_type":"code","source":"import pydicom\n\n\npatient_dir = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430\"\ndatasets = []\n\n# First, order the files in the dataset\nfiles = os.listdir(patient_dir)\nfiles.sort(key=lambda f: int(re.sub('\\D', '', f)))\n\n# Read in the Dataset\nfor dcm in files:\n    path = os.path.join(patient_dir, dcm)\n    datasets.append(pydicom.dcmread(path))\n\n# Plot the images\nfig = plt.figure(figsize=(16, 6))\ncolumns = 10\nrows = 3\n\nfor i in range(1, min(columns*rows, len(datasets)) + 1):\n    img = datasets[i-1].pixel_array\n    ax = fig.add_subplot(rows, columns, i)\n    ax.imshow(img, cmap=\"plasma\")\n    ax.set_title(i, fontsize=9)\n    ax.axis('off')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:32.838564Z","iopub.execute_input":"2024-04-23T05:27:32.838851Z","iopub.status.idle":"2024-04-23T05:27:37.128481Z","shell.execute_reply.started":"2024-04-23T05:27:32.838827Z","shell.execute_reply":"2024-04-23T05:27:37.127505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* PIL (Python Imaging Library) Import:\n    * from PIL import Image imports the Python Imaging Library, which provides capabilities for opening, manipulating, and saving many different image file formats.\n* IPython Display Import:\n    * from IPython.display import Image as show_gif imports the Image function from IPython's display module. This allows displaying images, including GIFs, directly in Jupyter notebooks.\n* Scipy Misc Import:\n    * import scipy.misc imports the misc module from SciPy, which includes various image processing functions. However, it's important to note that scipy.misc has been deprecated in newer versions of SciPy in favor of other image processing libraries like scikit-image.\n\n In summary, this code snippet sets up the environment for working with images, including opening, displaying, and potentially processing them using the Python Imaging Library (PIL), IPython's display functionality, and the deprecated scipy.misc module.","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nfrom IPython.display import Image as show_gif\nimport scipy.misc","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:37.129711Z","iopub.execute_input":"2024-04-23T05:27:37.130017Z","iopub.status.idle":"2024-04-23T05:27:37.137649Z","shell.execute_reply.started":"2024-04-23T05:27:37.129991Z","shell.execute_reply":"2024-04-23T05:27:37.136751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Selecting a Patient:\n    * The function selects a patient based on the number of CT scans provided. It retrieves the patient ID from the dataset based on the specified number of CT scans.\n* Reading and Ordering CT Scan Files:\n    * It reads the DICOM (.dcm) files for the selected patient, orders them numerically, and stores them in a list of datasets.\n* Converting to PNG:\n    * The function converts each DICOM image to a PNG format and saves them in a new directory named png_patientID.\n* Creating GIF:\n    * It reads the PNG images, orders them, creates frames from the images, and then saves them as a GIF file named gif_patientID.gif.\n    * The GIF is set to loop indefinitely with a duration of 200 milliseconds per frame.\n\nIn summary, this function automates the process of converting a series of CT scan images for a specific patient into a GIF for easier visualization and analysis.","metadata":{}},{"cell_type":"code","source":"import matplotlib.image as mpimg\n\ndef create_gif(number_of_CT=87):\n    \"\"\"Selects a patient based on the CT number and creates a GIF with their CT scans.\"\"\"\n    \n    # Select a patient based on the CT number\n    patient = train[train[\"CT_number\"] == number_of_CT].sample(random_state=1)[\"Patient\"].values[0]\n    \n    # Read in .dcm files\n    patient_dir = \"../input/osic-pulmonary-fibrosis-progression/train/\" + patient\n    datasets = []\n\n    # Order the files in the dataset\n    files = os.listdir(patient_dir)\n    files.sort(key=lambda f: int(re.sub('\\D', '', f)))\n\n    # Read in the dataset from the patient path\n    for dcm in files:\n        path = os.path.join(patient_dir, dcm)\n        datasets.append(pydicom.dcmread(path))\n        \n    # Save as .png\n    # Create directory to save the png files\n    if not os.path.isdir(f\"png_{patient}\"):\n        os.mkdir(f\"png_{patient}\")\n\n    # Save images to PNG\n    for i, dataset in enumerate(datasets):\n        img = dataset.pixel_array\n        mpimg.imsave(f'png_{patient}/img_{i}.png', img)\n        \n    # Create GIF\n    # Order the files in the dataset (again)\n    files = os.listdir(f\"../working/png_{patient}\")\n    files.sort(key=lambda f: int(re.sub('\\D', '', f)))\n\n    # Create the frames\n    frames = []\n\n    # Create frames\n    for file in files:\n        new_frame = Image.open(f\"../working/png_{patient}/\" + file)\n        frames.append(new_frame)\n\n    # Save as a GIF file that loops forever\n    frames[0].save(f'gif_{patient}.gif', format='GIF',\n                   append_images=frames[1:],\n                   save_all=True,\n                   duration=200, loop=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:37.138813Z","iopub.execute_input":"2024-04-23T05:27:37.139075Z","iopub.status.idle":"2024-04-23T05:27:37.150104Z","shell.execute_reply.started":"2024-04-23T05:27:37.139052Z","shell.execute_reply":"2024-04-23T05:27:37.149335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Creating GIFs:\n    * The create_gif function is called three times with different numbers of CT scans (12, 30, and 87) for different patients. This generates GIFs from the CT scan images for each patient.\n* Printing PNG File Counts:\n    * The code is commented out, but it intends to print the number of PNG files created for each patient after generating the GIFs.\n    * It checks the number of PNG files in the directories where the PNG images were saved for each patient.\n    In summary, this code snippet automates the creation of GIFs from CT scan images for different patients with varying numbers of CT scans and then checks the number of PNG files generated for each patient's GIF creation process.","metadata":{}},{"cell_type":"code","source":"create_gif(number_of_CT=12)\ncreate_gif(number_of_CT=30)\ncreate_gif(number_of_CT=87)\n\n#print(\"First file len:\", len(os.listdir(\"../working/png_ID00165637202237320314458\")), \"\\n\" +\n     #\"Second file len:\", len(os.listdir(\"../working/png_ID00199637202248141386743\")), \"\\n\" +\n      #\"Third file len:\", len(os.listdir(\"../working/png_ID00340637202287399835821\")))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:37.15144Z","iopub.execute_input":"2024-04-23T05:27:37.152177Z","iopub.status.idle":"2024-04-23T05:27:56.421754Z","shell.execute_reply.started":"2024-04-23T05:27:37.152151Z","shell.execute_reply":"2024-04-23T05:27:56.420966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* show_gif: This is the name of the function being called. It's likely a custom function that someone has defined elsewhere in their code.\n* filename=\"./gif_ID00165637202237320314458.gif\": This part specifies the path and name of the GIF file that the function should display. In this case, the GIF file is named \"gif_ID00165637202237320314458.gif\" and is located in the current directory (denoted by \"./\").\n* format='png': This parameter specifies the format in which the GIF should be displayed. However, it seems unusual because GIFs are typically displayed as GIFs themselves, not as PNG images. This might be a mistake or a specific requirement of the show_gif function.\n* width=400, height=400: These parameters specify the width and height, respectively, of the displayed image. In this case, the image will be displayed with a width of 400 pixels and a height of 400 pixels.\n* So, in simple terms, this line of code is telling a function called show_gif to display the GIF file named \"gif_ID00165637202237320314458.gif\" with a width of 400 pixels and a height of 400 pixels. However, it's unusual that the format is specified as 'png' since GIFs are typically displayed as GIFs.","metadata":{}},{"cell_type":"code","source":"show_gif(filename=\"./gif_ID00165637202237320314458.gif\", format='png', width=400, height=400)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:56.422802Z","iopub.execute_input":"2024-04-23T05:27:56.423075Z","iopub.status.idle":"2024-04-23T05:27:56.440182Z","shell.execute_reply.started":"2024-04-23T05:27:56.42305Z","shell.execute_reply":"2024-04-23T05:27:56.439203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* show_gif: Again, this is likely the name of a function that displays animated GIF files.\n* filename: This parameter specifies the path and name of the GIF file that the function should display. In the first line, it's showing a GIF named \"gif_ID00340637202287399835821.gif\", and in the second line, it's showing a GIF named \"gif_ID00199637202248141386743.gif\". Both files are assumed to be located in the current directory (denoted by \"./\").\n* format='png': This parameter specifies the format in which the GIF should be displayed. As previously mentioned, it's unusual to specify 'png' since GIFs are typically displayed as GIFs themselves.\n* width=400, height=400: These parameters specify the width and height, respectively, of the displayed images. In both cases, the images will be displayed with a width of 400 pixels and a height of 400 pixels.\n* In simple terms, these lines of code are telling a function called show_gif to display two different GIF files with identical dimensions of 400 pixels by 400 pixels. Again, it's noteworthy that the format is specified as 'png', which is uncommon for displaying GIFs.","metadata":{}},{"cell_type":"code","source":"show_gif(filename=\"./gif_ID00340637202287399835821.gif\", format='png', width=400, height=400)\nshow_gif(filename=\"./gif_ID00199637202248141386743.gif\", format='png', width=400, height=400)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:56.441296Z","iopub.execute_input":"2024-04-23T05:27:56.441575Z","iopub.status.idle":"2024-04-23T05:27:56.514549Z","shell.execute_reply.started":"2024-04-23T05:27:56.441551Z","shell.execute_reply":"2024-04-23T05:27:56.513366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Image Preprocessing:\n    * It calculates the mean and standard deviation of the image, then standardizes the image by subtracting the mean and dividing by the standard deviation. This step helps in normalizing the image data.\n    * It selects a region of interest (ROI) within the image, likely containing the lungs, for further processing.\n* Thresholding and Segmentation:\n    * It uses K-means clustering to separate foreground (soft tissue/bone) from background (lung/air) based on intensity values.\n    * After thresholding, it performs erosion and dilation operations to remove noise and refine the mask.\n* Labeling and Filtering:\n    * It labels different regions in the image and filters out the regions that are unlikely to be lungs based on their size and position within the image.\n* Mask Creation:\n \n    * It creates a binary mask where pixels corresponding to the lungs are marked as 1 and others as 0.\n* Visualization (Optional): \n    * If the display parameter is set to True, it visualizes the intermediate steps of the process, including the original image, thresholded image, eroded and dilated image, color-labeled regions, final mask, and the original image with the mask applied.\n* Return:\n    * It returns the masked image, where only the pixels corresponding to the lungs are retained.\n This function is designed to automate the process of segmenting lungs from medical images, which is often a crucial step in various medical image analysis tasks. It employs image processing techniques such as thresholding, morphological operations, and region labeling to achieve this segmentation.","metadata":{}},{"cell_type":"code","source":"def make_lungmask(img, display=False):\n    row_size= img.shape[0]\n    col_size = img.shape[1]\n    \n    mean = np.mean(img)\n    std = np.std(img)\n    img = img-mean\n    img = img/std\n    \n    # Find the average pixel value near the lungs\n        # to renormalize washed out images\n    middle = img[int(col_size/5):int(col_size/5*4),int(row_size/5):int(row_size/5*4)] \n    mean = np.mean(middle)  \n    max = np.max(img)\n    min = np.min(img)\n    \n    # To improve threshold finding, I'm moving the \n    # underflow and overflow on the pixel spectrum\n    img[img==max]=mean\n    img[img==min]=mean\n    \n    # Using Kmeans to separate foreground (soft tissue / bone) and background (lung/air)\n    \n    kmeans = KMeans(n_clusters=2).fit(np.reshape(middle,[np.prod(middle.shape),1]))\n    centers = sorted(kmeans.cluster_centers_.flatten())\n    threshold = np.mean(centers)\n    thresh_img = np.where(img<threshold,1.0,0.0)  # threshold the image\n\n    # First erode away the finer elements, then dilate to include some of the pixels surrounding the lung.  \n    # We don't want to accidentally clip the lung.\n\n    eroded = morphology.erosion(thresh_img,np.ones([3,3]))\n    dilation = morphology.dilation(eroded,np.ones([8,8]))\n\n    labels = measure.label(dilation) # Different labels are displayed in different colors\n    label_vals = np.unique(labels)\n    regions = measure.regionprops(labels)\n    good_labels = []\n    for prop in regions:\n        B = prop.bbox\n        if B[2]-B[0]<row_size/10*9 and B[3]-B[1]<col_size/10*9 and B[0]>row_size/5 and B[2]<col_size/5*4:\n            good_labels.append(prop.label)\n    mask = np.ndarray([row_size,col_size],dtype=np.int8)\n    mask[:] = 0\n\n\n    #  After just the lungs are left, we do another large dilation\n    #  in order to fill in and out the lung mask \n    \n    for N in good_labels:\n        mask = mask + np.where(labels==N,1,0)\n    mask = morphology.dilation(mask,np.ones([10,10])) # one last dilation\n\n    if (display):\n        fig, ax = plt.subplots(3, 2, figsize=[12, 12])\n        ax[0, 0].set_title(\"Original\")\n        ax[0, 0].imshow(img, cmap='gray')\n        ax[0, 0].axis('off')\n        ax[0, 1].set_title(\"Threshold\")\n        ax[0, 1].imshow(thresh_img, cmap='gray')\n        ax[0, 1].axis('off')\n        ax[1, 0].set_title(\"After Erosion and Dilation\")\n        ax[1, 0].imshow(dilation, cmap='gray')\n        ax[1, 0].axis('off')\n        ax[1, 1].set_title(\"Color Labels\")\n        ax[1, 1].imshow(labels)\n        ax[1, 1].axis('off')\n        ax[2, 0].set_title(\"Final Mask\")\n        ax[2, 0].imshow(mask, cmap='gray')\n        ax[2, 0].axis('off')\n        ax[2, 1].set_title(\"Apply Mask on Original\")\n        ax[2, 1].imshow(mask*img, cmap='gray')\n        ax[2, 1].axis('off')\n        \n        plt.show()\n    return mask*img","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:56.523584Z","iopub.execute_input":"2024-04-23T05:27:56.523901Z","iopub.status.idle":"2024-04-23T05:27:56.542674Z","shell.execute_reply.started":"2024-04-23T05:27:56.523873Z","shell.execute_reply":"2024-04-23T05:27:56.541814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Selecting a Sample:\n    * The code selects a sample DICOM image file from a specified path. DICOM is a standard format used for medical imaging data.\n* Reading the DICOM File:\n    * It reads the DICOM file using the pydicom library. This library is commonly used to work with DICOM files in Python.\n* After reading the file, it extracts the pixel array from the DICOM dataset, which contains the actual image data.\n* Creating a Lung Mask: \n    * It calls a function named make_lungmask to generate a mask for isolating the lungs from the rest of the image.\n    * The img variable, which contains the pixel array of the DICOM image, is passed to the make_lungmask function.\n    * Additionally, the display=True parameter indicates that the function should display intermediate steps of the lung masking process.\n* Visualization:\n    * The make_lungmask function displays intermediate steps of the lung masking process, likely including the original image, thresholded image, eroded and dilated image, color-labeled regions, final mask, and the original image with the mask applied.\n    \n    In summary, this code snippet reads a DICOM image, processes it to isolate the lungs using a lung masking function, and displays the intermediate steps of the masking process for visualization purposes. It's part of a larger process for analyzing medical images, particularly for tasks related to pulmonary fibrosis progression analysis.","metadata":{}},{"cell_type":"code","source":"# Select a sample\npath = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/19.dcm\"\ndataset = pydicom.dcmread(path)\nimg = dataset.pixel_array\n\n# Masked image\nmask_img = make_lungmask(img, display=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:27:56.544053Z","iopub.execute_input":"2024-04-23T05:27:56.544371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Setting up the Environment:\n\n    * patient_dir: This variable contains the path to the directory containing DICOM files related to a specific patient. DICOM files are a standard format for medical images.\n* datasets: This list will store the DICOM datasets read from the files.\n* Ordering DICOM Files: \n    * The code first collects all the DICOM files present in the patient_dir directory and sorts them based on their numerical order. This ensures that the files are processed in the correct sequence.\n* Reading DICOM Files:\n    * It iterates over the sorted list of DICOM files, reads each file using pydicom.dcmread(), and appends the resulting DICOM dataset to the datasets list.\n* Extracting Pixel Arrays:\n    * For each dataset in the datasets list, it extracts the pixel array using the pixel_array attribute of the DICOM dataset. This pixel array represents the image data.\n* Displaying Masks:\n    * It sets up a figure to display the masks generated from the lung segmentation process.\n    * It iterates over the datasets and generates a lung mask for each image using the make_lungmask() function.\n    * Each mask is displayed in a separate subplot of the figure using plt.imshow().\n    * The loop continues for columns * rows iterations, creating a grid of subplots to display the masks for each image.\n    * plt.title() assigns a title to each subplot, indicating the index of the image.\n    * plt.axis('off') removes axis ticks and labels from each subplot.\n\nOverall, this code reads DICOM files related to a specific patient, processes them to extract image data, generates lung masks using a segmentation function, and displays the resulting masks in a grid for visual inspection. It's a common workflow in medical image analysis tasks, particularly for analyzing pulmonary fibrosis progression.","metadata":{}},{"cell_type":"code","source":"patient_dir = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430\"\ndatasets = []\n\n# First Order the files in the dataset\nfiles = []\nfor dcm in list(os.listdir(patient_dir)):\n    files.append(dcm) \nfiles.sort(key=lambda f: int(re.sub('\\D', '', f)))\n\n# Read in the Dataset\nfor dcm in files:\n    path = patient_dir + \"/\" + dcm\n    datasets.append(pydicom.dcmread(path))\n    \nimgs = []\nfor data in datasets:\n    img = data.pixel_array\n    imgs.append(img)\n    \n    \n# Show masks\nfig=plt.figure(figsize=(16, 6))\ncolumns = 10\nrows = 3\n\nfor i in range(1, columns*rows +1):\n    img = make_lungmask(datasets[i-1].pixel_array)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(i, fontsize = 9)\n    plt.axis('off');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\n!cp ../input/rapids/rapids.0.14.0 /opt/conda/envs/rapids.tar.gz\n!cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n!cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_observation_data(path):\n    \"\"\"Get information from the .dcm files.\n    path: complete path to the .dcm file\"\"\"\n\n    image_data = pydicom.read_file(path)\n    \n    # Dictionary to store the information from the image\n    observation_data = {\n        \"FileNumber\" : path.split(\"/\")[5],\n        \"Rows\" : image_data.get(\"Rows\"),\n        \"Columns\" : image_data.get(\"Columns\"),\n        \"PatientID\" : image_data.get(\"PatientID\"),\n        \"BodyPartExamined\" : image_data.get(\"BodyPartExamined\"),\n        \"RotationDirection\" : image_data.get(\"RotationDirection\"),\n        \"ConvolutionKernel\" : image_data.get(\"ConvolutionKernel\"),\n        \"PatientPosition\" : image_data.get(\"PatientPosition\"),\n        \"PhotometricInterpretation\" : image_data.get(\"PhotometricInterpretation\"),\n        \"Modality\" : image_data.get(\"Modality\"),\n        \"StudyInstanceUID\" : image_data.get(\"StudyInstanceUID\"),\n        \"PixelPaddingValue\" : image_data.get(\"PixelPaddingValue\"),\n        \"SamplesPerPixel\" : image_data.get(\"SamplesPerPixel\"),\n        \"BitsAllocated\" : image_data.get(\"BitsAllocated\"),\n        \"BitsStored\" : image_data.get(\"BitsStored\"),\n        \"HighBit\" : image_data.get(\"HighBit\"),\n        \"PixelRepresentation\" : image_data.get(\"PixelRepresentation\"),\n        \"RescaleType\" : image_data.get(\"RescaleType\"),\n    }\n\n    # Integer columns\n    int_columns = [\"SliceThickness\", \"KVP\", \"DistanceSourceToDetector\", \n        \"DistanceSourceToPatient\", \"GantryDetectorTilt\", \"TableHeight\", \n        \"XRayTubeCurrent\", \"GeneratorPower\", \"WindowCenter\", \"WindowWidth\", \n        \"SliceLocation\", \"RescaleIntercept\", \"RescaleSlope\"]\n    for k in int_columns:\n        observation_data[k] = int(image_data.get(k)) if k in image_data else None\n\n    # String columns\n    str_columns = [\"ImagePositionPatient\", \"ImageOrientationPatient\", \"ImageType\", \"PixelSpacing\"]\n    for k in str_columns:\n        observation_data[k] = str(image_data.get(k)) if k in image_data else None\n\n    \n    return observation_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/10.dcm\"\nexample = get_observation_data(p)\n# Get full paths for the images\npaths = []\nfor path in train[\"Path\"]:\n    for doc in os.listdir(path):\n        paths.append(path + \"/\" + doc)\n        \n# How many paths?\nprint(\"There are {:,} paths in total.\".format(len(paths)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**From the above code we infer the required information we get from dcm files for example the above code gets a total of 289826 paths in total for a given directory.**","metadata":{}},{"cell_type":"markdown","source":"tqdm library\n* Progress Tracking: tqdm is a Python library designed for tracking the progress of iterative tasks, providing users with real-time feedback on the completion status.\n\n* Versatile Compatibility: It seamlessly integrates with various iterable types, including lists, tuples, dictionaries, and custom iterators, catering to diverse programming needs.\n\n* Customization Options: Users can easily customize the appearance and behavior of progress bars, tailoring them to specific preferences through features such as style selection and format adjustment.\n\n* Concurrency Support: tqdm supports concurrent programming paradigms like multiprocessing and multithreading, allowing for efficient tracking of progress in concurrent tasks.","metadata":{}},{"cell_type":"markdown","source":"* After tracking tqdm paths (289826) we are finding exceptions in meta data.\n* This tqdm helps us finding that exceptions.\n* Here we get around 279 exceptions in meta data.","metadata":{}},{"cell_type":"code","source":"import tqdm\n# Create a dictionary with the data\nexceptions = 0\ndicts = []\n\nfor path in tqdm.tqdm(paths):\n    # Get info in dict format\n    try:\n        d = get_observation_data(path)\n        dicts.append(d)\n    except Exception as e:\n        exceptions += 1\n        continue\n\n# Convert into a cudf dataframe\n# meta_train_data = cudf.DataFrame(data=dicts, columns=example.keys())\nmeta_train_data = pd.DataFrame(data=dicts, columns=example.keys())\n\n# Export information to a .csv\nmeta_train_data.to_csv(\"meta_train.csv\", index=False)","metadata":{"execution":{"iopub.status.idle":"2024-04-23T05:51:33.311624Z","shell.execute_reply.started":"2024-04-23T05:28:12.392278Z","shell.execute_reply":"2024-04-23T05:51:33.310603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Exceptions: {}\".format(exceptions))\nmeta_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:33.312856Z","iopub.execute_input":"2024-04-23T05:51:33.31316Z","iopub.status.idle":"2024-04-23T05:51:33.339548Z","shell.execute_reply.started":"2024-04-23T05:51:33.313135Z","shell.execute_reply":"2024-04-23T05:51:33.33871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate missing percentages for specific columns\nmissing_columns = [\"PixelPaddingValue\", \"RescaleType\", \"DistanceSourceToDetector\", \n                   \"DistanceSourceToPatient\", \"GantryDetectorTilt\", \"GeneratorPower\", \n                   \"SliceLocation\", \"ImagePositionPatient\", \"ImageOrientationPatient\"]\n\nmissing_percentages = {col: meta_train_data[col].isna().sum() / len(meta_train_data[col]) for col in missing_columns}\n\n# Print missing percentages\nprint(\"---Missing Columns---\")\nfor col, percentage in missing_percentages.items():\n    print(f\"{col} missing percentage: {percentage:.2f}\")\n\n# Visualize missing values using matrix plot\nax = msno.matrix(meta_train_data, color=(152/255, 241/255, 198/255), fontsize=10)\nplt.title(\"Metadata Missing Values\", fontsize=25)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:33.340666Z","iopub.execute_input":"2024-04-23T05:51:33.340941Z","iopub.status.idle":"2024-04-23T05:51:37.678242Z","shell.execute_reply.started":"2024-04-23T05:51:33.340916Z","shell.execute_reply":"2024-04-23T05:51:37.677324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Laplace Log Likelihood\n![](https://i.imgur.com/tEIZvli.png)\n*Image Credits: https://en.wikipedia.org/wiki/Laplace_distribution*\n\nThe evaluation metric of this competition is a modified version of Laplace Log Likelihood. Read more about it on the [Evaluation Page](https://www.kaggle.com/c/osic-pulmonary-fibrosis-progression/overview/evaluation).\n\n![LLL.png](attachment:LLL.png)\n\nThis notebook explores the functioning of this metric with some examples and visualizations.\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\n\nfrom bokeh.layouts import row, column\nfrom bokeh.models import ColumnDataSource, CustomJS, Label, Range1d, Slider, Span\nfrom bokeh.plotting import figure, output_notebook, show\n\noutput_notebook()\n\n## reading data\ndf_train = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:37.67934Z","iopub.execute_input":"2024-04-23T05:51:37.679605Z","iopub.status.idle":"2024-04-23T05:51:38.134248Z","shell.execute_reply.started":"2024-04-23T05:51:37.679582Z","shell.execute_reply":"2024-04-23T05:51:38.133531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## FVC Distribution\nLet's look at the distribution of the target variable **FVC**.\n","metadata":{}},{"cell_type":"code","source":"sns.distplot(df_train.FVC, color = \"brown\");","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:38.135758Z","iopub.execute_input":"2024-04-23T05:51:38.136164Z","iopub.status.idle":"2024-04-23T05:51:38.593843Z","shell.execute_reply.started":"2024-04-23T05:51:38.13613Z","shell.execute_reply":"2024-04-23T05:51:38.593058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FVC ranges from **827 to 6399** and seems to have a (visually) normal distribution with **mean 2690** and **standard deviation 833**.\n\n## Evaluation Metric\nNote that not only do we need to predict the FVC but also a confidence value of the prediction. The term *confidence* is a bit confusing (as lower value means more confident) and it might be better to just consider it as the *standard deviation* (or the *uncertainty*).   \n\nLet's define a function and then use it with some examples and understand the metric.\n","metadata":{}},{"cell_type":"code","source":"def laplace_log_likelihood(actual_fvc, predicted_fvc, confidence, return_values = False):\n    \"\"\"\n    Calculates the modified Laplace Log Likelihood score for this competition.\n    \"\"\"\n    sd_clipped = np.maximum(confidence, 70)\n    delta = np.minimum(np.abs(actual_fvc - predicted_fvc), 1000)\n    metric = - np.sqrt(2) * delta / sd_clipped - np.log(np.sqrt(2) * sd_clipped)\n\n    if return_values:\n        return metric\n    else:\n        return np.mean(metric)\n\n\n## default benchmark\nlaplace_log_likelihood(df_train.FVC, np.mean(df_train.FVC), np.std(df_train.FVC))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:38.594908Z","iopub.execute_input":"2024-04-23T05:51:38.595182Z","iopub.status.idle":"2024-04-23T05:51:38.605224Z","shell.execute_reply.started":"2024-04-23T05:51:38.595157Z","shell.execute_reply":"2024-04-23T05:51:38.604438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**-8.023** is the default score to beat while cross-validating models on train data. Any model scoring worse than this is not useful. You can get this default score to beat for each (fold of) validation data as well.\n\n## Constant Prediction\nLet's try out a few combinations of predicting a constant value for FVC as well as confidence to compare how the errors vary with the actual FVC values.\n","metadata":{}},{"cell_type":"code","source":"def plot_metric_constants(actual_fvc, constant_fvc, constant_confidence):\n    \"\"\"\n    Generates a Bokeh plot for a constant value of predicted FVC and confidence.\n    \"\"\"\n    lll = laplace_log_likelihood(actual_fvc, constant_fvc, constant_confidence)\n\n    df = pd.DataFrame({\n        \"actual_FVC\": actual_fvc,\n        \"predicted_FVC\": constant_fvc,\n        \"confidence\": constant_confidence,\n        \"metric\": laplace_log_likelihood(actual_fvc, constant_fvc, constant_confidence, return_values=True)\n    }).sort_values(\"actual_FVC\")\n    \n    source = ColumnDataSource(df)\n    \n    tooltips = [\n        (\"Actual FVC\", \"@actual_FVC\"),\n        (\"Predicted FVC\", \"@predicted_FVC\"),\n        (\"Confidence\", \"@confidence\"),\n        (\"Metric\", \"@metric{0.000}\")\n    ]\n    \n    v = figure(\n        width=345,\n        height=345,\n        y_range=Range1d(-4, -25),\n        tooltips=tooltips,\n        title=f\"Metric values for FVC = {constant_fvc}, Confidence = {constant_confidence}\"\n    )\n\n    v.circle(\"actual_FVC\", \"metric\", source=source, size=3, color=\"deepskyblue\", alpha=0.8)\n    \n    mean = Span(\n        location=lll,\n        dimension=\"width\",\n        line_color=\"red\",\n        line_dash=\"dashed\",\n        line_width=1.5\n    )\n\n    v.add_layout(mean)\n    \n    score = Label(\n        x=3500,\n        y=lll + 1.25,\n        text=f\"Laplace Log Likelihood = {round(lll, 3)}\",\n        text_font_size=\"7pt\"\n    )\n\n    v.add_layout(score)\n    \n    v.xaxis.axis_label = \"Actual FVC\"\n    v.yaxis.axis_label = \"Metric Value\"\n\n    return v\n\n\nv1 = plot_metric_constants(df_train.FVC, 2690, 100)\nv2 = plot_metric_constants(df_train.FVC, 2000, 100)\nv3 = plot_metric_constants(df_train.FVC, 3000, 100)\nv4 = plot_metric_constants(df_train.FVC, 4000, 100)\n\nv5 = plot_metric_constants(df_train.FVC, 2000, 833)\nv6 = plot_metric_constants(df_train.FVC, 2690, 833)\nv7 = plot_metric_constants(df_train.FVC, 3000, 833)\nv8 = plot_metric_constants(df_train.FVC, 4000, 833)\n\nv9 = plot_metric_constants(df_train.FVC, 2690, 70)\nv10 = plot_metric_constants(df_train.FVC, 2690, 100)\nv11 = plot_metric_constants(df_train.FVC, 2690, 833)\nv12 = plot_metric_constants(df_train.FVC, 2690, 1000)\n\nshow(column(row(v1, v2), row(v3, v4), row(v5, v6), row(v7, v8), row(v9, v10), row(v11, v12)))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:38.606542Z","iopub.execute_input":"2024-04-23T05:51:38.607246Z","iopub.status.idle":"2024-04-23T05:51:39.549106Z","shell.execute_reply.started":"2024-04-23T05:51:38.607221Z","shell.execute_reply":"2024-04-23T05:51:39.548264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Interactive Simulation\nYou can play around with combinations of FVC values to see how the metric value varies over confidence.\n\n* Any difference between the FVC values greater than 1000 will yield the same metric values since the delta is capped at 1000. \n* The metric value is only dependent on the difference of FVC values and not on the absolute values.\n","metadata":{}},{"cell_type":"code","source":"x_values = list(range(70, 2001))\nsource = ColumnDataSource(data=dict(\n    x=x_values,\n    y=[-np.sqrt(2) * 500 / x - np.log(np.sqrt(2) * x) for x in x_values]\n))\n\ntooltips = [\n    (\"Confidence\", \"@x\"),\n    (\"Metric\", \"@y\")\n]\n\nv1 = figure(\n    width=300,\n    height=300,\n    tooltips=tooltips,\n    title=\"Metric values across confidence\"\n)\n\nv1.line(\"x\", \"y\", source=source, line_width=4, color=\"coral\", alpha=0.8)\n\nv1.y_range.flipped = True\n\nv1.xaxis.axis_label = \"Confidence (Uncertainty)\"\nv1.yaxis.axis_label = \"Metric Value\"\n\nv2 = figure(\n    width=300,\n    height=300,\n    tooltips=tooltips,\n    y_range=Range1d(-4, -25),\n    title=\"Metric values across confidence\"\n)\n\nv2.line(\"x\", \"y\", source=source, line_width=4, color=\"coral\", alpha=0.8)\n\nv2.xaxis.axis_label = \"Confidence (Uncertainty)\"\nv2.yaxis.axis_label = \"Metric Value\"\n\nslider_actual_FVC = Slider(start=827, end=6399, value=2500, step=1, title=\"Actual FVC\")\nslider_predicted_FVC = Slider(start=827, end=6399, value=3000, step=1, title=\"Predicted FVC\")\n\ncallback = CustomJS(args=dict(\n    source=source,\n    actual_FVC=slider_actual_FVC,\n    predicted_FVC=slider_predicted_FVC\n), code=\"\"\"\n    var data = source.data;\n    var actual_FVC = actual_FVC.value;\n    var predicted_FVC = predicted_FVC.value;\n\n    var x = data['x'];\n    var y = data['y'];\n    var delta = Math.min(Math.abs(actual_FVC - predicted_FVC), 1000);\n\n    for (var i = 0; i < x.length; i++) {\n        y[i] = -Math.sqrt(2) * delta / x[i] - Math.log(Math.sqrt(2) * x[i]);\n    }\n\n    source.change.emit();\n\"\"\")\n\nslider_actual_FVC.js_on_change(\"value\", callback)\nslider_predicted_FVC.js_on_change(\"value\", callback)\n\nshow(column(slider_actual_FVC, slider_predicted_FVC, row(v1, v2)))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:39.550269Z","iopub.execute_input":"2024-04-23T05:51:39.550567Z","iopub.status.idle":"2024-04-23T05:51:39.74637Z","shell.execute_reply.started":"2024-04-23T05:51:39.550543Z","shell.execute_reply":"2024-04-23T05:51:39.745599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ML Prediction**","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mpimg\nfrom tabulate import tabulate\nimport missingno as msno\nfrom IPython.display import display_html\nfrom PIL import Image\nimport gc\nimport cv2\nfrom scipy.stats import pearsonr","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:39.747556Z","iopub.execute_input":"2024-04-23T05:51:39.747914Z","iopub.status.idle":"2024-04-23T05:51:39.755878Z","shell.execute_reply.started":"2024-04-23T05:51:39.747882Z","shell.execute_reply":"2024-04-23T05:51:39.755146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While ML models optimize on the point predictions of a target variable (like FVC), it will be interesting to see how the confidence predictions can be optimized. I'm sure through the course of the competition we will see some interesting approaches for the same.\n\n## ML Prediction\nLet's look at the out-of-fold predictions of some of the ML models. The example used below are the predictions from this kernel: https://www.kaggle.com/yasufuminakama/osic-lgb-baseline\n","metadata":{}},{"cell_type":"code","source":"df_preds = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/sample_submission.csv\")\ndf_preds.columns=df_preds.columns.to_series().apply(lambda x:x.strip())\n\ndef plot_fvc_metric_model(actual_fvc, predicted_fvc, predicted_confidence, constant = False):\n    \"\"\"\n    Generatings a bokeh plot for the metric values across actual values of FVC.\n    \"\"\"\n    lll = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence)\n    metric_values = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence, return_values = True)\n\n    df = pd.DataFrame({\n        \"actual_FVC\": actual_fvc,\n        \"predicted_FVC\": predicted_fvc,\n        \"confidence\": predicted_confidence,\n        \"metric\": metric_values\n    }).sort_values(\"actual_FVC\")\n    \n    source = ColumnDataSource(df)\n    \n    tooltips = [\n        (\"Actual FVC\", \"@actual_FVC{0}\"),\n        (\"Predicted FVC\", \"@predicted_FVC{0}\"),\n        (\"Confidence\", \"@confidence{0}\"),\n        (\"Metric\", \"@metric{0.000}\")\n    ]\n    \n    if type(predicted_confidence) != int:\n        title = f\"Metric values over FVC\"\n        offset = 0.35\n    else:\n        title = f\"Metric values over FVC (Confidence = {predicted_confidence})\"\n        offset = 0.11\n    \n    v = figure(\n        plot_width = 600,\n        plot_height = 600,\n        y_range = Range1d(max(metric_values) + 0.5, min(metric_values) - 0.5),\n        tooltips = tooltips,\n        title = title\n    )\n\n    v.circle(\"actual_FVC\", \"metric\", source = source, size = 3, color = \"mediumseagreen\", alpha = 0.8)\n    \n    mean = Span(\n        location = lll,\n        dimension = \"width\",\n        line_color = \"red\",\n        line_dash = \"dashed\",\n        line_width = 1.5\n    )\n\n    v.add_layout(mean)\n    \n    score = Label(\n        x = 3750,\n        y = lll + offset,\n        text = f\"Laplace Log Likelihood = {round(lll, 3)}\",\n        text_font_size = \"12pt\"\n    )\n\n    v.add_layout(score)\n    \n    v.xaxis.axis_label = \"Actual FVC\"\n    v.yaxis.axis_label = \"Metric Value\"\n\n    return v\n\n\ndef plot_confidence_metric_model(actual_fvc, predicted_fvc, predicted_confidence):\n    \"\"\"\n    Generatings a bokeh plot for the metric values across predicted values of confidence.\n    \"\"\"\n    lll = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence)\n    metric_values = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence, return_values = True)\n\n    df = pd.DataFrame({\n        \"actual_FVC\": actual_fvc,\n        \"predicted_FVC\": predicted_fvc,\n        \"confidence\": predicted_confidence,\n        \"metric\": metric_values\n    }).sort_values(\"confidence\")\n    \n    source = ColumnDataSource(df)\n    \n    tooltips = [\n        (\"Actual FVC\", \"@actual_FVC{0}\"),\n        (\"Predicted FVC\", \"@predicted_FVC{0}\"),\n        (\"Confidence\", \"@confidence{0}\"),\n        (\"Metric\", \"@metric{0.000}\")\n    ]\n    \n    v = figure(\n        plot_width = 600,\n        plot_height = 600,\n        y_range = Range1d(max(metric_values) + 0.5, min(metric_values) - 0.5),\n        tooltips = tooltips,\n        title = \"Metric values over Confidence\"\n    )\n\n    v.circle(\"confidence\", \"metric\", source = source, size = 3, color = \"gold\", alpha = 0.8)\n    \n    mean = Span(\n        location = lll,\n        dimension = \"width\",\n        line_color = \"red\",\n        line_dash = \"dashed\",\n        line_width = 1.5\n    )\n\n    v.add_layout(mean)\n    \n    score = Label(\n        x = 600,\n        y = lll + 0.35,\n        text = f\"Laplace Log Likelihood = {round(lll, 3)}\",\n        text_font_size = \"12pt\"\n    )\n\n    v.add_layout(score)\n    \n    v.xaxis.axis_label = \"Predicted Confidence\"\n    v.yaxis.axis_label = \"Metric Value\"\n\n    return v\n\n\ndef plot_pred_metric_model(actual_fvc, predicted_fvc, predicted_confidence):\n    \"\"\"\n    Generatings a bokeh plot for the metric values across predicted values of FVC.\n    \"\"\"\n    lll = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence)\n    metric_values = laplace_log_likelihood(actual_fvc, predicted_fvc, predicted_confidence, return_values = True)\n\n    df = pd.DataFrame({\n        \"actual_FVC\": actual_fvc,\n        \"predicted_FVC\": predicted_fvc,\n        \"confidence\": predicted_confidence,\n        \"metric\": metric_values\n    }).sort_values(\"predicted_FVC\")\n    \n    source = ColumnDataSource(df)\n    \n    tooltips = [\n        (\"Actual FVC\", \"@actual_FVC{0}\"),\n        (\"Predicted FVC\", \"@predicted_FVC{0}\"),\n        (\"Confidence\", \"@confidence{0}\"),\n        (\"Metric\", \"@metric{0.000}\")\n    ]\n    \n    v = figure(\n        plot_width = 600,\n        plot_height = 600,\n        y_range = Range1d(max(metric_values) + 0.5, min(metric_values) - 0.5),\n        tooltips = tooltips,\n        title = \"Metric values over predicted FVC\"\n    )\n\n    v.circle(\"predicted_FVC\", \"metric\", source = source, size = 3, color = \"turquoise\", alpha = 0.8)\n    \n    mean = Span(\n        location = lll,\n        dimension = \"width\",\n        line_color = \"red\",\n        line_dash = \"dashed\",\n        line_width = 1.5\n    )\n\n    v.add_layout(mean)\n    \n    score = Label(\n        x = 3500,\n        y = lll + 0.35,\n        text = f\"Laplace Log Likelihood = {round(lll, 3)}\",\n        text_font_size = \"12pt\"\n    )\n\n    v.add_layout(score)\n    \n    v.xaxis.axis_label = \"Predicted FVC\"\n    v.yaxis.axis_label = \"Metric Value\"\n\n    return v\n\n\nsns.pairplot(df_preds[[\"FVC\", \"Confidence\"]]);","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:39.757251Z","iopub.execute_input":"2024-04-23T05:51:39.757611Z","iopub.status.idle":"2024-04-23T05:51:41.198614Z","shell.execute_reply.started":"2024-04-23T05:51:39.75758Z","shell.execute_reply":"2024-04-23T05:51:41.197758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''''\n#v1 = plot_fvc_metric_model(df_preds.FVC,df_preds.Confidence)\nv2 = plot_fvc_metric_model(df_preds.FVC, df_preds.FVC, 833)\n\nshow(column(v1, v2))\n'''''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:41.199884Z","iopub.execute_input":"2024-04-23T05:51:41.20026Z","iopub.status.idle":"2024-04-23T05:51:41.20623Z","shell.execute_reply.started":"2024-04-23T05:51:41.200226Z","shell.execute_reply":"2024-04-23T05:51:41.205526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nv1 = plot_confidence_metric_model(df_preds.FVC, df_preds.FVC_pred, df_preds.Confidence)\nv2 = plot_pred_metric_model(df_preds.FVC, df_preds.FVC_pred, df_preds.Confidence)\n\nshow(column(v1, v2))\n'''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:41.207398Z","iopub.execute_input":"2024-04-23T05:51:41.208017Z","iopub.status.idle":"2024-04-23T05:51:41.217151Z","shell.execute_reply.started":"2024-04-23T05:51:41.20799Z","shell.execute_reply":"2024-04-23T05:51:41.216327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# using rmse and mse as metrics","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\n\n       \nimport numpy as np\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.preprocessing import MinMaxScaler\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport math\nfrom torch.nn import TransformerEncoder, TransformerEncoderLayer\n\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:41.218142Z","iopub.execute_input":"2024-04-23T05:51:41.2184Z","iopub.status.idle":"2024-04-23T05:51:45.089048Z","shell.execute_reply.started":"2024-04-23T05:51:41.218378Z","shell.execute_reply":"2024-04-23T05:51:45.088229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path = \"../input/osic-pulmonary-fibrosis-progression\"\ndf_train= pd.read_csv(f\"{Path}/train.csv\")\ndf_train.drop_duplicates(keep=False, inplace=True,subset=['Patient','Weeks'])\ndf_test= pd.read_csv(f\"{Path}/test.csv\")\nPatient_list= df_train[\"Patient\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.090202Z","iopub.execute_input":"2024-04-23T05:51:45.091168Z","iopub.status.idle":"2024-04-23T05:51:45.106101Z","shell.execute_reply.started":"2024-04-23T05:51:45.091139Z","shell.execute_reply":"2024-04-23T05:51:45.105328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.107223Z","iopub.execute_input":"2024-04-23T05:51:45.107833Z","iopub.status.idle":"2024-04-23T05:51:45.120589Z","shell.execute_reply.started":"2024-04-23T05:51:45.107799Z","shell.execute_reply":"2024-04-23T05:51:45.119836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.121565Z","iopub.execute_input":"2024-04-23T05:51:45.121882Z","iopub.status.idle":"2024-04-23T05:51:45.136498Z","shell.execute_reply.started":"2024-04-23T05:51:45.121854Z","shell.execute_reply":"2024-04-23T05:51:45.135773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for patient, group in df_train.groupby('Patient'):\n    print(f\"Patient {patient}: Weeks values {group['Weeks'].tolist()}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.137482Z","iopub.execute_input":"2024-04-23T05:51:45.137745Z","iopub.status.idle":"2024-04-23T05:51:45.160633Z","shell.execute_reply.started":"2024-04-23T05:51:45.137723Z","shell.execute_reply":"2024-04-23T05:51:45.159921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.groupby('Patient').size()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.161644Z","iopub.execute_input":"2024-04-23T05:51:45.16191Z","iopub.status.idle":"2024-04-23T05:51:45.170009Z","shell.execute_reply.started":"2024-04-23T05:51:45.161887Z","shell.execute_reply":"2024-04-23T05:51:45.169226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datapoints_per_patient = df_train.groupby('Patient').size()\n\n# Filter to get patients with less than 7 datapoints.\npatients_with_less_than_7 = datapoints_per_patient[datapoints_per_patient < 7]\n\n# Count the number of such patients.\nnumber_of_patients = patients_with_less_than_7.count()\n\nprint(f'Number of patients with less than 7 datapoints: {number_of_patients}')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.171038Z","iopub.execute_input":"2024-04-23T05:51:45.17133Z","iopub.status.idle":"2024-04-23T05:51:45.178104Z","shell.execute_reply.started":"2024-04-23T05:51:45.171299Z","shell.execute_reply":"2024-04-23T05:51:45.177342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.groupby('Patient').filter(lambda x: len(x) >= 7)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.179108Z","iopub.execute_input":"2024-04-23T05:51:45.179393Z","iopub.status.idle":"2024-04-23T05:51:45.195029Z","shell.execute_reply.started":"2024-04-23T05:51:45.179368Z","shell.execute_reply":"2024-04-23T05:51:45.194163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datapoints_per_patient = df_train.groupby('Patient').size()\n\n# Filter to get patients with less than 7 datapoints.\npatients_with_less_than_7 = datapoints_per_patient[datapoints_per_patient < 7]\n\n# Count the number of such patients.\nnumber_of_patients = patients_with_less_than_7.count()\n\nprint(f'Number of patients with less than 7 datapoints: {number_of_patients}')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.196081Z","iopub.execute_input":"2024-04-23T05:51:45.196356Z","iopub.status.idle":"2024-04-23T05:51:45.202957Z","shell.execute_reply.started":"2024-04-23T05:51:45.196332Z","shell.execute_reply":"2024-04-23T05:51:45.202214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Seven(group):\n    if len(group) > 7:\n        return group.iloc[-7:]  # Selects the last 7 rows\n    return group","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.204049Z","iopub.execute_input":"2024-04-23T05:51:45.204498Z","iopub.status.idle":"2024-04-23T05:51:45.212685Z","shell.execute_reply.started":"2024-04-23T05:51:45.204468Z","shell.execute_reply":"2024-04-23T05:51:45.211936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport itertools\nimport pandas as pd\n\nfrom itertools import combinations\n\ndef select_evenly_spaced_points(df_group, num_points=7):\n    weeks = df_group['Weeks'].tolist()\n    \n    if len(weeks) <= num_points:\n        return df_group\n\n    best_combination = None\n    best_std = float('inf')\n\n    for combination in combinations(weeks, num_points):\n        intervals = np.diff(sorted(combination))\n        std = np.std(intervals)\n\n        if std < best_std:\n            best_std = std\n            best_combination = combination\n\n    # Ensure that best_combination is not None and has unique points\n    if best_combination and len(set(best_combination)) == num_points:\n        return df_group[df_group['Weeks'].isin(best_combination)]\n    else:\n        # Fallback: select points with maximum intervals between them\n        sorted_weeks = sorted(weeks)\n        selected_indices = [0]  # Always include the first point\n        interval = len(sorted_weeks) // (num_points - 1)\n\n        for i in range(1, num_points - 1):\n            selected_indices.append(i * interval)\n        selected_indices.append(len(sorted_weeks) - 1)  # Always include the last point\n\n        return df_group[df_group['Weeks'].isin([sorted_weeks[i] for i in selected_indices])]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.21365Z","iopub.execute_input":"2024-04-23T05:51:45.21392Z","iopub.status.idle":"2024-04-23T05:51:45.223202Z","shell.execute_reply.started":"2024-04-23T05:51:45.213897Z","shell.execute_reply":"2024-04-23T05:51:45.222408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.groupby('Patient').apply(select_evenly_spaced_points).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.224375Z","iopub.execute_input":"2024-04-23T05:51:45.224689Z","iopub.status.idle":"2024-04-23T05:51:45.571904Z","shell.execute_reply.started":"2024-04-23T05:51:45.22466Z","shell.execute_reply":"2024-04-23T05:51:45.571152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_have_seven = all(df_train.groupby('Patient').size() == 7)\nprint(f'Do all patients have exactly 7 datapoints? {all_have_seven}')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.572891Z","iopub.execute_input":"2024-04-23T05:51:45.573135Z","iopub.status.idle":"2024-04-23T05:51:45.579667Z","shell.execute_reply.started":"2024-04-23T05:51:45.573113Z","shell.execute_reply":"2024-04-23T05:51:45.578917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_patient_group(df_group):\n    df_group = df_group.sort_values(by='Weeks')\n    patient_data = {\n        'Patient': df_group['Patient'].iloc[0],\n        'Weeks': ','.join(df_group['Weeks'].astype(str)),\n        'Percent': ','.join(df_group['Percent'].astype(str)),\n        'FVC': ','.join(df_group['FVC'].astype(str)),\n        'Age': df_group['Age'].iloc[0], \n        'Sex': df_group['Sex'].iloc[0],  \n        'SmokingStatus': df_group['SmokingStatus'].iloc[0]  # Assuming smoking status is constant for all rows of the same patient\n    }\n    return pd.Series(patient_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.580661Z","iopub.execute_input":"2024-04-23T05:51:45.580935Z","iopub.status.idle":"2024-04-23T05:51:45.590305Z","shell.execute_reply.started":"2024-04-23T05:51:45.580911Z","shell.execute_reply":"2024-04-23T05:51:45.589464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_df = df_train.groupby('Patient').apply(process_patient_group).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.591332Z","iopub.execute_input":"2024-04-23T05:51:45.591612Z","iopub.status.idle":"2024-04-23T05:51:45.7655Z","shell.execute_reply.started":"2024-04-23T05:51:45.591588Z","shell.execute_reply":"2024-04-23T05:51:45.764688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_df.head(7)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.766714Z","iopub.execute_input":"2024-04-23T05:51:45.767329Z","iopub.status.idle":"2024-04-23T05:51:45.778687Z","shell.execute_reply.started":"2024-04-23T05:51:45.767279Z","shell.execute_reply":"2024-04-23T05:51:45.777821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'processed_df' is your preprocessed DataFrame\npercent_values = processed_df['Percent'].apply(lambda x: np.fromstring(x, dtype=float, sep=','))\nmin_percent_values = percent_values.apply(min)\n\n# Check if any percent value is less than 0\nif (min_percent_values < 0).any():\n    print(\"There are percent values less than 0.\")\nelse:\n    print(\"All percent values are greater than or equal to 0.\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.779972Z","iopub.execute_input":"2024-04-23T05:51:45.780375Z","iopub.status.idle":"2024-04-23T05:51:45.790778Z","shell.execute_reply.started":"2024-04-23T05:51:45.780345Z","shell.execute_reply":"2024-04-23T05:51:45.789891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom tqdm import tqdm\nfrom PIL import Image\n\ndef preprocess_ct_images(patient_list, data_type=\"train\", image_size=(128, 128), show_first_image=False):\n    x = []\n    for patient in tqdm(patient_list):\n        image_dir = f\"{Path}/{data_type}/{patient}\"\n        week = df_train[df_train[\"Patient\"] == patient][\"Weeks\"]\n        image_count = 0  # Initialize a counter for the number of images\n        for i, w in enumerate(week):\n            try:\n                ds = pydicom.dcmread(os.path.join(image_dir, f\"{w}.dcm\"))\n                im = Image.fromarray(ds.pixel_array)\n                im = im.resize(image_size, resample=Image.NEAREST)\n                im_array = np.array(im).reshape((*image_size, 1))  # Grayscale image\n                im_array = im_array / 255.0  # Normalization\n                x.append(im_array)\n                image_count += 1  # Increment the counter\n                \n            except Exception as e:\n                print(f\"{e}\")\n        print(f\"Patient {patient} has {image_count} images.\")  # Print the total number of images found for each patient\n    return np.array(x)\n\n# Preprocess CT scan images\nx_train = preprocess_ct_images(Patient_list, \"train\", show_first_image=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:45.792146Z","iopub.execute_input":"2024-04-23T05:51:45.792487Z","iopub.status.idle":"2024-04-23T05:51:49.89137Z","shell.execute_reply.started":"2024-04-23T05:51:45.792456Z","shell.execute_reply":"2024-04-23T05:51:49.890337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers as L\nfrom tensorflow.keras import models as M\n\ndef extract_image_features():\n    ct = L.Input((128, 128, 1), name=\"Ctinput\")\n    x = L.Conv2D(64, (6, 6), activation=\"relu\", name=\"conv1\")(ct)\n    x = L.MaxPooling2D(pool_size=(3, 3), name='pool1')(x)\n    x = L.Conv2D(64, (6, 6), activation=\"relu\", name=\"conv2\")(x)\n    x = L.MaxPooling2D(pool_size=(3, 3), name='pool2')(x)\n    x = L.Conv2D(128, (6, 6), activation=\"relu\", name=\"conv3\")(x)\n    x = L.MaxPooling2D(pool_size=(2, 2), name='pool3')(x)\n    x = L.Flatten(name=\"features\")(x)\n    model = M.Model(ct, x, name=\"ImageFeatureExtractor\")\n    return model\n\n# Instantiate the image feature extractor\nimage_feature_extractor = extract_image_features()\nprint(image_feature_extractor.summary())\n\n# Extract image features from the training data\nimage_features = image_feature_extractor.predict(x_train, batch_size=50, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:51:49.892732Z","iopub.execute_input":"2024-04-23T05:51:49.89304Z","iopub.status.idle":"2024-04-23T05:52:04.939677Z","shell.execute_reply.started":"2024-04-23T05:51:49.893013Z","shell.execute_reply":"2024-04-23T05:52:04.938754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_feature_extractor.summary())\nprint(image_features.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:04.94097Z","iopub.execute_input":"2024-04-23T05:52:04.941233Z","iopub.status.idle":"2024-04-23T05:52:04.965123Z","shell.execute_reply.started":"2024-04-23T05:52:04.94121Z","shell.execute_reply":"2024-04-23T05:52:04.964339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'processed_df' is your preprocessed DataFrame\nprocessed_df = df_train.groupby('Patient').apply(process_patient_group).reset_index(drop=True)\n\n# Duplicate the dataframe and add a suffix to the patient IDs\nduplicated_df = processed_df.copy()\nduplicated_df['Patient'] = duplicated_df['Patient'] + '_dup'\n\n# Concatenate the original and duplicated dataframes\naugmented_df = pd.concat([processed_df, duplicated_df], ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:04.966173Z","iopub.execute_input":"2024-04-23T05:52:04.966462Z","iopub.status.idle":"2024-04-23T05:52:05.151244Z","shell.execute_reply.started":"2024-04-23T05:52:04.966438Z","shell.execute_reply":"2024-04-23T05:52:05.150266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import Dataset\nimport numpy as np\nimport torch\nfrom sklearn.preprocessing import MinMaxScaler\n\nclass FVCDataset(Dataset):\n    def __init__(self, dataframe):\n        fvc_values = [np.fromstring(row['FVC'], dtype=float, sep=',') for _, row in dataframe.iterrows()]\n        self.scaler = MinMaxScaler()\n        # Reshape to fit the scaler properly\n        self.fvc_values = self.scaler.fit_transform(np.array(fvc_values).reshape(-1, 1)).reshape(-1, 7)\n\n    def __len__(self):\n        return len(self.fvc_values)\n\n    def __getitem__(self, idx):\n        fvc = torch.tensor(self.fvc_values[idx][:5], dtype=torch.float32)  # First 5 FVC values as input\n        target = torch.tensor(self.fvc_values[idx][-2:], dtype=torch.float32)  # Last 2 FVC values as target\n        return fvc, target\n\n    def inverse_transform(self, scaled_fvc):\n        return self.scaler.inverse_transform(scaled_fvc)\n\n# Now you can proceed to create the dataset and split it into training and validation sets\ndataset = FVCDataset(augmented_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:05.152406Z","iopub.execute_input":"2024-04-23T05:52:05.152686Z","iopub.status.idle":"2024-04-23T05:52:05.185166Z","shell.execute_reply.started":"2024-04-23T05:52:05.152662Z","shell.execute_reply":"2024-04-23T05:52:05.184425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(processed_df['Age'].unique())\nprint(processed_df['SmokingStatus'].unique())\nprint(processed_df['Sex'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:05.186203Z","iopub.execute_input":"2024-04-23T05:52:05.186517Z","iopub.status.idle":"2024-04-23T05:52:05.192537Z","shell.execute_reply.started":"2024-04-23T05:52:05.186491Z","shell.execute_reply":"2024-04-23T05:52:05.191713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate indices: an array of integers from 0 to the length of the dataset - 1\ndataset_indices = [i for i in range(len(dataset))]\n\n# Split indices into training and validation sets\ntrain_indices, val_indices, _, _ = train_test_split(dataset_indices, dataset_indices, test_size=0.2, random_state=42)\n\n# Create subsets for training and validation using Subset\nfrom torch.utils.data import Subset\n\ntrain_dataset = Subset(dataset, train_indices)\nval_dataset = Subset(dataset, val_indices)\n\nfrom torch.utils.data import DataLoader\n\ntrain_loader = DataLoader(train_dataset, batch_size=10, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=10, shuffle=False)\n\n\n\nclass PositionalEncoding(nn.Module):\n    def __init__(self, d_model, dropout=0.1, max_len=5000):\n        super(PositionalEncoding, self).__init__()\n        self.dropout = nn.Dropout(p=dropout)\n\n        position = torch.arange(max_len).unsqueeze(1)\n        div_term = torch.exp(torch.arange(0, d_model, 2) * -(math.log(10000.0) / d_model))\n        pe = torch.zeros(max_len, d_model)\n        pe[:, 0::2] = torch.sin(position * div_term)\n        pe[:, 1::2] = torch.cos(position * div_term)\n        pe = pe.unsqueeze(0)\n        self.register_buffer('pe', pe)\n\n    def forward(self, x):\n        x = x + self.pe[:, :x.size(1), :]\n        return self.dropout(x)\n\n    \n    \n    \nimport torch\nimport torch.nn as nn\n\nclass FVCTransformer(nn.Module):\n    def __init__(self, fvc_input_dim, output_dim):\n        super(FVCTransformer, self).__init__()\n        self.fvc_linear = nn.Linear(fvc_input_dim, 64)\n        self.output_linear = nn.Linear(64, output_dim)\n\n    def forward(self, fvc):\n        fvc_output = self.fvc_linear(fvc)\n        output = self.output_linear(fvc_output)\n        return output\n\nfvc_input_dim = 5  # Only 5 FVC values as input\noutput_dim = 2  # Predicting the last 2 FVC values\nmodel = FVCTransformer(fvc_input_dim, output_dim)\n\n\nfvc_input_dim = 5  # Only 5 FVC values as input\noutput_dim = 2  # Predicting the last 2 FVC values\nmodel = FVCTransformer(fvc_input_dim, output_dim)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:05.193608Z","iopub.execute_input":"2024-04-23T05:52:05.193873Z","iopub.status.idle":"2024-04-23T05:52:05.22992Z","shell.execute_reply.started":"2024-04-23T05:52:05.19385Z","shell.execute_reply":"2024-04-23T05:52:05.229201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=20):\n    train_losses = []\n    val_losses = []\n    train_accuracies = []\n    val_accuracies = []\n\n    for epoch in range(num_epochs):\n        model.train()\n        total_train_loss = 0\n        correct_train_predictions = 0\n\n        for inputs, targets in train_loader:\n            optimizer.zero_grad()\n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            loss.backward()\n            optimizer.step()\n            total_train_loss += loss.item()\n\n            # Calculate accuracy for training\n            train_accuracy = torch.mean(((torch.abs(targets - outputs) / targets) < 0.15).float()).item()\n            correct_train_predictions += train_accuracy * len(targets)\n\n        avg_train_loss = total_train_loss / len(train_loader)\n        train_losses.append(avg_train_loss)\n        train_accuracies.append(correct_train_predictions / len(train_loader.dataset))\n\n        model.eval()\n        total_val_loss = 0\n        correct_val_predictions = 0\n\n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                total_val_loss += loss.item()\n\n                # Calculate accuracy for validation\n                val_accuracy = torch.mean(((torch.abs(targets - outputs) / targets) < 0.15).float()).item()\n                correct_val_predictions += val_accuracy * len(targets)\n\n        avg_val_loss = total_val_loss / len(val_loader)\n        val_losses.append(avg_val_loss)\n        val_accuracies.append(correct_val_predictions / len(val_loader.dataset))\n\n        print(f'Epoch {epoch+1}, Training Loss: {avg_train_loss}, Validation Loss: {avg_val_loss}, '\n              f'Training Accuracy: {train_accuracies[-1]:.2f}, Validation Accuracy: {val_accuracies[-1]:.2f}')\n\n    # Plot the training and validation losses and accuracies\n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.plot(range(1, num_epochs + 1), train_losses, label='Training Loss')\n    plt.plot(range(1, num_epochs + 1), val_losses, label='Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.title('Epoch vs. Loss')\n    plt.legend()\n\n    plt.subplot(1, 2, 2)\n    plt.plot(range(1, num_epochs + 1), train_accuracies, label='Training Accuracy')\n    plt.plot(range(1, num_epochs + 1), val_accuracies, label='Validation Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.title('Epoch vs. Accuracy')\n    plt.legend()\n\n    plt.tight_layout()\n    plt.show()\n\n    return train_losses, val_losses, train_accuracies, val_accuracies\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:05.231363Z","iopub.execute_input":"2024-04-23T05:52:05.231693Z","iopub.status.idle":"2024-04-23T05:52:05.245903Z","shell.execute_reply.started":"2024-04-23T05:52:05.231663Z","shell.execute_reply":"2024-04-23T05:52:05.244968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.optim as optim\nimport matplotlib.pyplot as plt\n\ncriterion = nn.MSELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)  # Start with a learning rate of 0.001\n\n# You can try different learning rates to see which one works best for your model\n# For example:\n# optimizer = optim.Adam(model.parameters(), lr=0.0001)  # Decrease the learning rate if needed\n\n# Train the model and plot the loss and accuracy\ntrain_losses, val_losses, train_accuracies, val_accuracies = train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=200)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:05.246999Z","iopub.execute_input":"2024-04-23T05:52:05.247235Z","iopub.status.idle":"2024-04-23T05:52:15.038494Z","shell.execute_reply.started":"2024-04-23T05:52:05.247214Z","shell.execute_reply":"2024-04-23T05:52:15.037561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import pearsonr\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nimport numpy as np\n\ndef evaluate_model(model, dataloader, scaler):\n    model.eval()  # Put the model in evaluation mode\n    \n    predictions, actuals = [], []\n    \n    with torch.no_grad():\n        for inputs, targets in dataloader:\n            outputs = model(inputs)\n            outputs = outputs.squeeze(-1)  # Adjust dimensions if necessary\n            \n            # Select only the last output for comparison\n            predicted_fvc = outputs.detach().cpu().numpy()\n            actual_fvc = targets.detach().cpu().numpy()\n            \n            # Unscale the predictions and actual values\n            predicted_fvc = scaler.inverse_transform(predicted_fvc.reshape(-1, 1)).flatten()\n            actual_fvc = scaler.inverse_transform(actual_fvc.reshape(-1, 1)).flatten()\n            \n            predictions.extend(predicted_fvc)\n            actuals.extend(actual_fvc)\n    \n    predictions = np.array(predictions)\n    actuals = np.array(actuals)\n    \n    # Calculate metrics\n    mae = mean_absolute_error(actuals, predictions)\n    rmse = mean_squared_error(actuals, predictions, squared=False)  # Pass squared=False for RMSE\n    r2 = r2_score(actuals, predictions)\n    pearson_corr = pearsonr(predictions, actuals)[0]\n    \n    return mae, rmse, r2, pearson_corr\n\n\n\n\n\nscaler = dataset.scaler  # Assuming 'dataset' is an instance of FVCDataset\nmae, rmse, r2, pearson_corr = evaluate_model(model, val_loader, scaler)\nprint(f\"Validation MAE: {mae}, Validation RMSE: {rmse}, Validation R^2: {r2}, Validation Pearson Correlation: {pearson_corr}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:15.039605Z","iopub.execute_input":"2024-04-23T05:52:15.039938Z","iopub.status.idle":"2024-04-23T05:52:15.059637Z","shell.execute_reply.started":"2024-04-23T05:52:15.039912Z","shell.execute_reply":"2024-04-23T05:52:15.058702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lgb baseline","metadata":{}},{"cell_type":"code","source":"import os\nfrom logging import getLogger, INFO, StreamHandler, FileHandler, Formatter\nfrom functools import partial\n\nimport random\nimport math\n\nfrom tqdm.notebook import tqdm\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\nfrom sklearn.metrics import mean_squared_error\nimport category_encoders as ce\n\nfrom PIL import Image\nimport cv2\nimport pydicom\n\nimport torch\n\nimport lightgbm as lgb\nfrom sklearn.linear_model import Ridge\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:15.061042Z","iopub.execute_input":"2024-04-23T05:52:15.061424Z","iopub.status.idle":"2024-04-23T05:52:18.827593Z","shell.execute_reply.started":"2024-04-23T05:52:15.06139Z","shell.execute_reply":"2024-04-23T05:52:18.826629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# utils","metadata":{}},{"cell_type":"code","source":"def get_logger(filename='log'):\n    logger = getLogger(__name__)\n    logger.setLevel(INFO)\n    handler1 = StreamHandler()\n    handler1.setFormatter(Formatter(\"%(message)s\"))\n    handler2 = FileHandler(filename=f\"{filename}.log\")\n    handler2.setFormatter(Formatter(\"%(message)s\"))\n    logger.addHandler(handler1)\n    logger.addHandler(handler2)\n    return logger\n\nlogger = get_logger()\n\n\ndef seed_everything(seed=777):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.828862Z","iopub.execute_input":"2024-04-23T05:52:18.829149Z","iopub.status.idle":"2024-04-23T05:52:18.837038Z","shell.execute_reply.started":"2024-04-23T05:52:18.829125Z","shell.execute_reply":"2024-04-23T05:52:18.836159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# config","metadata":{}},{"cell_type":"code","source":"OUTPUT_DICT = './'\n\nID = 'Patient_Week'\nTARGET = 'FVC'\nSEED = 42\nseed_everything(seed=SEED)\n\nN_FOLD = 4","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.837966Z","iopub.execute_input":"2024-04-23T05:52:18.838221Z","iopub.status.idle":"2024-04-23T05:52:18.856356Z","shell.execute_reply.started":"2024-04-23T05:52:18.838198Z","shell.execute_reply":"2024-04-23T05:52:18.855486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntrain[ID] = train['Patient'].astype(str) + '_' + train['Weeks'].astype(str)\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.857552Z","iopub.execute_input":"2024-04-23T05:52:18.858381Z","iopub.status.idle":"2024-04-23T05:52:18.883173Z","shell.execute_reply.started":"2024-04-23T05:52:18.858349Z","shell.execute_reply":"2024-04-23T05:52:18.882331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\\\n        .rename(columns={'Weeks': 'base_Week', 'FVC': 'base_FVC', 'Percent': 'base_Percent', 'Age': 'base_Age'})\nsubmission = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nsubmission['Patient'] = submission['Patient_Week'].apply(lambda x: x.split('_')[0])\nsubmission['predict_Week'] = submission['Patient_Week'].apply(lambda x: x.split('_')[1]).astype(int)\ntest = submission.drop(columns=['FVC', 'Confidence']).merge(test, on='Patient')\ntest['Week_passed'] = test['predict_Week'] - test['base_Week']\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.884388Z","iopub.execute_input":"2024-04-23T05:52:18.884731Z","iopub.status.idle":"2024-04-23T05:52:18.91218Z","shell.execute_reply.started":"2024-04-23T05:52:18.884684Z","shell.execute_reply":"2024-04-23T05:52:18.911481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nprint(submission.shape)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.913344Z","iopub.execute_input":"2024-04-23T05:52:18.913615Z","iopub.status.idle":"2024-04-23T05:52:18.926131Z","shell.execute_reply.started":"2024-04-23T05:52:18.913592Z","shell.execute_reply":"2024-04-23T05:52:18.925272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# prepare folds","metadata":{}},{"cell_type":"code","source":"folds = train[[ID, 'Patient', TARGET]].copy()\n#Fold = KFold(n_splits=N_FOLD, shuffle=True, random_state=SEED)\nFold = GroupKFold(n_splits=N_FOLD)\ngroups = folds['Patient'].values\nfor n, (train_index, val_index) in enumerate(Fold.split(folds, folds[TARGET], groups)):\n    folds.loc[val_index, 'fold'] = int(n)\nfolds['fold'] = folds['fold'].astype(int)\nfolds.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.927364Z","iopub.execute_input":"2024-04-23T05:52:18.927709Z","iopub.status.idle":"2024-04-23T05:52:18.948013Z","shell.execute_reply.started":"2024-04-23T05:52:18.927678Z","shell.execute_reply":"2024-04-23T05:52:18.94715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model","metadata":{}},{"cell_type":"code","source":"#===========================================================\n# model\n#===========================================================\ndef run_single_lightgbm(param, train_df, test_df, folds, features, target, fold_num=0, categorical=[]):\n    \n    trn_idx = folds[folds.fold != fold_num].index\n    val_idx = folds[folds.fold == fold_num].index\n    logger.info(f'len(trn_idx) : {len(trn_idx)}')\n    logger.info(f'len(val_idx) : {len(val_idx)}')\n    \n    if categorical == []:\n        trn_data = lgb.Dataset(train_df.iloc[trn_idx][features],\n                               label=target.iloc[trn_idx])\n        val_data = lgb.Dataset(train_df.iloc[val_idx][features],\n                               label=target.iloc[val_idx])\n    else:\n        trn_data = lgb.Dataset(train_df.iloc[trn_idx][features],\n                               label=target.iloc[trn_idx],\n                               categorical_feature=categorical)\n        val_data = lgb.Dataset(train_df.iloc[val_idx][features],\n                               label=target.iloc[val_idx],\n                               categorical_feature=categorical)\n\n    oof = np.zeros(len(train_df))\n    predictions = np.zeros(len(test_df))\n\n    num_round = 10000\n\n    clf = lgb.train(param,\n                    trn_data,\n                    num_round,\n                    valid_sets=[trn_data, val_data],\n                    verbose_eval=100,\n                    early_stopping_rounds=100)\n\n    oof[val_idx] = clf.predict(train_df.iloc[val_idx][features], num_iteration=clf.best_iteration)\n\n    fold_importance_df = pd.DataFrame()\n    fold_importance_df[\"Feature\"] = features\n    fold_importance_df[\"importance\"] = clf.feature_importance(importance_type='gain')\n    fold_importance_df[\"fold\"] = fold_num\n\n    predictions += clf.predict(test_df[features], num_iteration=clf.best_iteration)\n    \n    # RMSE\n    logger.info(\"fold{} RMSE score: {:<8.5f}\".format(fold_num, np.sqrt(mean_squared_error(target[val_idx], oof[val_idx]))))\n    \n    return oof, predictions, fold_importance_df\n\n\ndef run_kfold_lightgbm(param, train, test, folds, features, target, n_fold=5, categorical=[]):\n    \n    logger.info(f\"================================= {n_fold}fold lightgbm =================================\")\n    \n    oof = np.zeros(len(train))\n    predictions = np.zeros(len(test))\n    feature_importance_df = pd.DataFrame()\n\n    for fold_ in range(n_fold):\n        print(\"Fold {}\".format(fold_))\n        _oof, _predictions, fold_importance_df = run_single_lightgbm(param,\n                                                                     train,\n                                                                     test,\n                                                                     folds,\n                                                                     features,\n                                                                     target,\n                                                                     fold_num=fold_,\n                                                                     categorical=categorical)\n        feature_importance_df = pd.concat([feature_importance_df, fold_importance_df], axis=0)\n        oof += _oof\n        predictions += _predictions / n_fold\n\n    # RMSE\n    logger.info(\"CV RMSE score: {:<8.5f}\".format(np.sqrt(mean_squared_error(target, oof))))\n\n    logger.info(f\"=========================================================================================\")\n    \n    return feature_importance_df, predictions, oof\n\n    \ndef show_feature_importance(feature_importance_df, name):\n    cols = (feature_importance_df[[\"Feature\", \"importance\"]]\n            .groupby(\"Feature\")\n            .mean()\n            .sort_values(by=\"importance\", ascending=False)[:50].index)\n    best_features = feature_importance_df.loc[feature_importance_df.Feature.isin(cols)]\n\n    #plt.figure(figsize=(8, 16))\n    plt.figure(figsize=(6, 4))\n    sns.barplot(x=\"importance\", y=\"Feature\", data=best_features.sort_values(by=\"importance\", ascending=False))\n    plt.title('Features importance (averaged/folds)')\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DICT+f'feature_importance_{name}.png')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.949137Z","iopub.execute_input":"2024-04-23T05:52:18.94942Z","iopub.status.idle":"2024-04-23T05:52:18.969044Z","shell.execute_reply.started":"2024-04-23T05:52:18.949396Z","shell.execute_reply":"2024-04-23T05:52:18.96824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# predict fvc","metadata":{}},{"cell_type":"code","source":"'''''\ntarget = train[TARGET]\ntest[TARGET] = np.nan\n\n# features\ncat_features = ['Sex', 'SmokingStatus']\nnum_features = [c for c in test.columns if (test.dtypes[c] != 'object') & (c not in cat_features)]\nfeatures = num_features + cat_features\ndrop_features = [ID, TARGET, 'predict_Week', 'base_Week']\nfeatures = [c for c in features if c not in drop_features]\n\nif cat_features:\n    ce_oe = ce.OrdinalEncoder(cols=cat_features, handle_unknown='impute')\n    ce_oe.fit(train)\n    train = ce_oe.transform(train)\n    test = ce_oe.transform(test)\n        \nlgb_param = {'objective': 'regression',\n             'metric': 'rmse',\n             'boosting_type': 'gbdt',\n             'learning_rate': 0.01,\n             'seed': SEED,\n             'max_depth': -1,\n             'verbosity': -1,\n            }\n\nfeature_importance_df, predictions, oof = run_kfold_lightgbm(lgb_param, train, test, folds, features, target, \n                                                             n_fold=N_FOLD, categorical=cat_features)\n    \nshow_feature_importance(feature_importance_df, TARGET)\n'''''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.970195Z","iopub.execute_input":"2024-04-23T05:52:18.970472Z","iopub.status.idle":"2024-04-23T05:52:18.98518Z","shell.execute_reply.started":"2024-04-23T05:52:18.970449Z","shell.execute_reply":"2024-04-23T05:52:18.984275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train['FVC_pred'] = \"\"\n#test['FVC_pred'] = predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:18.9959Z","iopub.execute_input":"2024-04-23T05:52:18.996298Z","iopub.status.idle":"2024-04-23T05:52:19.00081Z","shell.execute_reply.started":"2024-04-23T05:52:18.99625Z","shell.execute_reply":"2024-04-23T05:52:18.999917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''''\n# baseline score\ntrain['Confidence'] = 100\ntrain['sigma_clipped'] = train['Confidence'].apply(lambda x: max(x, 70))\ntrain['diff'] = abs(train['FVC'] - train['FVC_pred'])\ntrain['delta'] = train['diff'].apply(lambda x: min(x, 1000))\ntrain['score'] = -math.sqrt(2)*train['delta']/train['sigma_clipped'] - np.log(math.sqrt(2)*train['sigma_clipped'])\nscore = train['score'].mean()\nprint(score)\n'''''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.001917Z","iopub.execute_input":"2024-04-23T05:52:19.002246Z","iopub.status.idle":"2024-04-23T05:52:19.011638Z","shell.execute_reply.started":"2024-04-23T05:52:19.002215Z","shell.execute_reply":"2024-04-23T05:52:19.010712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''''\nimport scipy as sp\n\ndef loss_func(weight, row):\n    confidence = weight\n    sigma_clipped = max(confidence, 70)\n    diff = abs(row['FVC'] - row['FVC_pred'])\n    delta = min(diff, 1000)\n    score = -math.sqrt(2)*delta/sigma_clipped - np.log(math.sqrt(2)*sigma_clipped)\n    return -score\n\nresults = []\ntk0 = tqdm(train.iterrows(), total=len(train))\nfor _, row in tk0:\n    loss_partial = partial(loss_func, row=row)\n    weight = [100]\n    #bounds = [(70, 100)]\n    #result = sp.optimize.minimize(loss_partial, weight, method='SLSQP', bounds=bounds)\n    result = sp.optimize.minimize(loss_partial, weight, method='SLSQP')\n    x = result['x']\n    results.append(x[0])\n'''''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.012926Z","iopub.execute_input":"2024-04-23T05:52:19.013184Z","iopub.status.idle":"2024-04-23T05:52:19.022437Z","shell.execute_reply.started":"2024-04-23T05:52:19.013162Z","shell.execute_reply":"2024-04-23T05:52:19.021506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  algorithm\n\n","metadata":{}},{"cell_type":"code","source":"'''''\n# Importing necessary libraries\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\n\n\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n# Assuming you have your features in X and the target variable (FVC) in y\n# X should be a 2D array where each row represents a sample and each column represents a feature\n# y should be a 1D array containing the target variable (FVC) values\n\n# Splitting the dataset into training and testing sets\n\nX_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.2, random_state=42)\n\n# Converting the dataset into LightGBM Dataset format\ntrain_data = lgb.Data(X_train, label=y_train)\ntest_data = lgb.Data(X_test, label=y_test)\n\n# Parameters for LightGBM model\nparams = {\n    'objective': 'regression',  # regression task\n    'metric': 'rmse',           # evaluation metric\n    'boosting': 'gbdt',         # gradient boosting decision tree\n    'num_leaves': 31,           # maximum number of leaves in one tree\n    'learning_rate': 0.05,      # learning rate\n    'feature_fraction': 0.9,    # subsample ratio of features\n    'bagging_fraction': 0.8,    # subsample ratio of data\n    'bagging_freq': 5,          # frequency for bagging\n    'verbose': 0                # no output when fitting the model\n}\n\n# Training the LightGBM model\nnum_round = 1000  # Number of boosting iterations\nlgb_model = lgb.train(params, train_data, num_round, valid_sets=[test_data], early_stopping_rounds=50)\n\n# Making predictions on the testing set\ny_pred = lgb_model.predict(X_test, num_iteration=lgb_model.best_iteration)\n\n# Evaluating the model\nrmse = mean_squared_error(y_test, y_pred, squared=False)\nprint(\"Root Mean Squared Error (RMSE):,\",rmse)\n'''''","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.023662Z","iopub.execute_input":"2024-04-23T05:52:19.024305Z","iopub.status.idle":"2024-04-23T05:52:19.038481Z","shell.execute_reply.started":"2024-04-23T05:52:19.024256Z","shell.execute_reply":"2024-04-23T05:52:19.037631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LGBM ALGORITHM","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\nfrom sklearn.preprocessing import LabelEncoder\n\n# Load your data (replace with your path)\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Convert categorical features into numerical format\nlabel_encoders = {}\nfor feature in [\"Patient\", \"Sex\", \"SmokingStatus\"]:\n    label_encoders[feature] = LabelEncoder()\n    data[feature] = label_encoders[feature].fit_transform(data[feature])\n\n# Separate features and target variable\nfeatures = data.drop(\"FVC\", axis=1)  # Assuming \"FVCrate\" is the target\ntarget = data[\"FVC\"]\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(features, target, test_size=0.2, random_state=42)\n\n# Define LightGBM parameters (adjust as needed)\nlgb_params = {\n    \"objective\": \"regression\",\n    \"metric\": [\"l2\", \"l1\", \"l2_root\"],  # Mean squared error, mean absolute error, root mean squared error\n    \"learning_rate\": 0.1,\n    \"n_estimators\": 100,  # Number of trees to grow\n    \"num_leaves\": 31,  # Maximum number of leaves per tree\n}\n\n# Create and train the LightGBM model\nmodel = LGBMRegressor(**lgb_params)\nmodel.fit(X_train, y_train)\n\n# Make predictions on the testing set\ny_pred = model.predict(X_test)\n\n# Evaluate model performance\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error (MSE) on testing set: {mse:.4f}\")\nprint(f\"Mean Absolute Error (MAE) on testing set: {mae:.4f}\")\nprint(f\"R-squared (R2) on testing set: {r2:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.178449Z","iopub.execute_input":"2024-04-23T05:52:19.178796Z","iopub.status.idle":"2024-04-23T05:52:19.268736Z","shell.execute_reply.started":"2024-04-23T05:52:19.17877Z","shell.execute_reply":"2024-04-23T05:52:19.267702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SVM","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.svm import SVR\n\n# Load your data (replace with your path)\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Convert categorical features into numerical format\nlabel_encoders = {}\nfor feature in [\"Patient\", \"Sex\", \"SmokingStatus\"]:\n    label_encoders[feature] = LabelEncoder()\n    data[feature] = label_encoders[feature].fit_transform(data[feature])\n\n# Separate features and target variable\nfeatures = data.drop(\"FVC\", axis=1)  # Assuming \"FVCrate\" is the target\ntarget = data[\"FVC\"]\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(features, target, test_size=0.2, random_state=42)\n\n# Create and train the SVM model\nsvm_model = SVR(kernel='linear')\nsvm_model.fit(X_train, y_train)\n\n# Make predictions on the testing set\ny_pred = svm_model.predict(X_test)\n\n# Evaluate model performance\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error (MSE) on testing set: {mse:.4f}\")\nprint(f\"Mean Absolute Error (MAE) on testing set: {mae:.4f}\")\nprint(f\"R-squared (R2) on testing set: {r2:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.26994Z","iopub.execute_input":"2024-04-23T05:52:19.270221Z","iopub.status.idle":"2024-04-23T05:52:19.414485Z","shell.execute_reply.started":"2024-04-23T05:52:19.270196Z","shell.execute_reply":"2024-04-23T05:52:19.413545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random forest","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import LabelEncoder\n\n# Load your data (replace with your path)\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Convert FVC to discrete classes for classification\n# You may need to adjust the binning strategy according to your data distribution\ndata['FVC_class'] = pd.qcut(data['FVC'], q=5, labels=False)\n\n# Convert categorical features into numerical format\nlabel_encoders = {}\nfor feature in [\"Patient\", \"Sex\", \"SmokingStatus\"]:\n    label_encoders[feature] = LabelEncoder()\n    data[feature] = label_encoders[feature].fit_transform(data[feature])\n\n# Separate features and target variable\nfeatures = data.drop([\"FVC\", \"FVC_class\"], axis=1)  # Assuming \"FVCrate\" is the target\ntarget = data[\"FVC_class\"]\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(features, target, test_size=0.2, random_state=42)\n\n# Create and train the Random Forest Classifier model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42)\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the testing set\ny_pred = rf_model.predict(X_test)\n\n# Evaluate model performance\naccuracy = accuracy_score(y_test, y_pred)\n\nprint(f\"Accuracy on testing set: {accuracy:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.415672Z","iopub.execute_input":"2024-04-23T05:52:19.41596Z","iopub.status.idle":"2024-04-23T05:52:19.793849Z","shell.execute_reply.started":"2024-04-23T05:52:19.415935Z","shell.execute_reply":"2024-04-23T05:52:19.793062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# resnet","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n\n# Load your data (replace with your path)\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Convert categorical features into numerical format\nlabel_encoders = {}\nfor feature in [\"Patient\", \"Sex\", \"SmokingStatus\"]:\n    label_encoders[feature] = LabelEncoder()\n    data[feature] = label_encoders[feature].fit_transform(data[feature])\n\n# Separate features and target variable\nX = data.drop([\"FVC\"], axis=1)  # Assuming \"FVCrate\" is the target\ny = data[\"FVC\"]\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Standardize features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Define the ResNet-like neural network architecture\ndef residual_block(inputs, num_neurons):\n    x = tf.keras.layers.Dense(num_neurons, activation='relu')(inputs)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Dense(num_neurons, activation='relu')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.add([inputs, x])\n    x = tf.keras.layers.Activation('relu')(x)\n    return x\n\ninputs = tf.keras.layers.Input(shape=(X_train_scaled.shape[1],))\nx = tf.keras.layers.Dense(64, activation='relu')(inputs)\nx = tf.keras.layers.BatchNormalization()(x)\n\n# Add residual blocks\nfor _ in range(3):  # Add 3 residual blocks\n    x = residual_block(x, 64)\n\nx = tf.keras.layers.Dense(1)(x)  # Output layer\n\nmodel = tf.keras.Model(inputs=inputs, outputs=x)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model\nhistory = model.fit(X_train_scaled, y_train, epochs=50, batch_size=32, validation_split=0.2, verbose=1)\n\n# Evaluate the model\ny_pred = model.predict(X_test_scaled)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error (MSE) on testing set: {mse:.4f}\")\nprint(f\"Mean Absolute Error (MAE) on testing set: {mae:.4f}\")\nprint(f\"R-squared (R2) on testing set: {r2:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:19.795261Z","iopub.execute_input":"2024-04-23T05:52:19.79557Z","iopub.status.idle":"2024-04-23T05:52:42.809746Z","shell.execute_reply.started":"2024-04-23T05:52:19.795544Z","shell.execute_reply":"2024-04-23T05:52:42.808765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# dense net","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n\n# Load your data (replace with your path)\ndata = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Convert categorical features into numerical format\nlabel_encoders = {}\nfor feature in [\"Patient\", \"Sex\", \"SmokingStatus\"]:\n    label_encoders[feature] = LabelEncoder()\n    data[feature] = label_encoders[feature].fit_transform(data[feature])\n\n# Separate features and target variable\nX = data.drop([\"FVC\"], axis=1)  # Assuming \"FVCrate\" is the target\ny = data[\"FVC\"]\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Standardize features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Reshape input data for Conv1D layer\nX_train_scaled_reshaped = np.expand_dims(X_train_scaled, axis=1)\nX_test_scaled_reshaped = np.expand_dims(X_test_scaled, axis=1)\n\n# Define the DenseNet-like neural network architecture\ndef dense_block(x, num_layers, growth_rate):\n    for _ in range(num_layers):\n        y = tf.keras.layers.BatchNormalization()(x)\n        y = tf.keras.layers.Activation('relu')(y)\n        y = tf.keras.layers.Conv1D(filters=growth_rate, kernel_size=3, padding='same')(y)\n        x = tf.keras.layers.Concatenate()([x, y])\n    return x\n\ninputs = tf.keras.layers.Input(shape=(X_train_scaled_reshaped.shape[1], X_train_scaled_reshaped.shape[2]))\n\n# Initial convolutional layer\nx = tf.keras.layers.Conv1D(filters=64, kernel_size=3, padding='same')(inputs)\n\n# Dense blocks\nx = dense_block(x, 3, 32)  # 3 dense blocks with 32 growth rate\n\n# Global average pooling\nx = tf.keras.layers.GlobalAveragePooling1D()(x)\n\n# Output layer\nx = tf.keras.layers.Dense(1)(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=x)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Train the model\nhistory = model.fit(X_train_scaled_reshaped, y_train, epochs=50, batch_size=32, validation_split=0.2, verbose=1)\n\n# Evaluate the model\ny_pred = model.predict(X_test_scaled_reshaped)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error (MSE) on testing set: {mse:.4f}\")\nprint(f\"Mean Absolute Error (MAE) on testing set: {mae:.4f}\")\nprint(f\"R-squared (R2) on testing set: {r2:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:52:42.811069Z","iopub.execute_input":"2024-04-23T05:52:42.811459Z","iopub.status.idle":"2024-04-23T05:52:58.239696Z","shell.execute_reply.started":"2024-04-23T05:52:42.811429Z","shell.execute_reply":"2024-04-23T05:52:58.23889Z"},"trusted":true},"execution_count":null,"outputs":[]}]}