{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"gpuType":"T4"},"accelerator":"GPU","kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20604,"databundleVersionId":1357052,"sourceType":"competition"},{"sourceId":648816,"sourceType":"modelInstanceVersion","modelInstanceId":489130,"modelId":504546}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"7EqCA2RigPil","cell_type":"code","source":"# !pip install python-gdcm pylibjpeg==2.0 pylibjpeg-libjpeg==2.1 pydicom==2.3.0","metadata":{"id":"7EqCA2RigPil","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:24.320181Z","iopub.execute_input":"2025-11-17T13:55:24.320467Z","iopub.status.idle":"2025-11-17T13:55:24.324717Z","shell.execute_reply.started":"2025-11-17T13:55:24.320441Z","shell.execute_reply":"2025-11-17T13:55:24.324007Z"}},"outputs":[],"execution_count":null},{"id":"vblA","cell_type":"code","source":"import os, glob\nfrom tqdm import tqdm\n\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nimport keras\nfrom keras import layers\nfrom keras.optimizers import Adam\nfrom keras.losses import MeanSquaredError\n\nimport pydicom\n\nIMG_WIDTH, IMG_HEIGHT, IMG_DEPTH = 128, 128, 64\nMIN_WEEK, MAX_WEEK = -12, 133\nSIGMA_MIN, DELTA_MAX = 70, 1000\nSQRT_2 = tf.constant(tf.sqrt(2.))\n\nimport kagglehub\n# kagglehub.login()\n# kagglehub.competition_download('osic-pulmonary-fibrosis-progression')\nDIR = '../input/osic-pulmonary-fibrosis-progression'\n# DIR = 'data/osic-pulmonary-fibrosis-progression'\n# SUBMISSION_DIR = 'data/osic-pulmonary-fibrosis-progression'\n# DIR = '/root/.cache/kagglehub/competitions/osic-pulmonary-fibrosis-progression'\nSUBMISSION_DIR = '.'\n# MODEL = ''","metadata":{"id":"vblA","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:24.328780Z","iopub.execute_input":"2025-11-17T13:55:24.329222Z","iopub.status.idle":"2025-11-17T13:55:41.228871Z","shell.execute_reply.started":"2025-11-17T13:55:24.329206Z","shell.execute_reply":"2025-11-17T13:55:41.228293Z"}},"outputs":[],"execution_count":null},{"id":"bkHC","cell_type":"code","source":"DEVs = tf.config.list_physical_devices()\ntf.config.set_visible_devices(DEVs)\nprint(tf.config.get_visible_devices())","metadata":{"id":"bkHC","outputId":"d6b7d8b9-8ae1-4fb9-c53c-02aaa532eace","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:41.230068Z","iopub.execute_input":"2025-11-17T13:55:41.230507Z","iopub.status.idle":"2025-11-17T13:55:41.234965Z","shell.execute_reply.started":"2025-11-17T13:55:41.230477Z","shell.execute_reply":"2025-11-17T13:55:41.234234Z"}},"outputs":[],"execution_count":null},{"id":"lEQa","cell_type":"markdown","source":"# OSIC Pulmonary Fibrosis Progression - Computer Vision","metadata":{"marimo":{"config":{"hide_code":true}},"id":"lEQa"}},{"id":"PKri","cell_type":"code","source":"df = pd.read_csv(f'{DIR}/train.csv') \\\n    .reset_index(drop=True) \\\n    .groupby(['Patient', 'Weeks']) \\\n    .agg({\n        'FVC': 'mean',\n        'Percent': 'mean',\n        'Age': 'first',\n        'Sex': 'first',\n        'SmokingStatus': 'first',\n    }) \\\n    .reset_index()\n\ndf","metadata":{"id":"PKri","outputId":"cceecacc-ffa4-49e8-a5c6-5652ef3f75ae","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:41.235884Z","iopub.execute_input":"2025-11-17T13:55:41.236136Z","iopub.status.idle":"2025-11-17T13:55:41.297012Z","shell.execute_reply.started":"2025-11-17T13:55:41.236112Z","shell.execute_reply":"2025-11-17T13:55:41.296401Z"}},"outputs":[],"execution_count":null},{"id":"Xref","cell_type":"code","source":"def path_scans(patient_id):\n    paths = glob.glob(f'{DIR}/train/{patient_id}/*.dcm')\n    keys = (int(path.split('/')[-1].split('.')[0]) for path in paths)\n    paths_keys = sorted(zip(paths, keys), key=lambda x: x[1])\n    paths = [elem[0] for elem in paths_keys]\n\n    img = pydicom.dcmread(paths[0])\n    return paths, (img.Columns, img.Rows), len(paths)\n\ndef patients(df):\n    def fvc_agg(g):\n        pos = g[g['Weeks'] >= 0].sort_values(by='Weeks', ascending=True)\n        neg = g[g['Weeks'] < 0].sort_values(by='Weeks', ascending=False)\n\n        fvc = 0\n        if pos.iloc[0]['Weeks'] == 0:\n            # has data at week 0\n            fvc = pos.iloc[0]['FVC']\n        else:\n            if neg.shape[0] > 0:\n                # has data before and after week 0\n                (x1, y1) = neg.iloc[0][['Weeks', 'FVC']]\n                (x2, y2) = pos.iloc[0][['Weeks', 'FVC']]\n\n                # Linear interpolation\n                fvc = (x2 * y1 - x1 * y2) / (x2 - x1)\n            else:\n                # only has data after week 0\n                (x1, y1) = pos.iloc[0][['Weeks', 'FVC']]\n                fvc = y1\n\n        fvc_full = pos.iloc[0]['FVC'] / pos.iloc[0]['Percent'] * 100\n        return pd.Series({'FVC_0': fvc, 'FVC_full': fvc_full, 'Ratio_0': fvc / fvc_full})\n\n    df_fvc_0 = df.groupby(['Patient'])[['Weeks', 'FVC', 'Percent']].apply(fvc_agg).reset_index()\n    df_patients = df[['Patient', 'Age', 'Sex', 'SmokingStatus']].drop_duplicates().reset_index(drop=True)\n    df_patients[['ScanDim', 'ScanDepth']] = df_patients['Patient'].apply(lambda _id: path_scans(_id)[1:]).apply(pd.Series)\n    return df_patients.merge(df_fvc_0, on='Patient').set_index('Patient')\n\ndf_patients = patients(df)\ndf_patients","metadata":{"id":"Xref","outputId":"cf210656-25ff-4bb8-8e7a-999c3a28a256","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:41.298480Z","iopub.execute_input":"2025-11-17T13:55:41.298659Z","iopub.status.idle":"2025-11-17T13:55:46.711642Z","shell.execute_reply.started":"2025-11-17T13:55:41.298645Z","shell.execute_reply":"2025-11-17T13:55:46.710966Z"}},"outputs":[],"execution_count":null},{"id":"SFPL","cell_type":"code","source":"def eda_scans():\n    fig, axes = plt.subplots(ncols=2, nrows=1, figsize=(10, 5))\n    df_patients.value_counts('ScanDim').plot.barh(ax=axes[0])\n    df_patients['ScanDepth'].hist(ax=axes[1], bins=25)\n\n    avg_depth = df_patients['ScanDepth'].mean()\n    axes[1].axvline(avg_depth, color='r')\n\n    axes[0].set_title('Image dimensions (W x H)')\n    axes[1].set_title('Image scan depths (D)')\n    plt.show()\n\neda_scans()\n# df_patients['ScanDepth'].mean()","metadata":{"id":"SFPL","outputId":"3a8d1c6c-5297-4045-d3e5-bc26e1d40ed6","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:46.712386Z","iopub.execute_input":"2025-11-17T13:55:46.712702Z","iopub.status.idle":"2025-11-17T13:55:47.117505Z","shell.execute_reply.started":"2025-11-17T13:55:46.712674Z","shell.execute_reply":"2025-11-17T13:55:47.116941Z"}},"outputs":[],"execution_count":null},{"id":"BYtC","cell_type":"markdown","source":"## EDA","metadata":{"marimo":{"config":{"hide_code":true}},"id":"BYtC"}},{"id":"RGSE","cell_type":"code","source":"def eda():\n    fig, axes = plt.subplots(nrows=1, ncols=4, figsize=(20, 5))\n    ages = df_patients['Age']\n    ages_m = df_patients.loc[df_patients['Sex'] ==   'Male', 'Age']\n    ages_f = df_patients.loc[df_patients['Sex'] == 'Female', 'Age']\n    sexes = df_patients['Sex'].value_counts()\n    statuses = df_patients['SmokingStatus'].value_counts()\n\n    axes[0].hist(x=ages)\n    axes[1].hist(x=[ages_m, ages_f], label=['Male', 'Female'])\n    axes[1].legend()\n    axes[2].bar(x=sexes.index, height=sexes.values, color=['tab:blue', 'tab:orange'])\n    axes[3].bar(x=statuses.index, height=statuses.values)\n\n    axes[0].set_title('Distribution of age')\n    axes[1].set_title('Distribution of age for each gender')\n    axes[2].set_title('Distribution of gender')\n    axes[3].set_title('Distribution of SmokingStatus')\n    plt.show()\n\neda()","metadata":{"id":"RGSE","outputId":"3b94b441-afce-4dac-85fc-34b50260048d","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:47.118213Z","iopub.execute_input":"2025-11-17T13:55:47.118442Z","iopub.status.idle":"2025-11-17T13:55:47.596729Z","shell.execute_reply.started":"2025-11-17T13:55:47.118416Z","shell.execute_reply":"2025-11-17T13:55:47.596145Z"}},"outputs":[],"execution_count":null},{"id":"Kclp","cell_type":"code","source":"def progression(status, color, ax=None):\n    patients_with_status = df_patients.loc[df_patients['SmokingStatus'] == status] \\\n        .sample(10, replace=True).index\n\n    for patient_id in patients_with_status:\n        df_patient = df.loc[df['Patient'] == patient_id]\n\n        weeks, fvcs = df_patient[['Weeks', 'Percent']].T.values\n        ax.set_xlabel('Weeks')\n        ax.set_ylabel('Percent')\n        ax.plot(weeks, fvcs, color=color)\n        ax.tick_params(axis='y', labelcolor=color)\n\nfig, ax1 = plt.subplots()\nprogression('Ex-smoker', color='tab:orange', ax=ax1)\nprogression('Never smoked', color='tab:blue', ax=ax1)\nprogression('Currently smokes', color='tab:red', ax=ax1)\n\nfig.tight_layout()  # otherwise the right y-label is slightly clipped\nplt.show()","metadata":{"id":"Kclp","outputId":"96e535c6-c283-42a0-b85d-7542cac06d24","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:47.597440Z","iopub.execute_input":"2025-11-17T13:55:47.597743Z","iopub.status.idle":"2025-11-17T13:55:47.798684Z","shell.execute_reply.started":"2025-11-17T13:55:47.597725Z","shell.execute_reply":"2025-11-17T13:55:47.798081Z"}},"outputs":[],"execution_count":null},{"id":"b2806d54-342a-4794-9b6b-f6badf1ae268","cell_type":"code","source":"df_patients","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:00:40.712855Z","iopub.execute_input":"2025-11-17T14:00:40.713574Z","iopub.status.idle":"2025-11-17T14:00:40.726399Z","shell.execute_reply.started":"2025-11-17T14:00:40.713549Z","shell.execute_reply":"2025-11-17T14:00:40.725760Z"}},"outputs":[],"execution_count":null},{"id":"emfo","cell_type":"code","source":"df_fvc_ratios = df.pivot(index='Patient', columns=['Weeks'], values=['Percent']) \\\n    .droplevel(0, axis=1) \\\n    .reindex(columns=range(MIN_WEEK, MAX_WEEK + 1)) \\\n    .interpolate(axis='columns') \\\n    .bfill(axis=1) \\\n    .map(lambda x: x / 100) \\\n    .merge(df_patients[['FVC_full']], left_index=True, right_index=True)\n\ndf_fvc_ratios","metadata":{"id":"emfo","outputId":"7261da94-0850-4ddb-a3ec-1d84dfd12739","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:01:32.078387Z","iopub.execute_input":"2025-11-17T14:01:32.079115Z","iopub.status.idle":"2025-11-17T14:01:32.124960Z","shell.execute_reply.started":"2025-11-17T14:01:32.079082Z","shell.execute_reply":"2025-11-17T14:01:32.124353Z"}},"outputs":[],"execution_count":null},{"id":"Hstk","cell_type":"code","source":"class OSICDataset(tf.keras.utils.Sequence):\n    def __init__(self, mode, DIR, csv_file,\n                 depth=64,\n                 batch_size=1,\n                 input_size=(128, 128, 64),\n                 shuffle=True):\n\n        self.mode = mode\n        self.df = pd.read_csv(f'{DIR}/{csv_file}') \\\n            .reset_index(drop=True) \\\n            .groupby(['Patient', 'Weeks']) \\\n            .agg({\n                'FVC': 'mean',\n                'Percent': 'mean',\n                'Age': 'first',\n                'Sex': 'first',\n                'SmokingStatus': 'first',\n            }) \\\n            .reset_index()\n\n        self.df_patients = self._patients()\n        self.df_fvc_ratios = self._fvc_ratios()\n\n        self.depth = depth\n        self.batch_size = batch_size\n        self.input_size = input_size\n        self.shuffle = shuffle\n        self.n = len(self.df_patients)\n\n    def _patients(self):\n        def __path_scans(patient_id):\n            paths = glob.glob(f'{DIR}/{self.mode}/{patient_id}/*.dcm')\n\n            keys = (int(path.split('/')[-1].split('.')[0]) for path in paths)\n            paths_keys = sorted(zip(paths, keys), key=lambda x: x[1])\n            paths = [elem[0] for elem in paths_keys]\n\n            img = pydicom.dcmread(paths[0])\n            return paths, (img.Columns, img.Rows), len(paths)\n\n        def __fvc_full(g):\n        #     pos = g[g['Weeks'] >= 0].sort_values(by='Weeks', ascending=True)\n        #     neg = g[g['Weeks'] < 0].sort_values(by='Weeks', ascending=False)\n\n        #     fvc = 0\n        #     if pos.iloc[0]['Weeks'] == 0:\n        #         # has data at week 0\n        #         fvc = pos.iloc[0]['FVC']\n        #     else:\n        #         if neg.shape[0] > 0:\n        #             # has data before and after week 0\n        #             (x1, y1) = neg.iloc[0][['Weeks', 'FVC']]\n        #             (x2, y2) = pos.iloc[0][['Weeks', 'FVC']]\n\n        #             # Linear interpolation\n        #             fvc = (x2 * y1 - x1 * y2) / (x2 - x1)\n        #         else:\n        #             # only has data after week 0\n        #             (x1, y1) = pos.iloc[0][['Weeks', 'FVC']]\n        #             fvc = y1\n\n            # patient = g.iloc[0]['Patient']\n            fvc_full = g.iloc[0]['FVC'] / g.iloc[0]['Percent'] * 100\n            # return pd.Series({'FVC_0': fvc, 'FVC_full': fvc_full, 'Ratio_0': fvc / fvc_full})\n            return pd.Series({'FVC_full': fvc_full})\n\n        df_fvc_full = self.df.groupby(['Patient'])[['Weeks', 'FVC', 'Percent']] \\\n            .apply(__fvc_full).reset_index()\n        df_patients = self.df[['Patient', 'Age', 'Sex', 'SmokingStatus']] \\\n            .drop_duplicates().reset_index(drop=True)\n        df_patients[['ScanDim', 'ScanDepth']] = df_patients['Patient'] \\\n            .apply(lambda _id: __path_scans(_id)[1:]) \\\n            .apply(pd.Series)\n\n        df_patients = df_patients.merge(df_fvc_full, on='Patient').set_index('Patient')\n        return df_patients\n\n    def _fvc_ratios(self):\n        return self.df.pivot(index='Patient', columns=['Weeks'], values=['Percent']) \\\n            .droplevel(0, axis=1) \\\n            .reindex(columns=range(MIN_WEEK, MAX_WEEK + 1)) \\\n            .interpolate(axis='columns') \\\n            .bfill(axis=1) \\\n            .map(lambda x: x / 100) \\\n            .merge(self.df_patients[['FVC_full']], left_index=True, right_index=True)\n\n    def on_epoch_end(self):\n        pass\n\n    def __getitem__(self, idx):\n        sta = idx * self.batch_size\n        fin = min(sta + self.batch_size, len(self.df_patients))\n        df_patients = self.df_patients.iloc[sta:fin]\n\n        imgs_3d = None\n        for patient_id in df_patients.index:\n            paths = glob.glob(f'{DIR}/{self.mode}/{patient_id}/*.dcm')\n            keys = (int(path.split('/')[-1].split('.')[0]) for path in paths)\n            paths = sorted(zip(paths, keys), key=lambda x: x[1])\n\n            img_3d = None\n            for idx in range(self.depth):\n                pos = int(idx * len(paths) / self.depth)\n                filename = paths[pos][0]\n                img = pydicom.dcmread(filename).pixel_array.astype(np.float32)\n                img = img / (np.max(img) + 0.001)\n                img = cv2.resize(img, (IMG_WIDTH, IMG_HEIGHT))\n                img = np.expand_dims(img, -1)\n\n                if img_3d is None:\n                    img_3d = img\n                else:\n                    img_3d = np.concatenate((img_3d, img), axis=-1)\n\n            img_3d = np.expand_dims(img_3d, 0)\n            if imgs_3d is None:\n                imgs_3d = img_3d\n            else:\n                imgs_3d = np.concatenate((imgs_3d, img_3d), axis=0)\n\n        if self.mode == 'train':\n            df_fvc_ratios = self.df_fvc_ratios.loc[df_patients.index.tolist()]\n            return (imgs_3d, df_patients, df_fvc_ratios)\n        else:\n            return (imgs_3d, df_patients)\n\n    def __len__(self):\n        return (self.n + self.batch_size - 1) // self.batch_size\n\nosic_dataset = OSICDataset('train', DIR, 'train.csv')\nosic_dataset[3][0].shape","metadata":{"id":"Hstk","outputId":"aec75855-3c56-4251-d9c8-de1ace8cd572","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:15:56.393594Z","iopub.execute_input":"2025-11-17T14:15:56.394363Z","iopub.status.idle":"2025-11-17T14:15:57.798145Z","shell.execute_reply.started":"2025-11-17T14:15:56.394338Z","shell.execute_reply":"2025-11-17T14:15:57.797193Z"}},"outputs":[],"execution_count":null},{"id":"nWHF","cell_type":"code","source":"# class OSICInitialFVCDataset(OSICDataset):\n#     def __getitem__(self, idx):\n#         (img_3d, df_patients, _) = super().__getitem__(idx)\n#         return img_3d, df_patients['Ratio_0']\n\n# osic_fvc0_dataset = OSICInitialFVCDataset('train', df_fvc_ratios, df_patients, batch_size=8)\n# osic_fvc0_dataset[0][0].shape","metadata":{"id":"nWHF","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:55:49.724614Z","iopub.execute_input":"2025-11-17T13:55:49.724824Z","iopub.status.idle":"2025-11-17T13:55:49.728271Z","shell.execute_reply.started":"2025-11-17T13:55:49.724808Z","shell.execute_reply":"2025-11-17T13:55:49.727503Z"}},"outputs":[],"execution_count":null},{"id":"iLit","cell_type":"code","source":"# https://keras.io/examples/vision/3D_image_classification/#define-a-3d-convolutional-neural-network\n\ndef cnn_3d_regression_features(width, height, depth):\n    \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n    inputs = keras.Input((width, height, depth, 1))\n\n    x = layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=128, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=256, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.GlobalAveragePooling3D()(x)\n    feature_vector = layers.Dense(units=256, activation=\"relu\", name=\"feature_vector\")(x)\n    x = layers.Dropout(0.3)(feature_vector)\n\n    outputs = layers.Dense(units=1, name=\"fvc_0\")(x)\n\n    # Define the model.\n    model = keras.Model(inputs, [outputs, feature_vector], name=\"cnn_3d_regression_features\")\n    return model\n\ncnn_3d_model = cnn_3d_regression_features(width=IMG_WIDTH, height=IMG_HEIGHT, depth=IMG_DEPTH)\n# cnn_3d_model = keras.saving.load_model('./weights/251010-cnn-3d.keras')\ncnn_3d_model.summary()","metadata":{"id":"iLit","outputId":"ca2a223b-e808-40f1-98be-9a6fe8a31139","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:16:00.790877Z","iopub.execute_input":"2025-11-17T14:16:00.791572Z","iopub.status.idle":"2025-11-17T14:16:00.908632Z","shell.execute_reply.started":"2025-11-17T14:16:00.791549Z","shell.execute_reply":"2025-11-17T14:16:00.907894Z"}},"outputs":[],"execution_count":null},{"id":"ZHCJ","cell_type":"code","source":"class OSICTimeSeriesDataset(OSICDataset):\n    def __init__(self, mode, DIR, csv_file, **kwargs):\n        super().__init__(mode, DIR, csv_file, **kwargs)\n\n    def __getitem__(self, idx):\n        if self.mode == 'train':\n            (img_3d, df_patients, df_fvc_ratios) = super().__getitem__(idx)\n            return img_3d, df_fvc_ratios.to_numpy()\n        else:\n            (img_3d, df_patients) = super().__getitem__(idx)\n            return img_3d, df_patients\n\nosic_time_series_dataset = OSICTimeSeriesDataset('train', DIR, 'train.csv', batch_size=8)\nprint(osic_time_series_dataset[0][0].shape)\nprint(osic_time_series_dataset[0][1].shape)","metadata":{"id":"ZHCJ","outputId":"6d5e2da9-5e14-4e99-b4ae-a6d8ed968d31","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:16:03.393621Z","iopub.execute_input":"2025-11-17T14:16:03.394363Z","iopub.status.idle":"2025-11-17T14:16:08.417885Z","shell.execute_reply.started":"2025-11-17T14:16:03.394336Z","shell.execute_reply":"2025-11-17T14:16:08.417090Z"}},"outputs":[],"execution_count":null},{"id":"ROlb","cell_type":"code","source":"def full_model(width, height, depth, activation=['linear', 'linear'], hidden_units=64, dense_units=112):\n    \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n    # cnn_3d_model = keras.saving.load_model('../input/cnn-3d/keras/default/1/251010-cnn-3d.keras')\n\n    cnn_3d_model = cnn_3d_regression_features(width, height, depth)\n    inputs = cnn_3d_model.input\n    _, x = cnn_3d_model.outputs\n    x = layers.Reshape((1, 256))(x)\n    x = layers.LSTM(hidden_units, activation=activation[0])(x)\n    x = layers.Dense(units=dense_units, activation='sigmoid')(x)\n    \n    x = tf.keras.layers.Reshape((dense_units, 1))(x)\n    x = layers.ZeroPadding1D(padding=(0, 1))(x)\n    outputs = tf.keras.layers.Reshape((dense_units + 1,), name=\"output\")(x)\n\n    # Define the model.\n    model = keras.Model(inputs, outputs, name=\"full_model\")\n    return model\n\nmodel = full_model(width=IMG_WIDTH, height=IMG_HEIGHT, depth=IMG_DEPTH, dense_units=MAX_WEEK-MIN_WEEK+1)\n# model = keras.models.load_model(MODEL)\nmodel.summary()","metadata":{"id":"ROlb","outputId":"d09a984e-ecfa-4219-f067-bbc5b6f3fdba","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:21:03.692181Z","iopub.execute_input":"2025-11-17T14:21:03.692454Z","iopub.status.idle":"2025-11-17T14:21:03.833949Z","shell.execute_reply.started":"2025-11-17T14:21:03.692435Z","shell.execute_reply":"2025-11-17T14:21:03.833323Z"}},"outputs":[],"execution_count":null},{"id":"UaFTwXNUSeAI","cell_type":"code","source":"@keras.saving.register_keras_serializable()\ndef competition_metric(y_true, y_pred):\n    \"\"\"OSIC Competition Metric\"\"\"\n    y_true = tf.cast(y_true, tf.float32)\n    y_pred = tf.cast(y_pred, tf.float32)\n    \n    fvc_full = y_true[:, -1:]\n    y_fvc_true = y_true[:, :-1] * fvc_full\n    y_fvc_pred = y_pred[:, :-1] * fvc_full\n\n    # print(f'y_rate_true.shape = {y_fvc_true.shape}')\n    # print(f'y_rate_pred.shape = {y_fvc_pred.shape}')\n    # print(f'y_rate_true = {y_fvc_true[0]}')\n    # print(f'y_rate_pred = {y_fvc_pred[0]}')\n\n    sigma = y_fvc_pred[:, 2] - y_fvc_pred[:, 0]\n    fvc_pred = y_fvc_pred[:, 1]\n\n    sigma_clipped = tf.maximum(sigma, SIGMA_MIN)\n    delta = tf.abs(y_fvc_true[:, 0] - fvc_pred)\n    delta_clipped = tf.minimum(delta, DELTA_MAX)\n\n    metric = (delta_clipped / sigma_clipped) * SQRT_2 + \\\n                tf.math.log(sigma_clipped * SQRT_2)\n\n    return tf.keras.backend.mean(metric)\n\n@keras.saving.register_keras_serializable()\ndef modified_mae_loss(y_true, y_pred):\n    \"\"\"OSIC Competition Metric\"\"\"\n    y_true = tf.cast(y_true, tf.float32)[:, :-1]\n    y_pred = tf.cast(y_pred, tf.float32)[:, :-1]\n\n    return tf.reduce_mean(tf.abs(y_true - y_pred))","metadata":{"id":"UaFTwXNUSeAI","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:20:58.379667Z","iopub.execute_input":"2025-11-17T14:20:58.380467Z","iopub.status.idle":"2025-11-17T14:20:58.386807Z","shell.execute_reply.started":"2025-11-17T14:20:58.380440Z","shell.execute_reply":"2025-11-17T14:20:58.385965Z"}},"outputs":[],"execution_count":null},{"id":"qnkX","cell_type":"code","source":"tf.keras.backend.clear_session()\nwith tf.device('/GPU:0'):\n    model.compile(optimizer='adam', loss=modified_mae_loss, metrics=[competition_metric], run_eagerly=True)\n    history = model.fit(osic_time_series_dataset, epochs=32)\n    model.save('251117-cnn-rnn.keras')","metadata":{"id":"qnkX","outputId":"b8592398-40c8-4b94-8faf-7716e822133c","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:21:15.574480Z","iopub.execute_input":"2025-11-17T14:21:15.575243Z","iopub.status.idle":"2025-11-17T14:22:12.418957Z","shell.execute_reply.started":"2025-11-17T14:21:15.575218Z","shell.execute_reply":"2025-11-17T14:22:12.418138Z"}},"outputs":[],"execution_count":null},{"id":"XyoisrZLryFE","cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.title('Model Loss per epochs')\nplt.xticks(range(len(history.history['loss'])))\nplt.xlabel('Epoch')\nplt.ylabel('MAE')\nplt.savefig('loss_per_epochs.png')\nplt.show()","metadata":{"id":"XyoisrZLryFE","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:22:19.116161Z","iopub.execute_input":"2025-11-17T14:22:19.116785Z","iopub.status.idle":"2025-11-17T14:22:19.266481Z","shell.execute_reply.started":"2025-11-17T14:22:19.116760Z","shell.execute_reply":"2025-11-17T14:22:19.265911Z"}},"outputs":[],"execution_count":null},{"id":"Vxnm","cell_type":"code","source":"x, y = osic_time_series_dataset[0]\ny_pred = model.predict(x)\n\nplt.plot(y_pred[0][:-1])\nplt.plot(y[0][:-1])\n# plt.ylim(0, 200)\nplt.show()","metadata":{"id":"Vxnm","outputId":"04dd0c10-ec1e-4817-8b5f-b58746e5e73e","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:23:29.346373Z","iopub.execute_input":"2025-11-17T14:23:29.347361Z","iopub.status.idle":"2025-11-17T14:23:32.129716Z","shell.execute_reply.started":"2025-11-17T14:23:29.347322Z","shell.execute_reply":"2025-11-17T14:23:32.129087Z"}},"outputs":[],"execution_count":null},{"id":"FMOvvYrbrzzc","cell_type":"markdown","source":"## Predicting the test dataset","metadata":{"id":"FMOvvYrbrzzc"}},{"id":"DnEU","cell_type":"code","source":"osic_time_series_dataset_test = OSICTimeSeriesDataset('test', DIR, 'test.csv', batch_size=2)\nosic_time_series_dataset_test","metadata":{"id":"DnEU","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:23:41.664797Z","iopub.execute_input":"2025-11-17T14:23:41.665564Z","iopub.status.idle":"2025-11-17T14:23:41.936306Z","shell.execute_reply.started":"2025-11-17T14:23:41.665537Z","shell.execute_reply":"2025-11-17T14:23:41.935521Z"}},"outputs":[],"execution_count":null},{"id":"ulZA","cell_type":"code","source":"patients_tests = []\n\nfor idx in range(len(osic_time_series_dataset_test)):\n    imgs, patients = osic_time_series_dataset_test[idx]\n    with tf.device('/GPU:0'):\n        y_test_pred = model.predict(imgs)\n        y_test_pred = y_test_pred[:, :-1]\n    patients_test = patients.reset_index()[['Patient', 'FVC_full']]\n\n    df_y_test_pred = pd.DataFrame(y_test_pred)\n    df_y_test_pred.columns += MIN_WEEK\n\n    patients_test = pd.concat([patients_test, df_y_test_pred], axis=1) \\\n        .melt(id_vars=['Patient', 'FVC_full'], var_name='Week', value_name='Percent')\n    patients_test['FVC'] = patients_test['FVC_full'] * patients_test['Percent']\n    patients_test['Confidence'] = 100\n\n    patients_test['Patient_Week'] = patients_test['Patient'] + '_' + patients_test['Week'].astype('str')\n    patients_tests.append(patients_test)\n\ndf_patients_test = pd.concat(patients_tests).set_index('Patient_Week')\ndf_patients_test","metadata":{"id":"ulZA","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T14:24:21.135608Z","iopub.execute_input":"2025-11-17T14:24:21.136231Z","iopub.status.idle":"2025-11-17T14:24:22.533268Z","shell.execute_reply.started":"2025-11-17T14:24:21.136206Z","shell.execute_reply":"2025-11-17T14:24:22.532650Z"}},"outputs":[],"execution_count":null},{"id":"ZBYS","cell_type":"code","source":"df_patients_test[['FVC', 'Confidence']] \\\n    .to_csv(f'{SUBMISSION_DIR}/submission.csv')","metadata":{"id":"ZBYS","trusted":true,"execution":{"iopub.status.busy":"2025-11-17T13:56:17.941211Z","iopub.status.idle":"2025-11-17T13:56:17.941421Z","shell.execute_reply.started":"2025-11-17T13:56:17.941326Z","shell.execute_reply":"2025-11-17T13:56:17.941335Z"}},"outputs":[],"execution_count":null},{"id":"SJFP42oSznI_","cell_type":"code","source":"","metadata":{"id":"SJFP42oSznI_","trusted":true},"outputs":[],"execution_count":null}]}