{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Work in Progress\nOsic-pulmonary-fibrosis-progression\nData Preprocessing for Patients and their Images\nmodel selection and parameter settings"},{"metadata":{"trusted":true},"cell_type":"code","source":"!conda install -c conda-forge gdcm -y","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"scrolled":true},"cell_type":"code","source":"from __future__ import print_function\n\nimport os\nfrom os import listdir\nimport IPython\nimport IPython.display\nimport copy\nimport pandas as pd\nimport numpy as np\nimport pydicom as dicom\nfrom pydicom import dcmread\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tqdm\nimport glob\nfrom typing import Dict\n\nfrom sklearn.preprocessing import RobustScaler\nfrom scipy import ndimage\nfrom scipy.ndimage.interpolation import zoom\n\nfrom skimage import measure, morphology, segmentation\nfrom skimage.measure import label, regionprops\nfrom skimage.morphology import binary_closing\nfrom skimage.segmentation import clear_border\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_path = \"../input/osic-pulmonary-fibrosis-progression/\"\n\nlistdir(file_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(file_path + \"train.csv\")\ntest_df = pd.read_csv(file_path + \"test.csv\")\nsub_df = pd.read_csv(file_path + \"sample_submission.csv\")\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"duplicate_data = train_df[train_df.duplicated(subset=[\"Patient\", \"Weeks\"], keep=False)]\n\nduplicate_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.drop_duplicates(subset=[\"Patient\", \"Weeks\"], keep=\"last\", inplace=True)\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df[[\"Patient\", \"Weeks\"]] = sub_df[\"Patient_Week\"].str.split(\"_\", expand=True)\nsub_df = sub_df[[\"Patient\", \"Weeks\", \"Patient_Week\"]]\nsub_df = sub_df.merge(test_df.drop(\"Weeks\", axis=1), on=\"Patient\")\n\ntrain_df[\"Source\"] = \"train\"\nsub_df[\"Source\"] = \"test\"\n\ndataset = train_df.append([sub_df])\ndataset.reset_index(drop=True, inplace=True)\ndataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset[\"FVC_ave\"] = (dataset[\"FVC\"] ) / dataset[\"Percent\"] * 100\ndataset.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def baseline_week(df):\n    df = df.copy()\n    df[\"Weeks\"] = df[\"Weeks\"].astype(int)\n    df.loc[df[\"Source\"] == \"test\", \"min_weeks\"] = np.nan\n    df[\"min_weeks\"] = df.groupby(\"Patient\")[\"Weeks\"].transform(\"min\")\n    df[\"baseline_week\"] = df[\"Weeks\"] - df[\"min_weeks\"]\n    \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset = baseline_week(dataset)\ndataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_baseline_fvc(df):\n    df = df.copy()\n    base = df.loc[df[\"Weeks\"] == df[\"min_weeks\"]].copy()\n    base = df[[\"Patient\", \"FVC\"]].copy()\n    base.columns = [\"Patient\", \"base_fvc\"]\n    base[\"no\"] = 1\n    base[\"no\"] = base.groupby(\"Patient\")[\"no\"].transform(\"cumsum\")\n    base = base[base.no == 1]\n    base.drop(\"no\", axis=1, inplace=True)\n    df = df.merge(base, on = \"Patient\", how = \"left\")\n    \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset = get_baseline_fvc(dataset)\ndataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset[\"Sex\"] = pd.Categorical(dataset[\"Sex\"])\ndataset[\"Sex\"] = dataset.Sex.cat.codes\ndataset[\"SmokingStatus\"] = pd.Categorical(dataset[\"SmokingStatus\"])\ndataset[\"SmokingStatus\"] = dataset.SmokingStatus.cat.codes\n\ndataset.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"true_pat_res = test_df.Patient.unique()\ntrue_pat_res.sort()\n\ntrue_result = train_df.loc[train_df[\"Patient\"].isin(true_pat_res)].copy()\ntrue_result.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset.drop_duplicates(subset=[\"Patient\", \"Weeks\"], keep=\"last\", inplace=True)\n\ntrain_df = dataset.loc[dataset[\"Source\"] == \"train\"].copy()\n\ntest_df = dataset.loc[dataset[\"Source\"] == \"test\"].copy()\ntrain_df.drop(\"Source\", axis=1, inplace=True)\ntest_df.drop(\"Source\", axis=1, inplace=True)\n\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.drop([\"Patient_Week\", \"Percent\", \"min_weeks\"], axis=1, inplace=True)\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.drop([\"Percent\", \"Patient_Week\", \"min_weeks\"], axis=1, inplace=True)\n\ntest_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if file_path == \"../input/osic-pulmonary-fibrosis-progression/\":\n    train_df[\"dcm_path\"] = file_path + \"train/\" + train_df.Patient + \"/\"\nelse:\n    train_df[\"dcm_path\"] = file_path + \"train/\" + train_df.StudyInstanceUID + \"/\" + train_df.SeriesInstanceUID","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if file_path == \"../input/osic-pulmonary-fibrosis-progression/\":\n    test_df[\"dcm_path\"] = file_path + \"test/\" + test_df.Patient + \"/\"\nelse:\n    test_df[\"dcm_path\"] = file_path + \"test/\" + test_df.StudyInstanceUID + \"/\" + test_df.SeriesInstanceUID","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = []\npatient_path = []\n\nif file_path == \"../input/osic-pulmonary-fibrosis-progression/\":\n    patients = train_df.Patient.unique()\nelse:\n    patients = train_df.StudyInstanceUID.unique()\n\nfor patient in patients:\n    patient_id.append(patient)\n    if file_path == \"../input/osic-pulmonary-fibrosis-progression/\":\n        path = train_df[train_df.Patient == patient].dcm_path.values[0]\n    else:\n        path = train_df[train_df.StudyInstanceUID == patient].dcm_path.values[0]\n    ex_dcm = listdir(path)[0]\n    patient_path.append(path)\n    ds = dcmread(path + \"/\" + ex_dcm)\n\n\n\npatient_df = pd.DataFrame(data=patient_id, columns=[\"patient\"])\npatient_df.loc[:, \"patient_path\"] = patient_path\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"def load_scan(dcm_path):\n    if file_path == \"/..input/osic-pulmonary-fibrosis-progression/\":\n        files = listdir(dcm_path)\n        file_no = [np.int(files.split(\".\")[0]) for file in files]\n        sorted_files = np.sort(file_no)[::-1]\n        slices = [dcmread(dcm_path  + \"/\" + str(file_no) + \".dcm\") for file_no in sorted_files]\n    else:\n            \n        slices = [dcmread(dcm_path + \"/\" + s) for s in listdir(dcm_path)]\n        slices = [s for s in slices if \"SliceLocation\" in s]\n        slices.sort(key=lambda x: int(x.InstanceNumber))\n        try:\n            slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n        except:\n            slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n            \n        for s in slices:\n            s.SliceThickness = slice_thickness\n            \n    return slices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"def convert_to_hu(slices):\n    image = np.stack([s.pixel_array for s in slices])\n    image = image.astype(np.int16)\n    \n    image[image == -2000] = 0\n    \n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    \n    if slope != 1:\n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n        \n    image += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ex = train_df.dcm_path.values[0]\nscans = load_scan(ex)\nhu_scans = convert_to_hu(scans)\n\nplt.figure()\nplt.imshow(hu_scans[13], cmap=plt.cm.gray)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def resample(image, scan, new_spacing=[1,1,1]):\n    spacing = np.array([scan[0].SliceThickness] + list(scan[0].PixelSpacing), dtype=np.float32)\n    resize_factor = spacing / new_spacing\n    new_real_shape = image.shape / resize_factor\n    new_shape = np.round(new_real_shape)\n    real_resize_factor = new_shape / image.shape\n    new_spacing = spacing / real_resize_factor\n    \n    image = ndimage.interpolation.zoom(image, real_resize_factor, mode=\"nearest\")\n    \n    return image, new_spacing","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"im_res, spacing = resample(hu_scans, scans, [1,1,1])\nhu_scans.shape, im_res.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(11, 7))\nax[0].imshow(hu_scans[20], cmap=plt.cm.gray)\nax[1].imshow(im_res[1], cmap=plt.cm.gray)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_multival_vals(feature):\n    if type(feature) == dicom.multival.MultiValue:\n        return np.int(feature[0])\n    else:\n        return np.int(feature)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def generate_markers(image):\n    marker_internal = image < -400\n    marker_internal = segmentation.clear_border(marker_internal)\n    marker_internal_labels = measure.label(marker_internal)\n    areas = [r.area for r in measure.regionprops(marker_internal_labels)]\n    areas.sort()\n    if len(areas) > 2:\n        for region in measure.regionprops(marker_internal_labels):\n            if region.area < areas[-2]:\n                for coordinates in region.coords:\n                    marker_internal_labels[coordinates[0], coordinates[1]] = 0\n    marker_internal = marker_internal_labels > 0\n    \n    external_a = ndimage.binary_dilation(marker_internal, iterations=10)\n    external_b = ndimage.binary_dilation(marker_internal, iterations=55)\n    marker_external = external_b ^ external_a\n    \n    marker_watershed = np.zeros((image.shape), dtype=np.int)\n    marker_watershed += marker_internal * 255\n    marker_watershed += marker_external * 128\n    \n    return marker_internal, marker_external, marker_watershed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_internal, patient_external, patient_watershed = generate_markers(hu_scans[13])\n\nfig, ax = plt.subplots(1, 3, figsize=(17, 6))\nax[0].set_title(\"Internel marker\")\nax[0].imshow(patient_internal, cmap=\"gray\")\nax[1].set_title(\"External Marker\")\nax[1].imshow(patient_external, cmap=\"gray\")\nax[2].set_title(\"watershed image\")\nax[2].imshow(patient_watershed, cmap=\"gray\")\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def separate_lungs(image):\n    marker_internal, marker_external, marker_watershed = generate_markers(image)\n    \n    sobel_filtered_dx = ndimage.sobel(image, 0)\n    sobel_filtered_dy = ndimage.sobel(image, 1)\n    sobel_gradient = np.hypot(sobel_filtered_dx, sobel_filtered_dy)\n    sobel_gradient *= 255.0 / np.max(sobel_gradient)\n    \n    watershed = morphology.watershed(sobel_gradient, marker_watershed)\n    \n    outline = ndimage.morphological_gradient(watershed, size=(3, 3))\n    outline = outline.astype(bool)\n    \n    blackhat_structure = [[0, 0, 1, 1, 1, 0, 0],\n                          [0, 1, 1, 1, 1, 1, 0],\n                          [1, 1, 1, 1, 1, 1, 1],\n                          [1, 1, 1, 1, 1, 1, 1],\n                          [1, 1, 1, 1, 1, 1, 1],\n                          [0, 1, 1, 1, 1, 1, 0],\n                          [0, 0, 1, 1, 1, 0, 0]]\n    \n    blackhat_structure = ndimage.iterate_structure(blackhat_structure, iterations=7)\n    outline += ndimage.black_tophat(outline, structure=blackhat_structure)\n    \n    lungfilter = np.bitwise_or(marker_internal, outline)\n    lungfilter = ndimage.morphology.binary_closing(lungfilter, structure=np.ones((5, 5)), iterations = 3)\n    \n    segmented = np.where(lungfilter == 1, image, -2000*np.ones((image.shape)))\n    \n    return segmented","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_segmented = separate_lungs(hu_scans[13])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(7, 7))\nplt.title(\"Segmented Lung\")\nplt.imshow(train_segmented, cmap=plt.cm.gray)\n\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def img_hu_processing(patient_df):\n    \n    for i, patient in enumerate(tqdm.tqdm(patient_df[\"patient\"].values)):\n        try:\n            path = patient_df.loc[patient_df[\"patient\"] == patient].patient_path.values[0]\n            scans = load_scan(path)\n            n = len(scans)\n            if n >= 30:\n                m = int(n/10.0)\n                scans = scans[int(n*0.1):int(n*0.9):int(m)*2]\n                hu_scans = convert_to_hu(scans)\n            else:\n                hu_scans = convert_to_hu(scans)\n                \n            for patient in path:\n                b = hu_scans\n                np.savez(\"imgs.npz\", b)   \n        except Exception as e:\n            continue\n            ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_hu_processing(patient_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dict_a = np.load(\"imgs.npz\")\nprint(dict_a.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"column_indices = {name: i for i, name in enumerate(train_df.columns)}\n\nn = len(train_df)\n\ntrain_ds = train_df[0:int(n*0.7)]\nval_ds = train_df[int(n*0.7):int(n*0.9)]\ntest_ds = train_df[int(n*0.9):]\n\nnum_features = train_df.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.describe().transpose()\n\ntrain_mean = train_ds.mean()\ntrain_std = train_ds.std()\n\ntrain_ds = (train_ds - train_mean) / train_std\nval_ds = (val_ds - train_mean) / train_std\ntest_ds = (test_ds - train_mean) / train_std","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class WindowGenerator():\n    def __init__(self, input_width, label_width, shift,\n                 train_ds=train_ds, val_ds=val_ds, test_ds=test_ds,\n                 label_columns=None):\n        \n        self.train_ds = train_ds\n        self.val_ds = val_ds\n        self.test_ds = test_ds\n        \n        self.label_columns = label_columns\n        if label_columns is not None:\n            self.label_columns_indices = {name: i for i, name in enumerate(label_columns)}\n            \n        self.column_indices = {name: i for i, name in enumerate(train_ds.columns)}\n        \n        self.input_width = input_width\n        self.label_width = label_width\n        self.shift = shift\n        \n        self.total_window_size = input_width + shift\n        \n        self.input_slice = slice(0, input_width)\n        self.input_indices = np.arange(self.total_window_size)[self.input_slice]\n        \n        self.label_start = self.total_window_size - self.label_width\n        self.labels_slice = slice(self.label_start, None)\n        self.label_indices = np.arange(self.total_window_size)[self.labels_slice]\n        \n    def __repr__(self):\n        return \"\\n\".join([\n            f\"TotalWindowSpread: {self.total_window_size}\",\n            f\"Total Indices: {self.input_indices}\",\n            f\"Label Indices: {self.label_indices}\",\n            f\"Label Name: {self.label_columns}\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w1 = WindowGenerator(input_width=30, label_width=1, shift=30, label_columns=[\"FVC\"])\nw1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w2 = WindowGenerator(input_width=5, label_width=1, shift=1, label_columns=[\"FVC\"])\nw2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def split_window(self, features):\n    inputs = features[:, self.input_slice, :]\n    labels = features[:, self.labels_slice, :]\n    if self.label_columns is not None:\n        labels = tf.stack(\n            [labels[:, :, self.column_indices[name]] for name in self.label_columns],\n            axis=-1)\n        \n    inputs.set_shape([None, self.input_width, None])\n    labels.set_shape([None, self.label_width, None])\n    \n    return inputs, labels\n\nWindowGenerator.split_window = split_window","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ex_window = tf.stack([np.array(train_ds[:w2.total_window_size]),\n                      np.array(train_ds[:+w2.total_window_size]),\n                      np.array(train_ds[:+w2.total_window_size])])\n\n\nex_inputs, ex_labels = w2.split_window(ex_window)\n\nprint(f\"Window shape: {ex_window.shape}\")\nprint(f\"Inputs shape: {ex_inputs.shape}\")\nprint(f\"Labels shape: {ex_labels.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w2.example = ex_inputs, ex_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot(self, model=None, plot_column=\"FVC\", max_subplots=3):\n    inputs, labels = self.example\n    plt.figure(figsize=(11, 7))\n    plot_column_index = self.column_indices[plot_column]\n    max_n = min(max_subplots, len(inputs))\n    for n in range(max_n):\n        plt.subplot(3, 1, n+1)\n        plt.ylabel(f\"{plot_column} [normed]\")\n        plt.plot(self.input_indices, inputs[n, :, plot_column_index],\n                 label=\"Inputs\", marker=\".\", zorder=-10)\n        \n        if self.label_columns:\n            label_column_index = self.label_columns_indices.get(plot_column, None)\n        else:\n            label_column_index = plot_column_index\n            \n        if label_column_index is None:\n            continue\n            \n        plt.scatter(self.label_indices, labels[n, :, label_column_index],\n                    edgecolors=\"k\", label=\"Labels\", c=\"#2ca02c\", s=64)\n        if model is not None:\n            predictions = model(inputs)\n            plt.scatter(self.label_indices, predictions[n, :, label_column_index],\n                        marker=\"X\", edgecolors=\"k\", label=\"Predictions\",\n                        c=\"#ff7f0e\", s=64)\n        \n        if n == 0:\n            plt.legend()\n        \n    plt.xlabel(\"Weeks\")\n    \nWindowGenerator.plot = plot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w2.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_dataset(self, data):\n    data = np.array(data, dtype=np.float32)\n    ds = tf.keras.preprocessing.timeseries_dataset_from_array(\n        data=data, targets=None, sequence_length=self.total_window_size,\n        sequence_stride=1, shuffle=True, batch_size=32)\n    \n    ds = ds.map(self.split_window)\n    \n    return ds\n\nWindowGenerator.create_dataset = create_dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"@property\ndef train(self):\n    return self.create_dataset(self.train_ds)\n\n@property\ndef val(self):\n    return self.create_dataset(self.val_ds)\n\n@property\ndef test(self):\n    return self.create_dataset(self.test_ds)\n\n@property\ndef example(self):\n    result = getattr(self, \"_example\", None)\n    if result is None:\n        result = next(iter(self.train))\n        self._example = result\n    return result\n\nWindowGenerator.train = train\nWindowGenerator.val = val\nWindowGenerator.test = test\nWindowGenerator.example = example","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w2.train.element_spec","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for example_inputs, example_labels in w2.train.take(1):\n    print(f\"exInput shape: {example_inputs.shape}\")\n    print(f\"exLabel shape: {example_labels.shape}\")\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"singlestep_wind = WindowGenerator(input_width=1, label_width=1, shift=1, label_columns=[\"FVC\"])\nsinglestep_wind","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for example_inputs, example_labels in singlestep_wind.train.take(1):\n    print(f\"Inputs: {example_inputs.shape}\")\n    print(f\"ouputs: {example_labels.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Baseline(tf.keras.Model):\n    def __init__(self, label_index=None):\n        super().__init__()\n        self.label_index = label_index\n        \n    def call(self, inputs):\n        if self.label_index is None:\n            return inputs\n        \n        result = inputs[:, :, self.label_index]\n        return result[:, :, tf.newaxis]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"max_epochs = 100\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor=\"val_loss\", \n                                                  patience=2, \n                                                  mode=\"min\")\n\ndef compile_and_fit(model, window):\n    \n    model.compile(loss=tf.keras.losses.MeanSquaredError(),\n                  optimizer=tf.keras.optimizers.Adam(lr=0.001, beta_1=0.9, beta_2=0.999, epsilon=1e-8),\n                  metrics=[\"mae\", \"mse\"])\n    model.fit(window.train, epochs = max_epochs,\n              validation_data = window.val, \n              callbacks=[early_stopping])\n    history = model\n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"baseline = Baseline(label_index=column_indices[\"FVC\"])\n\nbaseline.compile(loss=tf.keras.losses.MeanSquaredError(),\n                 metrics=[\"mse\"])\n\nval_performance = {}\nperformance = {}\n\nval_performance[\"Baseline\"] = baseline.evaluate(singlestep_wind.val)\nperformance[\"Baseline\"] = baseline.evaluate(singlestep_wind.test, verbose=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"wide_wind = WindowGenerator(input_width=30, label_width=30, shift=1, label_columns=[\"FVC\"])\nwide_wind","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Input shape:\", singlestep_wind.example[0].shape)\nprint(\"output shape:\", baseline(singlestep_wind.example[0]).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"wide_wind.plot(baseline)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lstm = tf.keras.models.Sequential([\n    tf.keras.layers.LSTM(32, return_sequences=True),\n    tf.keras.layers.Dense(units=1)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Lstm Input shape:\", wide_wind.example[0].shape)\nprint(\"Lstm Output shape:\", lstm(wide_wind.example[0]).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = compile_and_fit(lstm, wide_wind)\n\nIPython.display.clear_output()\nval_performance[\"LSTM\"] = lstm.evaluate(wide_wind.val)\nperformance[\"LSTM\"] = lstm.evaluate(wide_wind.test, verbose=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"wide_wind.plot(lstm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for name, value in performance.items():\n    print(f\"{name:12s}: {value[1]:0.4f}\")\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"output_steps = 140\nmulti_window = WindowGenerator(input_width=140,\n                               label_width=output_steps,\n                               shift=output_steps)\n\nmulti_window.plot()\nmulti_window","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class MultiStepLastBaseline(tf.keras.Model):\n    def call(self, inputs):\n        return tf.tile(inputs[:, -1:, :], [1, output_steps, 1])\n    \n\nlast_baseline = MultiStepLastBaseline()\nlast_baseline.compile(loss=tf.keras.losses.MeanSquaredError(),\n                      metrics=[\"mse\", \"mae\"])\n\n\nmulti_val_performance = {}\nmulti_performance = {}\n\nmulti_val_performance[\"Last\"] = last_baseline.evaluate(multi_window.val)\nmulti_performance[\"Last\"] = last_baseline.evaluate(multi_window.test, verbose=0)\nmulti_window.plot(last_baseline)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class RepeatBaseline(tf.keras.Model):\n    def call(self, inputs):\n        return inputs\n\n    \nrepeat_baseline = RepeatBaseline()\nrepeat_baseline.compile(loss=tf.keras.losses.MeanSquaredError(),\n                        metrics=[\"mse\", tf.keras.metrics.Accuracy()])\n\nmulti_val_performance[\"Repeat\"] = repeat_baseline.evaluate(multi_window.val)\nmulti_performance[\"Repeat\"] = repeat_baseline.evaluate(multi_window.test, verbose=0)\nmulti_window.plot(repeat_baseline)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"multi_lstm = tf.keras.models.Sequential([\n    tf.keras.layers.LSTM(32, return_sequences=True),\n    tf.keras.layers.Dense(output_steps*num_features, kernel_initializer=tf.initializers.zeros),\n    tf.keras.layers.Reshape([output_steps, num_features])])\n\nhistory = compile_and_fit(multi_lstm, multi_window)\n\nIPython.display.clear_output()\n\n\nmulti_val_performance[\"LSTM\"] = multi_lstm.evaluate(multi_window.val)\nmulti_performance[\"LSTM\"] = multi_lstm.evaluate(multi_window.test, verbose=0)\nmulti_window.plot(multi_lstm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class FeedBack(tf.keras.Model):\n    def __init__(self, units, output_steps):\n        super().__init__()\n        self.output_steps = output_steps\n        self.units = units\n        self.lstm_cell = tf.keras.layers.LSTMCell(units)\n        self.lstm_rnn = tf.keras.layers.RNN(self.lstm_cell, return_state=True)\n        self.dense = tf.keras.layers.Dense(num_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feedback_lstm = FeedBack(units=32, output_steps=output_steps)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def warmup(self, inputs):\n    x, *state = self.lstm_rnn(inputs)\n    \n    prediction = self.dense(x)\n    \n    return prediction, state\n\nFeedBack.warmup = warmup","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prediction, state = feedback_lstm.warmup(multi_window.example[0])\nprediction.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def call(self, inputs, training=None):\n    predictions = []\n    \n    prediction, state = self.warmup(inputs)\n    \n    predictions.append(prediction)\n    \n    for n in range(1, self.output_steps):\n        x = prediction\n        x, state = self.lstm_cell(x, states=state, training=training)\n        \n        prediction = self.dense(x)\n        predictions.append(prediction)\n        \n    predictions = tf.stack(predictions)\n    predictions = tf.transpose(predictions, [1, 0, 2])\n    return predictions\n    \nFeedBack.call = call","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Batch, Time, Features:\", feedback_lstm(multi_window.example[0]).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = compile_and_fit(feedback_lstm, multi_window)\n\nIPython.display.clear_output()\n\nmulti_val_performance[\"AR LSTM\"] = feedback_lstm.evaluate(multi_window.val)\nmulti_performance[\"AR LSTM\"] = feedback_lstm.evaluate(multi_window.test, verbose=0)\nmulti_window.plot(feedback_lstm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for name, value in multi_performance.items():\n    print(f\"{name:12s}: {value[1]:0.4f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def score(FVC_true, FVC_pred, sigma):\n    sigma_clipped = np.max(sigma, 70)\n    delta = np.abs(FVC_true - FVC_pred)\n    delta = np.min(delta, 1000)\n    sq_2 = np.sqrt(2)\n    metric = (delta / sigma_clipped) * sq_2 + (np.log(sigma_clipped * sq_2))\n    \n    return np.mean(metric)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}