{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os, sys\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\nfrom sklearn import preprocessing","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/osic-pulmonary-fibrosis-progression/\"\nMODEL_DIR = \"/kaggle/input/osicqrmodel/\"\nQUANTILES = [0.2, 0.5, 0.8]\n# columns to be scaled using min-max scaling\nSCALE_COLUMNS = ['Weeks', 'FVC', 'Percent', 'Age']\nSEX_COLUMNS = ['Male', 'Female']\nSMOKING_STATUS_COLUMNS = ['Currently smokes', 'Ex-smoker', 'Never smoked']\n\n# create the FV (feature vector) using the scaled columns + other columns\nFV = SEX_COLUMNS + SMOKING_STATUS_COLUMNS + SCALE_COLUMNS\nDEVICE = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# initialize the sklearn's min-max scaler\nMIN_MAX_SCALER = preprocessing.MinMaxScaler()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# read the train_df and initialize the MIN_MAX_SCALER with train data\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntrain_df.drop_duplicates(keep=False, inplace=True, subset=['Patient', 'Weeks'])\ntest_df = pd.read_csv(os.path.join(DATA_DIR, \"test.csv\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fit the scaler and transform the data using the fit_transform function\ntrain_df[SCALE_COLUMNS] = MIN_MAX_SCALER.fit_transform(train_df[SCALE_COLUMNS])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# specify the categorical columns and the categories incase any class is missing (in test)\n# convert into one-hot-encoding\ntrain_df['Sex'] = pd.Categorical(train_df['Sex'], categories=SEX_COLUMNS)\ntrain_df['SmokingStatus'] = pd.Categorical(train_df['SmokingStatus'], categories=SMOKING_STATUS_COLUMNS)\ntrain_df = train_df.join(pd.get_dummies(train_df['Sex']))\ntrain_df = train_df.join(pd.get_dummies(train_df['SmokingStatus']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = pd.read_csv(os.path.join(DATA_DIR, \"sample_submission.csv\"))\n# get the patient_id and the week from the Patient_Week column\nsub_df['Patient'] = sub_df['Patient_Week'].apply(lambda x: x.split('_')[0])\nsub_df['Weeks'] = sub_df['Patient_Week'].apply(lambda x: int(x.split('_')[-1]))\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = sub_df.drop(\"FVC\", axis=1).merge(test_df.drop('Weeks', axis=1), on='Patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# have to make it categorical coz sub's sex column has males only\nsub_df['Sex'] = pd.Categorical(sub_df['Sex'], categories=SEX_COLUMNS)\nsub_df['SmokingStatus'] = pd.Categorical(sub_df['SmokingStatus'], categories=SMOKING_STATUS_COLUMNS)\nsub_df = sub_df.join(pd.get_dummies(sub_df['Sex']))\nsub_df = sub_df.join(pd.get_dummies(sub_df['SmokingStatus']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df[SCALE_COLUMNS] = MIN_MAX_SCALER.transform(sub_df[SCALE_COLUMNS])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class PulmonaryDataset(Dataset):\n    def __init__(self, df, FV, test=False):\n        self.df = df\n        self.test = test\n        self.FV = FV\n\n    def __getitem__(self, idx):\n        return {\n            'features': torch.tensor(self.df[self.FV].iloc[idx].values),\n            'target': torch.tensor(self.df['FVC'].iloc[idx])\n        }\n\n    def __len__(self):\n        return len(self.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class PulmonaryModel(nn.Module):\n    def __init__(self, in_features=9, out_quantiles=3):\n        super(PulmonaryModel, self).__init__()\n        self.fc1 = nn.Linear(in_features, 100)\n        self.fc2 = nn.Linear(100, 100)\n        self.fc3 = nn.Linear(100, out_quantiles)\n    \n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = self.fc3(x)\n        return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dataset = PulmonaryDataset(sub_df, FV)\n\ntest_data_loader = DataLoader(\n    test_dataset,\n    batch_size=10,\n    drop_last=False,\n    num_workers=2\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"models = []\nfor fold in range(5):\n    model = PulmonaryModel(len(FV))\n    checkpoint = torch.load(os.path.join(MODEL_DIR, f\"model_fold_{fold}.pt\"))\n    model.load_state_dict(checkpoint['model_state_dict'])\n    model.to(DEVICE)\n    models.append(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"avg_preds = np.zeros((len(test_dataset), len(QUANTILES)))\nwith torch.no_grad():\n    for model in models:\n        preds = []\n        for j, test_data in enumerate(test_data_loader):\n            features = test_data['features']\n            targets = test_data['target']\n\n            features = features.to(DEVICE).float()\n            targets = targets.to(DEVICE).float()\n\n            out = model(features)\n            preds.append(out)\n        preds = torch.cat(preds, dim=0).cpu().numpy()\n        avg_preds += preds\n    avg_preds /= len(models)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"avg_preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# inverse the scaling operation for FVC\navg_preds -= MIN_MAX_SCALER.min_[SCALE_COLUMNS.index('FVC')]\navg_preds /= MIN_MAX_SCALER.scale_[SCALE_COLUMNS.index('FVC')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"avg_preds[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df['FVC'] = avg_preds[:, 1]\nsub_df['Confidence'] = np.abs(avg_preds[:, 2] - avg_preds[:, 0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.head(25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df[['Patient_Week', 'FVC', 'Confidence']].to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}