{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Here is the 5th solution (Citeseq part). \n**Our highest private score is obtained by integrating 4 MLP models and 2 LGBM models. The method is very simple. If you read it, you will not regret it.**\n\nThis notebook is our MLP training part. There are three MLP models in this notebook, and one in another notebook.\n\nThe training template in the notebook comes from the top method of last year opened by LS_lab.\n\nThanks for sharing！","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:06:29.073425Z","iopub.execute_input":"2022-11-17T12:06:29.074037Z","iopub.status.idle":"2022-11-17T12:06:32.138489Z","shell.execute_reply.started":"2022-11-17T12:06:29.073939Z","shell.execute_reply":"2022-11-17T12:06:32.137406Z"}}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom tqdm import tqdm\nimport os\n\nimport numpy as np\n\nfrom sklearn.decomposition import TruncatedSVD\n\nfrom scipy.sparse import csc_matrix\nimport logging\nfrom torch.utils.data.dataset import Dataset\nimport copy\nimport gc\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:12:24.700346Z","iopub.execute_input":"2022-11-17T12:12:24.700745Z","iopub.status.idle":"2022-11-17T12:12:24.709866Z","shell.execute_reply.started":"2022-11-17T12:12:24.700704Z","shell.execute_reply":"2022-11-17T12:12:24.708642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport scipy\nimport scipy.sparse\n\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:08:13.901485Z","iopub.execute_input":"2022-11-17T12:08:13.901857Z","iopub.status.idle":"2022-11-17T12:08:13.908011Z","shell.execute_reply.started":"2022-11-17T12:08:13.901825Z","shell.execute_reply":"2022-11-17T12:08:13.906808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.read_csv(FP_CELL_METADATA, index_col='cell_id')\nmetadata_df = metadata_df[metadata_df.technology==\"citeseq\"]\nmetadata_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:08:16.650707Z","iopub.execute_input":"2022-11-17T12:08:16.651261Z","iopub.status.idle":"2022-11-17T12:08:17.192608Z","shell.execute_reply.started":"2022-11-17T12:08:16.651225Z","shell.execute_reply":"2022-11-17T12:08:17.191656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(7)\n    ]\n\n# create a list of the values we want to assign for each condition\nvalues = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16]\n\n# create a new column and use np.select to assign values to it using our lists as arguments\nmetadata_df['comb'] = np.select(conditions, values)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:08:19.903752Z","iopub.execute_input":"2022-11-17T12:08:19.904192Z","iopub.status.idle":"2022-11-17T12:08:19.938121Z","shell.execute_reply.started":"2022-11-17T12:08:19.904148Z","shell.execute_reply":"2022-11-17T12:08:19.937006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df['comb'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:08:20.048668Z","iopub.execute_input":"2022-11-17T12:08:20.051277Z","iopub.status.idle":"2022-11-17T12:08:20.067770Z","shell.execute_reply.started":"2022-11-17T12:08:20.051236Z","shell.execute_reply":"2022-11-17T12:08:20.066741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = pd.read_hdf(FP_CITE_TRAIN_INPUTS)\ncell_index = X.index\nmeta = metadata_df.reindex(cell_index)\ndel X\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:08:23.900621Z","iopub.execute_input":"2022-11-17T12:08:23.901541Z","iopub.status.idle":"2022-11-17T12:09:19.414286Z","shell.execute_reply.started":"2022-11-17T12:08:23.901489Z","shell.execute_reply":"2022-11-17T12:09:19.412415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_ = 2","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:11:02.276617Z","iopub.execute_input":"2022-11-17T12:11:02.277313Z","iopub.status.idle":"2022-11-17T12:11:02.282085Z","shell.execute_reply.started":"2022-11-17T12:11:02.277275Z","shell.execute_reply":"2022-11-17T12:11:02.280829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if model_ == 1 or model_ == 2:\n    X = pd.read_csv('../input/cite-preprocessing/X_159.csv').values\n    Xt = pd.read_csv('../input/cite-preprocessing/Xt_159.csv').values\nelse:\n    X = pd.read_csv('../input/cite-preprocessing/X_274.csv').values\n    Xt = pd.read_csv('../input/cite-preprocessing/Xt_274.csv').values","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:03.092478Z","iopub.execute_input":"2022-11-17T12:10:03.092853Z","iopub.status.idle":"2022-11-17T12:10:12.667355Z","shell.execute_reply.started":"2022-11-17T12:10:03.092821Z","shell.execute_reply":"2022-11-17T12:10:12.666376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = pd.read_hdf(FP_CITE_TRAIN_TARGETS)\nY = Y.values\nY -= Y.mean(axis=1).reshape(-1, 1)\nY /= Y.std(axis=1).reshape(-1, 1)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:12.669153Z","iopub.execute_input":"2022-11-17T12:10:12.669741Z","iopub.status.idle":"2022-11-17T12:10:13.279359Z","shell.execute_reply.started":"2022-11-17T12:10:12.669702Z","shell.execute_reply":"2022-11-17T12:10:13.278277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NegativeCorrLoss(preds, targets):\n    \"\"\"Compute the correlation between each rows of the y_true and y_pred tensors.\n    Compatible with backpropagation.\n    \"\"\"\n    my = torch.mean(preds, dim=1)\n    my = torch.tile(torch.unsqueeze(my, dim=1), (1, targets.shape[1]))\n\n    ym = preds - my\n    r_num = torch.sum(torch.multiply(targets, ym), dim=1)\n    r_den = torch.sqrt(\n        torch.sum(torch.square(ym), dim=1) * float(targets.shape[-1])\n    )\n    r = torch.mean(r_num / r_den)\n    return -r\ndef correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:13.281533Z","iopub.execute_input":"2022-11-17T12:10:13.282159Z","iopub.status.idle":"2022-11-17T12:10:13.292925Z","shell.execute_reply.started":"2022-11-17T12:10:13.282119Z","shell.execute_reply":"2022-11-17T12:10:13.291881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net1(nn.Module):\n    def __init__(self):\n        super(Net1, self).__init__()\n        self.fc1 = nn.Linear(159,120)\n        self.fc2 = nn.Linear(120,200)\n        self.fc3 = nn.Linear(200,620)\n        self.fc4 = nn.Linear(620,440)\n        self.fc5 = nn.Linear(440,140)\n\n    def forward(self, x):\n        x = F.gelu(self.fc1(x))\n        x = F.gelu(self.fc2(x))\n        x = F.gelu(self.fc3(x))\n        x = F.gelu(self.fc4(x))\n        x = self.fc5(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:19.900808Z","iopub.execute_input":"2022-11-17T12:10:19.901162Z","iopub.status.idle":"2022-11-17T12:10:19.908809Z","shell.execute_reply.started":"2022-11-17T12:10:19.901130Z","shell.execute_reply":"2022-11-17T12:10:19.907623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net2(nn.Module):\n    def __init__(self):\n        super(Net2, self).__init__()\n        self.fc1 = nn.Linear(159,120)\n        self.fc2 = nn.Linear(120,200)\n        self.fc3 = nn.Linear(200,620)\n        self.fc4 = nn.Linear(620,440)\n        self.fc5 = nn.Linear(440,140)\n\n    def forward(self, x):\n        x = F.hardswish(self.fc1(x))\n        x = F.hardswish(self.fc2(x))\n        x = F.hardswish(self.fc3(x))\n        x = F.hardswish(self.fc4(x))\n        x = self.fc5(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:20.121129Z","iopub.execute_input":"2022-11-17T12:10:20.121396Z","iopub.status.idle":"2022-11-17T12:10:20.128866Z","shell.execute_reply.started":"2022-11-17T12:10:20.121370Z","shell.execute_reply":"2022-11-17T12:10:20.127738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net3(nn.Module):\n    def __init__(self):\n        super(Net3, self).__init__()\n        self.selu = nn.SELU()\n        self.fc1 = nn.Linear(274,120)\n        self.fc2 = nn.Linear(120,200)\n        self.fc3 = nn.Linear(200,620)\n        self.fc4 = nn.Linear(620,440)\n        self.fc5 = nn.Linear(440,140)\n\n    def forward(self, x):\n        x = self.selu(self.fc1(x))\n        x = self.selu(self.fc2(x))\n        x = self.selu(self.fc3(x))\n        x = self.selu(self.fc4(x))\n        x = self.fc5(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:20.424016Z","iopub.execute_input":"2022-11-17T12:10:20.424284Z","iopub.status.idle":"2022-11-17T12:10:20.431714Z","shell.execute_reply.started":"2022-11-17T12:10:20.424257Z","shell.execute_reply":"2022-11-17T12:10:20.430591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, split, X_train, X_val, X_test, y_train, y_val):\n        self.split = split\n\n        if self.split == \"train\":\n            self.data = X_train\n            self.gt = y_train\n        elif self.split == \"val\":\n            self.data = X_val\n            self.gt = y_val\n        else:\n            self.data = X_test\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        if self.split == \"train\":\n            return self.data[idx], self.gt[idx]\n        elif self.split == \"val\":\n            return self.data[idx], 0\n        else:\n            return self.data[idx]","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:23.298958Z","iopub.execute_input":"2022-11-17T12:10:23.299316Z","iopub.status.idle":"2022-11-17T12:10:23.309210Z","shell.execute_reply.started":"2022-11-17T12:10:23.299284Z","shell.execute_reply":"2022-11-17T12:10:23.308211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, optimizer, dataloaders_dict, true_test_mod2, num_epochs):\n    best_mse = 100\n    best_model = 0\n    best_cor = 0\n    for epoch in range(num_epochs):\n        y_pred = []\n\n        print('Epoch {}/{}'.format(epoch, num_epochs - 1))\n        print('-' * 10)\n\n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()\n            else:\n                model.eval()\n\n            running_loss, running_corrects = 0.0, 0\n\n            for inputs, gts in tqdm(dataloaders_dict[phase]):\n                inputs = inputs.cuda()\n                gts = gts.cuda()\n\n                optimizer.zero_grad()\n\n                with torch.set_grad_enabled(phase == 'train'):\n                    outputs = model(inputs)\n\n                    if phase == 'train':\n                        loss = NegativeCorrLoss(outputs, gts)\n                        running_loss += loss.item() * inputs.size(0)\n                        loss.backward()\n                        optimizer.step()\n                    else:\n                        y_pred.extend(outputs.cpu().numpy())\n\n\n            if phase == \"train\":\n                epoch_loss = running_loss / len(dataloaders_dict[phase].dataset)\n                print('{} Loss: {:.4f}'.format(phase, epoch_loss))\n            else:\n                y_pred = np.array(y_pred)\n\n                cor = correlation_score(true_test_mod2, y_pred)\n                print(cor)\n                if cor > best_cor:\n                    best_model = copy.deepcopy(model)\n                    best_cor = cor\n \n    print(\"Best cor: \", best_cor)\n    \n    return best_model","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:12:49.427391Z","iopub.execute_input":"2022-11-17T12:12:49.428406Z","iopub.status.idle":"2022-11-17T12:12:49.439865Z","shell.execute_reply.started":"2022-11-17T12:12:49.428363Z","shell.execute_reply":"2022-11-17T12:12:49.438749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(model, dataloader):\n    y_pred = []\n    model.eval()\n\n    for inputs in tqdm(dataloader):\n        if isinstance(inputs,list) == True:\n            inputs = inputs[0]\n        inputs = inputs.cuda()\n\n        with torch.set_grad_enabled(False):\n            outputs = model(inputs)\n\n            y_pred.extend(outputs.cpu().numpy())\n\n    y_pred = np.array(y_pred)\n\n    return y_pred","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:25.750259Z","iopub.execute_input":"2022-11-17T12:10:25.750605Z","iopub.status.idle":"2022-11-17T12:10:25.757292Z","shell.execute_reply.started":"2022-11-17T12:10:25.750573Z","shell.execute_reply":"2022-11-17T12:10:25.756179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.astype('float32')\nXt = Xt.astype('float32')","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:10:39.299565Z","iopub.execute_input":"2022-11-17T12:10:39.300254Z","iopub.status.idle":"2022-11-17T12:10:39.339099Z","shell.execute_reply.started":"2022-11-17T12:10:39.300215Z","shell.execute_reply":"2022-11-17T12:10:39.338082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold, train_test_split\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\ny_pred = 0\nN_SPLITS_ANN = len(meta['comb'].value_counts())\nkf = GroupKFold(n_splits=N_SPLITS_ANN)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, groups=meta.comb)):\n        model = None\n        gc.collect()\n        \n        X_train = X[idx_tr] \n        y_train = Y[idx_tr]\n        X_val = X[idx_va]\n        y_val = Y[idx_va]\n        X_test = Xt\n        \n        if model_ == 1:\n            model = Net1().cuda()\n        elif model_ == 2:\n            model = Net1().cuda()\n        else:\n            model = Net3().cuda()\n            \n        lr = 1e-4\n        optimizer_ft = optim.Adam(model.parameters(), lr=lr)\n        \n        data = {x: CustomDataset(x, X_train, X_val, X_test, y_train, y_val) for x in ['train', 'val', 'test']}\n        dataloaders_dict = {\"train\": torch.utils.data.DataLoader(data[\"train\"], batch_size=100, shuffle=True, num_workers=8),\n                        \"val\": torch.utils.data.DataLoader(data[\"val\"], batch_size=100, shuffle=False, num_workers=8),\n                        \"test\": torch.utils.data.DataLoader(data[\"test\"], batch_size=100, shuffle=False, num_workers=8)}\n        \n        best_model_net = train_model(model, optimizer_ft, dataloaders_dict, y_val, num_epochs=200)\n        y_pred = y_pred + infer(best_model_net, dataloaders_dict[\"test\"])\n        ","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:12:51.493399Z","iopub.execute_input":"2022-11-17T12:12:51.493795Z","iopub.status.idle":"2022-11-17T12:13:07.733763Z","shell.execute_reply.started":"2022-11-17T12:12:51.493760Z","shell.execute_reply":"2022-11-17T12:13:07.731905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/open-problems-multimodal/sample_submission.csv')\nsubmission.loc[:48663*140-1,'target'] = y_pred.reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:15:54.283985Z","iopub.execute_input":"2022-11-17T12:15:54.284335Z","iopub.status.idle":"2022-11-17T12:16:03.684287Z","shell.execute_reply.started":"2022-11-17T12:15:54.284304Z","shell.execute_reply":"2022-11-17T12:16:03.683295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T12:15:52.624789Z","iopub.execute_input":"2022-11-17T12:15:52.625495Z","iopub.status.idle":"2022-11-17T12:15:52.639400Z","shell.execute_reply.started":"2022-11-17T12:15:52.625456Z","shell.execute_reply":"2022-11-17T12:15:52.638325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}