{"cells":[{"metadata":{"_uuid":"efe9c6234e4adaad9aee1108fc8e72a61f57d844"},"cell_type":"markdown","source":"## Convert an existing kernel from Keras to Pytorch, accept three phases at once\nhttps://www.kaggle.com/fernandoramacciotti/cnn-with-class-weights "},{"metadata":{"trusted":true,"_uuid":"52fc56dd5128f9134bfacf6b8f419ecedbc225cc"},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport scipy.stats as stats\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import StratifiedShuffleSplit\n\nimport pyarrow.parquet as pq\nimport math\nimport os\nimport gc\nfrom tqdm import tqdm_notebook as tqdm, tnrange\n\nprint(os.listdir(\"../input\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91754d921e49eaa916bf0cca20ee8170b3c7fcca"},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e13e115c31541e2560249356d1e8779c77a8ca9d"},"cell_type":"code","source":"train = pq.read_pandas('../input/train.parquet').to_pandas()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f8cc5d2ee2a60e5068e20d67e5b5d1c0211fa07"},"cell_type":"code","source":"mdtrain = pd.read_csv('../input/metadata_train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a7081d0b6d83412f2f262424724ad878bc5a053"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60f2e1de2a5adf5313643336b80efa2efa65226a"},"cell_type":"code","source":"mdtrain.head()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a8945601c912a9aa9ec4fdd7139690942afc8cb6"},"cell_type":"markdown","source":"## Extract features"},{"metadata":{"trusted":true,"_uuid":"21f8683058756cc2aac1d1cb0b6689098d316173"},"cell_type":"code","source":"def feature_extractor(x, n_part=1000):\n    length = len(x)\n    n_feat = 7\n    pool = np.int32(np.ceil(length/n_part))\n    output = np.zeros((n_part, n_feat))\n    for j, i in enumerate(range(0,length, pool)):\n        if i+pool < length:\n            k = x[i:i+pool]\n        else:\n            k = x[i:]\n        output[j, 0] = np.mean(k, axis=0) #mean\n        output[j, 1] = np.min(k, axis=0) #min\n        output[j, 2] = np.max(k, axis=0) #max\n        output[j, 3] = np.std(k, axis=0) #std\n        output[j, 4] = np.median(k, axis=0) #median\n        output[j, 5] = stats.skew(k, axis=0) #skew\n        output[j, 6] = stats.kurtosis(k, axis=0) # kurtosis\n    return output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79d4bba5b5652082a74ed35db97c2726c38e1ff6"},"cell_type":"code","source":"X = []\ny = []\nfor i in tqdm(mdtrain.signal_id):\n    idx = mdtrain.loc[mdtrain.signal_id==i, 'signal_id'].values.tolist()\n    y.append(mdtrain.loc[mdtrain.signal_id==i, 'target'].values)\n    X.append(feature_extractor(train.iloc[:, idx].values, n_part=400))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"22770c55394fceaf9396e915188bf542b09a57cd"},"cell_type":"code","source":"X = np.array(X).reshape(-1, X[0].shape[0], X[0].shape[1])\nX = np.transpose(X, [0,2,1]) # Make X shape (batch size, channels, time steps) \n# because channels/time-steps are different in Keras\n\ny = np.array(y).reshape(-1,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e84dc6ae4e57e2141f8ed552e4b78a8b14368175"},"cell_type":"code","source":"X.shape, y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17d3d4947c602a49f4e8cd1c91bb9cbabc4c2b50"},"cell_type":"code","source":"del train; gc.collect()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"676d63657cf4bb7c9a4ed482ad9720b83ebd95bc"},"cell_type":"markdown","source":"## Split into test/val sets and normalize"},{"metadata":{"trusted":true,"_uuid":"e6623ce33dc03b41ce6af8edc76ca17cbf7690eb"},"cell_type":"code","source":"sss = StratifiedShuffleSplit(n_splits=1, test_size=3*int(0.25*X.shape[0]/3), random_state=0)\n(train_idx, val_idx) = next(sss.split(X, y))\n\nX_train, X_val = X[train_idx], X[val_idx]\ny_train, y_val = y[train_idx], y[val_idx]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8f08ca21ab81d4505fd33b368ea87dc684d8b82"},"cell_type":"code","source":"X_train.shape, X_val.shape, y_train.shape, y_val.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"204aac5497d56ea42b5cf5b150d8062e12bdcbf2"},"cell_type":"code","source":"scalers = {} # This code actually looked wrong for Keras but right for Pytorch already!\nfor i in range(X_train.shape[1]):\n    scalers[i] = MinMaxScaler(feature_range=(-1, 1))\n    X_train[:, i, :] = scalers[i].fit_transform(X_train[:, i, :]) \n\nfor i in range(X_val.shape[1]):\n    X_val[:, i, :] = scalers[i].transform(X_val[:, i, :]) \n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fd059e6330e750d5c8bb8dc37edccb95feef20c1"},"cell_type":"markdown","source":"## Evaluation Metrics"},{"metadata":{"trusted":true,"_uuid":"92b2b72a703f2108e6b0c273b8f69c288679c605"},"cell_type":"code","source":"def confusion(prediction, truth):\n    \"\"\" Returns the confusion matrix for the values in the `prediction` and `truth`\n    tensors, i.e. the amount of positions where the values of `prediction`\n    and `truth` are\n    - 1 and 1 (True Positive)\n    - 1 and 0 (False Positive)\n    - 0 and 0 (True Negative)\n    - 0 and 1 (False Negative)\n    \"\"\"\n\n    confusion_vector = torch.as_tensor(prediction, dtype=torch.float32) / torch.as_tensor(truth, dtype=torch.float32)\n    # Element-wise division of the 2 tensors returns a new tensor which holds a\n    # unique value for each case:\n    #   1     where prediction and truth are 1 (True Positive)\n    #   inf   where prediction is 1 and truth is 0 (False Positive)\n    #   nan   where prediction and truth are 0 (True Negative)\n    #   0     where prediction is 0 and truth is 1 (False Negative)\n\n    true_positives = torch.sum(confusion_vector == 1).item()\n    false_positives = torch.sum(confusion_vector == float('inf')).item()\n    true_negatives = torch.sum(torch.isnan(confusion_vector)).item()\n    false_negatives = torch.sum(confusion_vector == 0).item()\n\n    return true_positives, false_positives, true_negatives, false_negatives","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b0f466368416ef1bce26bfded8c7cab8a9cbf56"},"cell_type":"code","source":"def matthews(TP, FP, TN, FN):\n    nom = TP*TN - FP*FN\n    denom = math.sqrt((TP+FP)*(TP+FN)*(TN+FP)*(TN+FN))\n    return nom/denom\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3195fc9b36326decd6c41ad19c7155cb52fe93f6"},"cell_type":"markdown","source":"## Neural Network in Pytorch"},{"metadata":{"trusted":true,"_uuid":"2b3cdd37a02883f3057d094c6d8fb0f606cc9861"},"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\ntorch.cuda.set_device(0)\ntorch.backends.cudnn.benchmark=True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48b2d8044fda0f0b55de36c9ddf5625e0213bf7c"},"cell_type":"code","source":"class VSBNet1(nn.Module):\n    \n    def __init__(self):\n        super(VSBNet1, self).__init__()\n        \n        self.conv1 = nn.Conv1d(in_channels=7, out_channels=64, kernel_size=4)\n        self.conv2 = nn.Conv1d(in_channels=64, out_channels=64, kernel_size=4)\n        \n        self.mp1 = nn.MaxPool1d(2, padding=1)\n        \n        self.conv3 = nn.Conv1d(in_channels=64, out_channels=20, kernel_size=4)\n        self.conv4 = nn.Conv1d(in_channels=20, out_channels=20, kernel_size=4)\n        \n        #self.gap1 = nn.AdaptiveMaxPool1d(20)\n        self.gap1 = nn.AvgPool1d(192)\n        \n        self.do1 = nn.Dropout(0.2)\n        \n        # Flatten\n        \n        self.lin1 = nn.Linear(20,32)\n        self.do2 = nn.Dropout(0.5)\n        \n        self.lin2 = nn.Linear(32,8)\n        self.do3 = nn.Dropout(0.5)\n\n        self.lin3 = nn.Linear(8,1)\n        \n        def init_weights(m):\n            # Conv1d defaults to kaiming just like Keras does\n            if type(m) == nn.Linear:\n                # Keras defaults Dense to glorot_uniform (which is also called xavier uniform)\n                # Whereas Pytorch default for Linear is kaiming\n                torch.nn.init.xavier_uniform_(m.weight)\n                \n        self.apply(init_weights)\n        \n        \n    def forward(self,x):\n        x = F.relu(self.conv1(x))\n        x = F.relu(self.conv2(x))\n        \n        x = self.mp1(x)\n        \n        x = F.relu(self.conv3(x))\n        x = F.relu(self.conv4(x))\n\n        x = self.gap1(x)\n        \n        x = self.do1(x)\n        \n        x = x.view(x.shape[0],-1)\n        \n        x = torch.tanh(self.lin1(x))\n        x = self.do2(x)\n        \n        x = torch.tanh(self.lin2(x))\n        x = self.do3(x)\n        \n        x = self.lin3(x) # Leave sigmoid for the loss function\n            \n        return x\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab5f54234f708fa8c309f0a79d45343887d94a8c"},"cell_type":"code","source":"net = VSBNet1() ; net.cuda()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f8031a4b7c507c46183595ec5cfa59ba0498911f"},"cell_type":"markdown","source":"## Train the first model"},{"metadata":{"trusted":true,"_uuid":"dec53778994cd50f826b9977f7cf6d24445483da"},"cell_type":"code","source":"NUM_EPOCHS = 30\nBS = 16","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c104f1b6e8d685591ad8cc844005cc89972bea76"},"cell_type":"code","source":"X_train.shape, y_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c1935df9c6114091dc527525686670f1681af0a"},"cell_type":"code","source":"trainloader = DataLoader(list(zip(X_train,y_train)), batch_size=BS, shuffle=False, num_workers=2, pin_memory=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7a4770ea3779c906a2778cbc016387d82f0da87"},"cell_type":"code","source":"criterion = torch.nn.BCEWithLogitsLoss() #pos_weight=torch.Tensor([1.0,1.2]))\noptimizer = torch.optim.Adam(net.parameters(), lr=1e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a65757b21326489d88a46cb6ccdb60a51b5164be"},"cell_type":"code","source":"net.train()\n    \nfor t in tnrange(NUM_EPOCHS, desc='Epochs'):\n    \n    #running_loss = 0.0\n    for i, (X_batch, y_target_batch) in tqdm(enumerate(trainloader), total=len(trainloader)):\n        \n        X_batch, y_target_batch = X_batch.to('cuda'), y_target_batch.to('cuda')\n              \n        # Forward pass: Compute predicted y by passing x to the model\n        y_pred = net(X_batch.float())\n        # y_pred = net(X.view(X.shape[0], 1, X.shape[1]))\n\n        # Compute and print loss\n        loss = criterion(y_pred, y_target_batch.float())\n\n\n        # Zero gradients, perform a backward pass, and update the weights.\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25c64926dfe7002068f7992eb901af4c2bceb72e"},"cell_type":"code","source":"MODEL_PATH='vsb2c.model'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3f272b120fb635276164e76a16a2041205b45d6"},"cell_type":"code","source":"torch.save(net.state_dict(), MODEL_PATH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f99d01e2e1e1128377fb682c26a94d6ee57e467"},"cell_type":"code","source":"net.load_state_dict(torch.load(MODEL_PATH))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e73f684ff571eba3212d90ab0ebec0687b482263"},"cell_type":"markdown","source":"## Evaluate"},{"metadata":{"trusted":true,"_uuid":"f8ac48e0cf8883e4c642d22d1668e0f7a16e3c74"},"cell_type":"code","source":"net.eval()\ny_preds = net(torch.Tensor(X_train).to('cuda', torch.float32)).detach()\nloss = criterion(y_preds, torch.Tensor(y_train).to('cuda', torch.float32)).detach(); loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dcac1dd08d9277d2f917ad2b93e78c2364350ba8"},"cell_type":"code","source":"# Number of target==0, no of target==1\n(y_preds <= 0).sum().item(), (y_preds > 0).sum().item()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d825134f099d14fd9dedd16d1779512fbdc21b8"},"cell_type":"code","source":"net.eval()\ny_val_preds = net(torch.Tensor(X_val).to('cuda', torch.float32)).detach()\nloss_val = criterion(y_val_preds, torch.Tensor(y_val).to('cuda', torch.float32)).detach(); loss_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f72633b0f866bc6a24c312496aed019456a766f"},"cell_type":"code","source":"(y_val_preds <= 0).sum().item(), (y_val_preds > 0).sum().item()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3ebba2be5c76298016c75747a2b58eb65ea145d7"},"cell_type":"markdown","source":"### Confusion and Matthews"},{"metadata":{"trusted":true,"_uuid":"c30f6542105d08180241c157012d62cc3f95cef5"},"cell_type":"code","source":"c_train = confusion(y_preds > 0, y_train); c_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0ef43d6497742fed0ce95a6bb10cccb9563c1de"},"cell_type":"code","source":"matthews(*c_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8aba9e85605f612329bb0900ff4f43cd7eab30aa"},"cell_type":"code","source":"c_val = confusion(y_val_preds > 0, y_val); c_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c02b8617dd58e0af479f533ce5f8a740c89abdbb"},"cell_type":"code","source":"matthews(*c_val)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"68e17786ab93b509117c298e0a445aef79b743b4"},"cell_type":"markdown","source":"## Try a network on all three phases at once"},{"metadata":{"trusted":true,"_uuid":"03d0f8e740ed586830c29caced400ed830cc7962"},"cell_type":"code","source":"X_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4dc1c75b635b9d91e110bf9d8419863d1267e62c"},"cell_type":"code","source":"def squeeze_phases(input):\n    return np.stack([input[0::3,:], input[1::3,:], input[2::3,:]], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33ebbd99dde91d448a10aeab8e8281c7bdbdc9df"},"cell_type":"code","source":"X_train_s = squeeze_phases(X_train)\ny_train_s = squeeze_phases(y_train)\nX_val_s = squeeze_phases(X_val)\ny_val_s = squeeze_phases(y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ec0e3db40ec16943be52ef8fa9fe84ede1179ca"},"cell_type":"code","source":"class VSBPhaseNet1(nn.Module):\n    \n    def __init__(self, singlenet):\n        super(VSBPhaseNet1, self).__init__()\n\n        self.singlenet = singlenet\n\n        self.lin3 = nn.Linear(8,1)\n        \n        def init_weights(m):\n            # Conv1d defaults to kaiming just like Keras does\n            if type(m) == nn.Linear:\n                # Keras defaults Dense to glorot_uniform (which is also called xavier uniform)\n                # Whereas Pytorch default for Linear is kaiming\n                torch.nn.init.xavier_uniform_(m.weight)\n                \n        self.apply(init_weights)\n        \n        \n    def forward(self,x):\n        # Split input into the three phases (channels)\n        channels = torch.split(x, 1, dim=1)\n        channels = [torch.squeeze(c, dim=1) for c in channels]\n        \n        # Run each channel through the original network separately\n        xs = [self.singlenet(c) for c in channels]\n        \n        # Very last layer was replaced with Identity so should output 8x1 instead of 1x1 matrix\n        # Join them together and apply a new linear layer\n        xs = torch.stack(xs, dim=1)\n        \n        x = self.lin3(xs) # Leave sigmoid for the loss function\n            \n        return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aff684c4bb6d02c705289507cfdeac87e42c8251"},"cell_type":"code","source":"singlenettrunc = VSBNet1()\nsinglenettrunc.load_state_dict(torch.load(MODEL_PATH))\nsinglenettrunc.lin3 = nn.Sequential() # Replace last linear layer with Identity so remain with 8 features not 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"967d413b422c0dcdfd4a6dc74b5e78a168e04a29"},"cell_type":"code","source":"singlenettrunc.cuda()\n#for p in list(singlenettrunc.parameters())[:-2]:\n#    p.requires_grad = False # Freeze all layers\nphasenet = VSBPhaseNet1(singlenettrunc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63726c636079d4aa824f700cc86baa0ec74c9019"},"cell_type":"code","source":"phasenet.cuda() ;","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc48081e278ca000312d740be6e7aa1fe155808e"},"cell_type":"code","source":"NUM_EPOCHS = 30\nBS = 16","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85b95c09629c6525220475afb496df1f7cf56f06"},"cell_type":"code","source":"X_train_s.shape, y_train_s.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de8099ec0e0fc91e5394ae26ab6a79dd345e248b"},"cell_type":"code","source":"trainsloader = DataLoader(list(zip(X_train_s,y_train_s)), batch_size=BS, shuffle=False, num_workers=2, pin_memory=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4225f65da823719027bfa214e3f584db9e8160e7"},"cell_type":"code","source":"criterion = torch.nn.BCEWithLogitsLoss() #pos_weight=torch.Tensor([1.0,1.2]))\noptimizer = torch.optim.Adam(phasenet.parameters(), lr=1e-3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d527d5db023942566e7231c579c5edef57d57a6b"},"cell_type":"code","source":"phasenet.train()\n\nfor t in tnrange(NUM_EPOCHS, desc='Epochs'):\n    \n    #running_loss = 0.0\n    for i, (X_batch, y_target_batch) in tqdm(enumerate(trainsloader), total=len(trainsloader)):\n        X_batch = X_batch.to('cuda')\n        y_target_batch = y_target_batch.to('cuda')\n        \n        # Forward pass: Compute predicted y by passing x to the model\n        y_pred = phasenet(X_batch.float())\n\n        # Compute and print loss\n        loss = criterion(y_pred, y_target_batch.float())\n\n        # Zero gradients, perform a backward pass, and update the weights.\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35870501149bd8ede76c80373cc6f4750ae613f7"},"cell_type":"code","source":"PHASE_MODEL_PATH='vsbphase1c.model'\n\ntorch.save(net.state_dict(), PHASE_MODEL_PATH)\n#net.load_state_dict(torch.load(PHASE_MODEL_PATH))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"15e39c34e01d4f5cf2c496e0f811b01638b9ddb4"},"cell_type":"markdown","source":"### Evaluate Phase model"},{"metadata":{"trusted":true,"_uuid":"b01479c73f4f1380726f8a75837da4d4f3b35c52"},"cell_type":"code","source":"phasenet.eval()\ny_preds_s = phasenet(torch.Tensor(X_train_s).float().to('cuda')).detach()\nloss = criterion(y_preds_s, torch.Tensor(y_train_s).float().to('cuda')).detach(); loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1e0689f8a2a407138ff2aaeecf429bd9b40d4e2"},"cell_type":"code","source":"# Number of target==0, no of target==1\n(y_preds_s <= 0).sum().item(), (y_preds_s > 0).sum().item()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0605c9b0819567af086f2b69735869546c05d3d"},"cell_type":"code","source":"phasenet.eval()\ny_val_preds_s = phasenet(torch.Tensor(X_val_s).float().to('cuda')).detach()\nloss_val = criterion(y_val_preds_s, torch.Tensor(y_val_s).float().to('cuda')).detach(); loss_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dfb240c0346489666d72812f90033da09d7fc61d"},"cell_type":"code","source":"(y_val_preds_s <= 0).sum().item(), (y_val_preds_s > 0).sum().item()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66c02c3cc97a54f46246a552ce4cd71cad59314e"},"cell_type":"code","source":"c_train = confusion(y_preds_s > 0, y_train_s); c_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d81ba9a88e4422a573f40a3627b4c9909f351b8"},"cell_type":"code","source":"matthews(*c_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8338a9d15d83c669ed637879230f26fb3f69fe63"},"cell_type":"code","source":"c_val = confusion(y_val_preds_s > 0, y_val_s); c_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45fa7187d7223fc9828f2ac678e4a9f7f4a03e57"},"cell_type":"code","source":"matthews(*c_val)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2fc4a0db1610f02873d6b3d1798c87828f86d2a0"},"cell_type":"markdown","source":"## Now apply to test dataset"},{"metadata":{"trusted":true,"_uuid":"96253b69627a35527d36bb5650ee27d422e4d3d2"},"cell_type":"code","source":"mdtest = pd.read_csv('../input/metadata_test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51096b222ee5a41eaaa9d419dd76245c86d1600c"},"cell_type":"code","source":"mdtest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6952c501ea9974707cf7562fc95d9910a5c22b16"},"cell_type":"code","source":"start_test = mdtest.signal_id.min()\nend_test = mdtest.signal_id.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b05bf7e8aa024652236e4de8d0c4d21eec8852e4"},"cell_type":"code","source":"X_test = []\n\npool_test = 2000\n\nfor start_col in tqdm(range(start_test, end_test + 1, pool_test)):\n    end_col = min(start_col + pool_test, end_test + 1)\n    #print('cols {}-{}'.format(start_col, end_col-1))\n    test = pq.read_pandas('../input/test.parquet',\n                          columns=[str(c) for c in range(start_col, end_col)]).to_pandas()\n    #print(test.shape)\n    for i in tqdm(test.columns, desc=str(start_test)):\n        X_test.append(feature_extractor(test[i].values, n_part=400))\n        #test.drop([i], axis=1, inplace=True); gc.collect()\n    del test; gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b10d1f4869082e09ffb8d0e6c7852f531325d95"},"cell_type":"code","source":"X_test = np.array(X_test).reshape(-1, X_test[0].shape[0], X_test[0].shape[1])\nX_test = np.transpose(X_test, [0,2,1]) # Make X shape (batch size, channels, time steps) \n# because channels/time-steps are different in Keras","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02f77c6fa44978423bf2c6bbf462a101d0a0b41a"},"cell_type":"code","source":"for i in range(X_test.shape[1]):\n    X_test[:, i, :] = scalers[i].transform(X_test[:, i, :]) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27b707498c61f65153f070957c45d1535dd454df"},"cell_type":"code","source":"X_test_s = squeeze_phases(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5575deff07e2d852f636a61deaf61c0af32f2fad"},"cell_type":"code","source":"testloader = DataLoader(X_test_s, batch_size=100, shuffle=False, num_workers=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca047f9b8f23ea25c7e540d9dad8cda9b4a4f89b"},"cell_type":"code","source":"phasenet.eval()\n\ny_test_preds_s = None\nfor i, X_test_batch in tqdm(enumerate(testloader), total=len(testloader)):\n    y_test_preds_batch = phasenet(torch.as_tensor(X_test_batch, device='cuda', dtype=torch.float32)).detach()\n    if y_test_preds_s is None:\n        y_test_preds_s = y_test_preds_batch\n    else:\n        y_test_preds_s = torch.cat([y_test_preds_s, y_test_preds_batch])\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ec1aedf9c25938e2bef1686b1401e9a29ddd10c"},"cell_type":"code","source":"y_test_preds = y_test_preds_s.view(-1,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b1b2b3a1fe284a3ad5c6a91fe9a1c592b7e320a"},"cell_type":"code","source":"y_test_classes = y_test_preds.cpu() > 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a1e04fbd80f308ac78fa2621751056d6789d704"},"cell_type":"code","source":"submission = pd.read_csv('../input/sample_submission.csv')\nsubmission['signal_id'] = mdtest.signal_id.values\nsubmission['target'] = y_test_classes.data.numpy().astype(int)\nsubmission.to_csv('submission_phases.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d867b2f15d0ee47ae05ada46454935912c05ba3e"},"cell_type":"code","source":"(y_test_classes <= 0).sum().item(), (y_test_classes > 0).sum().item()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42aeb8421db3841e43cbc8dc9314bc9980eec7de"},"cell_type":"code","source":"from IPython.display import FileLink\nFileLink('submission_phases.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6b4470a648bb79e199d63d5b824d5ed77140e71"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ce57c93640f600b9ec265383f1fe9e150ede5bf"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d62252859422dcffe9be24f4962a04302a1582cc"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac26da2896968704cd7537c8ce50526e7878bf48"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"notify_time":"30"},"nbformat":4,"nbformat_minor":1}