{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load feature and target","metadata":{}},{"cell_type":"code","source":"%%time\nimport numpy as np \nimport pandas as pd \nimport gc\n\nfeature_path = '/kaggle/input/single-cell-features/'\n\ntrain_df = pd.read_feather(feature_path+'train_cite_inputs_id.feather')\ntest_df = pd.read_feather(feature_path+'test_cite_inputs_id.feather')\n\ntrain_cite_X = np.load(feature_path+'train_cite_X.npy')\ntest_cite_X = np.load(feature_path+'test_cite_X.npy')\n\ntrain_cite_y = np.load(feature_path+'train_cite_targets.npy')  ","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:32:36.861634Z","iopub.execute_input":"2022-11-18T00:32:36.862044Z","iopub.status.idle":"2022-11-18T00:32:44.729325Z","shell.execute_reply.started":"2022-11-18T00:32:36.861967Z","shell.execute_reply":"2022-11-18T00:32:44.72821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape,test_df.shape,train_cite_X.shape,test_cite_X.shape,train_cite_y.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:32:44.730654Z","iopub.execute_input":"2022-11-18T00:32:44.73092Z","iopub.status.idle":"2022-11-18T00:32:44.738648Z","shell.execute_reply.started":"2022-11-18T00:32:44.730898Z","shell.execute_reply":"2022-11-18T00:32:44.737772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:32:47.619838Z","iopub.execute_input":"2022-11-18T00:32:47.620177Z","iopub.status.idle":"2022-11-18T00:32:47.633196Z","shell.execute_reply.started":"2022-11-18T00:32:47.620146Z","shell.execute_reply":"2022-11-18T00:32:47.631805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import StratifiedKFold,KFold,GroupKFold\n\ndef correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)\n\ndef zscore(x):\n    x_zscore = []\n    for i in range(x.shape[0]):\n        x_row = x[i]\n        x_row = (x_row - np.mean(x_row)) / np.std(x_row)\n        x_zscore.append(x_row)\n    x_std = np.array(x_zscore)    \n    return x_std\n\ndef cosine_similarity_loss(y_true, y_pred):\n    x = y_true\n    y = y_pred\n    mx = tf.reduce_mean(x, axis=1, keepdims=True)\n    my = tf.reduce_mean(y, axis=1, keepdims=True)\n    xm, ym = x - mx, y - my\n    t1_norm = tf.math.l2_normalize(xm, axis = 1)\n    t2_norm = tf.math.l2_normalize(ym, axis = 1)\n    cosine = tf.keras.losses.CosineSimilarity(axis = 1)(t1_norm, t2_norm)\n    return cosine\n\ndef rotate_vector(data, angle):\n    # source: \n    # https://datascience.stackexchange.com/questions/57226/how-to-rotate-the-plot-and-find-minimum-point    \n    # make rotation matrix\n    theta = np.radians(angle)\n    co = np.cos(theta)\n    si = np.sin(theta)\n    rotation_matrix = np.array(((co, -si), (si, co)))\n    # rotate data vector\n    rotated_vector = data.dot(rotation_matrix)\n    return rotated_vector\n\ndef get_rotated_vector(x, rot_angle):\n    y = range(len(x))\n    d = np.hstack((np.vstack(y), np.vstack(x)))\n    x = rotate_vector(d, rot_angle)[:, 1]\n    return x\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, train_X, train_y, list_IDs, shuffle, batch_size, labels, do_aug: bool = False): \n        self.train_X = train_X\n        self.train_y = train_y\n        self.list_IDs = list_IDs        \n        self.shuffle = shuffle\n        self.batch_size = batch_size\n        self.labels = labels\n        self.do_aug = do_aug\n        self.on_epoch_end()\n        \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = len(self.list_IDs) // self.batch_size\n        return ct\n    \n    def __getitem__(self, idx):\n        'Generate one batch of data'\n        indexes = self.list_IDs[idx*self.batch_size:(idx+1)*self.batch_size]\n    \n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n\n        # Generate data\n        X, y = self.__data_generation(list_IDs_temp)\n        \n        if self.do_aug:\n            idxs = np.arange(X.shape[0])\n            # Flip aug\n            flip_p = 0.1\n            idxs_to_flip = np.random.uniform(0, 1, len(idxs)) < flip_p\n            X[idxs_to_flip] = np.fliplr(X[idxs_to_flip])\n#             # Rotation aug\n#             rot_p = 0.1\n#             rot_angle = np.random.uniform(-2, 2)\n#             rot_angle = np.round(rot_angle, 2)\n#             idxs_to_rot = np.random.uniform(0, 1, len(idxs)) < rot_p\n#             X[idxs_to_rot] = [get_rotated_vector(x, rot_angle) for x in X[idxs_to_rot]]\n\n        if self.labels: return X, y\n        else: return X\n \n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange( len(self.list_IDs) )\n        if self.shuffle: \n            np.random.shuffle(self.indexes)\n            \n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples'    \n        X = self.train_X[list_IDs_temp]\n        y = self.train_y[list_IDs_temp]        \n        return X, y\n    \ndef nn_kfold(train_df, train_cite_X, train_cite_y, test_df, test_cite_X, network, folds, model_name, do_aug=False):\n    oof_preds = np.zeros((train_df.shape[0],140))\n    sub_preds = np.zeros((test_df.shape[0],140))\n    cv_corr = []\n    for n_fold, (train_idx, valid_idx) in enumerate(folds.split(train_df,)):          \n        print (n_fold)\n        train_x = train_cite_X[train_idx]\n        valid_x = train_cite_X[valid_idx]\n        train_y = train_cite_y[train_idx]\n        valid_y = train_cite_y[valid_idx]\n\n        train_x_index = train_df.iloc[train_idx].reset_index(drop=True).index\n        valid_x_index = train_df.iloc[valid_idx].reset_index(drop=True).index\n        \n        model = network(train_cite_X.shape[1])\n        filepath = model_name+'_'+str(n_fold)+'.h5'\n        es = tf.keras.callbacks.EarlyStopping(patience=10, mode='min', verbose=1) \n        checkpoint = tf.keras.callbacks.ModelCheckpoint(monitor='val_loss', filepath=filepath, save_best_only=True,save_weights_only=True,mode='min') \n        reduce_lr_loss = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=LR_FACTOR, patience=6, verbose=1)\n    \n        train_dataset = DataGenerator(\n            train_x,\n            train_y,\n            list_IDs=train_x_index, \n            shuffle=True, \n            batch_size=BATCH_SIZE, \n            labels=True,\n            do_aug=do_aug\n        )\n        \n        valid_dataset = DataGenerator(\n            valid_x,\n            valid_y,\n            list_IDs=valid_x_index, \n            shuffle=False, \n            batch_size=BATCH_SIZE, \n            labels=True,\n        )\n    \n        hist = model.fit(train_dataset,\n                        validation_data=valid_dataset,  \n                        epochs=EPOCHS, \n                        callbacks=[checkpoint,es,reduce_lr_loss],\n                        workers=4,\n                        verbose=1)  \n    \n        model.load_weights(filepath)\n        \n        oof_preds[valid_idx] = model.predict(valid_x, \n                                batch_size=BATCH_SIZE,\n                                verbose=1)\n        \n        oof_corr = correlation_score(valid_y,  oof_preds[valid_idx])\n        cv_corr.append(oof_corr)\n        print (cv_corr)       \n        \n        sub_preds += model.predict(test_cite_X, \n                                batch_size=BATCH_SIZE,\n                                verbose=1) / folds.n_splits \n            \n        del model\n        gc.collect()\n        tf.keras.backend.clear_session()    \n    cv = correlation_score(train_cite_y,  oof_preds)\n    print ('Overall:',cv)           \n    return oof_preds,sub_preds    ","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:33:12.808816Z","iopub.execute_input":"2022-11-18T00:33:12.80915Z","iopub.status.idle":"2022-11-18T00:33:18.505163Z","shell.execute_reply.started":"2022-11-18T00:33:12.809124Z","shell.execute_reply":"2022-11-18T00:33:18.504502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model1 - cosine similarity loss","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom tensorflow.keras.layers import *\n\ndef cite_cos_sim_model(len_num):\n    \n    #######################  svd  #######################   \n    inp = Input(shape=(len_num,))\n    \n    reshape = Reshape((len_num, 1))(inp)\n    x_conv = Conv1D(4, \n                     5,\n                     strides=1,\n                     dilation_rate=3,\n                     activation=\"swish\")(reshape)\n    x_conv = GaussianDropout(0.1)(x_conv)\n    x_conv = Conv1D(4, \n                     4,\n                     strides=2,\n                     dilation_rate=1,\n                     activation=\"swish\")(x_conv)\n    x_conv = Flatten()(x_conv)\n    \n    x = Dense(512, activation=\"swish\")(inp)\n    x = GaussianDropout(0.1)(x)\n    x = Dense(1024, activation=\"swish\")(x)\n    x = Reshape([32,32,1])(x)\n    x_ = Conv2D(4, (5,5), activation=\"swish\", dilation_rate=3, padding=\"same\")(x)\n    x = Conv2D(4, (7,7), activation=\"swish\", padding=\"same\")(x)\n    x = Concatenate()([x_, x])\n    x = GaussianDropout(0.1)(x)\n    x = Conv2D(4, (3,3), activation=\"swish\", padding=\"same\")(x)\n    x = Flatten()(x)\n    \n    x = Concatenate()([x, x_conv])\n    \n    x = Dense(512, activation=\"swish\")(x)\n#     x = BatchNormalization()(x)\n    out = Dense(140, activation=\"linear\")(x)  \n    model = tf.keras.models.Model(inputs=inp, outputs=[out])\n    lr=0.001\n    adam = tf.keras.optimizers.Adam(learning_rate=lr, beta_1=0.9, beta_2=0.999, epsilon=None, )\n    model.compile(loss=cosine_similarity_loss, optimizer=adam,)\n    model.summary()\n    return model\n\n\nBATCH_SIZE = 620\nEPOCHS = 100\nLR_FACTOR = 0.05\nSEED = 666\nN_FOLD = 5\nfolds = KFold(n_splits= N_FOLD, shuffle=True, random_state=SEED)     \noof_preds_cos,sub_preds_cos = nn_kfold(train_df, train_cite_X, train_cite_y,test_df, test_cite_X, cite_cos_sim_model, folds, 'cite_cos_model', do_aug=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:39:05.289054Z","iopub.execute_input":"2022-11-18T00:39:05.289409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model2 - mse loss","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef cite_mse_model(len_num):\n    \n    #######################  svd  #######################   \n    input_num = tf.keras.Input(shape=(len_num))     \n\n    x = input_num\n    x = tf.keras.layers.Dense(1500,activation ='swish',)(x)    \n    x = tf.keras.layers.GaussianDropout(0.1)(x)   \n    x = tf.keras.layers.Dense(1500,activation ='swish',)(x) \n    x = tf.keras.layers.GaussianDropout(0.1)(x)   \n    x = tf.keras.layers.Dense(1500,activation ='swish',)(x) \n    x = tf.keras.layers.GaussianDropout(0.1)(x)    \n    x =  tf.keras.layers.Reshape((1,x.shape[1]))(x)\n    x = tf.keras.layers.Bidirectional(tf.keras.layers.GRU(700, activation='swish',return_sequences=False))(x)\n    x = tf.keras.layers.GaussianDropout(0.1)(x)  \n    \n    output = tf.keras.layers.Dense(140, activation='linear')(x) \n\n    model = tf.keras.models.Model(input_num, output)\n    \n    lr=0.0005\n    weight_decay = 0.0001\n    \n    opt = tfa.optimizers.AdamW(\n        learning_rate=lr, weight_decay=weight_decay\n    )    \n\n    model.compile(loss=tf.keras.losses.MeanSquaredError(), optimizer=opt,)\n    model.summary()\n    return model\n\nBATCH_SIZE = 600\nEPOCHS = 100\nLR_FACTOR = 0.1\nSEED = 666\nfolds = KFold(n_splits= 5, shuffle=True, random_state=SEED)    \n\n# zscore for target\ntrain_cite_y = zscore(train_cite_y)\n\noof_preds_mse,sub_preds_mse = nn_kfold(train_df, train_cite_X, train_cite_y, test_df, test_cite_X, cite_mse_model, folds, 'cite_mse_model')","metadata":{"execution":{"iopub.status.busy":"2022-11-17T14:57:18.177113Z","iopub.execute_input":"2022-11-17T14:57:18.177545Z","iopub.status.idle":"2022-11-17T14:57:22.490789Z","shell.execute_reply.started":"2022-11-17T14:57:18.177493Z","shell.execute_reply":"2022-11-17T14:57:22.489648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted Average","metadata":{}},{"cell_type":"code","source":"%%time\noof_preds_cos = zscore(oof_preds_cos)\noof_preds_mse = zscore(oof_preds_mse)\noof_preds = oof_preds_cos*0.55 + oof_preds_mse*0.45\ncv = correlation_score(train_cite_y,  oof_preds)\nprint ('Blend:',cv)     \n\nsub_preds_cos = zscore(sub_preds_cos)\nsub_preds_mse = zscore(sub_preds_mse)\nsub_preds = sub_preds_cos*0.55 + sub_preds_mse*0.45\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T14:57:22.492223Z","iopub.execute_input":"2022-11-17T14:57:22.492557Z","iopub.status.idle":"2022-11-17T14:57:22.522469Z","shell.execute_reply.started":"2022-11-17T14:57:22.49252Z","shell.execute_reply":"2022-11-17T14:57:22.521317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df,test_df,train_cite_X,test_cite_X,train_cite_y\ndel oof_preds_cos,oof_preds_mse,sub_preds_cos,sub_preds_mse\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T14:57:22.523861Z","iopub.execute_input":"2022-11-17T14:57:22.524301Z","iopub.status.idle":"2022-11-17T14:57:22.560752Z","shell.execute_reply.started":"2022-11-17T14:57:22.524257Z","shell.execute_reply":"2022-11-17T14:57:22.559224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merge Multi part submission","metadata":{}},{"cell_type":"code","source":"%%time\nsub_preds = sub_preds.reshape(-1)\nmulti_preds = np.load(feature_path+'senkin_multi_predictions.npy')\n\nsub = pd.read_csv('../input/open-problems-multimodal/sample_submission.csv')\nsub['target'] = np.concatenate([sub_preds,multi_preds])\nsub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T06:47:13.506268Z","iopub.execute_input":"2022-11-16T06:47:13.506634Z","iopub.status.idle":"2022-11-16T06:47:13.514048Z","shell.execute_reply.started":"2022-11-16T06:47:13.506602Z","shell.execute_reply":"2022-11-16T06:47:13.512823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2022-11-16T06:52:08.477903Z","iopub.execute_input":"2022-11-16T06:52:08.478358Z","iopub.status.idle":"2022-11-16T06:52:08.491587Z","shell.execute_reply.started":"2022-11-16T06:52:08.478303Z","shell.execute_reply":"2022-11-16T06:52:08.490395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}