{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# MSCI Baseline\n\nSimple keras nn baseline that I intend to improve over time to match competitive models.\n\nRely on:\n- My Cite Baseline: \n- Sbunzini work on spare matrices: [this](https://www.kaggle.com/code/sbunzini/reduce-memory-usage-by-95-with-sparse-matrices) and integration in keras: [here](https://www.kaggle.com/code/sbunzini/multiome-simple-nn-with-sparse-matrices)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nimport glob, os, gc\nfrom IPython.core.display import display, HTML\nimport matplotlib.pyplot as plt, seaborn as sns\n\nimport math\nimport scipy\nimport scipy.sparse\n\nfrom sklearn import preprocessing, model_selection\nimport tensorflow as tf\nimport tensorflow_probability as tfp\nfrom tensorflow import keras\nfrom keras import backend as K\n\nnp.random.seed(42)\n\nDEBUG = False\nTEST = True","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:32:53.066142Z","iopub.execute_input":"2022-09-04T09:32:53.066689Z","iopub.status.idle":"2022-09-04T09:33:03.576041Z","shell.execute_reply.started":"2022-09-04T09:32:53.066602Z","shell.execute_reply":"2022-09-04T09:33:03.574245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet tables","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-04T09:33:03.580027Z","iopub.execute_input":"2022-09-04T09:33:03.582597Z","iopub.status.idle":"2022-09-04T09:33:19.200362Z","shell.execute_reply.started":"2022-09-04T09:33:03.582552Z","shell.execute_reply":"2022-09-04T09:33:19.198552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Upgrade Tensorflow to the latest version\n# !conda install -c conda-forge cudatoolkit=11.2 cudnn=8.1.0\n!export LD_LIBRARY_PATH=$LD_LIBRARY_PATH:$CONDA_PREFIX/lib/\n!pip install --upgrade tensorflow","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:33:19.206662Z","iopub.execute_input":"2022-09-04T09:33:19.207100Z","iopub.status.idle":"2022-09-04T09:35:11.984214Z","shell.execute_reply.started":"2022-09-04T09:33:19.207041Z","shell.execute_reply":"2022-09-04T09:35:11.982690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"%%time \n\ntrain = scipy.sparse.load_npz('../input/open-problems-msci-multiome-sparse-matrices/train_multiome_input_sparse.npz')\nlabels = scipy.sparse.load_npz('../input/open-problems-msci-multiome-sparse-matrices/train_multi_targets_sparse.npz')","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:35:11.992921Z","iopub.execute_input":"2022-09-04T09:35:11.996228Z","iopub.status.idle":"2022-09-04T09:36:56.986457Z","shell.execute_reply.started":"2022-09-04T09:35:11.996178Z","shell.execute_reply":"2022-09-04T09:36:56.985246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 50000\n\ntrain = train[:n,:]\nlabels = labels[:n,:]\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:36:56.989317Z","iopub.execute_input":"2022-09-04T09:36:56.990302Z","iopub.status.idle":"2022-09-04T09:37:10.010598Z","shell.execute_reply.started":"2022-09-04T09:36:56.990255Z","shell.execute_reply":"2022-09-04T09:37:10.009215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generator\n\nGenerator from Sbunzini work.","metadata":{}},{"cell_type":"code","source":"class MultiomeSequence(tf.keras.utils.Sequence):\n\n    def __init__(self, x_set, y_set=None, batch_size=64):\n        self.x, self.y = x_set, y_set\n        self.batch_size = batch_size\n\n    def __len__(self):\n        return math.ceil(self.x.shape[0] / self.batch_size)\n\n    def __getitem__(self, idx):\n        \"\"\"\n        Return the idx-th batch\n        \"\"\"\n        batch_x = self.x[idx * self.batch_size:(idx + 1) * self.batch_size]\n        # Convert batch_x to TensorFlow COO Format\n        batch_x = batch_x.tocoo()\n        batch_x.data = batch_x.data.astype(np.float16)\n        pairs = np.column_stack((batch_x.row, batch_x.col)).astype(np.int64)\n        \n        # Convert batch_y to dense\n        if self.y is None:\n            batch_y = None\n        else:\n            batch_y = self.y[idx * self.batch_size:(idx + 1) * self.batch_size].toarray()\n\n        return tf.sparse.to_dense(tf.SparseTensor(indices=pairs, values=batch_x.data, dense_shape=batch_x.shape)), batch_y","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:37:10.012692Z","iopub.execute_input":"2022-09-04T09:37:10.014150Z","iopub.status.idle":"2022-09-04T09:37:10.027567Z","shell.execute_reply.started":"2022-09-04T09:37:10.014074Z","shell.execute_reply":"2022-09-04T09:37:10.025961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom Objective and Metric","metadata":{}},{"cell_type":"markdown","source":"reimplementation of custom correlation (seems better and mine's doesn work) ","metadata":{}},{"cell_type":"code","source":"lam = 0.03\n\n# def correlation_metric(y_true, y_pred):\n#     x = tf.convert_to_tensor(y_true)\n#     y = tf.convert_to_tensor(y_pred)\n#     mx = K.mean(x,axis=1)\n#     my = K.mean(y,axis=1)\n#     mx = tf.tile(tf.expand_dims(mx,axis=1),(1,x.shape[1]))\n#     my = tf.tile(tf.expand_dims(my,axis=1),(1,x.shape[1]))\n#     xm, ym = (x-mx)/100, (y-my)/100\n#     r_num = K.sum(tf.multiply(xm,ym),axis=1)\n#     r_den = tf.sqrt(tf.multiply(K.sum(K.square(xm),axis=1), K.sum(K.square(ym),axis=1)))\n#     r = tf.reduce_mean(r_num / r_den)\n#     r = K.maximum(K.minimum(r, 1.0), -1.0)\n#     return r\n\n# def correlation_loss(y_true, y_pred):\n#     return 1 - correlation_metric(y_true, y_pred) + lam * tf.keras.losses.MeanSquaredError()(tf.convert_to_tensor(y_true),tf.convert_to_tensor(y_pred))\n\nclass CorrelationLoss(tf.keras.losses.Loss):\n    \n    def call(self, y_true, y_pred):\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n        return 1 - tfp.stats.correlation(y_true, y_pred, event_axis=None) + lam * tf.keras.losses.MeanSquaredError()(y_true,y_pred)\n\n    \nclass CorrelationMetric(tf.keras.metrics.Mean):\n    def __init__(self, name='correlation_metric', **kwargs):\n        super(CorrelationMetric, self).__init__(name=name, **kwargs)\n        # self.correlation = self.add_weight(name='corr', initializer='zeros')\n    def update_state(self, y_true, y_pred, **kwargs):\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n        corr = tfp.stats.correlation(y_true, y_pred, event_axis=None)\n        super().update_state(corr, **kwargs)\n        # self.correlation.assign_add(corr)\n    # def result(self):\n    #     return self.correlation","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:37:10.029981Z","iopub.execute_input":"2022-09-04T09:37:10.030464Z","iopub.status.idle":"2022-09-04T09:37:10.044701Z","shell.execute_reply.started":"2022-09-04T09:37:10.030419Z","shell.execute_reply":"2022-09-04T09:37:10.043149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"N_ROWS = train.shape[0]\nN_COLS = train.shape[1]\nN_TARGETS = labels.shape[1]\nnoise = 0.05\n\nhidden_units = (256, 256, 256, 128, 64)\n\ndef base_model():\n    inp = tf.keras.Input(shape=(N_COLS))\n    out = tf.keras.layers.Dense(64, activation='selu')(inp)\n    out = tf.keras.layers.BatchNormalization()(out)\n    out = tf.keras.layers.GaussianNoise(noise)(out)\n    \n    for n_hidden in hidden_units:\n        out = tf.keras.layers.Dense(n_hidden, activation='selu', kernel_regularizer = tf.keras.regularizers.L2(l2=0.005))(out)\n        out = tf.keras.layers.BatchNormalization()(out)\n        out = tf.keras.layers.GaussianNoise(noise)(out)\n        \n    out = tf.keras.layers.Dense(N_TARGETS, activation='selu', name='prediction')(out)\n    \n    model = tf.keras.Model(\n        inputs = inp,\n        outputs = out\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:37:10.046382Z","iopub.execute_input":"2022-09-04T09:37:10.048289Z","iopub.status.idle":"2022-09-04T09:37:10.063329Z","shell.execute_reply.started":"2022-09-04T09:37:10.048244Z","shell.execute_reply":"2022-09-04T09:37:10.061668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CV + Training","metadata":{}},{"cell_type":"code","source":"gc.collect()\n\nepochs = 3 if DEBUG else 1000\nn_folds = 1 if DEBUG else (1 if TEST else 3)\n\nes = tf.keras.callbacks.EarlyStopping(\n    monitor='val_correlation_metric', min_delta=1e-05, patience=5, verbose=1,\n    mode='max', restore_best_weights = True)\n\nplateau = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_correlation_metric', factor=0.2, patience=3, verbose=1,\n    mode='max')\n\nkf = model_selection.ShuffleSplit(n_splits=n_folds, random_state=2022, test_size = 0.3)\nscores_folds = []\n\nfor i, (cal_index, val_index) in enumerate(kf.split(range(train.shape[0]))):\n    print(f'CV {i}/{n_folds}')\n    \n    X_train = train[cal_index]\n    y_train = labels[cal_index]\n    \n    X_test = train[val_index]\n    y_test = labels[val_index]\n    \n    gc.collect()\n    \n    model = base_model()\n    \n    model.compile(\n    tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss = CorrelationLoss(),\n    metrics =  CorrelationMetric(),\n    )\n\n    training_generator = MultiomeSequence(X_train, y_train, batch_size=128)\n    validation_generator = MultiomeSequence(X_test, y_test, batch_size=128)\n    \n    hist = model.fit(\n        x=training_generator,\n        batch_size=64,\n        epochs=epochs,\n        callbacks=[es, plateau],\n        validation_data=validation_generator,\n        shuffle=True,\n        verbose = 1\n    )\n    \n    score = hist.history['correlation_metric'][-1]\n    print(f'Fold {i}: {score:.2%}')\n    \n    scores_folds.append(score)    \n    model.save(f'model_multi_nn_fold_{i}')\n\n    tf.keras.backend.clear_session()\n    del  X_test, y_test,  X_train, y_train, training_generator, validation_generator\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-04T09:37:10.068132Z","iopub.execute_input":"2022-09-04T09:37:10.068586Z","iopub.status.idle":"2022-09-04T10:08:52.087385Z","shell.execute_reply.started":"2022-09-04T09:37:10.068559Z","shell.execute_reply":"2022-09-04T10:08:52.086007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, labels","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:08:52.102885Z","iopub.execute_input":"2022-09-04T10:08:52.103768Z","iopub.status.idle":"2022-09-04T10:08:52.307828Z","shell.execute_reply.started":"2022-09-04T10:08:52.103717Z","shell.execute_reply":"2022-09-04T10:08:52.305777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nevaluation_ids = pd.read_csv('../input/open-problems-multimodal/evaluation_ids.csv').set_index('row_id')\nunique_ids = np.unique(evaluation_ids.cell_id)\nsubmission = pd.Series(name='target', index=pd.MultiIndex.from_frame(evaluation_ids), dtype=np.float16)\n\ndel evaluation_ids\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:08:52.310125Z","iopub.execute_input":"2022-09-04T10:08:52.310668Z","iopub.status.idle":"2022-09-04T10:11:15.477188Z","shell.execute_reply.started":"2022-09-04T10:08:52.310625Z","shell.execute_reply":"2022-09-04T10:11:15.475653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_multi = tf.keras.models.load_model('./model_multi_nn_fold_0/', compile=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:11:15.480979Z","iopub.execute_input":"2022-09-04T10:11:15.482307Z","iopub.status.idle":"2022-09-04T10:11:16.751581Z","shell.execute_reply.started":"2022-09-04T10:11:15.482244Z","shell.execute_reply":"2022-09-04T10:11:16.750349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nchunksize = 1000\nstart = 0\nend = 55934\nnb_it = int(np.ceil(end/chunksize))\n\nmeta_data = pd.read_csv('../input/open-problems-multimodal/metadata.csv')\nmap_cat = { 'BP':0, 'EryP':1, 'HSC':2, 'MasP':3, 'MkP':4, 'MoP':5, 'NeuP':6, 'hidden':7}\ncolumns = pd.read_hdf(\"../input/open-problems-multimodal/train_multi_targets.h5\", start=0, stop=0).astype('float16').columns\n\nfor i in np.arange(nb_it):\n    \n    if DEBUG:\n        if i>1:\n            continue\n    \n    print(f'{i/nb_it:.2%}')\n    start_it = chunksize*i\n\n    print(f'start: {start_it}')\n    print(f'end: {min(start_it + chunksize, end)}')\n    \n    test = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/test_multi_inputs.h5\", start=start_it, stop=min(start_it + chunksize, end)).astype('float16')\n    gc.collect()\n    \n    test = test[test.index.isin(unique_ids)]\n\n    test_index = test.index\n    test_meta_data = meta_data.set_index('cell_id').loc[test_index]\n\n    test = test.values\n#     test_cat =  test_meta_data.cell_type.values\n#     test_cat = np.array([map_cat[t] for t in test_cat])\n\n    test_pred = model_multi.predict(test)\n\n    test_pred = pd.DataFrame(test_pred,index=test_index,columns=columns).astype('float16').stack()\n\n    sub = submission.loc[np.unique(test_pred.index.get_level_values(0))]\n    sub = sub.to_frame().merge(test_pred.to_frame(),left_index=True,right_index=True,how='left')\n\n    sub.target = sub[0]\n    sub = sub.drop(columns=[0])\n\n    submission.loc[sub.index] = sub.target\n    \n    del test, test_pred, sub\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:11:16.756178Z","iopub.execute_input":"2022-09-04T10:11:16.756564Z","iopub.status.idle":"2022-09-04T10:35:30.192422Z","shell.execute_reply.started":"2022-09-04T10:11:16.756533Z","shell.execute_reply":"2022-09-04T10:35:30.190777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_cite = pd.read_csv('../input/msci-citeseq-tf-keras-nn-custom-loss/submission_cite.csv', nrows=6812820).target\nsubmission.iloc[:len(submission_cite)] = submission_cite","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:35:30.194667Z","iopub.execute_input":"2022-09-04T10:35:30.196290Z","iopub.status.idle":"2022-09-04T10:35:34.834661Z","shell.execute_reply.started":"2022-09-04T10:35:30.196240Z","shell.execute_reply":"2022-09-04T10:35:34.833267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.reset_index(drop=True).reset_index().rename(columns={'index':'row_id'})\nsubmission.fillna(0).to_csv('submission_all.csv',index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-04T10:35:34.836649Z","iopub.execute_input":"2022-09-04T10:35:34.837192Z","iopub.status.idle":"2022-09-04T10:37:33.592308Z","shell.execute_reply.started":"2022-09-04T10:35:34.837158Z","shell.execute_reply":"2022-09-04T10:37:33.590933Z"},"trusted":true},"execution_count":null,"outputs":[]}]}