{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# MSCI Baseline\n\nSimple keras nn baseline that I intend to improve over time to match competitive models.\n\nRely on:\n- My Cite Baseline: \n- Sbunzini work on spare matrices: [this](https://www.kaggle.com/code/sbunzini/reduce-memory-usage-by-95-with-sparse-matrices) and integration in keras: [here](https://www.kaggle.com/code/sbunzini/multiome-simple-nn-with-sparse-matrices)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nimport glob, os, gc\n\n\nimport math\nimport scipy\nimport scipy.sparse\n\nfrom sklearn import preprocessing, model_selection\nimport tensorflow as tf\nimport tensorflow_probability as tfp\nfrom tensorflow import keras\nfrom keras import backend as K\n\nnp.random.seed(42)\n\nDEBUG = False\nTEST = True","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:06:33.293303Z","iopub.execute_input":"2022-11-07T06:06:33.293744Z","iopub.status.idle":"2022-11-07T06:06:43.174881Z","shell.execute_reply.started":"2022-11-07T06:06:33.293708Z","shell.execute_reply":"2022-11-07T06:06:43.173975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet tables","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-07T06:06:43.176966Z","iopub.execute_input":"2022-11-07T06:06:43.178447Z","iopub.status.idle":"2022-11-07T06:06:58.531192Z","shell.execute_reply.started":"2022-11-07T06:06:43.178385Z","shell.execute_reply":"2022-11-07T06:06:58.529791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Upgrade Tensorflow to the latest version\n# !conda install -c conda-forge cudatoolkit=11.2 cudnn=8.1.0\n!export LD_LIBRARY_PATH=$LD_LIBRARY_PATH:$CONDA_PREFIX/lib/\n# !pip install --upgrade tensorflow","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:06:58.533469Z","iopub.execute_input":"2022-11-07T06:06:58.533879Z","iopub.status.idle":"2022-11-07T06:06:59.641319Z","shell.execute_reply.started":"2022-11-07T06:06:58.533843Z","shell.execute_reply":"2022-11-07T06:06:59.639518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SVD = \"../input/binarized-data-for-multiome-part\"","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:06:59.642931Z","iopub.execute_input":"2022-11-07T06:06:59.643342Z","iopub.status.idle":"2022-11-07T06:06:59.649858Z","shell.execute_reply.started":"2022-11-07T06:06:59.643306Z","shell.execute_reply":"2022-11-07T06:06:59.648222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"%%time \nimport pickle\n# train = scipy.sparse.load_npz('../input/open-problems-msci-multiome-sparse-matrices/train_multiome_input_sparse.npz')\nwith open(f\"{INPUT_SVD}/multiome_train_x.pickle\", \"rb\") as f:\n    train = pickle.load(f)\nlabels = scipy.sparse.load_npz('../input/open-problems-msci-multiome-sparse-matrices/train_multi_targets_sparse.npz')","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:06:59.653673Z","iopub.execute_input":"2022-11-07T06:06:59.654065Z","iopub.status.idle":"2022-11-07T06:07:27.748494Z","shell.execute_reply.started":"2022-11-07T06:06:59.654032Z","shell.execute_reply":"2022-11-07T06:07:27.747094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.shape)\nprint(labels.dtype)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:27.750089Z","iopub.execute_input":"2022-11-07T06:07:27.750519Z","iopub.status.idle":"2022-11-07T06:07:27.757816Z","shell.execute_reply.started":"2022-11-07T06:07:27.750483Z","shell.execute_reply":"2022-11-07T06:07:27.756332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 120000\nnp.random.seed(1234)\nchuck_split = 23\nchuck_size = 23418//chuck_split\nchuck_number = 0\n\nall_row_indices = np.arange(train.shape[0])\nnp.random.shuffle(all_row_indices)\nselected_rows_indices = all_row_indices[: n]\n\ntrain = train[selected_rows_indices,:]\n# labels = labels[selected_rows_indices, chuck_number*chuck_size:(chuck_number + 1)*chuck_size]\nlabels = labels[selected_rows_indices]\n# labels = labels[selected_rows_indices]\ndel selected_rows_indices\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:27.759989Z","iopub.execute_input":"2022-11-07T06:07:27.760445Z","iopub.status.idle":"2022-11-07T06:07:29.576129Z","shell.execute_reply.started":"2022-11-07T06:07:27.760387Z","shell.execute_reply":"2022-11-07T06:07:29.574730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generator\n\nGenerator from Sbunzini work.","metadata":{}},{"cell_type":"code","source":"class MultiomeSequence(tf.keras.utils.Sequence):\n\n    def __init__(self, x_set, y_set=None, non_zero_indices=None, batch_size=64):\n        self.x, self.y = x_set, y_set\n        self.non_zero_indices = non_zero_indices\n        self.batch_size = batch_size\n\n    def __len__(self):\n        return math.ceil(self.x.shape[0] / self.batch_size)\n\n    def __getitem__(self, idx):\n        \"\"\"\n        Return the idx-th batch\n        \"\"\"\n        gc.collect()\n        batch_x = self.x[idx * self.batch_size:(idx + 1) * self.batch_size]\n        # Convert batch_x to TensorFlow COO Format\n#         batch_x = batch_x.tocoo()\n#         batch_x.data = batch_x.data.astype(np.float16)\n#         pairs = np.column_stack((batch_x.row, batch_x.col)).astype(np.int64)\n        \n        # Convert batch_y to dense\n        if self.y is None:\n            batch_y = None\n        else:\n            batch_y = self.y[idx * self.batch_size:(idx + 1) * self.batch_size].toarray()\n        if self.non_zero_indices is not None:\n            batch_y = batch_y[:, self.non_zero_indices[0]]\n#         global N_TARGETS\n#         bat\n        return batch_x, batch_y\n#         return tf.sparse.to_dense(tf.SparseTensor(indices=pairs, values=batch_x.data, dense_shape=batch_x.shape)), batch_y","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:29.577856Z","iopub.execute_input":"2022-11-07T06:07:29.578290Z","iopub.status.idle":"2022-11-07T06:07:29.588420Z","shell.execute_reply.started":"2022-11-07T06:07:29.578241Z","shell.execute_reply":"2022-11-07T06:07:29.587145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom Objective and Metric","metadata":{}},{"cell_type":"markdown","source":"reimplementation of custom correlation (seems better and mine's doesn work) ","metadata":{}},{"cell_type":"code","source":"lam = 0.05\n\n# def correlation_metric(y_true, y_pred):\n#     x = tf.convert_to_tensor(y_true)\n#     y = tf.convert_to_tensor(y_pred)\n#     mx = K.mean(x,axis=1)\n#     my = K.mean(y,axis=1)\n#     mx = tf.tile(tf.expand_dims(mx,axis=1),(1,x.shape[1]))\n#     my = tf.tile(tf.expand_dims(my,axis=1),(1,x.shape[1]))\n#     xm, ym = (x-mx)/100, (y-my)/100\n#     r_num = K.sum(tf.multiply(xm,ym),axis=1)\n#     r_den = tf.sqrt(tf.multiply(K.sum(K.square(xm),axis=1), K.sum(K.square(ym),axis=1)))\n#     r = tf.reduce_mean(r_num / r_den)\n#     r = K.maximum(K.minimum(r, 1.0), -1.0)\n#     return r\n\n# def correlation_loss(y_true, y_pred):\n#     return 1 - correlation_metric(y_true, y_pred) + lam * tf.keras.losses.MeanSquaredError()(tf.convert_to_tensor(y_true),tf.convert_to_tensor(y_pred))\n\nclass CorrelationLoss(tf.keras.losses.Loss):\n    \n    def call(self, y_true, y_pred):\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n        return (\n            1 - tfp.stats.correlation(y_true, y_pred, event_axis=None)\n             + lam * tf.keras.losses.MeanAbsoluteError()(y_true,y_pred)\n        )\n\n    \nclass CorrelationMetric(tf.keras.metrics.Mean):\n    def __init__(self, name='correlation_metric', **kwargs):\n        super(CorrelationMetric, self).__init__(name=name, **kwargs)\n        # self.correlation = self.add_weight(name='corr', initializer='zeros')\n    def update_state(self, y_true, y_pred, **kwargs):\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n        corr = tfp.stats.correlation(y_true, y_pred, event_axis=None)\n        super().update_state(corr, **kwargs)\n        # self.correlation.assign_add(corr)\n    # def result(self):\n    #     return self.correlation\n    \ndef negative_correlation_loss(y_true, y_pred):\n    \"\"\"Negative correlation loss function for Keras\n    \n    Precondition:\n    y_true.mean(axis=1) == 0\n    y_true.std(axis=1) == 1\n    \n    Returns:\n    -1 = perfect positive correlation\n    1 = totally negative correlation\n    \"\"\"\n    my = K.mean(tf.convert_to_tensor(y_pred), axis=1)\n    my = tf.tile(tf.expand_dims(my, axis=1), (1, y_true.shape[1]))\n    ym = y_pred - my\n    r_num = K.sum(tf.multiply(y_true, ym), axis=1)\n    r_den = tf.sqrt(K.sum(K.square(ym), axis=1) * float(y_true.shape[-1]))\n    r = tf.reduce_mean(r_num / r_den)\n    return - r","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:29.590354Z","iopub.execute_input":"2022-11-07T06:07:29.591164Z","iopub.status.idle":"2022-11-07T06:07:29.608849Z","shell.execute_reply.started":"2022-11-07T06:07:29.591128Z","shell.execute_reply":"2022-11-07T06:07:29.607088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"N_ROWS = train.shape[0]\nN_COLS = train.shape[1]\nN_TARGETS = labels.shape[1]\nnoise = 0.1/4\n\nhidden_units = (1024, ) * 5\n# hidden_units = (365, 731, 1463, 2927, 5854, 11709)\n# hidden_units = (2927, 5854, 11709)\nactivation = tf.keras.layers.ReLU(\n    max_value=13, negative_slope=0.0, threshold=0.0\n)\ndef base_model():\n    inp = tf.keras.Input(shape=(N_COLS))\n    \n    out = tf.keras.layers.BatchNormalization()(inp)\n    out = tf.keras.layers.LayerNormalization(axis=1)(out)\n    out = tf.keras.layers.GaussianNoise(noise)(out)\n    \n    for n_hidden in hidden_units:\n        out = tf.keras.layers.Dense(\n            n_hidden, activation=activation, kernel_regularizer = tf.keras.regularizers.L2(l2=0.05/4))(out)\n        out = tf.keras.layers.BatchNormalization()(out)\n        out = tf.keras.layers.GaussianNoise(noise)(out)\n\n    out = tf.keras.layers.Dense(N_TARGETS, activation=activation, name='prediction')(out)\n#     out = tf.keras.layers.LayerNormalization(axis=1)(out)\n\n    model = tf.keras.Model(\n        inputs = inp,\n        outputs = out\n    )\n    \n    return model\n\n\n# def base_model():\n#     inp = tf.keras.Input(shape=(N_COLS))\n#     out = tf.keras.layers.Reshape((N_COLS,1,1))(inp)\n#     out = tf.keras.layers.Conv1D(32, 1, activation='relu', input_shape=(N_COLS,1,1))(out)\n#     out = tf.keras.layers.Reshape((-1, 1))(out)\n#     out = tf.keras.layers.MaxPooling1D(pool_size=2, strides=1, padding='same')(out)\n    \n# #     out = tf.keras.layers.Conv1D(64, 1, activation='relu')(out)\n# #     out = tf.keras.layers.Reshape((-1, 1))(out)\n# #     out = tf.keras.layers.MaxPooling1D(pool_size=2, strides=1, padding='same')(out)\n    \n#     out = tf.keras.layers.Dense(N_TARGETS, activation='relu', name='prediction')(out)\n#     model = tf.keras.Model(\n#         inputs = inp,\n#         outputs = out\n#     )\n    \n#     return model\n\n# def base_model():\n#     inp = tf.keras.Input(shape=(N_COLS))\n#     out = tf.keras.layers.Reshape((N_COLS,1,1))(inp)\n#     out = tf.keras.layers.Conv1D(32, 1, activation='relu', input_shape=(N_COLS,1,1))(out)\n    \n#     out = tf.keras.layers.Dense(N_TARGETS, activation='relu', name='prediction')(out)\n#     out = tf.keras.layers.LayerNormalization(axis=1)(out)\n#     model = tf.keras.Model(\n#         inputs = inp,\n#         outputs = out\n#     )\n    \n#     return model","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:29.611070Z","iopub.execute_input":"2022-11-07T06:07:29.611793Z","iopub.status.idle":"2022-11-07T06:07:29.658594Z","shell.execute_reply.started":"2022-11-07T06:07:29.611686Z","shell.execute_reply":"2022-11-07T06:07:29.657576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CV + Training","metadata":{}},{"cell_type":"code","source":"model_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=\"checkpoint\",\n    save_weights_only=False,\n    monitor='val_correlation_metric',\n    mode='max',\n    save_best_only=True,\n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:29.660161Z","iopub.execute_input":"2022-11-07T06:07:29.660851Z","iopub.status.idle":"2022-11-07T06:07:29.665914Z","shell.execute_reply.started":"2022-11-07T06:07:29.660814Z","shell.execute_reply":"2022-11-07T06:07:29.665017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\nepochs = 30 if DEBUG else 1000\nepochs = 1000\nn_folds = 1 if DEBUG else (1 if TEST else 3)\nFOCUS_FOLD = 1\nes = tf.keras.callbacks.EarlyStopping(\n    monitor='val_correlation_metric', min_delta=1e-05, patience=15, verbose=1,\n    mode='max')\n\nplateau = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_correlation_metric', factor=0.4, patience=3, verbose=1,\n    mode='max')\n\nkf = model_selection.ShuffleSplit(n_splits=n_folds, random_state=2022, test_size = 0.2)\n# kf = model_selection.StratifiedShuffleSplit(n_splits=n_folds, random_state=2022, test_size = 0.3)\nscores_folds = []\n\nfor i, (cal_index, val_index) in enumerate(kf.split(range(train.shape[0])), 1):\n    if i == FOCUS_FOLD:\n        print(f'CV {i}/{n_folds}')\n\n        X_train = train[cal_index]\n        y_train = labels[cal_index]\n        \n        X_test = train[val_index]\n        y_test = labels[val_index]\n\n        gc.collect()\n\n        model = base_model()\n        print(model.summary())\n        model.compile(\n            tf.keras.optimizers.Nadam(learning_rate=1e-4),\n            loss = CorrelationLoss(),\n            metrics =  CorrelationMetric(),\n        )\n\n        training_generator = MultiomeSequence(X_train, y_train, None, batch_size=256//2)\n        validation_generator = MultiomeSequence(X_test, y_test, None, batch_size=256//2)\n\n        hist = model.fit(\n            x=training_generator,\n            batch_size=64,\n            epochs=epochs,\n            callbacks=[es, plateau, model_checkpoint_callback],\n            validation_data=validation_generator,\n            shuffle=True,\n            verbose = 1\n        )\n\n        score = hist.history['correlation_metric'][-1]\n        print(f'Fold {i}: {score:.2%}')\n\n        scores_folds.append(score)    \n        model.save(f'model_multi_nn_fold_{i}')\n\n        tf.keras.backend.clear_session()\n        ","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:07:29.668470Z","iopub.execute_input":"2022-11-07T06:07:29.669639Z","iopub.status.idle":"2022-11-07T06:15:04.789578Z","shell.execute_reply.started":"2022-11-07T06:07:29.669597Z","shell.execute_reply":"2022-11-07T06:15:04.788197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model(\n    \"checkpoint\",\n     custom_objects={\n         'CorrelationLoss': CorrelationLoss,\n         'CorrelationMetric': CorrelationMetric\n     }\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:04.797562Z","iopub.execute_input":"2022-11-07T06:15:04.798305Z","iopub.status.idle":"2022-11-07T06:15:06.682174Z","shell.execute_reply.started":"2022-11-07T06:15:04.798240Z","shell.execute_reply":"2022-11-07T06:15:06.681019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# zero_indices","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.703675Z","iopub.execute_input":"2022-11-07T06:15:06.704088Z","iopub.status.idle":"2022-11-07T06:15:06.709521Z","shell.execute_reply.started":"2022-11-07T06:15:06.704055Z","shell.execute_reply":"2022-11-07T06:15:06.707977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testtest = np.array([1,2,3])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.711396Z","iopub.execute_input":"2022-11-07T06:15:06.711811Z","iopub.status.idle":"2022-11-07T06:15:06.726446Z","shell.execute_reply.started":"2022-11-07T06:15:06.711750Z","shell.execute_reply":"2022-11-07T06:15:06.725466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testtest[[True, False, True]] = [5,6]","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.727589Z","iopub.execute_input":"2022-11-07T06:15:06.728717Z","iopub.status.idle":"2022-11-07T06:15:06.738331Z","shell.execute_reply.started":"2022-11-07T06:15:06.728667Z","shell.execute_reply":"2022-11-07T06:15:06.737451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testtest","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.739875Z","iopub.execute_input":"2022-11-07T06:15:06.740700Z","iopub.status.idle":"2022-11-07T06:15:06.752110Z","shell.execute_reply.started":"2022-11-07T06:15:06.740651Z","shell.execute_reply":"2022-11-07T06:15:06.750751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_train.max(axis=0).todense()[y_train.max(axis=0).todense() == 0]","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.754353Z","iopub.execute_input":"2022-11-07T06:15:06.755017Z","iopub.status.idle":"2022-11-07T06:15:06.767184Z","shell.execute_reply.started":"2022-11-07T06:15:06.754956Z","shell.execute_reply":"2022-11-07T06:15:06.766129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, labels","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.768622Z","iopub.execute_input":"2022-11-07T06:15:06.769367Z","iopub.status.idle":"2022-11-07T06:15:06.787672Z","shell.execute_reply.started":"2022-11-07T06:15:06.769324Z","shell.execute_reply":"2022-11-07T06:15:06.786313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del  X_test, y_test,  X_train, y_train, training_generator, validation_generator\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:06.789474Z","iopub.execute_input":"2022-11-07T06:15:06.790209Z","iopub.status.idle":"2022-11-07T06:15:07.178715Z","shell.execute_reply.started":"2022-11-07T06:15:06.790155Z","shell.execute_reply":"2022-11-07T06:15:07.177412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nevaluation_ids = pd.read_csv('../input/open-problems-multimodal/evaluation_ids.csv').set_index('row_id')\nunique_ids = np.unique(evaluation_ids.cell_id)\nsubmission = pd.Series(name='target', index=pd.MultiIndex.from_frame(evaluation_ids), dtype=np.float16)\n\ndel evaluation_ids\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:15:07.180374Z","iopub.execute_input":"2022-11-07T06:15:07.180864Z","iopub.status.idle":"2022-11-07T06:17:54.783040Z","shell.execute_reply.started":"2022-11-07T06:15:07.180817Z","shell.execute_reply":"2022-11-07T06:17:54.781736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_multi = model","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:17:54.784766Z","iopub.execute_input":"2022-11-07T06:17:54.785139Z","iopub.status.idle":"2022-11-07T06:17:54.791337Z","shell.execute_reply.started":"2022-11-07T06:17:54.785105Z","shell.execute_reply":"2022-11-07T06:17:54.789896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nchunksize = 1000\nstart = 0\nend = 55934 + 1\nnb_it = int(np.ceil(end/chunksize))\noutput_path = \"partial_submission.csv\"\n\nmeta_data = pd.read_csv('../input/open-problems-multimodal/metadata.csv')\nmap_cat = { 'BP':0, 'EryP':1, 'HSC':2, 'MasP':3, 'MkP':4, 'MoP':5, 'NeuP':6, 'hidden':7}\ncolumns = pd.read_hdf(\"../input/open-problems-multimodal/train_multi_targets.h5\", start=0, stop=0).astype('float16').columns\n\nwith open(f\"{INPUT_SVD}/pca.pickle\", \"rb\") as f:\n    pca = pickle.load(f)\n    \nfor i in np.arange(nb_it):\n    \n    if DEBUG:\n        if i>1:\n            continue\n    \n    print(f'{i/nb_it:.2%}')\n    start_it = chunksize*i\n\n    print(f'start: {start_it}')\n    print(f'end: {min(start_it + chunksize, end)}')\n    \n    test = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/test_multi_inputs.h5\", start=start_it, stop=min(start_it + chunksize, end)).astype('float16')\n    gc.collect()\n#     print(test.values.shape)\n    test = test[test.index.isin(unique_ids)]\n\n    test_index = test.index\n    test_meta_data = meta_data.set_index('cell_id').loc[test_index]\n#     print(test.values.shape)\n\n    test = test.values\n    test = test/18.76196\n    test = np.ceil(test)\n    print(\"predict...\")\n    test_pred = model.predict(pca.transform(test))\n\n\n\n    test_pred = pd.DataFrame(test_pred,index=test_index,columns=columns).astype('float16').stack()\n    test_pred.to_csv(output_path, mode='a', header=not os.path.exists(output_path))\n    sub = submission.loc[np.unique(test_pred.index.get_level_values(0))]\n    sub = sub.to_frame().merge(test_pred.to_frame(),left_index=True,right_index=True,how='left')\n\n    sub.target = sub[0]\n    sub = sub.drop(columns=[0])\n\n    submission.loc[sub.index] = sub.target\n    \n    del test, test_pred, sub\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-07T06:17:54.793339Z","iopub.execute_input":"2022-11-07T06:17:54.793846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_cite = pd.read_csv('../input/msci-citeseq-tf-keras-nn-custom-loss/submission_cite.csv', nrows=6812820).target\nsubmission.iloc[:len(submission_cite)] = submission_cite","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.reset_index(drop=True).reset_index().rename(columns={'index':'row_id'})\nsubmission.fillna(0).to_csv('submission_all.csv',index=False)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}