{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-12T09:00:32.287746Z","iopub.execute_input":"2023-04-12T09:00:32.288515Z","iopub.status.idle":"2023-04-12T09:00:32.346710Z","shell.execute_reply.started":"2023-04-12T09:00:32.288469Z","shell.execute_reply":"2023-04-12T09:00:32.345670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, gc, pickle, datetime, scipy.sparse\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import TruncatedSVD,PCA\nfrom sklearn.metrics import mean_squared_error\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.sparse\n\ndata_dir = \"../input/open-problems-multimodal\"\ncell_metadata = os.path.join(data_dir,\"metadata.csv\")\n\ncite_train_inputs = os.path.join(data_dir,\"train_cite_inputs.h5\")\ncite_train_targets = os.path.join(data_dir,\"train_cite_targets.h5\")\ncite_test_inputs = os.path.join(data_dir,\"test_cite_inputs.h5\")\nevaluation_ids = os.path.join(data_dir,\"evaluation_ids.csv\")\n\nVERBOSE = 0","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:00:32.348836Z","iopub.execute_input":"2023-04-12T09:00:32.349554Z","iopub.status.idle":"2023-04-12T09:00:34.197391Z","shell.execute_reply.started":"2023-04-12T09:00:32.349513Z","shell.execute_reply":"2023-04-12T09:00:34.195975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\ndf_cite","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:00:34.198881Z","iopub.execute_input":"2023-04-12T09:00:34.199234Z","iopub.status.idle":"2023-04-12T09:01:31.501592Z","shell.execute_reply.started":"2023-04-12T09:00:34.199201Z","shell.execute_reply":"2023-04-12T09:01:31.500438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_citetest = pd.read_hdf('/kaggle/input/open-problems-multimodal/test_cite_inputs.h5')\ndf_citetest","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:01:31.504263Z","iopub.execute_input":"2023-04-12T09:01:31.504696Z","iopub.status.idle":"2023-04-12T09:02:14.136446Z","shell.execute_reply.started":"2023-04-12T09:01:31.504657Z","shell.execute_reply":"2023-04-12T09:02:14.135179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_citetarget = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndf_citetarget","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:02:14.138168Z","iopub.execute_input":"2023-04-12T09:02:14.138533Z","iopub.status.idle":"2023-04-12T09:02:14.913097Z","shell.execute_reply.started":"2023-04-12T09:02:14.138497Z","shell.execute_reply":"2023-04-12T09:02:14.911798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=df_cite.iloc[:,:]\nX_test=df_citetest.iloc[:,:]\nY=df_citetarget.iloc[:,:]\nprint(X.columns,X_test.columns,Y.columns)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:02:14.914557Z","iopub.execute_input":"2023-04-12T09:02:14.914958Z","iopub.status.idle":"2023-04-12T09:02:14.924911Z","shell.execute_reply.started":"2023-04-12T09:02:14.914921Z","shell.execute_reply":"2023-04-12T09:02:14.923487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"constant_cols = list(X.columns[(X == 0).all(axis=0).values]) + list(X_test.columns[(X_test == 0).all(axis=0).values])\nlen(constant_cols)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:02:14.926823Z","iopub.execute_input":"2023-04-12T09:02:14.927207Z","iopub.status.idle":"2023-04-12T09:02:17.685333Z","shell.execute_reply.started":"2023-04-12T09:02:14.927167Z","shell.execute_reply":"2023-04-12T09:02:17.683798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"important_cols = []\n\nfor y_col in Y.columns:\n    important_cols += [x_col for x_col in X.columns if y_col in x_col and x_col not in constant_cols]\nlen(important_cols)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:02:17.687273Z","iopub.execute_input":"2023-04-12T09:02:17.687802Z","iopub.status.idle":"2023-04-12T09:02:18.204153Z","shell.execute_reply.started":"2023-04-12T09:02:17.687720Z","shell.execute_reply":"2023-04-12T09:02:18.202830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.read_csv(cell_metadata, index_col = 'cell_id')\nmetadata_df = metadata_df[metadata_df.technology == \"citeseq\"]\n\nX = pd.read_hdf(cite_train_inputs).drop(columns = constant_cols)\ncell_index = X.index\nmeta = metadata_df.reindex(cell_index)\nX0 = X[important_cols].values\n\ndel X\ngc.collect()\n\nXt = pd.read_hdf(cite_test_inputs).drop(columns = constant_cols)\ncell_index_test = Xt.index\nmeta_test = metadata_df.reindex(cell_index_test)\nX0t = Xt[important_cols].values\n\ndel Xt\ngc.collect()\n\nst = StandardScaler()\nX0 = st.fit_transform(X0)\nX0t = st.transform(X0t)\n\nprint(f'X0 shape {X0.shape} X0t shape {X0t.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:02:18.205657Z","iopub.execute_input":"2023-04-12T09:02:18.206079Z","iopub.status.idle":"2023-04-12T09:03:56.118878Z","shell.execute_reply.started":"2023-04-12T09:02:18.206041Z","shell.execute_reply":"2023-04-12T09:03:56.117511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import scipy.sparse\nfrom sklearn.decomposition import TruncatedSVD","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:03:56.123639Z","iopub.execute_input":"2023-04-12T09:03:56.124488Z","iopub.status.idle":"2023-04-12T09:03:56.130422Z","shell.execute_reply.started":"2023-04-12T09:03:56.124442Z","shell.execute_reply":"2023-04-12T09:03:56.129394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=df_cite.drop(columns = constant_cols)\nX=scipy.sparse.csr_matrix(X.values)\nX_test=df_citetest.drop(columns = constant_cols)\nX_test=scipy.sparse.csr_matrix(X_test.values)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:03:56.131897Z","iopub.execute_input":"2023-04-12T09:03:56.132457Z","iopub.status.idle":"2023-04-12T09:05:27.131765Z","shell.execute_reply.started":"2023-04-12T09:03:56.132422Z","shell.execute_reply":"2023-04-12T09:05:27.130143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnew=scipy.sparse.vstack([X, X_test])\nprint(f\"Shape of dataset before SVD: {new.shape}\")\ntsvd = TruncatedSVD(n_components=128,random_state=1)\nnew=tsvd.fit_transform(new)\nprint(f\"Shape of dataset after SVD:  {new.shape}\")\n#Splitting X and X_test after performing dim reduction\nX = new[:70988] \nX_test = new[70988:]\ndel new\n","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:05:27.133880Z","iopub.execute_input":"2023-04-12T09:05:27.134462Z","iopub.status.idle":"2023-04-12T09:13:50.522498Z","shell.execute_reply.started":"2023-04-12T09:05:27.134403Z","shell.execute_reply":"2023-04-12T09:13:50.520621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = pd.read_hdf(cite_train_targets)\nY = Y.values\nY -= Y.mean(axis=1).reshape(-1, 1)\nY /= Y.std(axis=1).reshape(-1, 1)\nY.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:13:50.524777Z","iopub.execute_input":"2023-04-12T09:13:50.525337Z","iopub.status.idle":"2023-04-12T09:13:51.294631Z","shell.execute_reply.started":"2023-04-12T09:13:50.525269Z","shell.execute_reply":"2023-04-12T09:13:51.293226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.hstack((X[:,:75],X0))\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:13:51.296045Z","iopub.execute_input":"2023-04-12T09:13:51.296388Z","iopub.status.idle":"2023-04-12T09:13:51.345426Z","shell.execute_reply.started":"2023-04-12T09:13:51.296357Z","shell.execute_reply":"2023-04-12T09:13:51.344075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = np.hstack((X_test[:,:75],X0t))\nX_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:13:51.347218Z","iopub.execute_input":"2023-04-12T09:13:51.347572Z","iopub.status.idle":"2023-04-12T09:13:51.393278Z","shell.execute_reply.started":"2023-04-12T09:13:51.347537Z","shell.execute_reply":"2023-04-12T09:13:51.391828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, Concatenate, Dropout, BatchNormalization","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:13:51.394855Z","iopub.execute_input":"2023-04-12T09:13:51.396060Z","iopub.status.idle":"2023-04-12T09:14:02.495237Z","shell.execute_reply.started":"2023-04-12T09:13:51.396012Z","shell.execute_reply":"2023-04-12T09:14:02.493683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)\n\ndef negative_correlation_loss(y_true, y_pred):\n    my = K.mean(tf.convert_to_tensor(y_pred), axis=1)\n    my = tf.tile(tf.expand_dims(my, axis=1), (1, y_true.shape[1]))\n    ym = y_pred - my\n    r_num = K.sum(tf.multiply(y_true, ym), axis=1)\n    r_den = tf.sqrt(K.sum(K.square(ym), axis=1) * float(y_true.shape[-1]))\n    r = tf.reduce_mean(r_num / r_den)\n    return - r","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:14:02.496947Z","iopub.execute_input":"2023-04-12T09:14:02.497769Z","iopub.status.idle":"2023-04-12T09:14:02.509011Z","shell.execute_reply.started":"2023-04-12T09:14:02.497713Z","shell.execute_reply":"2023-04-12T09:14:02.507508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\nLR_START = 0.01\nBATCH_SIZE = 512\ndef create_model():  \n    reg1 = 9.613e-06\n    reg2 = 1e-07\n    REG1 = tf.keras.regularizers.l2(reg1)\n    REG2 = tf.keras.regularizers.l2(reg2)\n    DROP = 0.1\n    activation = 'selu'\n    inputs = Input(shape =(X.shape[1],))\n    x0 = Dense(256, \n              kernel_regularizer = REG1,\n              activation = activation,\n             )(inputs)\n    x0 = Dropout(DROP)(x0)\n    \n    \n    x1 = Dense(512, \n               kernel_regularizer = REG1,\n               activation = activation,\n             )(x0)\n    x1 = Dropout(DROP)(x1)\n    \n    \n    x2 = Dense(512, \n               kernel_regularizer = REG1,\n               activation = activation,\n             )(x1) \n    x2= Dropout(DROP)(x2)\n    \n    x3 = Dense(Y.shape[1],\n               kernel_regularizer = REG1,\n               activation = activation,\n             )(x2)\n    x3 = Dropout(DROP)(x3)\n\n         \n    x = Concatenate()([\n                x0, \n                x1, \n                x2, \n                x3\n                ])\n    x = Dense(Y.shape[1], \n                kernel_regularizer = REG2,\n                activation='linear',\n                )(x)\n    model = Model(inputs, x)\n    \n    return model\ndisplay(plot_model(create_model(), show_layer_names=True, show_shapes=True, dpi=64))","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:22:12.762998Z","iopub.execute_input":"2023-04-12T09:22:12.764524Z","iopub.status.idle":"2023-04-12T09:22:13.464679Z","shell.execute_reply.started":"2023-04-12T09:22:12.764470Z","shell.execute_reply":"2023-04-12T09:22:13.463356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nEPOCHS = 300 \nN_SPLITS = 3\n\npred_train = np.zeros((Y.shape[0],Y.shape[1]))\n\nnp.random.seed(1)\ntf.random.set_seed(1)\nscore_list = []\nkf = GroupKFold(n_splits=N_SPLITS)\nscore_list = []\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, groups=meta.donor)):\n    start_time = datetime.datetime.now()\n    model = None\n    gc.collect()\n    \n    X_tr = X[idx_tr]\n    y_tr = Y[idx_tr]\n    X_va = X[idx_va]\n    y_va = Y[idx_va]\n\n    lr = ReduceLROnPlateau(\n                    monitor = \"val_loss\",\n                    factor = 0.9, \n                    patience = 4, \n                    verbose = VERBOSE)\n\n    es = EarlyStopping(\n                    monitor = \"val_loss\",\n                    patience = 40, \n                    verbose = VERBOSE,\n                    mode = \"min\", \n                    restore_best_weights = True)\n\n    model_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n                    filepath = './citeseq',\n                    save_weights_only = True,\n                    monitor = 'val_loss',\n                    mode = 'min',\n                    save_best_only = True)\n\n    callbacks = [\n                    lr, \n                    es, \n                    model_checkpoint_callback\n                    ]\n    \n    model = create_model()\n    \n    model.compile(\n                optimizer = tf.keras.optimizers.Adam(learning_rate=LR_START),\n                metrics = [negative_correlation_loss],\n                loss = negative_correlation_loss\n                 )\n    # Training\n    history=model.fit(\n                X_tr,\n                y_tr, \n                validation_data=(\n                                X_va,\n                                y_va), \n                epochs = EPOCHS,\n                verbose = VERBOSE,\n                batch_size = BATCH_SIZE,\n                shuffle = True,\n                callbacks = callbacks)\n    plt.plot(history.history['val_loss'])\n    plt.title('model loss')\n    plt.ylabel('negative correlation loss')\n    plt.xlabel('epoch')\n    plt.legend(['validation'], loc='upper left')\n    plt.show()\n    del X_tr, y_tr \n    gc.collect()\n    \n    model.load_weights('./citeseq')\n    model.save(f\"./submissions/model_{fold}\")\n    print('model saved')\n    \n    #  Model validation\n    y_va_pred = model.predict(X_va)\n    corrscore = correlation_score(y_va, y_va_pred)\n    pred_train[idx_va] = y_va_pred\n    \n    \n    print(f'\\n FOLD {fold} ')\n    print(f\"Correlation =  {corrscore:.5f}\")\n    print(f'Mean squared error = {np.round(mean_squared_error(y_va,y_va_pred),2)}')\n    del X_va, y_va, y_va_pred\n    gc.collect()\n    score_list.append(corrscore)\n\n# Show overall score\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Mean corr = {np.array(score_list).mean():.5f}{Style.RESET_ALL}\")\nscore_total = correlation_score(Y, pred_train)\n# print(f\"{Fore.BLUE}{Style.BRIGHT}Oof corr   = {score_total:.5f}{Style.RESET_ALL}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-12T09:44:38.850974Z","iopub.execute_input":"2023-04-12T09:44:38.851482Z","iopub.status.idle":"2023-04-12T10:20:22.004316Z","shell.execute_reply.started":"2023-04-12T09:44:38.851440Z","shell.execute_reply":"2023-04-12T10:20:22.002993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = np.zeros((len(X_test), 140), dtype=np.float32)\nfor fold in range(N_SPLITS):\n    print(f\"Predicting with fold {fold}\")\n    model = load_model(f\"./submissions/model_{fold}\",\n                       custom_objects={'negative_correlation_loss': negative_correlation_loss})\n    test_pred += model.predict(X_test)\n\n# Copy the targets for the data leak but useless since the change in the public LB...\ntest_pred[:7476] = Y[:7476]\n# from Juan Smith Perera to complete with the Multiome part :\nsubmission = pd.read_csv('../input/citeseq-keras-multiome-5x5/submission.csv',index_col='row_id', squeeze=True)\nsubmission.iloc[:len(test_pred.ravel())] = test_pred.ravel()\nassert not submission.isna().any()\n\nsubmission.to_csv('submission_lolo_1.csv')\ndisplay(submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nEPOCHS = 300 \nN_SPLITS = 3\n\npred_train = np.zeros((Y.shape[0],Y.shape[1]))\n\nnp.random.seed(1)\ntf.random.set_seed(1)\nscore_list = []\nkf = GroupKFold(n_splits=N_SPLITS)\nscore_list = []\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, groups=meta.day)):\n    start_time = datetime.datetime.now()\n    model = None\n    gc.collect()\n    \n    X_tr = X[idx_tr]\n    y_tr = Y[idx_tr]\n    X_va = X[idx_va]\n    y_va = Y[idx_va]\n\n    lr = ReduceLROnPlateau(\n                    monitor = \"val_loss\",\n                    factor = 0.9, \n                    patience = 4, \n                    verbose = VERBOSE)\n\n    es = EarlyStopping(\n                    monitor = \"val_loss\",\n                    patience = 40, \n                    verbose = VERBOSE,\n                    mode = \"min\", \n                    restore_best_weights = True)\n\n    model_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n                    filepath = './citeseq',\n                    save_weights_only = True,\n                    monitor = 'val_loss',\n                    mode = 'min',\n                    save_best_only = True)\n\n    callbacks = [\n                    lr, \n                    es, \n                    model_checkpoint_callback\n                    ]\n    \n    model = create_model()\n    \n    model.compile(\n                optimizer = tf.keras.optimizers.Adam(learning_rate=LR_START),\n                metrics = [negative_correlation_loss],\n                loss = negative_correlation_loss\n                 )\n    # Training\n    model.fit(\n                X_tr,\n                y_tr, \n                validation_data=(\n                                X_va,\n                                y_va), \n                epochs = EPOCHS,\n                verbose = VERBOSE,\n                batch_size = BATCH_SIZE,\n                shuffle = True,\n                callbacks = callbacks)\n\n    del X_tr, y_tr \n    gc.collect()\n    \n    model.load_weights('./citeseq')\n    model.save(f\"./submissions/model_{fold}\")\n    print('model saved')\n    \n    #  Model validation\n    y_va_pred = model.predict(X_va)\n    corrscore = correlation_score(y_va, y_va_pred)\n    pred_train[idx_va] = y_va_pred\n    \n    \n    print(f'\\n FOLD {fold} ')\n    print(f\"Correlation =  {corrscore:.5f}\")\n    print(f'Mean squared error = {np.round(mean_squared_error(y_va,y_va_pred),2)}')\n    del X_va, y_va, y_va_pred\n    gc.collect()\n    score_list.append(corrscore)\n\n# Show overall score\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Mean corr = {np.array(score_list).mean():.5f}{Style.RESET_ALL}\")\nscore_total = correlation_score(Y, pred_train)\n# print(f\"{Fore.BLUE}{Style.BRIGHT}Oof corr   = {score_total:.5f}{Style.RESET_ALL}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('../input/targets-multiome-sparse-scaled/INDEX_train_multiome.pkl','rb') as f: INDEX_train_multiome = pickle.load(f)\nwith open('../input/targets-multiome-sparse-scaled/train_512.pkl','rb') as f: X = pickle.load(f)\nwith open('../input/targets-multiome-sparse-scaled/pca_train_512.pkl','rb') as f: pca_train = pickle.load(f)\nwith open('../input/targets-multiome-sparse-scaled/pca_target_512.pkl','rb') as f: pca_target = pickle.load(f)\nwith open('../input/targets-multiome-sparse-scaled/Y_512.pkl','rb') as f: Y = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T10:28:13.714870Z","iopub.execute_input":"2023-04-12T10:28:13.715380Z","iopub.status.idle":"2023-04-12T10:28:23.519558Z","shell.execute_reply.started":"2023-04-12T10:28:13.715335Z","shell.execute_reply":"2023-04-12T10:28:23.518456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X[:,:40]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-12T10:28:23.521239Z","iopub.execute_input":"2023-04-12T10:28:23.522444Z","iopub.status.idle":"2023-04-12T10:28:23.531501Z","shell.execute_reply.started":"2023-04-12T10:28:23.522383Z","shell.execute_reply":"2023-04-12T10:28:23.529996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.read_csv('../input/open-problems-multimodal/metadata.csv',index_col='cell_id')\nmetadata_df = metadata_df[metadata_df.technology==\"multiome\"]\nmeta = metadata_df.reindex(INDEX_train_multiome)","metadata":{"execution":{"iopub.status.busy":"2023-04-12T10:28:23.533039Z","iopub.execute_input":"2023-04-12T10:28:23.533485Z","iopub.status.idle":"2023-04-12T10:28:24.162357Z","shell.execute_reply.started":"2023-04-12T10:28:23.533430Z","shell.execute_reply":"2023-04-12T10:28:24.160841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y.shape, X.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-12T10:28:24.165828Z","iopub.execute_input":"2023-04-12T10:28:24.166394Z","iopub.status.idle":"2023-04-12T10:28:24.175017Z","shell.execute_reply.started":"2023-04-12T10:28:24.166318Z","shell.execute_reply":"2023-04-12T10:28:24.173684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nN_SPLIT = 3\n#kf = KFold(n_splits=N_SPLIT, shuffle=True, random_state=42)\nkf = GroupKFold(n_splits = 3)\n\nfor fold,(idx_tr, idx_va) in enumerate(kf.split(X,groups=meta.donor)):\n    \n    X_tr = X[idx_tr]\n    y_tr = Y[idx_tr]\n    \n    X_va = X[idx_va]\n    y_va = Y[idx_va] \n    \n    model = create_model()\n    \n    lr = ReduceLROnPlateau(\n                monitor = \"val_loss\",\n                factor = 0.9, \n                patience = 4, \n                verbose = VERBOSE)\n    \n    es = EarlyStopping(\n                monitor = \"val_loss\",\n                patience = 30, \n                verbose = VERBOSE,\n                mode = \"min\", \n                restore_best_weights = True)\n\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n                  loss = 'mse',\n                  metrics=None)\n    history1=model.fit(X_tr,\n              y_tr,\n              validation_data=(X_va,y_va),\n              epochs =300,\n              verbose = 1,\n              batch_size=256,\n              callbacks = [es,lr]\n             )\n    plt.plot(history1.history['val_loss'])\n    plt.title('model loss')\n    plt.ylabel('negative correlation loss')\n    plt.xlabel('epoch')\n    plt.legend(['validation'], loc='upper left')\n    plt.show()\n    pred = model.predict(X_va)\n    corrscore = correlation_score(y_va, pred)\n    \n    print(f'\\n FOLD {fold}')\n    print(f\"Correlation =  {corrscore:.5f}\")\n    print(f'Mean squared error = {np.round(mean_squared_error(y_va,pred),2)}')\n   \n    filename = f\"model_{fold}\"\n    model.save(filename)\n    print('model saved :',filename)\n        \n    del X_tr,X_va,y_tr,y_va\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-12T10:29:46.236818Z","iopub.execute_input":"2023-04-12T10:29:46.237300Z","iopub.status.idle":"2023-04-12T13:05:59.542553Z","shell.execute_reply.started":"2023-04-12T10:29:46.237241Z","shell.execute_reply":"2023-04-12T13:05:59.541057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nN_SPLIT = 3\n#kf = KFold(n_splits=N_SPLIT, shuffle=True, random_state=42)\nkf = GroupKFold(n_splits = 3)\n\nfor fold,(idx_tr, idx_va) in enumerate(kf.split(X,groups=meta.day)):\n    \n    X_tr = X[idx_tr]\n    y_tr = Y[idx_tr]\n    \n    X_va = X[idx_va]\n    y_va = Y[idx_va] \n    \n    model = create_model()\n    \n    lr = ReduceLROnPlateau(\n                monitor = \"val_loss\",\n                factor = 0.9, \n                patience = 4, \n                verbose = VERBOSE)\n    \n    es = EarlyStopping(\n                monitor = \"val_loss\",\n                patience = 30, \n                verbose = VERBOSE,\n                mode = \"min\", \n                restore_best_weights = True)\n\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n                  loss = 'mse',\n                  metrics=None)\n    model.fit(X_tr,\n              y_tr,\n              validation_data=(X_va,y_va),\n              epochs =300,\n              verbose = VERBOSE,\n              batch_size=256,\n              callbacks = [es,lr]\n             )\n    pred = model.predict(X_va)\n    corrscore = correlation_score(y_va, pred)\n    \n    print(f'\\n FOLD {fold}')\n    print(f\"Correlation =  {corrscore:.5f}\")\n    print(f'Mean squared error = {np.round(mean_squared_error(y_va,pred),2)}')\n   \n    filename = f\"model_{fold}\"\n    model.save(filename)\n    print('model saved :',filename)\n        \n    del X_tr,X_va,y_tr,y_va\n    gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}