{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\n\n# define numpy seed for reproducability\n#np.random.seed(100)\n\ntrain_filename = '/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv'\n\n# parameters before loading data\nN_ROWS = 806_573 # from competation page (https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data?select=train_data.csv)\nCHUNKSIZE = 400_000\nSKIPROWS = np.random.randint(1, N_ROWS-CHUNKSIZE)\n\n# load data\ntrain_df = pd.read_csv(train_filename, skiprows=0, chunksize=1).get_chunk()\ntrain_df = pd.read_csv(train_filename, skiprows=SKIPROWS, chunksize=CHUNKSIZE, names=train_df.keys()).get_chunk()\n\nfor i,j in zip(['A','G','U','C'], ['1','2','3','4']):\n    train_df['sequence'] = train_df['sequence'].str.replace(i,j)\n#train_df['sequence'] = train_df['sequence'].astype('uint8')\n\n# load data (full data)\n#train_filename = '/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv'\n#train_df = pd.read_csv(train_filename)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:20:41.167058Z","iopub.execute_input":"2023-09-20T11:20:41.167456Z","iopub.status.idle":"2023-09-20T11:21:00.172947Z","shell.execute_reply.started":"2023-09-20T11:20:41.167428Z","shell.execute_reply":"2023-09-20T11:21:00.171703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fill list with index of reactivity and reactivity error from dataframe","metadata":{}},{"cell_type":"code","source":"react_list = []\nreact_err_list = []\nfor i,k in enumerate(train_df.keys()):\n    if 'reactivity' in k and 'error' not in k:\n        react_list.append(i)\n    elif 'reactivity_error' in k:\n        react_err_list.append(i)\n\n#print(react_list)\n#print(react_err_list)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:21:04.251135Z","iopub.execute_input":"2023-09-20T11:21:04.252128Z","iopub.status.idle":"2023-09-20T11:21:04.261132Z","shell.execute_reply.started":"2023-09-20T11:21:04.252076Z","shell.execute_reply":"2023-09-20T11:21:04.258991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fill list with inputs (X_train) and targets (y_train)","metadata":{}},{"cell_type":"code","source":"inputs_length = 457\ninputs = []\ntargets = []\n\n# fill list with type of experiment (DMS_MaP or 2A3_MaP)\nexp_type = []\nexp_type_uniques = ['DMS_MaP', '2A3_MaP']\n\n# loop through dataframe\nfor i, row in train_df.iterrows():\n\n    # get reactivities \n    r = np.array(row[react_list].values).astype('float')\n\n    # get sequence\n    seq = row.sequence#.replace('A', '1').replace('G', '2').replace('U', '3').replace('C', '4')\n    seq = np.array([*seq]).astype('int')\n    \n    # store input values\n    #input = np.zeros(len(r)).astype('int')\n    input = np.zeros(inputs_length).astype('int')\n    input[:len(seq)] = seq\n\n    # store target values\n    target = np.zeros(inputs_length).astype('int')\n    target[:len(r)] = np.nan_to_num(r).astype('int')\n    wt = np.where(target)[0]\n    if len(wt)>0:\n        target = np.roll(target, -wt[0])\n\n    # clip target values between 0 and 1\n    target[np.where(target<0)] = 0\n    target[np.where(target>1)] = 1\n\n    # append values\n    exp_type.append(row.experiment_type)\n    inputs.append(input)\n    targets.append(target)\n\n# convert lists to numpy arrays\nexp_type = np.array(exp_type)\ninputs = np.array(inputs)\ntargets = np.array(targets)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:21:07.922689Z","iopub.execute_input":"2023-09-20T11:21:07.923058Z","iopub.status.idle":"2023-09-20T11:24:24.837173Z","shell.execute_reply.started":"2023-09-20T11:21:07.923035Z","shell.execute_reply":"2023-09-20T11:24:24.834970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define and train ML models","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import binary_crossentropy\n\n# parameters to tune models\nEPOCHS = 1\nBATCH_SIZE = 5\nINPUT_SIZE = inputs.shape[1]\nADAM_VAL = 0.0001\n\n# model on DMS maps\nmodel_DMS = Sequential()\nmodel_DMS.add(Dense(units=32, activation='relu', input_dim=INPUT_SIZE))\nmodel_DMS.add(Dense(units=64, activation='relu'))\nmodel_DMS.add(Dense(units=inputs.shape[1], activation='sigmoid'))\nmodel_DMS.compile(optimizer=Adam(ADAM_VAL), loss=binary_crossentropy, metrics='mae')\n\n# model on 2A3 maps\nmodel_2A3 = Sequential()\nmodel_2A3.add(Dense(units=32, activation='relu', input_dim=inputs.shape[1]))\nmodel_2A3.add(Dense(units=64, activation='relu'))\nmodel_2A3.add(Dense(units=inputs.shape[1], activation='sigmoid'))\nmodel_2A3.compile(optimizer=Adam(ADAM_VAL), loss=binary_crossentropy, metrics='mae')\n\n# fit models\nwhere_DMS = np.where(exp_type == exp_type_uniques[0])[0]\nif len(where_DMS) > 0:\n    print(f'Training DMS model with {len(where_DMS)}/{train_df.shape[0]} datapoints.')\n    model_DMS.fit(x=inputs[where_DMS], y=targets[where_DMS], epochs=EPOCHS, batch_size=BATCH_SIZE)\n\nwhere_2A3 = np.where(exp_type == exp_type_uniques[1])[0]\nif len(where_2A3) > 0:\n    print(f'Training 2A3 model with {len(where_2A3)}/{train_df.shape[0]} datapoints.')\n    model_2A3.fit(x=inputs[where_2A3], y=targets[where_2A3], epochs=EPOCHS, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:25:18.441866Z","iopub.execute_input":"2023-09-20T11:25:18.442358Z","iopub.status.idle":"2023-09-20T11:29:07.683677Z","shell.execute_reply.started":"2023-09-20T11:25:18.442299Z","shell.execute_reply":"2023-09-20T11:29:07.681386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_DMS.save('model_DMS.keras')\nmodel_2A3.save('model_2A3.keras')\ndel train_df, exp_type, inputs, targets, model_DMS, model_2A3","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:29:15.922925Z","iopub.execute_input":"2023-09-20T11:29:15.923378Z","iopub.status.idle":"2023-09-20T11:29:16.051499Z","shell.execute_reply.started":"2023-09-20T11:29:15.923340Z","shell.execute_reply":"2023-09-20T11:29:16.050260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict on test data","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n\nmodel_DMS = load_model('model_DMS.keras')\nmodel_2A3 = load_model('model_2A3.keras')","metadata":{"execution":{"iopub.status.busy":"2023-09-20T12:14:47.141519Z","iopub.execute_input":"2023-09-20T12:14:47.142069Z","iopub.status.idle":"2023-09-20T12:14:47.350606Z","shell.execute_reply.started":"2023-09-20T12:14:47.142003Z","shell.execute_reply":"2023-09-20T12:14:47.349520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load test data\ntest_filename = '/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv'\ntest_df = pd.read_csv(test_filename)\n\n# make input numerixc\nfor i,j in zip(['A','G','U','C'], ['1','2','3','4']):\n    test_df['sequence'] = test_df['sequence'].str.replace(i,j)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T12:15:30.926481Z","iopub.execute_input":"2023-09-20T12:15:30.926916Z","iopub.status.idle":"2023-09-20T12:15:41.276454Z","shell.execute_reply.started":"2023-09-20T12:15:30.926881Z","shell.execute_reply":"2023-09-20T12:15:41.275386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\n\nt0,t00 = time.time(),time.time()\n\nCHUNKSIZE = 2_000\nEST_TOT = 269_796_671\n\n# define dataframe with predictions\npred_df = pd.DataFrame(columns=['id', 'reactivity_DMS_MaP', 'reactivity_2A3_MaP'])\n\n\nX = np.zeros((0, INPUT_SIZE)).astype('float16')\nX_temp = np.zeros((1,INPUT_SIZE)).astype('float16')\nid_extr = []\n\n\nI,II = 0,0\nII_end = 5\n\n# loop through dataframe\nfor i, row in test_df.iterrows():\n\n    # get numeric inputs\n    seq = row.sequence\n    seq = [*seq]\n    \n    # fill input values\n    X_temp = 0*X_temp\n    X_temp[0,:len(seq)] = np.array(seq).astype('int')\n    X = np.append(X, X_temp, axis=0)\n    id_extr.append([row.id_min, row.id_max+1])\n    \n    if not (i+1)%CHUNKSIZE:\n        I += CHUNKSIZE*len(seq)\n        II += 1\n        print(\"%i %% : %i / %i\" % (100*I/EST_TOT, I, EST_TOT))#, end='\\r')\n        \n        if II==II_end:\n        \n            # do predictions\n            p_DMS = model_DMS.predict(X)\n            p_2A3 = model_2A3.predict(X)\n\n            # add predictions to dataframe\n            for j in range(len(id_extr)):\n                ids = np.arange(id_extr[j][0], id_extr[j][1])\n                df = pd.DataFrame({'id': ids,\n                                  'reactivity_DMS_MaP': p_DMS[j][:len(ids)],\n                                  'reactivity_2A3_MaP': p_2A3[j][:len(ids)]})\n                pred_df = pd.concat([pred_df, df])\n\n            # make id the index column\n            pred_df = pred_df.set_index('id')\n\n            # save predictions\n            if i+1==II*CHUNKSIZE:\n                print('New predict_submission_new.csv file...')\n                pred_df.to_csv('predict_submission_new.csv', header=pred_df.keys())\n            else:\n                print('Appending to predict_submission_new.csv file...')\n                pred_df.to_csv('predict_submission_new.csv', mode='a', header=False)\n\n            print(pred_df.shape)\n            print('Total/interval Time used: %.1f/ %.1f s' % (time.time()-t00,time.time()-t0))\n            t0 = time.time()\n            print()\n            \n            del X, pred_df, id_extr\n            X = np.zeros((0, INPUT_SIZE)).astype('float16')\n            pred_df = pd.DataFrame(columns=['id', 'reactivity_DMS_MaP', 'reactivity_2A3_MaP'])\n            id_extr = []\n            II = 0\n\nif II!=II_end:\n    p_DMS = model_DMS.predict(X)\n    p_2A3 = model_2A3.predict(X)\n    \n    # add predictions to dataframe\n    for j in range(len(id_extr)):\n        ids = np.arange(id_extr[j][0], id_extr[j][1])\n        df = pd.DataFrame({'id': ids,\n                          'reactivity_DMS_MaP': p_DMS[j][:len(ids)],\n                          'reactivity_2A3_MaP': p_2A3[j][:len(ids)]})\n        pred_df = pd.concat([pred_df, df])\n\n    # make id the index column\n    pred_df = pred_df.set_index('id')\n\n    # save predictions\n    print('Appending to predict_submission_new.csv file...')\n    pred_df.to_csv('predict_submission_new.csv', mode='a', header=False)\n    \nprint('DONE!!')","metadata":{"execution":{"iopub.status.busy":"2023-09-20T12:19:36.960243Z","iopub.execute_input":"2023-09-20T12:19:36.960768Z","iopub.status.idle":"2023-09-20T17:07:34.345061Z","shell.execute_reply.started":"2023-09-20T12:19:36.960729Z","shell.execute_reply":"2023-09-20T17:07:34.342351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pwd\n!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!rm predict_submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.read_csv('/kaggle/working/predict_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-20T17:25:51.350883Z","iopub.execute_input":"2023-09-20T17:25:51.351446Z","iopub.status.idle":"2023-09-20T17:28:03.935934Z","shell.execute_reply.started":"2023-09-20T17:25:51.351344Z","shell.execute_reply":"2023-09-20T17:28:03.934927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(pred_df.shape)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T17:28:03.937584Z","iopub.execute_input":"2023-09-20T17:28:03.938138Z","iopub.status.idle":"2023-09-20T17:28:03.966881Z","shell.execute_reply.started":"2023-09-20T17:28:03.938108Z","shell.execute_reply":"2023-09-20T17:28:03.963816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EST_TOT","metadata":{"execution":{"iopub.status.busy":"2023-09-20T17:28:33.030520Z","iopub.execute_input":"2023-09-20T17:28:33.031009Z","iopub.status.idle":"2023-09-20T17:28:33.039562Z","shell.execute_reply.started":"2023-09-20T17:28:33.030973Z","shell.execute_reply":"2023-09-20T17:28:33.038478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(pred_df.reactivity_DMS_MaP.min())\nprint(pred_df.reactivity_DMS_MaP.max())\nprint(pred_df.reactivity_2A3_MaP.min())\nprint(pred_df.reactivity_2A3_MaP.max())","metadata":{"execution":{"iopub.status.busy":"2023-09-20T17:28:03.969387Z","iopub.execute_input":"2023-09-20T17:28:03.969989Z","iopub.status.idle":"2023-09-20T17:28:05.835150Z","shell.execute_reply.started":"2023-09-20T17:28:03.969941Z","shell.execute_reply":"2023-09-20T17:28:05.832760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle competitions submit -c stanford-ribonanza-rna-folding -f predict_submission.csv -m \"My first prediction\"","metadata":{"execution":{"iopub.status.busy":"2023-09-20T17:36:50.407482Z","iopub.execute_input":"2023-09-20T17:36:50.409884Z","iopub.status.idle":"2023-09-20T17:36:51.622575Z","shell.execute_reply.started":"2023-09-20T17:36:50.409771Z","shell.execute_reply":"2023-09-20T17:36:51.620099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}