{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notebook to get important multiome features by predicting sex by RNA\n\nY - sex, X - RNA gene expression levels\n\nresults are stored in tables\n\n<..._weights.csv> - table with obtained coeffs after training\n\n<..._topN_abs.csv> - table top N genes by coeffs\n    \n<..._zeroN.csv> - table with zero genes by coeffs\n    \n#### Results:\n##### **The first ten largest weights affecting sex obtained for proteins**:\nENSG00000129824_RPS4Y1, ENSG00000229807_XIST, ENSG00000198034_RPS4X, ENSG00000198692_EIF1AY, ENSG00000230202_AL450405.1, ENSG00000164587_RPS14, ENSG00000198502_HLA-DRB5, ENSG00000251562_MALAT1, ENSG00000111640_GAPDH, ENSG00000197728_RPS26\n\n##### **The 10 weights NOT affecting sex obtained for proteins**:\nENSG00000121410_A1BG, ENSG00000229474_PATL2, ENSG00000166889_PATL1, ENSG00000132849_PATJ, ENSG00000115687_PASK, ENSG00000138964_PARVG, ENSG00000188677_PARVB, ENSG00000152931_PART1, ENSG00000162396_PARS2, ENSG00000185480_PARPBP<br> ","metadata":{"execution":{"iopub.status.busy":"2023-01-16T15:28:26.648165Z","iopub.execute_input":"2023-01-16T15:28:26.648692Z","iopub.status.idle":"2023-01-16T15:28:26.658298Z","shell.execute_reply.started":"2023-01-16T15:28:26.648648Z","shell.execute_reply":"2023-01-16T15:28:26.656397Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.linear_model import LogisticRegression, Lasso\nfrom sklearn.model_selection import GridSearchCV, cross_val_score\nfrom sklearn.metrics import r2_score, mean_squared_error, roc_auc_score\nimport matplotlib.colors as mcolors\nimport matplotlib.pyplot as plt\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:23:02.895596Z","iopub.execute_input":"2023-04-19T09:23:02.896457Z","iopub.status.idle":"2023-04-19T09:23:05.413814Z","shell.execute_reply.started":"2023-04-19T09:23:02.896344Z","shell.execute_reply":"2023-04-19T09:23:05.411764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv('/kaggle/input/open-problems-multimodal/metadata.csv', index_col = 0)\ndisplay(meta_df) ","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:23:07.709782Z","iopub.execute_input":"2023-04-19T09:23:07.710167Z","iopub.status.idle":"2023-04-19T09:23:08.272755Z","shell.execute_reply.started":"2023-04-19T09:23:07.710136Z","shell.execute_reply":"2023-04-19T09:23:08.271921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5') \n# cite_inputs = lgb.Dataset('/kaggle/input/open-problems-multimodal/test_cite_inputs.h5')\ndisplay(X)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:23:12.362352Z","iopub.execute_input":"2023-04-19T09:23:12.362771Z","iopub.status.idle":"2023-04-19T09:24:23.980162Z","shell.execute_reply.started":"2023-04-19T09:23:12.362738Z","shell.execute_reply":"2023-04-19T09:24:23.978934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df['sex'] = np.where(meta_df['donor'] > 30000, 'Male', 'Female')\ny = meta_df[meta_df.index.isin(X.index)].set_index(X.index)[['sex']]\ny = pd.get_dummies(y)\nprint(y, type(y))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:25:52.24421Z","iopub.execute_input":"2023-04-19T09:25:52.244646Z","iopub.status.idle":"2023-04-19T09:25:52.388068Z","shell.execute_reply.started":"2023-04-19T09:25:52.24461Z","shell.execute_reply":"2023-04-19T09:25:52.386845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna_index = X.columns\nrna_index","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:25:55.853809Z","iopub.execute_input":"2023-04-19T09:25:55.854221Z","iopub.status.idle":"2023-04-19T09:25:55.864094Z","shell.execute_reply.started":"2023-04-19T09:25:55.854188Z","shell.execute_reply":"2023-04-19T09:25:55.862457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = y['sex_Male'].values\nprint(y.shape, type(y))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:25:59.701992Z","iopub.execute_input":"2023-04-19T09:25:59.702489Z","iopub.status.idle":"2023-04-19T09:25:59.711149Z","shell.execute_reply.started":"2023-04-19T09:25:59.70245Z","shell.execute_reply":"2023-04-19T09:25:59.709533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = X.astype('float64')\nX = X.values\nprint(X.shape, type(X))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:26:02.099146Z","iopub.execute_input":"2023-04-19T09:26:02.099573Z","iopub.status.idle":"2023-04-19T09:26:02.108679Z","shell.execute_reply.started":"2023-04-19T09:26:02.099541Z","shell.execute_reply":"2023-04-19T09:26:02.107025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX = scaler.fit_transform(X)\nprint(X.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:26:05.458473Z","iopub.execute_input":"2023-04-19T09:26:05.458911Z","iopub.status.idle":"2023-04-19T09:26:36.57754Z","shell.execute_reply.started":"2023-04-19T09:26:05.458876Z","shell.execute_reply":"2023-04-19T09:26:36.576313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:27:11.931819Z","iopub.execute_input":"2023-04-19T09:27:11.93227Z","iopub.status.idle":"2023-04-19T09:27:12.131882Z","shell.execute_reply.started":"2023-04-19T09:27:11.932236Z","shell.execute_reply":"2023-04-19T09:27:12.130634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic regression\n","metadata":{}},{"cell_type":"code","source":"train_size = 0.7\nprint('X.shape, y.shape:', X.shape, y.shape, 'train_size:', train_size)\n\n\np = np.random.permutation(len(y))\nN = int(len(y) *  train_size)\nIX_train = np.arange(len(y))[p][:N]\nIX_test =  np.arange(len(y))[p][N:]\nprint(\" len(IX_train), len(IX_test):\",  len(IX_train), len(IX_test) )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T09:27:16.51423Z","iopub.execute_input":"2023-04-19T09:27:16.51476Z","iopub.status.idle":"2023-04-19T09:27:16.529014Z","shell.execute_reply.started":"2023-04-19T09:27:16.514688Z","shell.execute_reply":"2023-04-19T09:27:16.527051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg = LogisticRegression(penalty='l1', C = 100, solver = 'saga', max_iter=500)\nlog_reg.fit(X[IX_train,:], y[IX_train])\n\ny_pred = log_reg.predict(X[IX_test])\n\nauc_score = roc_auc_score(y[IX_test], y_pred)\n\nprint(\"AUC Score: \", auc_score)\nweights_dict = {}\nweights_dict['coef'] = log_reg.coef_\nprint('non-zero coeffs:', (log_reg.coef_ != 0 ).sum())","metadata":{},"execution_count":null,"outputs":[]}]}