{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Perform Lasso ML for citeseqgse148127 data**\n# Analyse the GSE148127 data\n# Dataset engineering\n# Cross-val for Lasso\n# Lasso ML for every CD feature","metadata":{}},{"cell_type":"markdown","source":"# Input: GSE148127 data\n# Output: GSE148127_genes.csv","metadata":{}},{"cell_type":"markdown","source":"# *Download & investigate data*","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport time\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-13T08:18:06.472038Z","iopub.execute_input":"2022-12-13T08:18:06.472450Z","iopub.status.idle":"2022-12-13T08:18:06.491502Z","shell.execute_reply.started":"2022-12-13T08:18:06.472417Z","shell.execute_reply":"2022-12-13T08:18:06.490177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## TotalA_Ref and metadata files","metadata":{}},{"cell_type":"code","source":"total = pd.read_csv('../input/citeseqgse148127/GSE148127_TotalA_Ref.csv', index_col=0)\ntotal.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:18:09.127263Z","iopub.execute_input":"2022-12-13T08:18:09.128506Z","iopub.status.idle":"2022-12-13T08:18:09.189002Z","shell.execute_reply.started":"2022-12-13T08:18:09.128457Z","shell.execute_reply":"2022-12-13T08:18:09.187943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = pd.read_csv('../input/citeseqgse148127/GSE148127_metadata.csv', index_col=0)\ndisplay(meta.head(5))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:18:12.536376Z","iopub.execute_input":"2022-12-13T08:18:12.536949Z","iopub.status.idle":"2022-12-13T08:18:12.716780Z","shell.execute_reply.started":"2022-12-13T08:18:12.536900Z","shell.execute_reply":"2022-12-13T08:18:12.715614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## ADT and RNA files","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/citeseqgse148127/GSE148127_ADT.counts.csv', index_col=0)\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:18:14.867102Z","iopub.execute_input":"2022-12-13T08:18:14.867524Z","iopub.status.idle":"2022-12-13T08:18:15.364716Z","shell.execute_reply.started":"2022-12-13T08:18:14.867492Z","shell.execute_reply":"2022-12-13T08:18:15.363534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite = df.T\n# df_cite.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:18:17.737785Z","iopub.execute_input":"2022-12-13T08:18:17.738153Z","iopub.status.idle":"2022-12-13T08:18:17.745693Z","shell.execute_reply.started":"2022-12-13T08:18:17.738122Z","shell.execute_reply":"2022-12-13T08:18:17.744185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RNA = pd.read_csv('../input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv', index_col=0)\nRNA.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:18:19.840688Z","iopub.execute_input":"2022-12-13T08:18:19.841224Z","iopub.status.idle":"2022-12-13T08:19:23.200639Z","shell.execute_reply.started":"2022-12-13T08:18:19.841183Z","shell.execute_reply":"2022-12-13T08:19:23.199328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RNA.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:19:23.203153Z","iopub.execute_input":"2022-12-13T08:19:23.203696Z","iopub.status.idle":"2022-12-13T08:19:56.082006Z","shell.execute_reply.started":"2022-12-13T08:19:23.203645Z","shell.execute_reply":"2022-12-13T08:19:56.080692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Making useful joined dataframe & Extracting features","metadata":{}},{"cell_type":"code","source":"# df.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:40.118288Z","iopub.execute_input":"2022-11-30T19:51:40.118724Z","iopub.status.idle":"2022-11-30T19:51:40.121685Z","shell.execute_reply.started":"2022-11-30T19:51:40.118699Z","shell.execute_reply":"2022-11-30T19:51:40.120955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RNA.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:40.122667Z","iopub.execute_input":"2022-11-30T19:51:40.122936Z","iopub.status.idle":"2022-11-30T19:51:40.132479Z","shell.execute_reply.started":"2022-11-30T19:51:40.122914Z","shell.execute_reply":"2022-11-30T19:51:40.131926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_RNA = RNA.T\ndf_RNA.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:19:56.083998Z","iopub.execute_input":"2022-12-13T08:19:56.084490Z","iopub.status.idle":"2022-12-13T08:19:56.109102Z","shell.execute_reply.started":"2022-12-13T08:19:56.084443Z","shell.execute_reply":"2022-12-13T08:19:56.107955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_cite.head(3))\ndisplay(df_RNA.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:19:56.111698Z","iopub.execute_input":"2022-12-13T08:19:56.112072Z","iopub.status.idle":"2022-12-13T08:19:56.150824Z","shell.execute_reply.started":"2022-12-13T08:19:56.112023Z","shell.execute_reply":"2022-12-13T08:19:56.149545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Building one joined DataFrame\ndf_new = df_cite.join(df_RNA)\n# df_CD = df_new.T\n# df_CD['CD'] = df_CD.index\n# df_CD = df_CD.reset_index(level=0)\n# df_CD.rename(columns={'index':'CD'}, inplace=True)\n# display(df_CD.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:19:56.152567Z","iopub.execute_input":"2022-12-13T08:19:56.152911Z","iopub.status.idle":"2022-12-13T08:19:58.150922Z","shell.execute_reply.started":"2022-12-13T08:19:56.152879Z","shell.execute_reply":"2022-12-13T08:19:58.149549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making list of features\nfeatures_list = df_cite.columns.to_list()\nprint('Type of feature_list:', type(features_list))\nprint('Features:', features_list)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:19:58.153037Z","iopub.execute_input":"2022-12-13T08:19:58.153555Z","iopub.status.idle":"2022-12-13T08:19:58.160767Z","shell.execute_reply.started":"2022-12-13T08:19:58.153507Z","shell.execute_reply":"2022-12-13T08:19:58.159672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-06T12:24:35.501977Z","iopub.execute_input":"2022-12-06T12:24:35.503649Z","iopub.status.idle":"2022-12-06T12:24:35.534333Z","shell.execute_reply.started":"2022-12-06T12:24:35.503587Z","shell.execute_reply":"2022-12-06T12:24:35.533275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Specific Lasso run for CD proteins","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:28:06.541519Z","iopub.execute_input":"2022-12-13T08:28:06.542658Z","iopub.status.idle":"2022-12-13T08:28:07.249632Z","shell.execute_reply.started":"2022-12-13T08:28:06.542613Z","shell.execute_reply":"2022-12-13T08:28:07.247957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lasso_run(feature):\n    # train_df = df_new.drop(columns=features_list[idx+1:])   # I am not sure, that it is meaningful step. It is also possible to just choose feature_column and drop features for X.\n    y = df_new[feature]\n    X = df_new.drop(columns=[feature])\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42)\n    model = Lasso(alpha=0.04838482756585978)  \n    model.fit(X=X_train, y=y_train)\n    \n    # Squared tests\n    train_score = round(model.score(X_train, y_train), 4)\n    test_score = round(model.score(X_test, y_test), 4)\n    run_score = round(model.score(X, y), 4)\n    print(f'Lasso for {feature}:')\n    print('R squared training set:', train_score)\n    print('R squared test set:', test_score)\n    print('R squared for whole dataset:', run_score)\n    \n#     # Prediction on the whole data\n#     pred = model.predict(X_test)\n#     print('Min:', np.min(pred), '\\nMax:', np.max(pred), '\\nMedian:', np.median(pred), \n#           '\\n90%:', np.quantile(pred, 0.9))\n    \n#     genes = list(zip(pred, X))\n#     list_genes = []\n#     list_genes_score = []\n#     k = np.quantile(pred, 0.99)\n#     for gene in genes:\n#         if gene[0] > k:\n#             list_genes.append(gene[1])\n#             list_genes_score.append(gene[0])\n            \n    \n    return train_score, test_score, run_score","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:28:08.902752Z","iopub.execute_input":"2022-12-13T08:28:08.904323Z","iopub.status.idle":"2022-12-13T08:28:08.916354Z","shell.execute_reply.started":"2022-12-13T08:28:08.904272Z","shell.execute_reply":"2022-12-13T08:28:08.914755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfinal = pd.DataFrame(columns=['CD', 'Score', 'Train Score', 'Test Score'])\n\nfor feature in features_list:\n    train_score, test_score, run_score = lasso_run(feature)\n    feature_df = pd.DataFrame([[feature, run_score, train_score, test_score]],\n                              columns=['CD', 'Score', 'Train Score', 'Test Score'])\n    final = pd.concat([final, feature_df])\n    \nfinal_sorted = final.sort_values(by=['Score'])  \nprint('RAM consumed:' , df.memory_usage().sum())","metadata":{"execution":{"iopub.status.busy":"2022-12-13T08:28:11.441779Z","iopub.execute_input":"2022-12-13T08:28:11.442847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.to_csv('CD_lasso_scores.csv')\nfinal_sorted.to_csv('CD_lasso_scores_sorted.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Lasso mastering**","metadata":{}},{"cell_type":"markdown","source":"## First try","metadata":{}},{"cell_type":"code","source":"# from sklearn.linear_model import Lasso\n# from sklearn.model_selection import train_test_split\n# from sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2022-12-06T12:32:03.497593Z","iopub.execute_input":"2022-12-06T12:32:03.498233Z","iopub.status.idle":"2022-12-06T12:32:04.810185Z","shell.execute_reply.started":"2022-12-06T12:32:03.498192Z","shell.execute_reply":"2022-12-06T12:32:04.808550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Just an example feature df\n# example_feature = features_list[0]\n# train_df = df_new.drop(columns=features_list[1:])\n# train_df","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:41.987704Z","iopub.execute_input":"2022-11-30T19:51:41.987977Z","iopub.status.idle":"2022-11-30T19:51:41.992416Z","shell.execute_reply.started":"2022-11-30T19:51:41.987942Z","shell.execute_reply":"2022-11-30T19:51:41.991305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Preparing data for Lasso\n# y = train_df[example_feature]\n# X = train_df.drop(columns=[example_feature])","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:41.996272Z","iopub.execute_input":"2022-11-30T19:51:41.996552Z","iopub.status.idle":"2022-11-30T19:51:42.003177Z","shell.execute_reply.started":"2022-11-30T19:51:41.996529Z","shell.execute_reply":"2022-11-30T19:51:42.002356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.004206Z","iopub.execute_input":"2022-11-30T19:51:42.004993Z","iopub.status.idle":"2022-11-30T19:51:42.015094Z","shell.execute_reply.started":"2022-11-30T19:51:42.004961Z","shell.execute_reply":"2022-11-30T19:51:42.014174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Defining model\n# model = Lasso(alpha=0.01)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.016257Z","iopub.execute_input":"2022-11-30T19:51:42.017009Z","iopub.status.idle":"2022-11-30T19:51:42.023081Z","shell.execute_reply.started":"2022-11-30T19:51:42.016982Z","shell.execute_reply":"2022-11-30T19:51:42.022416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# # Running train\n# model.fit(X=X_train, y=y_train)\n# print('RAM consumed:' , df.memory_usage().sum())\n\n# # Squared tests\n# print('R squared training set', round(model.score(X_train, y_train)*100, 2))\n# print('R squared test set', round(model.score(X_test, y_test)*100, 2))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.024111Z","iopub.execute_input":"2022-11-30T19:51:42.024552Z","iopub.status.idle":"2022-11-30T19:51:42.032310Z","shell.execute_reply.started":"2022-11-30T19:51:42.024526Z","shell.execute_reply":"2022-11-30T19:51:42.031503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# # Running prediction\n# pred = model.predict(X)\n# print('Min:', np.min(pred), '\\nMax:', np.max(pred), '\\nMedian:', np.median(pred), \n#       '\\n90%:', np.quantile(pred, 0.9))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.033510Z","iopub.execute_input":"2022-11-30T19:51:42.033764Z","iopub.status.idle":"2022-11-30T19:51:42.041765Z","shell.execute_reply.started":"2022-11-30T19:51:42.033741Z","shell.execute_reply":"2022-11-30T19:51:42.040635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Finding target genes\n# genes = list(zip(pred, X))\n# list_genes = []\n# for gene in genes:\n#     if gene[0] > 3:\n#         list_genes.append(gene[1])\n# print(len(list_genes), '\\n', list_genes)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.044655Z","iopub.execute_input":"2022-11-30T19:51:42.045091Z","iopub.status.idle":"2022-11-30T19:51:42.049541Z","shell.execute_reply.started":"2022-11-30T19:51:42.045066Z","shell.execute_reply":"2022-11-30T19:51:42.048850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Crossval","metadata":{}},{"cell_type":"markdown","source":"> > ## LassoCV","metadata":{}},{"cell_type":"code","source":"# %%time\n# from sklearn.linear_model import LassoCV\n\n# # Lasso with 10 fold cross-validation\n# model_cross = LassoCV(cv=5, random_state=0, max_iter=10000)\n\n# # Fit model\n# model_cross.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.050562Z","iopub.execute_input":"2022-11-30T19:51:42.050801Z","iopub.status.idle":"2022-11-30T19:51:42.060035Z","shell.execute_reply.started":"2022-11-30T19:51:42.050778Z","shell.execute_reply":"2022-11-30T19:51:42.058856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_cross.alpha_","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.061332Z","iopub.execute_input":"2022-11-30T19:51:42.062231Z","iopub.status.idle":"2022-11-30T19:51:42.068115Z","shell.execute_reply.started":"2022-11-30T19:51:42.062194Z","shell.execute_reply":"2022-11-30T19:51:42.067251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# alpha = model_cross.alpha_","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.069449Z","iopub.execute_input":"2022-11-30T19:51:42.069777Z","iopub.status.idle":"2022-11-30T19:51:42.077287Z","shell.execute_reply.started":"2022-11-30T19:51:42.069754Z","shell.execute_reply":"2022-11-30T19:51:42.076419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> > ### Kfold","metadata":{}},{"cell_type":"code","source":"# %%time\n# from sklearn.model_selection import GridSearchCV\n# from sklearn.model_selection import RepeatedKFold\n# from numpy import arange\n\n# # Model\n# model = Lasso()\n# # Evaluation method\n# cv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n\n# # Grid\n# grid = dict()\n# grid['alpha'] = arange(0, 1, 0.01)\n\n# search = GridSearchCV(model, grid, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# results = search.fit(X, y)\n\n# # Summarize\n# print('MAE: %.3f' % results.best_score_)\n# print('Config: %s' % results.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.078421Z","iopub.execute_input":"2022-11-30T19:51:42.078680Z","iopub.status.idle":"2022-11-30T19:51:42.086706Z","shell.execute_reply.started":"2022-11-30T19:51:42.078657Z","shell.execute_reply":"2022-11-30T19:51:42.085906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Big run","metadata":{}},{"cell_type":"code","source":"# for idx, feature in features_list:\n#     train_df = df_new.drop(columns=features_list[idx+1:])\n    \n#     y = train_df[feature]\n#     X = train_df.drop(columns=[feature])\n    \n#     #X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42)\n#     model = Lasso(alpha=0.01)\n    \n#     %%time\n#     model.fit(X=X, y=y)\n#     print('RAM consumed:' , df.memory_usage().sum())\n\n#     # Squared tests\n#     print('R squared training set', round(model.score(X, y)*100, 2))\n#     print('R squared test set', round(model.score(X, y)*100, 2))\n    \n#     %%time\n#     pred = model.predict(X)\n#     print('Min:', np.min(pred), '\\nMax:', np.max(pred), '\\nMedian:', np.median(pred), \n#           '\\n90%:', np.quantile(pred, 0.9))\n    \n#     genes = list(zip(pred, X))\n#     list_genes = []\n#     for gene in genes:\n#         if gene[0] > 3:\n#             list_genes.append(gene[1])\n#     print(len(list_genes), '\\n', list_genes)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:51:42.087779Z","iopub.execute_input":"2022-11-30T19:51:42.088181Z","iopub.status.idle":"2022-11-30T19:51:42.096285Z","shell.execute_reply.started":"2022-11-30T19:51:42.088146Z","shell.execute_reply":"2022-11-30T19:51:42.095568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lasso_run(feature, idx):\n    train_df = df_new.drop(columns=features_list[idx+1:])\n    y = train_df[feature]\n    X = train_df.drop(columns=[feature])\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42)\n    model = Lasso(alpha=0.04838482756585978)  \n    model.fit(X=X_train, y=y_train)\n    \n    # Squared tests\n    print(f'Lasso for {feature}:')\n    print('R squared training set', round(model.score(X_train, y_train)*100, 2))\n    print('R squared test set', round(model.score(X_test, y_test)*100, 2))\n    \n    # Prediction on the whole data\n    pred = model.predict(X_test)\n    print('Min:', np.min(pred), '\\nMax:', np.max(pred), '\\nMedian:', np.median(pred), \n          '\\n90%:', np.quantile(pred, 0.9))\n    \n    genes = list(zip(pred, X))\n    list_genes = []\n    list_genes_score = []\n    k = np.quantile(pred, 0.99)\n    for gene in genes:\n        if gene[0] > k:\n            list_genes.append(gene[1])\n            list_genes_score.append(gene[0])\n            \n    \n    return np.min(pred), np.max(pred), np.median(pred), np.quantile(pred, 0.9), list_genes, list_genes_score","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:56:50.539794Z","iopub.execute_input":"2022-11-30T19:56:50.541082Z","iopub.status.idle":"2022-11-30T19:56:50.550210Z","shell.execute_reply.started":"2022-11-30T19:56:50.541023Z","shell.execute_reply":"2022-11-30T19:56:50.549452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# final = pd.DataFrame(columns=['Feature', 'Min.score', 'Max.score', 'Median.score', '0.9.score', 'Genes'])\n\n# for idx, feature in enumerate(features_list):\n#     minimum, maximum, median, quantile, list_genes, list_genes_score = lasso_run(feature, idx)\n#     feature_df = pd.DataFrame([[feature, minimum, maximum, median, quantile, list_genes, list_genes_score]],\n#                               columns=['Feature', 'Min.score', 'Max.score', 'Median.score', '0.9.score', 'Genes', 'Genes_score'])\n#     final = pd.concat([final, feature_df])\n#     final_sorted = final.sort_values(by=['Genes_score'])\n    \n# print('RAM consumed:' , df.memory_usage().sum())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T19:59:53.023183Z","iopub.execute_input":"2022-11-30T19:59:53.023604Z","iopub.status.idle":"2022-11-30T21:15:19.880136Z","shell.execute_reply.started":"2022-11-30T19:59:53.023575Z","shell.execute_reply":"2022-11-30T21:15:19.879073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final.to_csv('gene_features.csv')\n# final_sorted.to_csv('features_sorted.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T21:15:19.882073Z","iopub.execute_input":"2022-11-30T21:15:19.882444Z","iopub.status.idle":"2022-11-30T21:15:19.893726Z","shell.execute_reply.started":"2022-11-30T21:15:19.882408Z","shell.execute_reply":"2022-11-30T21:15:19.892846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T21:58:33.405326Z","iopub.execute_input":"2022-11-30T21:58:33.406685Z","iopub.status.idle":"2022-11-30T21:58:33.423731Z","shell.execute_reply.started":"2022-11-30T21:58:33.406649Z","shell.execute_reply":"2022-11-30T21:58:33.422287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dict = {}\n# for index, row in final_sorted.iterrows():\n#     feature = row['Feature']\n#     list_gen = []\n#     list_scor = []\n#     genes = row['Genes']\n#     genes_scores = row['Genes_score']\n#     for gen, scor in zip(genes, genes_scores):\n#         list_gen.append(gen)\n#         list_scor.append(scor)\n#     dict[f'{feature}_gen'] = list_gen\n#     dict[f'{feature}_score'] = list_scor\n# gen_scor = pd.DataFrame.from_dict(dict)\n    \n#     dict[feature] = genes_list","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:09:16.780156Z","iopub.execute_input":"2022-11-30T22:09:16.780558Z","iopub.status.idle":"2022-11-30T22:09:16.792866Z","shell.execute_reply.started":"2022-11-30T22:09:16.780531Z","shell.execute_reply":"2022-11-30T22:09:16.792009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import json\n# with open('/kaggle/input/dict-of-gse/dict.json', 'r') as f:\n#     dic = json.load(f)\n# df = pd.DataFrame.from_dict(dic, orient='index').T\n# df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-06T07:50:49.187746Z","iopub.execute_input":"2022-12-06T07:50:49.188075Z","iopub.status.idle":"2022-12-06T07:50:49.199484Z","shell.execute_reply.started":"2022-12-06T07:50:49.188050Z","shell.execute_reply":"2022-12-06T07:50:49.198521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gen_scor.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:13:06.570076Z","iopub.execute_input":"2022-11-30T22:13:06.570438Z","iopub.status.idle":"2022-11-30T22:13:06.599369Z","shell.execute_reply.started":"2022-11-30T22:13:06.570411Z","shell.execute_reply":"2022-11-30T22:13:06.598433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gen_scor.to_csv(\"gen_score.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:13:34.136091Z","iopub.execute_input":"2022-11-30T22:13:34.136482Z","iopub.status.idle":"2022-11-30T22:13:34.144543Z","shell.execute_reply.started":"2022-11-30T22:13:34.136449Z","shell.execute_reply":"2022-11-30T22:13:34.143510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dict.keys()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:10:29.589586Z","iopub.execute_input":"2022-11-30T22:10:29.589962Z","iopub.status.idle":"2022-11-30T22:10:29.595730Z","shell.execute_reply.started":"2022-11-30T22:10:29.589935Z","shell.execute_reply":"2022-11-30T22:10:29.594890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import json\n\n# json = json.dumps(dict)\n# with open(\"dict.json\",\"w\") as f:\n#     f.write(json)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:12:33.029463Z","iopub.execute_input":"2022-11-30T22:12:33.029827Z","iopub.status.idle":"2022-11-30T22:12:33.036604Z","shell.execute_reply.started":"2022-11-30T22:12:33.029801Z","shell.execute_reply":"2022-11-30T22:12:33.035231Z"},"trusted":true},"execution_count":null,"outputs":[]}]}