{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nCreate models on different subsamples and compare features importances.\n\nLasso models\n\n#### Findings:\n\n**alpha = 0.1** for CD44 does not depend on subsample size ( train_size - parameter ) \n\n**Score: 0.79-0.8** correlation (Pearson) with \"y_true\" (CD44) for all subsample sizes \n\n\n#### Versions: \n\n    1,2 - n_trials = 10\n    3 - n_trials = 100 \n    4 - edditing comments\n    6 - Analysis of genes in datasets\n    ","metadata":{}},{"cell_type":"markdown","source":"# Key Params","metadata":{}},{"cell_type":"code","source":"target_name = 'CD44'\n\ntrain_size = 0.95\n\nn_trials = 10\n\nstr_method = 'Lasso '\n\nfast_mode = 0\n\nalpha_selected = 0.1  # This value may vary!!!\n\nimport time\nt0start = time.time() ","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:56:22.000983Z","iopub.execute_input":"2023-01-14T15:56:22.002210Z","iopub.status.idle":"2023-01-14T15:56:22.008507Z","shell.execute_reply.started":"2023-01-14T15:56:22.002147Z","shell.execute_reply":"2023-01-14T15:56:22.007217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results ","metadata":{}},{"cell_type":"code","source":"# From the previous notebook with Ridge model: \nimport numpy as np\nimport pandas as pd ","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:56:24.677237Z","iopub.execute_input":"2023-01-14T15:56:24.677791Z","iopub.status.idle":"2023-01-14T15:56:24.684057Z","shell.execute_reply.started":"2023-01-14T15:56:24.677747Z","shell.execute_reply":"2023-01-14T15:56:24.682717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-14T15:56:26.874269Z","iopub.execute_input":"2023-01-14T15:56:26.874838Z","iopub.status.idle":"2023-01-14T15:56:26.889835Z","shell.execute_reply.started":"2023-01-14T15:56:26.874792Z","shell.execute_reply":"2023-01-14T15:56:26.888360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:56:30.730504Z","iopub.execute_input":"2023-01-14T15:56:30.731861Z","iopub.status.idle":"2023-01-14T15:56:30.737535Z","shell.execute_reply.started":"2023-01-14T15:56:30.731799Z","shell.execute_reply":"2023-01-14T15:56:30.736504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"%%time\ndf_rna = pd.read_csv('/kaggle/input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv', index_col = 0)\ndf_Rna = df_rna.T\nprint('RNA dataset')\ndisplay(df_Rna.head(3))\n\ndf_meta = pd.read_csv('/kaggle/input/citeseqgse148127/GSE148127_metadata.csv', index_col=0)\nprint('Meta information')\ndisplay(df_meta.head(3))\n\ndf_adt = pd.read_csv('/kaggle/input/citeseqgse148127/GSE148127_ADT.counts.csv', index_col=0)\ndf_Adt = df_adt.T\nprint('ADT dataset')\ndisplay(df_Adt.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:56:33.439733Z","iopub.execute_input":"2023-01-14T15:56:33.440250Z","iopub.status.idle":"2023-01-14T15:57:43.984970Z","shell.execute_reply.started":"2023-01-14T15:56:33.440202Z","shell.execute_reply":"2023-01-14T15:57:43.983989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_Rna.shape)\nprint(df_Adt.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:58:45.168247Z","iopub.execute_input":"2023-01-14T15:58:45.168701Z","iopub.status.idle":"2023-01-14T15:58:45.176043Z","shell.execute_reply.started":"2023-01-14T15:58:45.168650Z","shell.execute_reply":"2023-01-14T15:58:45.174723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfilename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\ndf_rna_kaggle = pd.read_hdf(filename_rna_data)\ndisplay(df_rna_kaggle.head(3))\n\n#%%time\ndf_y_kaggle = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndisplay(df_y_kaggle.head(3))\n\nfn = '/kaggle/input/open-problems-multimodal/metadata.csv'\ndf_meta_kaggle = pd.read_csv(fn, index_col = 0 )\ndisplay(df_meta_kaggle.head(3))\n# Cut only train cite-seq part: \nd_kaggle = pd.DataFrame(index = df_y_kaggle.index)\nprint(d_kaggle.shape)\ndf_meta_kaggle = d_kaggle.join(df_meta_kaggle, how = 'left')\ndisplay(df_meta_kaggle.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:01:04.366907Z","iopub.execute_input":"2023-01-14T16:01:04.368221Z","iopub.status.idle":"2023-01-14T16:01:04.460119Z","shell.execute_reply.started":"2023-01-14T16:01:04.368134Z","shell.execute_reply":"2023-01-14T16:01:04.458734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of proteins in datasets","metadata":{}},{"cell_type":"code","source":"print('Genes info:\\n', 'GSE -', df_Rna.shape, 'Kaggle -', df_rna_kaggle.shape)\nprint('Example:')\ndisplay(df_Rna.head(1))\n\nprint('\\n\\nProteins info:\\n', 'GSE -\\n', df_Adt.columns, '\\n Kaggle -\\n', df_y_kaggle.columns)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:09:00.375859Z","iopub.execute_input":"2023-01-14T16:09:00.376396Z","iopub.status.idle":"2023-01-14T16:09:00.402325Z","shell.execute_reply.started":"2023-01-14T16:09:00.376357Z","shell.execute_reply":"2023-01-14T16:09:00.400929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd_genes_kaggle = []\nfor gene in df_rna_kaggle.columns:\n    if 'cd' in gene or 'CD' in gene:\n        cd_genes_kaggle.append(gene[gene.find('_')+1:])\nprint(len(cd_genes_kaggle))\nprint(cd_genes_kaggle[:10])\n\ncd_genes_gse = []\nfor gene in df_Rna.columns:\n    if 'cd' in gene or 'CD' in gene:\n        cd_genes_gse.append(gene)\nprint(len(cd_genes_gse))\nprint(cd_genes_gse[:10])\n\nlower_genes_kaggle = [gene.lower() for gene in cd_genes_kaggle]\nshared_genes = []\nfor gse in cd_genes_gse:\n    if gse.lower() in lower_genes_kaggle:\n        shared_genes.append(gse)\nprint(len(shared_genes))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:18:56.771953Z","iopub.execute_input":"2023-01-14T16:18:56.772459Z","iopub.status.idle":"2023-01-14T16:18:56.795987Z","shell.execute_reply.started":"2023-01-14T16:18:56.772419Z","shell.execute_reply":"2023-01-14T16:18:56.794722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd44_lasso_genes = pd.read_csv('/kaggle/input/top-lasso-genes/CD44_stable_top_features_Lasso _n_trials_10.csv', index_col=0)\ndisplay(cd44_lasso_genes)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:24:17.206902Z","iopub.execute_input":"2023-01-14T16:24:17.207466Z","iopub.status.idle":"2023-01-14T16:24:17.249920Z","shell.execute_reply.started":"2023-01-14T16:24:17.207413Z","shell.execute_reply":"2023-01-14T16:24:17.248547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lst = [str(x).lower() for x in cd44_lasso_genes['Top 300 Intersection']]\nfor gene in shared_genes:\n    if gene.lower() in lst:\n        print(gene)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:26:11.798568Z","iopub.execute_input":"2023-01-14T16:26:11.799189Z","iopub.status.idle":"2023-01-14T16:26:11.830116Z","shell.execute_reply.started":"2023-01-14T16:26:11.799126Z","shell.execute_reply":"2023-01-14T16:26:11.828608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"y = df_Adt['CITE-CD44']\nprint(y.shape, type(y))\n\nX = df_Rna\nprint(X.shape, type(X))","metadata":{"execution":{"iopub.status.busy":"2023-01-10T20:13:33.331891Z","iopub.execute_input":"2023-01-10T20:13:33.332388Z","iopub.status.idle":"2023-01-10T20:13:33.340280Z","shell.execute_reply.started":"2023-01-10T20:13:33.332351Z","shell.execute_reply":"2023-01-10T20:13:33.338777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y = df_y[target_name].values\n# print(y.shape, type(y) )\n\n# X = ( df_rna.values )\n# print(X.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:43:59.943071Z","iopub.execute_input":"2023-01-10T16:43:59.943828Z","iopub.status.idle":"2023-01-10T16:43:59.96778Z","shell.execute_reply.started":"2023-01-10T16:43:59.943781Z","shell.execute_reply":"2023-01-10T16:43:59.966443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\n\nX = scaler.fit_transform( X )\nprint(X.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T20:14:01.799512Z","iopub.execute_input":"2023-01-10T20:14:01.800085Z","iopub.status.idle":"2023-01-10T20:14:09.120709Z","shell.execute_reply.started":"2023-01-10T20:14:01.800039Z","shell.execute_reply":"2023-01-10T20:14:09.119390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Determine optimal Alpha","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LassoCV\nfrom sklearn.linear_model import Lasso\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2023-01-10T20:14:13.895726Z","iopub.execute_input":"2023-01-10T20:14:13.897185Z","iopub.status.idle":"2023-01-10T20:14:14.066306Z","shell.execute_reply.started":"2023-01-10T20:14:13.897121Z","shell.execute_reply":"2023-01-10T20:14:14.064900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n\nimport time \nprint('X.shape, y.shape:', X.shape, y.shape, 'train_size:', train_size)\n\nif fast_mode and (target_name == 'CD36' ) and (filename_rna_data == '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5') and ( X.shape[1] == 22050) and ( 'Lasso' in str_method ) : \n     alpha_selected =  0.1 # We have already found that 0.1 is optimal alpha for CD36 full rna-data data for KaggleNIPS22 dataset, so we will just use it \nelse:\n    p = np.random.permutation(len(y))\n    N = int(len(y) *  train_size )\n    IX_train = np.arange(len(y))[p][:N]\n    IX_test =  np.arange(len(y))[p][N:]\n    print(\" len(IX_train), len(IX_test):\",  len(IX_train), len(IX_test) )\n\n    df_models_1 = pd.DataFrame()\n\n\n    for i,alpha in enumerate( [1e-3, 1e-2, 1e-1, 1,1e1,1e2 ,1e3,1e4,1e5,1e6,1e7] ):\n        t0 = time.time()\n\n        model = Lasso(alpha = alpha )\n\n        model.fit(X[IX_train,:], y[IX_train])\n\n        col = i\n        df_models_1.loc[col,'alpha'] = alpha\n        y_pred = model.predict(X[IX_train])\n        c = np.corrcoef(y[IX_train], y_pred)[0,1]\n        print(alpha, 'Corr Coef Train', c)\n        df_models_1.loc[col,'Corr Coef Train'] = c\n\n        y_pred = model.predict(X[IX_test])\n        c = np.corrcoef(y[IX_test], y_pred)[0,1]\n        print(alpha, 'Corr Coef Test', c)\n        df_models_1.loc[col,'Corr Coef'] = c\n\n        df_models_1.loc[col,'n_nonzeros'] = (model.coef_ != 0 ).sum()\n\n        print('%.1f secs passed'%(time.time()-t0))\n\n    alpha_selected = df_models_1.sort_values('Corr Coef', ascending = False)['alpha'].iat[0] \n    print('Best alpha: ', alpha_selected )    \n    display( df_models_1 )  \n    \n    \nprint('Best alpha: ', alpha_selected )    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:44:24.916692Z","iopub.execute_input":"2023-01-10T16:44:24.917612Z","iopub.status.idle":"2023-01-10T16:44:24.940493Z","shell.execute_reply.started":"2023-01-10T16:44:24.917564Z","shell.execute_reply":"2023-01-10T16:44:24.939634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Best alpha: ',  alpha_selected )    ","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:44:24.941845Z","iopub.execute_input":"2023-01-10T16:44:24.942453Z","iopub.status.idle":"2023-01-10T16:44:24.955959Z","shell.execute_reply.started":"2023-01-10T16:44:24.94242Z","shell.execute_reply":"2023-01-10T16:44:24.954732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('That standard  way with LassoCV - causes RAM crash. So we do not use it and  search for alpha by direct loop')  \nif 0:\n    from sklearn.linear_model import LassoCV\n    from sklearn.linear_model import RidgeCV\n    from sklearn.linear_model import Ridge\n    from sklearn.model_selection import cross_val_predict\n    from sklearn.model_selection import KFold \n    from sklearn.metrics import r2_score\n    from sklearn.metrics import mean_squared_error\n\n    #alpha_selected = 1e4\n\n    n_splits_for_cross_valdition = 2\n    random_state_cross_validation = 0\n    kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n\n    #model = RidgeCV(alphas= [ 1e6], cv= kf ).fit(X, y)\n    model = RidgeCV(alphas=[1e-3, 1e-2, 1e-1, 1,1e1,1e2,1e3,1e4,1e5,1e6,1e7], cv= kf ).fit(X, y) # Wall time: 1h 10min 36s\n\n    print(model)\n    alpha_selected = model.alpha_\n    print( alpha_selected )\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:44:24.957505Z","iopub.execute_input":"2023-01-10T16:44:24.958296Z","iopub.status.idle":"2023-01-10T16:44:24.969688Z","shell.execute_reply.started":"2023-01-10T16:44:24.958249Z","shell.execute_reply":"2023-01-10T16:44:24.968574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nif not fast_mode: \n#     alpha_selected = 0.1  # This value may vary!!!\n    model = Lasso(alpha = alpha_selected )\n    model.fit(X, y)\n    print('Count non zero coefs for training on the full sample:', (model.coef_ != 0 ).sum()  )\n    df_model_on_full_sample =  pd.DataFrame(index = df_Rna.columns)\n    df_model_on_full_sample['Coefs'] = model.coef_\n    df_model_on_full_sample = df_model_on_full_sample.sort_values('Coefs', key = abs, ascending = False )\n    display( df_model_on_full_sample.head(30) )\n    m = df_model_on_full_sample['Coefs'] != 0\n    display( df_model_on_full_sample[m].tail(15) )\n\n\n    fn = target_name+'_model_on_full_sample_'+str_method + '.csv'\n    print(fn); print()\n    df_model_on_full_sample.to_csv(fn)\n\n\n    m = df_model_on_full_sample['Coefs'] != 0\n    print('Total non-zero coefs (model on full data):', m.sum() )\n    fig = plt.figure(figsize = (20,5))\n    plt.plot( df_model_on_full_sample['Coefs'][m].abs().values, '*-', label = 'Abs Coefs')\n    plt.grid()\n    plt.legend()\n    plt.title('Sorted Lasso abs coefficients train on the full sample. Only non-zero coefs', fontsize = 20 )\n    plt.xlabel('Coefs', fontsize = 20)\n    plt.xlabel('Index', fontsize = 20)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T20:29:38.113712Z","iopub.execute_input":"2023-01-10T20:29:38.115667Z","iopub.status.idle":"2023-01-10T20:35:07.925819Z","shell.execute_reply.started":"2023-01-10T20:29:38.115589Z","shell.execute_reply":"2023-01-10T20:35:07.924287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling on different random subsamples ","metadata":{}},{"cell_type":"code","source":"%%time\n\nimport time\n\ndf_importances = pd.DataFrame(index = df_Rna.columns)\ndf_models = pd.DataFrame()\n\nprint('alpha_selected:', alpha_selected , 'model:', str_method)\n\nfor trial in range( n_trials):\n    t0 = time.time()\n\n    p = np.random.permutation(len(y))\n    N = int(len(y) * train_size )\n    IX_train = np.arange(len(y))[p][:N]\n    IX_test = np.arange(len(y))[p][N:]\n    print('Trial:', trial, \" len(IX_train), len(IX_test):\",  len(IX_train), len(IX_test), '%.1f secs passed'%(time.time() - t0 ) )\n\n    #model = Ridge(alpha = alpha_selected )\n    model = Lasso(alpha = alpha_selected )\n    model.fit(X[IX_train,:], y[IX_train])\n\n    \n    col = str_method + 'Trial'+str(trial)\n    df_importances[col] = model.coef_\n\n    y_pred = model.predict(X[IX_test])\n    c = np.corrcoef(y[IX_test], y_pred)[0,1]\n    print('Corr Coef:',c)\n    df_models.loc[col,'Corr Coef'] = c\n    \n    y_pred = model.predict(X[IX_train])\n    c = np.corrcoef(y[IX_train], y_pred)[0,1]\n    # print(c)\n    df_models.loc[col,'Corr Coef Train'] = c\n    \n    df_models.loc[col,'n_nonzeros'] = (model.coef_ != 0 ).sum()\n    \n    print('Trial', trial, '%.1f secs passed'%(time.time() - t0 ))\n    \ndisplay(df_models )    \ndisplay(df_importances.sort_values( df_importances.columns[0],ascending = False ).head(6))\nmedian_importances = df_importances.median(axis = 1) \nmedian_importances = median_importances.sort_values(key = abs, ascending = False)\nmedian_importances    ","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:44:24.996589Z","iopub.execute_input":"2023-01-10T16:44:24.997214Z","iopub.status.idle":"2023-01-10T16:46:13.372591Z","shell.execute_reply.started":"2023-01-10T16:44:24.997147Z","shell.execute_reply":"2023-01-10T16:46:13.370968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_importances.to_csv(target_name+'_importances_'+str_method+'_Trials' + str(df_importances.shape[1]) + '.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:13.375583Z","iopub.execute_input":"2023-01-10T16:46:13.377032Z","iopub.status.idle":"2023-01-10T16:46:13.446728Z","shell.execute_reply.started":"2023-01-10T16:46:13.376953Z","shell.execute_reply":"2023-01-10T16:46:13.445494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models.to_csv(target_name+'_models_stat_'+str_method+'_Trials' + str(df_importances.shape[1]) + '.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:13.448641Z","iopub.execute_input":"2023-01-10T16:46:13.44909Z","iopub.status.idle":"2023-01-10T16:46:13.456828Z","shell.execute_reply.started":"2023-01-10T16:46:13.449042Z","shell.execute_reply":"2023-01-10T16:46:13.455431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:13.459318Z","iopub.execute_input":"2023-01-10T16:46:13.459982Z","iopub.status.idle":"2023-01-10T16:46:13.489665Z","shell.execute_reply.started":"2023-01-10T16:46:13.459942Z","shell.execute_reply":"2023-01-10T16:46:13.488426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,5))\nfor col in ['Corr Coef', 'Corr Coef Train'  ]:\n    if col in df_models.columns:\n        plt.plot( df_models[col].values, label = col)\n    \nplt.grid()\nplt.legend()\nplt.title('Correlation of predictions with targets', fontsize = 20 )\nplt.xlabel('Correlation', fontsize = 20)\nplt.xlabel('Trial', fontsize = 20)\nplt.show()\n\nfig = plt.figure(figsize = (20,5))\nplt.plot( df_models['n_nonzeros'].values, label = 'n_nonzeros')\nplt.grid()\nplt.legend()\nplt.title('Count of non-zeros coefficients for '+str_method, fontsize = 20 )\nplt.xlabel('Count', fontsize = 20)\nplt.xlabel('Trial', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:13.491329Z","iopub.execute_input":"2023-01-10T16:46:13.491777Z","iopub.status.idle":"2023-01-10T16:46:14.028892Z","shell.execute_reply.started":"2023-01-10T16:46:13.49173Z","shell.execute_reply":"2023-01-10T16:46:14.027593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([ median_importances.iloc[:20].to_frame().reset_index(), \nmedian_importances.iloc[20:40].to_frame().reset_index(), median_importances.iloc[40:60].to_frame().reset_index()\n          , median_importances.iloc[60:80].to_frame().reset_index() ], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:14.030421Z","iopub.execute_input":"2023-01-10T16:46:14.030809Z","iopub.status.idle":"2023-01-10T16:46:14.061997Z","shell.execute_reply.started":"2023-01-10T16:46:14.030772Z","shell.execute_reply":"2023-01-10T16:46:14.06056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = (median_importances != 0 )\nprint('Count Non-zero median importances ', m.sum())\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:14.063279Z","iopub.execute_input":"2023-01-10T16:46:14.06376Z","iopub.status.idle":"2023-01-10T16:46:14.071618Z","shell.execute_reply.started":"2023-01-10T16:46:14.063704Z","shell.execute_reply":"2023-01-10T16:46:14.070234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = (median_importances != 0 )\nprint(m.sum())\nfig = plt.figure(figsize = (20,5))\nplt.plot( median_importances[m].abs().values, '*-',label = 'Abs Coefs')\nplt.grid()\nplt.legend( fontsize = 20)\nplt.title('Sorted '+str_method+' abs importances . Only non-zero.   Median over trials '+str(n_trials), fontsize = 20 )\nplt.xlabel('Coefs', fontsize = 20)\nplt.xlabel('Index', fontsize = 20)\nplt.show()\n\nm = (median_importances != 0 )\nfig = plt.figure(figsize = (20,5))\nplt.plot( median_importances[m].abs().values[1:100],'*-', label = 'Abs Coefs', )\nplt.grid()\nplt.legend( fontsize = 20)\nplt.title('Sorted  '+str_method+'  abs importances 1:100. Only non-zero.   Median over trials '+str(n_trials), fontsize = 20 )\nplt.xlabel('Coefs', fontsize = 20)\nplt.xlabel('Index', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:14.079402Z","iopub.execute_input":"2023-01-10T16:46:14.07979Z","iopub.status.idle":"2023-01-10T16:46:14.621019Z","shell.execute_reply.started":"2023-01-10T16:46:14.07975Z","shell.execute_reply":"2023-01-10T16:46:14.619667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Intersection of non-zero importances for all trials ","metadata":{}},{"cell_type":"code","source":"%%time\nprint('df_importances.shape', df_importances.shape)\n\nm = (median_importances != 0 )\nprint('Count Non-zero median importances ', m.sum())\n\ncc  = (df_importances != 0).sum(axis = 0).mean()  ##  == df_importances.shape[1]\nprint('Average count genes non-zero for trials: ', cc )\n\nm  = (df_importances != 0).sum(axis = 1) == df_importances.shape[1]\nprint('Count genes non-zero for all trials: ', m.sum())\n#print()\n\nif df_importances.shape[1] <= 10:\n    l = []\n    for col1 in df_importances.columns:\n        for col2 in df_importances.columns:\n            m  = (df_importances[[col1,col2]] != 0).sum(axis = 1) == df_importances[[col1,col2]].shape[1]\n            l.append(m.sum())\n    print('number of non-zero coeffients common for Pairs of trials:  ' , np.mean(l))        \n    #print()\n\nif df_importances.shape[1] <= 10:\n    import itertools\n\n    l = []\n    for cols in itertools.combinations( df_importances.columns , 3 ) :\n        m  = (df_importances[list(cols) ] != 0).sum(axis = 1) == df_importances[list(cols) ].shape[1]\n        l.append(m.sum())\n    print('number of non-zero coeffients common for TRIPLES of trials:  ' , np.mean(l))        \n    \nif df_importances.shape[1] <= 5:\n    import itertools\n    l = []\n    for cols in itertools.combinations( df_importances.columns , 4 ) :\n        m  = (df_importances[list(cols) ] != 0).sum(axis = 1) == df_importances[list(cols) ].shape[1]\n        l.append(m.sum())\n    print('number of non-zero coeffients common for QUADRUPLETS of trials:  ' , np.mean(l))        \n    for k in range(5, np.min([10, df_importances.shape[1]+1 ]) ):\n        l = []\n        for cols in itertools.combinations( df_importances.columns , k ) :\n            m  = (df_importances[list(cols) ] != 0).sum(axis = 1) == df_importances[list(cols) ].shape[1]\n            l.append(m.sum())\n        print('number of non-zero coeffients common for ', str(k)+'-tuples of trials:  ' , np.mean(l))        \n\n        \nprint()\nprint('Genes with non-zero importance for all trials: ' )\nprint( df_importances[m].sort_values(df_importances.columns[0], ascending = False, key = abs ).index[:200] )\nprint( [t.split('_')[1] for t in df_importances[m].sort_values(df_importances.columns[0], ascending = False, key = abs ).index[:200] ] )\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:14.622477Z","iopub.execute_input":"2023-01-10T16:46:14.622827Z","iopub.status.idle":"2023-01-10T16:46:14.654386Z","shell.execute_reply.started":"2023-01-10T16:46:14.622794Z","shell.execute_reply":"2023-01-10T16:46:14.653147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nl = []\nfor k in range(1, df_importances.shape[1]+1):\n    m  = (df_importances.iloc[:,:k] != 0).sum(axis = 1) == k # df_importances.i.shape[1]\n    l.append(m.sum())\n        \nplt.show()\n\nfig = plt.figure(figsize = (20,5))\nplt.plot(l, '*-')\nplt.grid()\nplt.legend( fontsize = 20)\nplt.title('Count of non-zeros elements common in the first Index trials ', fontsize = 20 )\nplt.xlabel('Count non-zero', fontsize = 20)\nplt.xlabel('Index', fontsize = 20)\nplt.show()\n\nprint()\nprint('Counts of intersections of non-zero genes list  for the first k trials for all k:')\nprint('All:', l)\nprint('First 30:', l[:30])\nprint('Last 30:', l[-30:])\nprint()\nprint('Interesecting first k trials gives common non-zero genes (for selected \"k\"):')\nfor k in [0,10, 20, 30, 40, 50, 100]:\n    if len(l)>k: print('for k=',k+2,' common non-zero genes:', l[k-1]  ) \nprint()\nprint(m.sum(), 'genes non-zero for all trials (show not to more than 300):')\ndd = df_importances[m].median(axis = 1).sort_values(ascending = False, key = abs)\nprint(dd.index[:300]  )\nprint()\nprint('Top15:')\ndisplay( dd.head(15).to_frame() ) \nprint()\nprint('Tail15:')\ndisplay( dd.tail(15).to_frame() ) \ndd.to_csv(target_name+'_non_zero_median_sorted_'+str_method+'_Trials' + str(df_importances.shape[1]) + '.csv')\n        \ni_start = 10\nfig = plt.figure(figsize = (20,5))\nplt.plot(range(i_start, len(l)), l[i_start:], '*-')\nplt.grid()\nplt.legend( fontsize = 20)\nplt.title('Count of non-zeros elements common in the first '+str(i_start ) + '+Index trials ', fontsize = 20 )\nplt.xlabel('Count non-zero', fontsize = 20)\nplt.xlabel('Index', fontsize = 20)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:14.655934Z","iopub.execute_input":"2023-01-10T16:46:14.656839Z","iopub.status.idle":"2023-01-10T16:46:15.169943Z","shell.execute_reply.started":"2023-01-10T16:46:14.6568Z","shell.execute_reply":"2023-01-10T16:46:15.168505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df_importances.iloc[:,1] != 0).sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.1718Z","iopub.execute_input":"2023-01-10T16:46:15.172194Z","iopub.status.idle":"2023-01-10T16:46:15.18113Z","shell.execute_reply.started":"2023-01-10T16:46:15.172135Z","shell.execute_reply":"2023-01-10T16:46:15.179776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N0 = df_importances.shape[1]\nN2 = int(N0  /2 )\nN2\nm1 = (df_importances.iloc[:,:N2] != 0 ).sum(axis = 1) == N2\nm2 = (df_importances.iloc[:,N2:] != 0 ).sum(axis = 1) == (N0 - N2)\n\nprint('Count common non-zeros for the first half and second half of trials: ', m1.sum() , m2.sum()  )\ns = set(df_importances.index[m1]) &  set(df_importances.index[m2]) \nprint('Count interesection between the first and the second halves: ', len(s) )\nsr = df_importances[m1&m2].median(axis = 1).sort_values( ascending = False , key = abs )\nprint('Top 20 of intersection:')\ndisplay( sr.head(20).to_frame() )\nprint('Tail 15 of intersection:')\ndisplay( sr.tail(15).to_frame() )","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.183598Z","iopub.execute_input":"2023-01-10T16:46:15.184196Z","iopub.status.idle":"2023-01-10T16:46:15.219201Z","shell.execute_reply.started":"2023-01-10T16:46:15.184108Z","shell.execute_reply":"2023-01-10T16:46:15.217933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis of the obtained importances ","metadata":{}},{"cell_type":"code","source":"df_importances.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.220612Z","iopub.execute_input":"2023-01-10T16:46:15.220945Z","iopub.status.idle":"2023-01-10T16:46:15.243872Z","shell.execute_reply.started":"2023-01-10T16:46:15.220914Z","shell.execute_reply":"2023-01-10T16:46:15.242561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_importances.corr()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.245549Z","iopub.execute_input":"2023-01-10T16:46:15.245902Z","iopub.status.idle":"2023-01-10T16:46:15.259622Z","shell.execute_reply.started":"2023-01-10T16:46:15.24587Z","shell.execute_reply":"2023-01-10T16:46:15.258273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nn_x_subplots = 5\nc = 0\n\nfor i in range(5):\n    if i >= df_importances.shape[1]: continue\n    for j in range(5):\n        if i <= j: continue \n        if j >= df_importances.shape[1]: continue\n        col1 = df_importances.columns[i]\n        col2 = df_importances.columns[j]\n\n        if c % n_x_subplots == 0:\n            if c > 0:\n                plt.show()\n            fig = plt.figure(figsize = (20,5) ); c = 0\n            #plt.suptitle(str(k),fontsize = 20 )# str_data_inf + ' n_cells: ' +str(mask.sum()) + ' RED > median expression, BLUE <= median ' )# +' ' + cell_type +' ' + drug )\n\n        c += 1; fig.add_subplot(1,n_x_subplots ,c)\n\n        sns.scatterplot(x = df_importances[col1], y=df_importances[col2] )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.262082Z","iopub.execute_input":"2023-01-10T16:46:15.262471Z","iopub.status.idle":"2023-01-10T16:46:15.522557Z","shell.execute_reply.started":"2023-01-10T16:46:15.262438Z","shell.execute_reply":"2023-01-10T16:46:15.521359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculations of intersections between ordered importances ","metadata":{}},{"cell_type":"code","source":"%%time \ndict_counts = {}\n\ndict_sorted = {} # pd.DataFrame()\nfor col in df_importances.columns:\n    dict_sorted[col] = df_importances[col].abs().sort_values(ascending = False )\n\nfor k in range(0, df_importances.shape[0]):\n    if (k%5000 == 1): print(k)\n    col = df_importances.columns[0]\n    s = set(  dict_sorted[col].index[:k] )\n    for col in df_importances.columns:\n        \n        s = s & set(dict_sorted[col].index[:k] )\n        if col not in dict_counts.keys(): \n            dict_counts[col] = []\n        dict_counts[col].append(len(s))\n\nplt.figure(figsize = (20,10))\nfor i,col in enumerate(dict_counts):\n    if len(dict_counts ) >= 50:\n        if (i%10) != 0: continue\n    plt.plot(dict_counts[col], label = 'intersection till: ' + col )\nplt.legend(fontsize = 20 )\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:46:15.524413Z","iopub.execute_input":"2023-01-10T16:46:15.524767Z","iopub.status.idle":"2023-01-10T16:48:39.591988Z","shell.execute_reply.started":"2023-01-10T16:46:15.524732Z","shell.execute_reply":"2023-01-10T16:48:39.590627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plots of parabolic approximation for full list of features","metadata":{}},{"cell_type":"code","source":"for i,col in  enumerate(dict_counts): # = '3 32606 EryP'\n    \n    if len(dict_counts ) >= 50:\n        if (i%10) != 0: continue \n\n    ll = dict_counts[col]\n    plt.figure(figsize = (20,6))\n    plt.plot(ll, label = 'intersection till: ' + col)\n    y = np.array(ll)\n    x = np.arange(len(y))\n    p = np.polyfit(x,y,2)\n    print(p)\n    plt.plot(np.polyval(p,x) )\n    plt.legend(fontsize = 20 )\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:39.59363Z","iopub.execute_input":"2023-01-10T16:48:39.5941Z","iopub.status.idle":"2023-01-10T16:48:40.210515Z","shell.execute_reply.started":"2023-01-10T16:48:39.594043Z","shell.execute_reply":"2023-01-10T16:48:40.209236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plots of linear and parabolic approximations for ONLY top genes ","metadata":{}},{"cell_type":"code","source":"list_slopes = []\nfor i, col in  enumerate(dict_counts): # = '3 32606 EryP'\n    ll = dict_counts[col]\n    \n    y = np.array(ll)\n    x = np.arange(len(y))\n    \n    M = 100\n    p = np.polyfit(x[:M],y[:M],1)\n    print(p)\n    list_slopes.append(p[0])\n    \n    if len(dict_counts ) >= 50:\n        if (i%10) != 0: continue \n        \n    plt.figure(figsize = (20,6))\n    plt.plot(ll, label = 'intersection till: ' + col)\n    y = np.array(ll)\n    x = np.arange(len(y))\n    p = np.polyfit(x,y,2)\n    print(p)\n    plt.plot(np.polyval(p,x), label = 'quadratic approx'  )\n\n    M = 100\n    p = np.polyfit(x[:M],y[:M],1)\n    print(p)\n    plt.plot(np.polyval(p,x), label = 'linear approx' )\n\n    \n    if i < 3:\n        plt.ylim([0,5000])\n    else:\n        plt.ylim([0,200])\n    plt.xlim([0,5000])\n    plt.legend(fontsize = 20 )\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:40.211989Z","iopub.execute_input":"2023-01-10T16:48:40.212365Z","iopub.status.idle":"2023-01-10T16:48:40.878395Z","shell.execute_reply.started":"2023-01-10T16:48:40.21233Z","shell.execute_reply":"2023-01-10T16:48:40.877421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of slopes (of linear approximations) changes with trial number changes  - is there stabilization or not ? ","metadata":{}},{"cell_type":"code","source":"    \nplt.figure(figsize = (20,10))\nplt.plot( list_slopes     )\nplt.title('Slopes for linear approximation ', fontsize = 20)\nplt.grid()\nplt.show()\n\nplt.figure(figsize = (20,10))\nx = np.arange( 100,len(list_slopes) )\nplt.plot(x,  np.array(list_slopes)[x]     )\nplt.title('Slopes for linear approximation ' , fontsize = 20)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:40.879683Z","iopub.execute_input":"2023-01-10T16:48:40.880545Z","iopub.status.idle":"2023-01-10T16:48:41.401125Z","shell.execute_reply.started":"2023-01-10T16:48:40.880506Z","shell.execute_reply":"2023-01-10T16:48:41.399699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"median_importances = df_importances.median(axis = 1 )\nmedian_importances_sorted = median_importances.sort_values(ascending = False, key = abs)\nmedian_importances_sorted","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.402991Z","iopub.execute_input":"2023-01-10T16:48:41.404191Z","iopub.status.idle":"2023-01-10T16:48:41.427755Z","shell.execute_reply.started":"2023-01-10T16:48:41.404119Z","shell.execute_reply":"2023-01-10T16:48:41.426563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\ndict_top_stable_features = {}\ndf_top_stable_features_counts = pd.DataFrame(); IX = 0\n\ndict_sorted = {} # pd.DataFrame()\nfor col in df_importances.columns:\n    dict_sorted[col] = df_importances[col].abs().sort_values(ascending = False )\n\nfor k in [1, 5, 10, 20, 30,40, 50 , 100, 200,300,500, 1000, 2000, 5000]: #  range(0, df_importances.shape[0]):\n    #if (k%5000 == 1): \n    #print(k)\n    col = df_importances.columns[0]\n    s = set(  dict_sorted[col].index[:k] )\n    for col in df_importances.columns:\n        \n        s = s & set(dict_sorted[col].index[:k] )\n\n    s2 = s & set(median_importances_sorted.index[:k])    \n    m = median_importances_sorted.index.isin(s)\n    list_top_stable_features_ordered_by_median_importances = list( median_importances_sorted[m].index )\n    dict_top_stable_features[k] = list_top_stable_features_ordered_by_median_importances\n    df_top_stable_features_counts.loc[IX, 'Top K' ] = k\n    df_top_stable_features_counts.loc[IX, 'Intersection' ] = len(s)\n    df_top_stable_features_counts.loc[IX, 'Intersection With Median Importances' ] = len(s2)\n    IX += 1 \n    \n    print('Interesection top ',k, 'features for all trials gives: ', len(s) ,' in common', 'Intersection with median importances', len(s2) )\n    \nplt.figure(figsize = (10,5 ))\nsns.lineplot(data = df_top_stable_features_counts,  x = 'Top K', y = 'Intersection'   )\nsns.lineplot(data = df_top_stable_features_counts,  x = 'Top K', y = 'Intersection With Median Importances'  )\nplt.legend()\nplt.grid()\nplt.show()\ndisplay(df_top_stable_features_counts)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.429149Z","iopub.execute_input":"2023-01-10T16:48:41.429548Z","iopub.status.idle":"2023-01-10T16:48:41.762899Z","shell.execute_reply.started":"2023-01-10T16:48:41.429515Z","shell.execute_reply":"2023-01-10T16:48:41.761648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Print top 100 for intersections obtained for different K","metadata":{}},{"cell_type":"code","source":"for k in dict_top_stable_features:\n    l = dict_top_stable_features[k]\n    print(k, len(l))\n    print(l[:100])\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.764325Z","iopub.execute_input":"2023-01-10T16:48:41.76466Z","iopub.status.idle":"2023-01-10T16:48:41.771746Z","shell.execute_reply.started":"2023-01-10T16:48:41.764629Z","shell.execute_reply":"2023-01-10T16:48:41.770398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.773337Z","iopub.execute_input":"2023-01-10T16:48:41.773674Z","iopub.status.idle":"2023-01-10T16:48:41.785234Z","shell.execute_reply.started":"2023-01-10T16:48:41.773643Z","shell.execute_reply":"2023-01-10T16:48:41.783749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataframe with stable topK features for different K ","metadata":{}},{"cell_type":"code","source":"mx = 0\nfor k in dict_top_stable_features:\n    l = dict_top_stable_features[k]\n    mx = max([len(l), mx])\nprint('max len:', mx)\n\ndf_stable_top_features = pd.DataFrame()\n\nfor k in dict_top_stable_features:\n    df_stable_top_features['Top ' +str(k) + ' Intersection'] = [np.nan]*mx\n    \n    l = dict_top_stable_features[k]\n    df_stable_top_features['Top ' +str(k) + ' Intersection'].iloc[:len(l)] = l\ndf_stable_top_features.head(100)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.786634Z","iopub.execute_input":"2023-01-10T16:48:41.787618Z","iopub.status.idle":"2023-01-10T16:48:41.885335Z","shell.execute_reply.started":"2023-01-10T16:48:41.787574Z","shell.execute_reply":"2023-01-10T16:48:41.884034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"str_method, n_trials","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.88677Z","iopub.execute_input":"2023-01-10T16:48:41.88765Z","iopub.status.idle":"2023-01-10T16:48:41.895267Z","shell.execute_reply.started":"2023-01-10T16:48:41.887613Z","shell.execute_reply":"2023-01-10T16:48:41.893952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = target_name+'_stable_top_features_'+str_method+'_n_trials_'+str(n_trials)+'.csv'\nprint(fn); print()\ndf_stable_top_features.to_csv(fn)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.896604Z","iopub.execute_input":"2023-01-10T16:48:41.897103Z","iopub.status.idle":"2023-01-10T16:48:41.927571Z","shell.execute_reply.started":"2023-01-10T16:48:41.897063Z","shell.execute_reply":"2023-01-10T16:48:41.926338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tt = time.time() - t0start\nprint('%.1f seconds ( = %.1f minutes, = %.2f hours) passed'%( tt, tt/60, tt/3600 ) )","metadata":{"execution":{"iopub.status.busy":"2023-01-10T16:48:41.928868Z","iopub.execute_input":"2023-01-10T16:48:41.929285Z","iopub.status.idle":"2023-01-10T16:48:41.935527Z","shell.execute_reply.started":"2023-01-10T16:48:41.929247Z","shell.execute_reply":"2023-01-10T16:48:41.934501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}