{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nFind alpha for each model. \n\n","metadata":{}},{"cell_type":"markdown","source":"# Key Params","metadata":{}},{"cell_type":"code","source":"target_name = 'CD44'\n\ntrain_size = 0.5\n\nstr_method = 'Lasso '\n\n# fast_mode = 0\n# n_trials = 100\n\n\nimport time\nt0start = time.time() ","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:53:11.987360Z","iopub.execute_input":"2023-01-21T20:53:11.987835Z","iopub.status.idle":"2023-01-21T20:53:12.018052Z","shell.execute_reply.started":"2023-01-21T20:53:11.987741Z","shell.execute_reply":"2023-01-21T20:53:12.017094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-21T20:53:12.020369Z","iopub.execute_input":"2023-01-21T20:53:12.021260Z","iopub.status.idle":"2023-01-21T20:53:12.046890Z","shell.execute_reply.started":"2023-01-21T20:53:12.021208Z","shell.execute_reply":"2023-01-21T20:53:12.046040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:53:12.048482Z","iopub.execute_input":"2023-01-21T20:53:12.049147Z","iopub.status.idle":"2023-01-21T20:53:12.710725Z","shell.execute_reply.started":"2023-01-21T20:53:12.049110Z","shell.execute_reply":"2023-01-21T20:53:12.709485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.set_option('display.max_rows', 500)\n# pd.set_option('display.max_columns', 500)\n# pd.set_option('display.width', 1000)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:53:12.714030Z","iopub.execute_input":"2023-01-21T20:53:12.714524Z","iopub.status.idle":"2023-01-21T20:53:12.720200Z","shell.execute_reply.started":"2023-01-21T20:53:12.714477Z","shell.execute_reply":"2023-01-21T20:53:12.718791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"%%time\nfilename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\ndf_rna = pd.read_hdf(filename_rna_data)\ndisplay(df_rna) \n\n#%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndisplay(df_y)\n\nfn = '/kaggle/input/open-problems-multimodal/metadata.csv'\ndf_meta = pd.read_csv(fn, index_col = 0 )\ndf_meta\n# Cut only train cite-seq part: \nd = pd.DataFrame(index = df_y.index)\nprint(d.shape)\ndf_meta = d.join(df_meta, how = 'left')\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:53:12.721723Z","iopub.execute_input":"2023-01-21T20:53:12.722136Z","iopub.status.idle":"2023-01-21T20:54:17.901968Z","shell.execute_reply.started":"2023-01-21T20:53:12.722103Z","shell.execute_reply":"2023-01-21T20:54:17.900696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"y = df_y[target_name].values\nprint(y.shape, type(y) )\n\nX = ( df_rna.values )\nprint(X.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:54:17.903775Z","iopub.execute_input":"2023-01-21T20:54:17.904267Z","iopub.status.idle":"2023-01-21T20:54:17.911716Z","shell.execute_reply.started":"2023-01-21T20:54:17.904220Z","shell.execute_reply":"2023-01-21T20:54:17.910476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\n\nX = scaler.fit_transform( X )\nprint(X.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:54:17.913646Z","iopub.execute_input":"2023-01-21T20:54:17.914171Z","iopub.status.idle":"2023-01-21T20:54:49.098367Z","shell.execute_reply.started":"2023-01-21T20:54:17.914126Z","shell.execute_reply":"2023-01-21T20:54:49.096241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Determine optimal Alpha","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LassoCV\nfrom sklearn.linear_model import Lasso\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.linear_model import Ridge\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import KFold \nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:54:49.100706Z","iopub.execute_input":"2023-01-21T20:54:49.101140Z","iopub.status.idle":"2023-01-21T20:54:49.256679Z","shell.execute_reply.started":"2023-01-21T20:54:49.101090Z","shell.execute_reply":"2023-01-21T20:54:49.255549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n\nimport time \nprint('X.shape, y.shape:', X.shape, y.shape, 'train_size:', train_size)\n\n# if fast_mode and (target_name == 'CD36' ) and (filename_rna_data == '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5') and ( X.shape[1] == 22050) and ( 'Lasso' in str_method ) : \n#      alpha_selected =  0.1 # We have already found that 0.1 is optimal alpha for CD36 full rna-data data for KaggleNIPS22 dataset, so we will just use it \n# else:\nif 1:\n    p = np.random.permutation(len(y))\n    N = int(len(y) *  train_size )\n    IX_train = np.arange(len(y))[p][:N]\n    IX_test =  np.arange(len(y))[p][N:]\n    print(\" len(IX_train), len(IX_test):\",  len(IX_train), len(IX_test) )\n\n    df_models_1 = pd.DataFrame()\n\n\n    for i,alpha in enumerate( [ 10, 1, 0.1, 0.01,0.001, 0.0001]): #  1e-3, 1e-2, 1e-1, 1,1e1,1e2 ,1e3,1e4,1e5,1e6,1e7] ):\n        t0 = time.time()\n\n        model = Lasso(alpha = alpha )\n\n        model.fit(X[IX_train,:], y[IX_train])\n\n        col = i\n        df_models_1.loc[col,'alpha'] = alpha\n        y_pred = model.predict(X[IX_train])\n        c = np.corrcoef(y[IX_train], y_pred)[0,1]\n        print('alpha=', alpha, 'Corr Coef Train', c)\n        df_models_1.loc[col,'Corr Coef Train'] = c\n\n        y_pred = model.predict(X[IX_test])\n        c = np.corrcoef(y[IX_test], y_pred)[0,1]\n        print('alpha', alpha, 'Corr Coef Test', c)\n        df_models_1.loc[col,'Corr Coef'] = c\n\n        df_models_1.loc[col,'n_nonzeros'] = (model.coef_ != 0 ).sum()\n\n        print('%.1f secs passed'%(time.time()-t0))\n\n    alpha_selected = df_models_1.sort_values('Corr Coef', ascending = False)['alpha'].iat[0] \n    print('Best alpha: ', alpha_selected )    \n    display( df_models_1 )  \n    \n    \nprint('Best alpha: ', alpha_selected )    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T20:54:49.258382Z","iopub.execute_input":"2023-01-21T20:54:49.258860Z","iopub.status.idle":"2023-01-21T21:38:49.226038Z","shell.execute_reply.started":"2023-01-21T20:54:49.258827Z","shell.execute_reply":"2023-01-21T21:38:49.224052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Enforce alpha value \n# alpha_selected =  0.1\n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T21:38:49.295873Z","iopub.execute_input":"2023-01-21T21:38:49.297041Z","iopub.status.idle":"2023-01-21T21:38:49.302848Z","shell.execute_reply.started":"2023-01-21T21:38:49.296954Z","shell.execute_reply":"2023-01-21T21:38:49.301875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Best alpha: ',  alpha_selected )    ","metadata":{"execution":{"iopub.status.busy":"2023-01-21T21:38:49.245989Z","iopub.execute_input":"2023-01-21T21:38:49.246947Z","iopub.status.idle":"2023-01-21T21:38:49.258432Z","shell.execute_reply.started":"2023-01-21T21:38:49.246891Z","shell.execute_reply":"2023-01-21T21:38:49.256919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 'That standard  way with LassoCV - causes RAM crash. So we do not use it and  search for alpha by direct loop' - see code above ","metadata":{}},{"cell_type":"code","source":"%%time\nprint('That standard  way with LassoCV - causes RAM crash. So we do not use it and  search for alpha by direct loop')  \nif 0:\n    from sklearn.linear_model import LassoCV\n    from sklearn.linear_model import RidgeCV\n    from sklearn.linear_model import Ridge\n    from sklearn.model_selection import cross_val_predict\n    from sklearn.model_selection import KFold \n    from sklearn.metrics import r2_score\n    from sklearn.metrics import mean_squared_error\n\n    #alpha_selected = 1e4\n\n    n_splits_for_cross_valdition = 2\n    random_state_cross_validation = 0\n    kf = KFold(n_splits=n_splits_for_cross_valdition,  shuffle=True, random_state= random_state_cross_validation )\n\n    #model = RidgeCV(alphas= [ 1e6], cv= kf ).fit(X, y)\n    model = RidgeCV(alphas=[1e-3, 1e-2, 1e-1, 1,1e1,1e2,1e3,1e4,1e5,1e6,1e7], cv= kf ).fit(X, y) # Wall time: 1h 10min 36s\n\n    print(model)\n    alpha_selected = model.alpha_\n    print( alpha_selected )\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T21:38:49.260562Z","iopub.execute_input":"2023-01-21T21:38:49.262055Z","iopub.status.idle":"2023-01-21T21:38:49.279035Z","shell.execute_reply.started":"2023-01-21T21:38:49.261985Z","shell.execute_reply":"2023-01-21T21:38:49.277528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tt = time.time() - t0start\nprint('%.1f seconds ( = %.1f minutes, = %.2f hours) passed'%( tt, tt/60, tt/3600 ) )","metadata":{"execution":{"iopub.status.busy":"2023-01-21T21:38:49.281848Z","iopub.execute_input":"2023-01-21T21:38:49.283023Z","iopub.status.idle":"2023-01-21T21:38:49.292958Z","shell.execute_reply.started":"2023-01-21T21:38:49.282944Z","shell.execute_reply":"2023-01-21T21:38:49.291441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}