{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\n### Briefly - downsample and make quick experiments\n\nHere we will downsample the data. I.e. select 10% of data as a kind of \"Playground\" - for quick experiments. \nAnd will train/tune different models on that playground first.\n\n\n### About CV scheme\n\nPublic and private test sets are quite different. (Private - contains new DAY (and donor), while public ONLY new donor).\nSo one should be careful with the validation schemes.\nSome proposals are described here: \n\nNotebook: https://www.kaggle.com/code/alexandervc/mmscel-crossvalidation-schemes\nTopic: https://www.kaggle.com/competitions/open-problems-multimodal/discussion/358860\n\nWe will be based on them. \n\n### Start with selection of \"Playground\" - 10% part of data - to quickly test models, ideas\n\nData is quite big, and so training models, tuning params might take long time. That is not always affordable.\nIn the present script we first choose some 10% part of data - to make quick experiments.\nThat part is chosen to have SAME proporitions of key characteristics: donors, days, cell types as the initial data\n\nSo we can create CV scheme for that \"playground\" part.\nBut we can also simplify even further - for start - use  not 6-fold scheme but just splite by days. And have only 1 train subset for quick experiments. \n\nTo simplify even further we can start play with just one target only. \n","metadata":{}},{"cell_type":"code","source":"tune_RF = False # True # Takes 4 minutes for 6 params ","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:42:12.588912Z","iopub.execute_input":"2022-11-07T11:42:12.589415Z","iopub.status.idle":"2022-11-07T11:42:12.595786Z","shell.execute_reply.started":"2022-11-07T11:42:12.589369Z","shell.execute_reply":"2022-11-07T11:42:12.594582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Install/import modules, load technical data\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-07T11:42:12.598036Z","iopub.execute_input":"2022-11-07T11:42:12.598484Z","iopub.status.idle":"2022-11-07T11:42:12.618588Z","shell.execute_reply.started":"2022-11-07T11:42:12.598443Z","shell.execute_reply":"2022-11-07T11:42:12.616591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:42:12.620930Z","iopub.execute_input":"2022-11-07T11:42:12.622141Z","iopub.status.idle":"2022-11-07T11:42:31.009660Z","shell.execute_reply.started":"2022-11-07T11:42:12.622090Z","shell.execute_reply":"2022-11-07T11:42:31.008052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:42:31.011810Z","iopub.execute_input":"2022-11-07T11:42:31.012175Z","iopub.status.idle":"2022-11-07T11:42:49.413221Z","shell.execute_reply.started":"2022-11-07T11:42:31.012137Z","shell.execute_reply":"2022-11-07T11:42:49.411473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load prepared Features for CITE-seq part of task","metadata":{}},{"cell_type":"code","source":"%%time\n\nprint('Load prepared features for CITE-seq')\n# These files contain both train and test parts .\n# For CITEseq part - first 70988 elements - train, and later 48663 - test. Overall 119651 samples.\n# fn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_TruncatedSVD200_niter7_rs42.csv'\n# fn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_PCA500.csv'\nfn = '/kaggle/input/3790-features-in-df/3790_features_in_df.csv'\ndf_cite = pd.read_csv(fn,index_col = 0)\ndf_cite = df_cite.set_index('cell_id')\ndisplay(df_cite)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:42:49.415552Z","iopub.execute_input":"2022-11-07T11:42:49.416634Z","iopub.status.idle":"2022-11-07T11:44:24.053740Z","shell.execute_reply.started":"2022-11-07T11:42:49.416569Z","shell.execute_reply":"2022-11-07T11:44:24.052779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.mean(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:24.055990Z","iopub.execute_input":"2022-11-07T11:44:24.056367Z","iopub.status.idle":"2022-11-07T11:44:24.825988Z","shell.execute_reply.started":"2022-11-07T11:44:24.056329Z","shell.execute_reply":"2022-11-07T11:44:24.824606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.std(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:24.827261Z","iopub.execute_input":"2022-11-07T11:44:24.827557Z","iopub.status.idle":"2022-11-07T11:44:27.471674Z","shell.execute_reply.started":"2022-11-07T11:44:24.827530Z","shell.execute_reply":"2022-11-07T11:44:27.470205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Targets for CITE-seq","metadata":{}},{"cell_type":"code","source":"%%time\n#if 1:\nprint('Load CITE-seq targets and ')\ndf_cite_train_y = pd.read_hdf(FP_CITE_TRAIN_TARGETS)\ndisplay(df_cite_train_y)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:27.473992Z","iopub.execute_input":"2022-11-07T11:44:27.475330Z","iopub.status.idle":"2022-11-07T11:44:28.267970Z","shell.execute_reply.started":"2022-11-07T11:44:27.475281Z","shell.execute_reply":"2022-11-07T11:44:28.266087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and start prepare some metadata (cut only CITE-seq train part)","metadata":{}},{"cell_type":"code","source":"%%time\nfn2 = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv'\ndf_meta_full = pd.read_csv(fn2,index_col = 0)\ndisplay(df_meta_full)\n#if 1:\ndf_meta = pd.DataFrame(index = df_cite_train_y.index) \ndf_meta = df_meta.join(df_cell.set_index('cell_id') )\ndisplay(df_meta)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:28.269759Z","iopub.execute_input":"2022-11-07T11:44:28.270113Z","iopub.status.idle":"2022-11-07T11:44:28.609760Z","shell.execute_reply.started":"2022-11-07T11:44:28.270084Z","shell.execute_reply":"2022-11-07T11:44:28.608896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create \"Playground\"  (i.e. downsample)\n\nSmall 10% of data of data where we can make prelimanary experiments. \n\nit will be labeled by special column \"Playground\" in df_meta (meta data for CITE-seq train only)","metadata":{}},{"cell_type":"code","source":"# Prepare for creation of additional holdout folds with 10% of samples \n# We will use stratified Kfold to achieve that days, cell_types and donors are equally distributed \nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nscol = 'donor&day&CT'\ndf_meta[scol] =df_meta['donor'].apply(lambda x:str(x)+'_') + df_meta['day'].apply(lambda x:str(x)+'_') + df_meta['cell_type']\n\n\nskf = StratifiedKFold(n_splits=10,  shuffle=True, random_state=40)\nskf.get_n_splits(df_meta, df_meta[scol] )\n\n\n\ny = df_meta[scol] \nfor train_index, test_index in skf.split(df_meta, df_meta[scol]):\n    print(\"TRAIN:\", len(train_index), \"TEST:\", len(test_index) ); \n    break\nprint(test_index)\nprint(df_meta[scol].value_counts().head(5)   )\nprint(df_meta.iloc[test_index,:][scol].value_counts().head(5)    )\n\n\nflagged_column_name = 'Playground'\ndf_meta[flagged_column_name] = 0 \ndf_meta.loc[df_meta.index[test_index],flagged_column_name]  = 1\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:28.611080Z","iopub.execute_input":"2022-11-07T11:44:28.612049Z","iopub.status.idle":"2022-11-07T11:44:28.959268Z","shell.execute_reply.started":"2022-11-07T11:44:28.612015Z","shell.execute_reply":"2022-11-07T11:44:28.958370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling Preparations","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:28.960551Z","iopub.execute_input":"2022-11-07T11:44:28.961466Z","iopub.status.idle":"2022-11-07T11:44:28.967870Z","shell.execute_reply.started":"2022-11-07T11:44:28.961434Z","shell.execute_reply":"2022-11-07T11:44:28.965875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat = pd.DataFrame()# columns = [ 'r2_score','mse','Time',  'n_feat', 'Target' ])\n#IXmdl = 0\ndf_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:28.972648Z","iopub.execute_input":"2022-11-07T11:44:28.973061Z","iopub.status.idle":"2022-11-07T11:44:28.987475Z","shell.execute_reply.started":"2022-11-07T11:44:28.973026Z","shell.execute_reply":"2022-11-07T11:44:28.985725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_save_predictions = {}\ndef update_models_stat(model, model_ID, t0 ):\n    \n    IXmdl = model_ID # df_models_stat.shape[0] + 1\n    #df_models_stat.loc[IXmdl,'Model'] = model_ID\n    y_pred_loc = model.predict(X_test)\n    df_models_stat.loc[IXmdl,'r2_score'] = r2_score(y_test, y_pred_loc) \n    df_models_stat.loc[IXmdl,'mse'] =  mean_squared_error(y_test, y_pred_loc)\n    df_models_stat.loc[IXmdl,'Time'] = np.round( time.time() - t0,3)\n    df_models_stat.loc[IXmdl,'Target'] = selected_target\n    df_models_stat.loc[IXmdl,'n_feat'] = X_train.shape[1]\n    df_models_stat.loc[IXmdl,'n_samples_train'] = X_train.shape[0]\n\n    dict_save_predictions[model_ID] = (model.predict(X_train), model.predict(X_test), \n                                       model.predict(X_test_public_like), model.predict(X_oop)   )\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:28.989301Z","iopub.execute_input":"2022-11-07T11:44:28.989626Z","iopub.status.idle":"2022-11-07T11:44:29.000591Z","shell.execute_reply.started":"2022-11-07T11:44:28.989584Z","shell.execute_reply":"2022-11-07T11:44:28.998651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:29.002659Z","iopub.execute_input":"2022-11-07T11:44:29.003159Z","iopub.status.idle":"2022-11-07T11:44:29.017446Z","shell.execute_reply.started":"2022-11-07T11:44:29.003110Z","shell.execute_reply":"2022-11-07T11:44:29.015416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso_40 ","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 40\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:29.019852Z","iopub.execute_input":"2022-11-07T11:44:29.021264Z","iopub.status.idle":"2022-11-07T11:44:29.186711Z","shell.execute_reply.started":"2022-11-07T11:44:29.021197Z","shell.execute_reply":"2022-11-07T11:44:29.184820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:29.189151Z","iopub.execute_input":"2022-11-07T11:44:29.189584Z","iopub.status.idle":"2022-11-07T11:44:30.667381Z","shell.execute_reply.started":"2022-11-07T11:44:29.189548Z","shell.execute_reply":"2022-11-07T11:44:30.666449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=0.01)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_40', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.672061Z","iopub.execute_input":"2022-11-07T11:44:30.672912Z","iopub.status.idle":"2022-11-07T11:44:30.751070Z","shell.execute_reply.started":"2022-11-07T11:44:30.672877Z","shell.execute_reply":"2022-11-07T11:44:30.748843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.752432Z","iopub.execute_input":"2022-11-07T11:44:30.753302Z","iopub.status.idle":"2022-11-07T11:44:30.760771Z","shell.execute_reply.started":"2022-11-07T11:44:30.753258Z","shell.execute_reply":"2022-11-07T11:44:30.759803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.762277Z","iopub.execute_input":"2022-11-07T11:44:30.764272Z","iopub.status.idle":"2022-11-07T11:44:30.772822Z","shell.execute_reply.started":"2022-11-07T11:44:30.764223Z","shell.execute_reply":"2022-11-07T11:44:30.771968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.774005Z","iopub.execute_input":"2022-11-07T11:44:30.775563Z","iopub.status.idle":"2022-11-07T11:44:30.786444Z","shell.execute_reply.started":"2022-11-07T11:44:30.775522Z","shell.execute_reply":"2022-11-07T11:44:30.785659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(20),\n                     coef.sort_values().tail(20)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.787717Z","iopub.execute_input":"2022-11-07T11:44:30.788898Z","iopub.status.idle":"2022-11-07T11:44:30.800056Z","shell.execute_reply.started":"2022-11-07T11:44:30.788860Z","shell.execute_reply":"2022-11-07T11:44:30.799034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.802306Z","iopub.execute_input":"2022-11-07T11:44:30.803834Z","iopub.status.idle":"2022-11-07T11:44:30.821665Z","shell.execute_reply.started":"2022-11-07T11:44:30.803791Z","shell.execute_reply":"2022-11-07T11:44:30.820602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 10.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:30.823627Z","iopub.execute_input":"2022-11-07T11:44:30.825289Z","iopub.status.idle":"2022-11-07T11:44:31.185453Z","shell.execute_reply.started":"2022-11-07T11:44:30.825246Z","shell.execute_reply":"2022-11-07T11:44:31.184413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:31.186899Z","iopub.execute_input":"2022-11-07T11:44:31.188250Z","iopub.status.idle":"2022-11-07T11:44:31.202909Z","shell.execute_reply.started":"2022-11-07T11:44:31.188213Z","shell.execute_reply":"2022-11-07T11:44:31.201176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso 50","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 50\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:31.205218Z","iopub.execute_input":"2022-11-07T11:44:31.205678Z","iopub.status.idle":"2022-11-07T11:44:31.377678Z","shell.execute_reply.started":"2022-11-07T11:44:31.205631Z","shell.execute_reply":"2022-11-07T11:44:31.376227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:31.380095Z","iopub.execute_input":"2022-11-07T11:44:31.380736Z","iopub.status.idle":"2022-11-07T11:44:33.558576Z","shell.execute_reply.started":"2022-11-07T11:44:31.380703Z","shell.execute_reply":"2022-11-07T11:44:33.557689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=0.01)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_50', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.560549Z","iopub.execute_input":"2022-11-07T11:44:33.561890Z","iopub.status.idle":"2022-11-07T11:44:33.645976Z","shell.execute_reply.started":"2022-11-07T11:44:33.561850Z","shell.execute_reply":"2022-11-07T11:44:33.645162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.651818Z","iopub.execute_input":"2022-11-07T11:44:33.655635Z","iopub.status.idle":"2022-11-07T11:44:33.669466Z","shell.execute_reply.started":"2022-11-07T11:44:33.655589Z","shell.execute_reply":"2022-11-07T11:44:33.668677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.673778Z","iopub.execute_input":"2022-11-07T11:44:33.676408Z","iopub.status.idle":"2022-11-07T11:44:33.683389Z","shell.execute_reply.started":"2022-11-07T11:44:33.676371Z","shell.execute_reply":"2022-11-07T11:44:33.682223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.702105Z","iopub.execute_input":"2022-11-07T11:44:33.702511Z","iopub.status.idle":"2022-11-07T11:44:33.713005Z","shell.execute_reply.started":"2022-11-07T11:44:33.702480Z","shell.execute_reply":"2022-11-07T11:44:33.712013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(25),\n                     coef.sort_values().tail(25)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.714345Z","iopub.execute_input":"2022-11-07T11:44:33.715624Z","iopub.status.idle":"2022-11-07T11:44:33.722104Z","shell.execute_reply.started":"2022-11-07T11:44:33.715569Z","shell.execute_reply":"2022-11-07T11:44:33.721337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.723486Z","iopub.execute_input":"2022-11-07T11:44:33.724018Z","iopub.status.idle":"2022-11-07T11:44:33.741900Z","shell.execute_reply.started":"2022-11-07T11:44:33.723986Z","shell.execute_reply":"2022-11-07T11:44:33.741145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 10.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:33.745812Z","iopub.execute_input":"2022-11-07T11:44:33.746399Z","iopub.status.idle":"2022-11-07T11:44:34.275086Z","shell.execute_reply.started":"2022-11-07T11:44:33.746366Z","shell.execute_reply":"2022-11-07T11:44:34.274109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:34.277123Z","iopub.execute_input":"2022-11-07T11:44:34.277554Z","iopub.status.idle":"2022-11-07T11:44:34.290904Z","shell.execute_reply.started":"2022-11-07T11:44:34.277519Z","shell.execute_reply":"2022-11-07T11:44:34.290008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso 100","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 100\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:34.292319Z","iopub.execute_input":"2022-11-07T11:44:34.292620Z","iopub.status.idle":"2022-11-07T11:44:34.490669Z","shell.execute_reply.started":"2022-11-07T11:44:34.292592Z","shell.execute_reply":"2022-11-07T11:44:34.489129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:34.492302Z","iopub.execute_input":"2022-11-07T11:44:34.492905Z","iopub.status.idle":"2022-11-07T11:44:37.597400Z","shell.execute_reply.started":"2022-11-07T11:44:34.492871Z","shell.execute_reply":"2022-11-07T11:44:37.596452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=0.01)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_100', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.602002Z","iopub.execute_input":"2022-11-07T11:44:37.604443Z","iopub.status.idle":"2022-11-07T11:44:37.787128Z","shell.execute_reply.started":"2022-11-07T11:44:37.604407Z","shell.execute_reply":"2022-11-07T11:44:37.786016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.792001Z","iopub.execute_input":"2022-11-07T11:44:37.792689Z","iopub.status.idle":"2022-11-07T11:44:37.800080Z","shell.execute_reply.started":"2022-11-07T11:44:37.792652Z","shell.execute_reply":"2022-11-07T11:44:37.799107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.803287Z","iopub.execute_input":"2022-11-07T11:44:37.804455Z","iopub.status.idle":"2022-11-07T11:44:37.810578Z","shell.execute_reply.started":"2022-11-07T11:44:37.804415Z","shell.execute_reply":"2022-11-07T11:44:37.809460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.812497Z","iopub.execute_input":"2022-11-07T11:44:37.813162Z","iopub.status.idle":"2022-11-07T11:44:37.823806Z","shell.execute_reply.started":"2022-11-07T11:44:37.813129Z","shell.execute_reply":"2022-11-07T11:44:37.822492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(50),\n                     coef.sort_values().tail(50)])\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.826693Z","iopub.execute_input":"2022-11-07T11:44:37.827992Z","iopub.status.idle":"2022-11-07T11:44:37.839732Z","shell.execute_reply.started":"2022-11-07T11:44:37.827931Z","shell.execute_reply":"2022-11-07T11:44:37.838243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 100)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.843084Z","iopub.execute_input":"2022-11-07T11:44:37.844233Z","iopub.status.idle":"2022-11-07T11:44:37.851459Z","shell.execute_reply.started":"2022-11-07T11:44:37.844193Z","shell.execute_reply":"2022-11-07T11:44:37.850026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.854942Z","iopub.execute_input":"2022-11-07T11:44:37.856268Z","iopub.status.idle":"2022-11-07T11:44:37.876373Z","shell.execute_reply.started":"2022-11-07T11:44:37.856230Z","shell.execute_reply":"2022-11-07T11:44:37.875376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 10.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:37.878489Z","iopub.execute_input":"2022-11-07T11:44:37.879784Z","iopub.status.idle":"2022-11-07T11:44:38.824940Z","shell.execute_reply.started":"2022-11-07T11:44:37.879750Z","shell.execute_reply":"2022-11-07T11:44:38.824299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:38.826140Z","iopub.execute_input":"2022-11-07T11:44:38.826670Z","iopub.status.idle":"2022-11-07T11:44:38.839897Z","shell.execute_reply.started":"2022-11-07T11:44:38.826642Z","shell.execute_reply":"2022-11-07T11:44:38.838548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso 200","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 200\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:38.841628Z","iopub.execute_input":"2022-11-07T11:44:38.842291Z","iopub.status.idle":"2022-11-07T11:44:39.097852Z","shell.execute_reply.started":"2022-11-07T11:44:38.842246Z","shell.execute_reply":"2022-11-07T11:44:39.096900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20,30,40]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:39.099174Z","iopub.execute_input":"2022-11-07T11:44:39.100740Z","iopub.status.idle":"2022-11-07T11:44:45.460807Z","shell.execute_reply.started":"2022-11-07T11:44:39.100664Z","shell.execute_reply":"2022-11-07T11:44:45.459807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=1)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_200', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.463072Z","iopub.execute_input":"2022-11-07T11:44:45.464501Z","iopub.status.idle":"2022-11-07T11:44:45.591002Z","shell.execute_reply.started":"2022-11-07T11:44:45.464462Z","shell.execute_reply":"2022-11-07T11:44:45.590174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.595839Z","iopub.execute_input":"2022-11-07T11:44:45.599176Z","iopub.status.idle":"2022-11-07T11:44:45.614298Z","shell.execute_reply.started":"2022-11-07T11:44:45.599137Z","shell.execute_reply":"2022-11-07T11:44:45.613313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.619163Z","iopub.execute_input":"2022-11-07T11:44:45.621750Z","iopub.status.idle":"2022-11-07T11:44:45.628362Z","shell.execute_reply.started":"2022-11-07T11:44:45.621705Z","shell.execute_reply":"2022-11-07T11:44:45.627413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.632847Z","iopub.execute_input":"2022-11-07T11:44:45.633616Z","iopub.status.idle":"2022-11-07T11:44:45.643291Z","shell.execute_reply.started":"2022-11-07T11:44:45.633581Z","shell.execute_reply":"2022-11-07T11:44:45.642453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(8),\n                     coef.sort_values().tail(12)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.644764Z","iopub.execute_input":"2022-11-07T11:44:45.646246Z","iopub.status.idle":"2022-11-07T11:44:45.661879Z","shell.execute_reply.started":"2022-11-07T11:44:45.646212Z","shell.execute_reply":"2022-11-07T11:44:45.660870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.663619Z","iopub.execute_input":"2022-11-07T11:44:45.664283Z","iopub.status.idle":"2022-11-07T11:44:45.674632Z","shell.execute_reply.started":"2022-11-07T11:44:45.664247Z","shell.execute_reply":"2022-11-07T11:44:45.673785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 10.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.677326Z","iopub.execute_input":"2022-11-07T11:44:45.681530Z","iopub.status.idle":"2022-11-07T11:44:45.902201Z","shell.execute_reply.started":"2022-11-07T11:44:45.681494Z","shell.execute_reply":"2022-11-07T11:44:45.900244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.904809Z","iopub.execute_input":"2022-11-07T11:44:45.905616Z","iopub.status.idle":"2022-11-07T11:44:45.924652Z","shell.execute_reply.started":"2022-11-07T11:44:45.905541Z","shell.execute_reply":"2022-11-07T11:44:45.922381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso 500","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 500\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:45.927314Z","iopub.execute_input":"2022-11-07T11:44:45.927830Z","iopub.status.idle":"2022-11-07T11:44:46.549157Z","shell.execute_reply.started":"2022-11-07T11:44:45.927781Z","shell.execute_reply":"2022-11-07T11:44:46.547394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:46.551081Z","iopub.execute_input":"2022-11-07T11:44:46.551467Z","iopub.status.idle":"2022-11-07T11:44:58.365219Z","shell.execute_reply.started":"2022-11-07T11:44:46.551432Z","shell.execute_reply":"2022-11-07T11:44:58.364410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=1)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_500', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.367099Z","iopub.execute_input":"2022-11-07T11:44:58.368434Z","iopub.status.idle":"2022-11-07T11:44:58.568922Z","shell.execute_reply.started":"2022-11-07T11:44:58.368393Z","shell.execute_reply":"2022-11-07T11:44:58.568163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.570359Z","iopub.execute_input":"2022-11-07T11:44:58.570897Z","iopub.status.idle":"2022-11-07T11:44:58.581996Z","shell.execute_reply.started":"2022-11-07T11:44:58.570865Z","shell.execute_reply":"2022-11-07T11:44:58.579784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.587277Z","iopub.execute_input":"2022-11-07T11:44:58.590369Z","iopub.status.idle":"2022-11-07T11:44:58.597249Z","shell.execute_reply.started":"2022-11-07T11:44:58.590320Z","shell.execute_reply":"2022-11-07T11:44:58.596298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.598890Z","iopub.execute_input":"2022-11-07T11:44:58.600254Z","iopub.status.idle":"2022-11-07T11:44:58.612176Z","shell.execute_reply.started":"2022-11-07T11:44:58.600217Z","shell.execute_reply":"2022-11-07T11:44:58.611272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(8),\n                     coef.sort_values().tail(12)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.614499Z","iopub.execute_input":"2022-11-07T11:44:58.615203Z","iopub.status.idle":"2022-11-07T11:44:58.632680Z","shell.execute_reply.started":"2022-11-07T11:44:58.615162Z","shell.execute_reply":"2022-11-07T11:44:58.631715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.634547Z","iopub.execute_input":"2022-11-07T11:44:58.635182Z","iopub.status.idle":"2022-11-07T11:44:58.648837Z","shell.execute_reply.started":"2022-11-07T11:44:58.635149Z","shell.execute_reply":"2022-11-07T11:44:58.647672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 7.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.650839Z","iopub.execute_input":"2022-11-07T11:44:58.651504Z","iopub.status.idle":"2022-11-07T11:44:58.892115Z","shell.execute_reply.started":"2022-11-07T11:44:58.651467Z","shell.execute_reply":"2022-11-07T11:44:58.889393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.896081Z","iopub.execute_input":"2022-11-07T11:44:58.896478Z","iopub.status.idle":"2022-11-07T11:44:58.911730Z","shell.execute_reply.started":"2022-11-07T11:44:58.896446Z","shell.execute_reply":"2022-11-07T11:44:58.910248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lasso 1000","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 1000\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:44:58.913352Z","iopub.execute_input":"2022-11-07T11:44:58.913734Z","iopub.status.idle":"2022-11-07T11:45:00.636991Z","shell.execute_reply.started":"2022-11-07T11:44:58.913699Z","shell.execute_reply":"2022-11-07T11:45:00.635409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:45:00.638348Z","iopub.execute_input":"2022-11-07T11:45:00.638654Z","iopub.status.idle":"2022-11-07T11:45:59.990583Z","shell.execute_reply.started":"2022-11-07T11:45:00.638626Z","shell.execute_reply":"2022-11-07T11:45:59.988706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=1)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_1000', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:45:59.995056Z","iopub.execute_input":"2022-11-07T11:45:59.995597Z","iopub.status.idle":"2022-11-07T11:46:00.347000Z","shell.execute_reply.started":"2022-11-07T11:45:59.995563Z","shell.execute_reply":"2022-11-07T11:46:00.346006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.348617Z","iopub.execute_input":"2022-11-07T11:46:00.349274Z","iopub.status.idle":"2022-11-07T11:46:00.363675Z","shell.execute_reply.started":"2022-11-07T11:46:00.349231Z","shell.execute_reply":"2022-11-07T11:46:00.362643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.368419Z","iopub.execute_input":"2022-11-07T11:46:00.369048Z","iopub.status.idle":"2022-11-07T11:46:00.377689Z","shell.execute_reply.started":"2022-11-07T11:46:00.369016Z","shell.execute_reply":"2022-11-07T11:46:00.376890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.384418Z","iopub.execute_input":"2022-11-07T11:46:00.386781Z","iopub.status.idle":"2022-11-07T11:46:00.400941Z","shell.execute_reply.started":"2022-11-07T11:46:00.386719Z","shell.execute_reply":"2022-11-07T11:46:00.399838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(11),\n                     coef.sort_values().tail(11)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.402440Z","iopub.execute_input":"2022-11-07T11:46:00.403180Z","iopub.status.idle":"2022-11-07T11:46:00.416369Z","shell.execute_reply.started":"2022-11-07T11:46:00.403146Z","shell.execute_reply":"2022-11-07T11:46:00.415510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.417925Z","iopub.execute_input":"2022-11-07T11:46:00.418491Z","iopub.status.idle":"2022-11-07T11:46:00.434147Z","shell.execute_reply.started":"2022-11-07T11:46:00.418457Z","shell.execute_reply":"2022-11-07T11:46:00.433042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 7.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.435639Z","iopub.execute_input":"2022-11-07T11:46:00.436212Z","iopub.status.idle":"2022-11-07T11:46:00.686284Z","shell.execute_reply.started":"2022-11-07T11:46:00.436178Z","shell.execute_reply":"2022-11-07T11:46:00.685215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.687433Z","iopub.execute_input":"2022-11-07T11:46:00.687688Z","iopub.status.idle":"2022-11-07T11:46:00.700475Z","shell.execute_reply.started":"2022-11-07T11:46:00.687664Z","shell.execute_reply":"2022-11-07T11:46:00.699733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### lasso 3800","metadata":{}},{"cell_type":"code","source":"# Create X,y, X_train, y_train\nselected_target = 'CD31'\nn_features = 3800\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\nX_test2 = X_test_public_like\ny_test2 = y_test_public_like\n\nmask_out_of_playground = (df_meta['Playground']==0)\nX_oop = df_cite.iloc[:70988,:n_features][mask_out_of_playground]\ny_oop = df_cite_train_y[mask_out_of_playground][ selected_target ]\n\n\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:00.701616Z","iopub.execute_input":"2022-11-07T11:46:00.702057Z","iopub.status.idle":"2022-11-07T11:46:06.740722Z","shell.execute_reply.started":"2022-11-07T11:46:00.702030Z","shell.execute_reply":"2022-11-07T11:46:06.739329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.model_selection import GridSearchCV\nlasso = Lasso()\n\nparameters = {\"alpha\":[1e-15, 1e-10, 1e-8, 1e-4, 1e-3, 1e-2, 1, 5, 10, 20]}\nlasso_regression = GridSearchCV(lasso, parameters, scoring='neg_mean_squared_error', cv=5)\nlasso_regression.fit(X_train , y_train)\n\nprint(lasso_regression.best_params_)\nprint(lasso_regression.best_score_)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:46:06.742071Z","iopub.execute_input":"2022-11-07T11:46:06.742399Z","iopub.status.idle":"2022-11-07T11:49:47.332213Z","shell.execute_reply.started":"2022-11-07T11:46:06.742371Z","shell.execute_reply":"2022-11-07T11:49:47.331373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.linear_model import Lasso\nt0 = time.time()\nmodel = Lasso(alpha=1)\nmodel.fit(X_train,y_train)\ny_train = model.predict(X_train)\ny_pred = model.predict(X_test)\nupdate_models_stat(model, 'Lasso_1000', t0 ) # Updates df_models_stat\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:47.333812Z","iopub.execute_input":"2022-11-07T11:49:47.334429Z","iopub.status.idle":"2022-11-07T11:49:48.678721Z","shell.execute_reply.started":"2022-11-07T11:49:47.334396Z","shell.execute_reply":"2022-11-07T11:49:48.677936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eps = 1e-6\nlasso_coef = model.coef_\nprint('Zero coefficients:', sum(np.abs(lasso_coef)< eps))\nprint('All coefficients:', lasso_coef.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.680232Z","iopub.execute_input":"2022-11-07T11:49:48.680807Z","iopub.status.idle":"2022-11-07T11:49:48.701206Z","shell.execute_reply.started":"2022-11-07T11:49:48.680774Z","shell.execute_reply":"2022-11-07T11:49:48.700128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = pd.Series(model.coef_, index = X_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.703266Z","iopub.execute_input":"2022-11-07T11:49:48.704035Z","iopub.status.idle":"2022-11-07T11:49:48.712842Z","shell.execute_reply.started":"2022-11-07T11:49:48.703995Z","shell.execute_reply":"2022-11-07T11:49:48.711799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Lasso picked \" + str(sum(coef != 0)) + \" variables and eliminated the other \" +  str(sum(coef == 0)) + \" variables\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.714402Z","iopub.execute_input":"2022-11-07T11:49:48.715170Z","iopub.status.idle":"2022-11-07T11:49:48.727154Z","shell.execute_reply.started":"2022-11-07T11:49:48.715128Z","shell.execute_reply":"2022-11-07T11:49:48.725866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_coef = pd.concat([coef.sort_values().head(11),\n                     coef.sort_values().tail(11)])","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.728677Z","iopub.execute_input":"2022-11-07T11:49:48.729187Z","iopub.status.idle":"2022-11-07T11:49:48.740359Z","shell.execute_reply.started":"2022-11-07T11:49:48.729154Z","shell.execute_reply":"2022-11-07T11:49:48.739223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 100)","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.741708Z","iopub.execute_input":"2022-11-07T11:49:48.742070Z","iopub.status.idle":"2022-11-07T11:49:48.750493Z","shell.execute_reply.started":"2022-11-07T11:49:48.742040Z","shell.execute_reply":"2022-11-07T11:49:48.749000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The most important coefficients are:')\nimp_coef","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.761575Z","iopub.execute_input":"2022-11-07T11:49:48.762302Z","iopub.status.idle":"2022-11-07T11:49:48.773371Z","shell.execute_reply.started":"2022-11-07T11:49:48.762261Z","shell.execute_reply":"2022-11-07T11:49:48.772437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matplotlib.rcParams['figure.figsize'] = (8.0, 7.0)\nimp_coef.plot(kind = \"barh\")\nplt.title(\"Coefficients in the Lasso Model\")","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:48.775005Z","iopub.execute_input":"2022-11-07T11:49:48.775563Z","iopub.status.idle":"2022-11-07T11:49:49.024414Z","shell.execute_reply.started":"2022-11-07T11:49:48.775528Z","shell.execute_reply":"2022-11-07T11:49:49.023560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat","metadata":{"execution":{"iopub.status.busy":"2022-11-07T11:49:49.025706Z","iopub.execute_input":"2022-11-07T11:49:49.026307Z","iopub.status.idle":"2022-11-07T11:49:49.042731Z","shell.execute_reply.started":"2022-11-07T11:49:49.026268Z","shell.execute_reply":"2022-11-07T11:49:49.041538Z"},"trusted":true},"execution_count":null,"outputs":[]}]}