{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-11T22:45:29.558262Z","iopub.execute_input":"2023-11-11T22:45:29.558690Z","iopub.status.idle":"2023-11-11T22:45:30.810083Z","shell.execute_reply.started":"2023-11-11T22:45:29.558659Z","shell.execute_reply":"2023-11-11T22:45:30.808707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Memory Reducer","metadata":{}},{"cell_type":"code","source":"#Thanks for https://qiita.com/kaggle_grandmaster-arai-san/items/d59b2fb7142ec7e270a5#reduce_mem_usage\n\nimport numpy as np\nimport pandas as pd\n\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:45:30.812860Z","iopub.execute_input":"2023-11-11T22:45:30.813698Z","iopub.status.idle":"2023-11-11T22:45:30.829988Z","shell.execute_reply.started":"2023-11-11T22:45:30.813659Z","shell.execute_reply":"2023-11-11T22:45:30.828273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings # suppress warnings\nwarnings.filterwarnings('ignore')\n\nimport pandas as pd\nimport dask\nimport dask.dataframe as ddf","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:45:30.832174Z","iopub.execute_input":"2023-11-11T22:45:30.832595Z","iopub.status.idle":"2023-11-11T22:45:32.333312Z","shell.execute_reply.started":"2023-11-11T22:45:30.832559Z","shell.execute_reply":"2023-11-11T22:45:32.332136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv')\nde_train = reduce_mem_usage(pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'))\nadata_train = reduce_mem_usage(pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet'))#heavy\nadata_obs_meta = reduce_mem_usage(pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/adata_obs_meta.csv'))\n\n#multiome_obs_meta = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/multiome_obs_meta.csv')\n#multiome_train = reduce_mem_usage(pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/multiome_train.parquet'))#heavy\n#multiome_var_meta = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/multiome_var_meta.csv')\n\nsample = reduce_mem_usage(pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:45:32.337275Z","iopub.execute_input":"2023-11-11T22:45:32.338249Z","iopub.status.idle":"2023-11-11T22:48:30.498824Z","shell.execute_reply.started":"2023-11-11T22:45:32.338188Z","shell.execute_reply":"2023-11-11T22:48:30.496782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.501493Z","iopub.execute_input":"2023-11-11T22:48:30.502028Z","iopub.status.idle":"2023-11-11T22:48:30.522971Z","shell.execute_reply.started":"2023-11-11T22:48:30.501987Z","shell.execute_reply":"2023-11-11T22:48:30.520964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.525448Z","iopub.execute_input":"2023-11-11T22:48:30.525866Z","iopub.status.idle":"2023-11-11T22:48:30.846201Z","shell.execute_reply.started":"2023-11-11T22:48:30.525835Z","shell.execute_reply":"2023-11-11T22:48:30.843868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.847842Z","iopub.execute_input":"2023-11-11T22:48:30.848194Z","iopub.status.idle":"2023-11-11T22:48:30.861835Z","shell.execute_reply.started":"2023-11-11T22:48:30.848170Z","shell.execute_reply":"2023-11-11T22:48:30.859563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata_obs_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.863512Z","iopub.execute_input":"2023-11-11T22:48:30.864082Z","iopub.status.idle":"2023-11-11T22:48:30.888571Z","shell.execute_reply.started":"2023-11-11T22:48:30.864036Z","shell.execute_reply":"2023-11-11T22:48:30.886436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiome_obs_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.890326Z","iopub.execute_input":"2023-11-11T22:48:30.890684Z","iopub.status.idle":"2023-11-11T22:48:30.897352Z","shell.execute_reply.started":"2023-11-11T22:48:30.890653Z","shell.execute_reply":"2023-11-11T22:48:30.896178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiome_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.902231Z","iopub.execute_input":"2023-11-11T22:48:30.902665Z","iopub.status.idle":"2023-11-11T22:48:30.909701Z","shell.execute_reply.started":"2023-11-11T22:48:30.902627Z","shell.execute_reply":"2023-11-11T22:48:30.908515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiome_var_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.911567Z","iopub.execute_input":"2023-11-11T22:48:30.912489Z","iopub.status.idle":"2023-11-11T22:48:30.921444Z","shell.execute_reply.started":"2023-11-11T22:48:30.912431Z","shell.execute_reply":"2023-11-11T22:48:30.919580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unique values\n### de_train\n","metadata":{}},{"cell_type":"code","source":"for col in de_train.columns[:5]:\n  print('--', col)\n  print(de_train[col].value_counts(), '\\n')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.923072Z","iopub.execute_input":"2023-11-11T22:48:30.924045Z","iopub.status.idle":"2023-11-11T22:48:30.947131Z","shell.execute_reply.started":"2023-11-11T22:48:30.924007Z","shell.execute_reply":"2023-11-11T22:48:30.946271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### id_map","metadata":{}},{"cell_type":"code","source":"for col in id_map.columns[1:]:\n  print('--', col)\n  print(id_map[col].value_counts(), '\\n')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.947982Z","iopub.execute_input":"2023-11-11T22:48:30.948312Z","iopub.status.idle":"2023-11-11T22:48:30.961707Z","shell.execute_reply.started":"2023-11-11T22:48:30.948287Z","shell.execute_reply":"2023-11-11T22:48:30.959706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### adata_train","metadata":{}},{"cell_type":"code","source":"# Done later","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.963508Z","iopub.execute_input":"2023-11-11T22:48:30.963864Z","iopub.status.idle":"2023-11-11T22:48:30.972858Z","shell.execute_reply.started":"2023-11-11T22:48:30.963836Z","shell.execute_reply":"2023-11-11T22:48:30.970816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### adata_obs_meta","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Data into X and Y","metadata":{}},{"cell_type":"code","source":"X=de_train.iloc[:, :2]\ny=de_train.iloc[:, 5:]\n\n#datas by conditions\ncelltype_dict = {}\nfor celltype in list(X.cell_type.unique()):\n    y_cell = y[X['cell_type']==celltype]\n    celltype_dict[celltype] = y_cell\n\n#NK cells              146\n#T cells CD4+          146\n#T regulatory cells    146\n#T cells CD8+          142\n#B cells                17\n#Myeloid cells          17\n\nX.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:30.976645Z","iopub.execute_input":"2023-11-11T22:48:30.977115Z","iopub.status.idle":"2023-11-11T22:48:33.876422Z","shell.execute_reply.started":"2023-11-11T22:48:30.977062Z","shell.execute_reply":"2023-11-11T22:48:33.874847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:33.878302Z","iopub.execute_input":"2023-11-11T22:48:33.878671Z","iopub.status.idle":"2023-11-11T22:48:33.972772Z","shell.execute_reply.started":"2023-11-11T22:48:33.878642Z","shell.execute_reply":"2023-11-11T22:48:33.971461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:33.974554Z","iopub.execute_input":"2023-11-11T22:48:33.974973Z","iopub.status.idle":"2023-11-11T22:48:34.834880Z","shell.execute_reply.started":"2023-11-11T22:48:33.974934Z","shell.execute_reply":"2023-11-11T22:48:34.833226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# cell type","metadata":{}},{"cell_type":"code","source":"#Train\nprint('--Cell types in de_train--')\nprint(de_train['cell_type'].value_counts(), '\\n')\n\n#id_map\nprint('--Cell types in id_map--')\nprint(id_map['cell_type'].value_counts(),'\\n')\n\n#adata\nprint('--Cell types in adata_obs--')\nprint(adata_obs_meta['cell_type'].value_counts(),'\\n')\n\n#multiome\n#print('--Cell types in multiome_obs--')\n#print(multiome_obs_meta['cell_type'].value_counts(),'\\n')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.837064Z","iopub.execute_input":"2023-11-11T22:48:34.837571Z","iopub.status.idle":"2023-11-11T22:48:34.853711Z","shell.execute_reply.started":"2023-11-11T22:48:34.837534Z","shell.execute_reply":"2023-11-11T22:48:34.852219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Few predicted cells are in de_train... we need some techniques here.","metadata":{}},{"cell_type":"markdown","source":"# small molecule","metadata":{}},{"cell_type":"code","source":"#Train\nTrain_mol = de_train['sm_name'].unique()\nprint('train',len(de_train['sm_name'].unique()))\n#id_map\nTest_mol = id_map['sm_name'].unique()\nprint('test', len(id_map['sm_name'].unique()))\n\nsm_common = list(set(Train_mol) & set(Test_mol))\nsm_uncommon = [mol for mol in Train_mol if mol not in sm_common]\n\nprint(len(sm_uncommon))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.855481Z","iopub.execute_input":"2023-11-11T22:48:34.855840Z","iopub.status.idle":"2023-11-11T22:48:34.869762Z","shell.execute_reply.started":"2023-11-11T22:48:34.855812Z","shell.execute_reply":"2023-11-11T22:48:34.868640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('--Molecules in adata_obs--')\nprint(adata_obs_meta['sm_name'].value_counts())\nadata_mol = adata_obs_meta['sm_name'].unique()\n\nsm_common_all = list(set(sm_common) & set(adata_mol))\nsm_common_adatatrain = list(set(Train_mol) & set(adata_mol))\nsm_common_adatatest =  list(set(Test_mol) & set(adata_mol))\nprint(len(sm_common_all))\nprint(len(sm_common_adatatrain))\nprint(len(sm_common_adatatest))\n\n# Now it figure out that mol in test and adata are the same","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.871660Z","iopub.execute_input":"2023-11-11T22:48:34.872185Z","iopub.status.idle":"2023-11-11T22:48:34.886839Z","shell.execute_reply.started":"2023-11-11T22:48:34.872144Z","shell.execute_reply":"2023-11-11T22:48:34.885506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# obs_id","metadata":{}},{"cell_type":"code","source":"#adata_meta_obsid = reduce_mem_usage(adata_obs_meta['obs_id'].unique())\n#print(len(adata_meta_obsid))\n#adata_train_obsid = reduce_mem_usage(adata_train['obs_id'].unique())\n#print(len(adata_train_obsid))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.889251Z","iopub.execute_input":"2023-11-11T22:48:34.889659Z","iopub.status.idle":"2023-11-11T22:48:34.894169Z","shell.execute_reply.started":"2023-11-11T22:48:34.889622Z","shell.execute_reply":"2023-11-11T22:48:34.893168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiome_meta_obsid = multiome_obs_meta['obs_id'].unique()\n#print(len(multiome_meta_obsid))\n#multiome_train_obsid = multiome_train['obs_id'].unique()\n#print(len(multiome_train_obsid))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.895243Z","iopub.execute_input":"2023-11-11T22:48:34.895607Z","iopub.status.idle":"2023-11-11T22:48:34.904880Z","shell.execute_reply.started":"2023-11-11T22:48:34.895574Z","shell.execute_reply":"2023-11-11T22:48:34.903999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# genes","metadata":{}},{"cell_type":"code","source":"train_genes = de_train.columns\nprint(len(train_genes))\nadata_genes =adata_train['gene'].unique()\nprint(len(adata_genes))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:34.905829Z","iopub.execute_input":"2023-11-11T22:48:34.906198Z","iopub.status.idle":"2023-11-11T22:48:36.320763Z","shell.execute_reply.started":"2023-11-11T22:48:34.906168Z","shell.execute_reply":"2023-11-11T22:48:36.319539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiome_genes = multiome_var_meta[multiome_var_meta['feature_type'] == 'Gene Expression']['location'].unique()\n#print(len(multiome_genes))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.322061Z","iopub.execute_input":"2023-11-11T22:48:36.325445Z","iopub.status.idle":"2023-11-11T22:48:36.329833Z","shell.execute_reply.started":"2023-11-11T22:48:36.325413Z","shell.execute_reply":"2023-11-11T22:48:36.328602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_common_trainadata = list(set(train_genes) & set(adata_genes))\n#gene_common_multiomeadata = list(set(multiome_genes) & set(adata_genes))\n#gene_common_all = list(set(train_genes) & set(gene_common_multiomeadata))\nprint(len(gene_common_trainadata))\n#print(len(gene_common_multiomeadata))\n#print(len(gene_common_all))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.331410Z","iopub.execute_input":"2023-11-11T22:48:36.331813Z","iopub.status.idle":"2023-11-11T22:48:36.353296Z","shell.execute_reply.started":"2023-11-11T22:48:36.331779Z","shell.execute_reply":"2023-11-11T22:48:36.351642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Subsetting Clusters of interest","metadata":{}},{"cell_type":"code","source":"#gb = adata_obs_meta.groupby('cell_type')\n#B_adata, Myeloid_adata, NK_adata, CD4_adata, CD8_adata, Treg_adata = [gb.get_group(x) for x in gb.groups]\n#gb = multiome_obs_meta.groupby('cell_type')\n#B_mlt, Myeloid_mlt, NK_mlt, CD4_mlt, CD8_mlt, Treg_mlt = [gb.get_group(x) for x in gb.groups]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.354802Z","iopub.execute_input":"2023-11-11T22:48:36.355247Z","iopub.status.idle":"2023-11-11T22:48:36.368271Z","shell.execute_reply.started":"2023-11-11T22:48:36.355216Z","shell.execute_reply":"2023-11-11T22:48:36.367289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA2\n### By cell_type","metadata":{}},{"cell_type":"code","source":"#B_merge, Myeloid_merge, NK_merge, CD4_merge, CD8_merge, Treg_merge = [adata_train.merge(adata, on='obs_id') for adata in [B_adata, Myeloid_adata, NK_adata, CD4_adata, CD8_adata, Treg_adata]]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.369867Z","iopub.execute_input":"2023-11-11T22:48:36.370609Z","iopub.status.idle":"2023-11-11T22:48:36.380494Z","shell.execute_reply.started":"2023-11-11T22:48:36.370578Z","shell.execute_reply":"2023-11-11T22:48:36.378773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#B_counts = B_merge[['count', 'normalized_count']]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.390117Z","iopub.execute_input":"2023-11-11T22:48:36.390473Z","iopub.status.idle":"2023-11-11T22:48:36.394846Z","shell.execute_reply.started":"2023-11-11T22:48:36.390450Z","shell.execute_reply":"2023-11-11T22:48:36.394119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import datashader as ds\n#from datashader import transfer_functions as tf\n\n#tf.shade(ds.Canvas().points(B_counts,'counts','normalized_count'))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.396165Z","iopub.execute_input":"2023-11-11T22:48:36.396629Z","iopub.status.idle":"2023-11-11T22:48:36.406410Z","shell.execute_reply.started":"2023-11-11T22:48:36.396604Z","shell.execute_reply":"2023-11-11T22:48:36.404910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import seaborn as sns\n#sns.clustermap(y, vmin=-50, vmax=50)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.408600Z","iopub.execute_input":"2023-11-11T22:48:36.409242Z","iopub.status.idle":"2023-11-11T22:48:36.416857Z","shell.execute_reply.started":"2023-11-11T22:48:36.409201Z","shell.execute_reply":"2023-11-11T22:48:36.416017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X and y set","metadata":{}},{"cell_type":"raw","source":"X and Y set\n- X VS y: basic\n- X_bl VS y_bl: mol appears both train and test\n- X VS y_PCA: PCA is performed\n- X_bl VS y_PCA_bl: Both mol selection and PCA are performed\n- X_dummies_bl VS y_bl: sm_name and cell type are represented by dummt variable, with mol selection","metadata":{}},{"cell_type":"code","source":"X_features = ['cell_type','sm_name']\n\ny_bl = y[X['sm_name'].isin(sm_common)]\nX_bl = X[X['sm_name'].isin(sm_common)]\n#drop_firstなんとかしたい\nX_dummies = pd.get_dummies(X, prefix = '', prefix_sep = '', columns=X_features)\nid_map_dummies = pd.get_dummies(id_map[X_features], prefix = '', prefix_sep = '', columns=X_features)\n\nuncommon = [f for f in X_dummies if f not in id_map_dummies]\nX_dummies_bl = X_dummies.drop(columns=uncommon)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.418048Z","iopub.execute_input":"2023-11-11T22:48:36.418372Z","iopub.status.idle":"2023-11-11T22:48:36.817925Z","shell.execute_reply.started":"2023-11-11T22:48:36.418349Z","shell.execute_reply":"2023-11-11T22:48:36.816839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_dummies_bl","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.819118Z","iopub.execute_input":"2023-11-11T22:48:36.819423Z","iopub.status.idle":"2023-11-11T22:48:36.848778Z","shell.execute_reply.started":"2023-11-11T22:48:36.819401Z","shell.execute_reply":"2023-11-11T22:48:36.847360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map_dummies","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.850784Z","iopub.execute_input":"2023-11-11T22:48:36.851157Z","iopub.status.idle":"2023-11-11T22:48:36.878320Z","shell.execute_reply.started":"2023-11-11T22:48:36.851128Z","shell.execute_reply":"2023-11-11T22:48:36.876912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fcluster(Z, t=100, criterion='maxclust')\nX","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.880678Z","iopub.execute_input":"2023-11-11T22:48:36.881068Z","iopub.status.idle":"2023-11-11T22:48:36.897182Z","shell.execute_reply.started":"2023-11-11T22:48:36.881040Z","shell.execute_reply":"2023-11-11T22:48:36.896207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"train = de_train.copy()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:36.898120Z","iopub.execute_input":"2023-11-11T22:48:36.898437Z","iopub.status.idle":"2023-11-11T22:48:37.467745Z","shell.execute_reply.started":"2023-11-11T22:48:36.898410Z","shell.execute_reply":"2023-11-11T22:48:37.466179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train = de_train[de_train['control'] == 0]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:37.469212Z","iopub.execute_input":"2023-11-11T22:48:37.469631Z","iopub.status.idle":"2023-11-11T22:48:37.863964Z","shell.execute_reply.started":"2023-11-11T22:48:37.469596Z","shell.execute_reply":"2023-11-11T22:48:37.862728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:37.866305Z","iopub.execute_input":"2023-11-11T22:48:37.866827Z","iopub.status.idle":"2023-11-11T22:48:37.942672Z","shell.execute_reply.started":"2023-11-11T22:48:37.866798Z","shell.execute_reply":"2023-11-11T22:48:37.941782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SMILES","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:37.943802Z","iopub.execute_input":"2023-11-11T22:48:37.944269Z","iopub.status.idle":"2023-11-11T22:48:49.579170Z","shell.execute_reply.started":"2023-11-11T22:48:37.944244Z","shell.execute_reply":"2023-11-11T22:48:49.578186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:49.581145Z","iopub.execute_input":"2023-11-11T22:48:49.581575Z","iopub.status.idle":"2023-11-11T22:48:50.499964Z","shell.execute_reply.started":"2023-11-11T22:48:49.581539Z","shell.execute_reply":"2023-11-11T22:48:50.498759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.502050Z","iopub.execute_input":"2023-11-11T22:48:50.502516Z","iopub.status.idle":"2023-11-11T22:48:50.520137Z","shell.execute_reply.started":"2023-11-11T22:48:50.502479Z","shell.execute_reply":"2023-11-11T22:48:50.518191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sm2smiles = dict(zip(de_train['sm_name'], de_train['SMILES']))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.522224Z","iopub.execute_input":"2023-11-11T22:48:50.523346Z","iopub.status.idle":"2023-11-11T22:48:50.532036Z","shell.execute_reply.started":"2023-11-11T22:48:50.523303Z","shell.execute_reply":"2023-11-11T22:48:50.530600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map['SMILES'] = id_map['sm_name'].map(sm2smiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.533438Z","iopub.execute_input":"2023-11-11T22:48:50.534536Z","iopub.status.idle":"2023-11-11T22:48:50.543297Z","shell.execute_reply.started":"2023-11-11T22:48:50.534496Z","shell.execute_reply":"2023-11-11T22:48:50.541828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit.Chem import Descriptors\ndef rtn_rdkit_desc(mol):\n    dic = {}\n    for tup in Descriptors._descList:\n        dic[tup[0]] = tup[1](mol)\n    return dic","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.544505Z","iopub.execute_input":"2023-11-11T22:48:50.544812Z","iopub.status.idle":"2023-11-11T22:48:50.600666Z","shell.execute_reply.started":"2023-11-11T22:48:50.544787Z","shell.execute_reply":"2023-11-11T22:48:50.598930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(Descriptors._descList)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.602666Z","iopub.execute_input":"2023-11-11T22:48:50.603076Z","iopub.status.idle":"2023-11-11T22:48:50.610841Z","shell.execute_reply.started":"2023-11-11T22:48:50.603037Z","shell.execute_reply":"2023-11-11T22:48:50.608970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for tup in Descriptors._descList:\n    de_train[tup[0]] = de_train['SMILES'].apply(lambda x: tup[1](Chem.MolFromSmiles(x))).astype(float)\n    id_map[tup[0]] = id_map['SMILES'].apply(lambda x: tup[1](Chem.MolFromSmiles(x))).astype(float)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:48:50.613527Z","iopub.execute_input":"2023-11-11T22:48:50.615013Z","iopub.status.idle":"2023-11-11T22:49:39.493605Z","shell.execute_reply.started":"2023-11-11T22:48:50.614977Z","shell.execute_reply":"2023-11-11T22:49:39.492531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nde_train.iloc[:, -210:]= scaler.fit_transform(de_train.iloc[:, -210:])\nid_map.iloc[:, 4:] = scaler.transform(id_map.iloc[:, 4:])","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:39.494869Z","iopub.execute_input":"2023-11-11T22:49:39.495173Z","iopub.status.idle":"2023-11-11T22:49:39.720360Z","shell.execute_reply.started":"2023-11-11T22:49:39.495148Z","shell.execute_reply":"2023-11-11T22:49:39.718568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"de_train['MF'] = de_train['SMILES'].apply(lambda x: AllChem.GetMorganFingerprintAsBitVect(Chem.MolFromSmiles(x), 3, nBits=2048))\nid_map['MF'] = id_map['SMILES'].apply(lambda x: AllChem.GetMorganFingerprintAsBitVect(Chem.MolFromSmiles(x), 3, nBits=2048))\nfgp_array_train = np.array(de_train['MF'].tolist())\nfgp_array_test = np.array(id_map['MF'].tolist())","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:39.722001Z","iopub.execute_input":"2023-11-11T22:49:39.722473Z","iopub.status.idle":"2023-11-11T22:49:41.043889Z","shell.execute_reply.started":"2023-11-11T22:49:39.722441Z","shell.execute_reply":"2023-11-11T22:49:41.041948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.drop(columns='MF', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:41.045447Z","iopub.execute_input":"2023-11-11T22:49:41.045807Z","iopub.status.idle":"2023-11-11T22:49:41.509612Z","shell.execute_reply.started":"2023-11-11T22:49:41.045777Z","shell.execute_reply":"2023-11-11T22:49:41.508727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:41.510815Z","iopub.execute_input":"2023-11-11T22:49:41.511929Z","iopub.status.idle":"2023-11-11T22:49:42.873225Z","shell.execute_reply.started":"2023-11-11T22:49:41.511882Z","shell.execute_reply":"2023-11-11T22:49:42.872243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_cell_dum = pd.get_dummies(de_train['cell_type'], prefix = '', prefix_sep = '')\nid_map_cell_dum = pd.get_dummies(id_map['cell_type'], prefix = '', prefix_sep = '')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:42.874788Z","iopub.execute_input":"2023-11-11T22:49:42.876122Z","iopub.status.idle":"2023-11-11T22:49:42.894380Z","shell.execute_reply.started":"2023-11-11T22:49:42.876054Z","shell.execute_reply":"2023-11-11T22:49:42.892153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_cell_dum","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:42.896581Z","iopub.execute_input":"2023-11-11T22:49:42.897113Z","iopub.status.idle":"2023-11-11T22:49:42.919918Z","shell.execute_reply.started":"2023-11-11T22:49:42.897054Z","shell.execute_reply":"2023-11-11T22:49:42.917618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map_cell_dum[['NK cells', 'T cells CD4','T cells CD8','T regulatory cells']]=False\nid_map_cell_dum","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:42.922041Z","iopub.execute_input":"2023-11-11T22:49:42.923502Z","iopub.status.idle":"2023-11-11T22:49:42.943531Z","shell.execute_reply.started":"2023-11-11T22:49:42.923459Z","shell.execute_reply":"2023-11-11T22:49:42.941872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.iloc[0:5, 18210:]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:42.945667Z","iopub.execute_input":"2023-11-11T22:49:42.947269Z","iopub.status.idle":"2023-11-11T22:49:43.010331Z","shell.execute_reply.started":"2023-11-11T22:49:42.947201Z","shell.execute_reply":"2023-11-11T22:49:43.009163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.iloc[:, 4:-1]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.011731Z","iopub.execute_input":"2023-11-11T22:49:43.012139Z","iopub.status.idle":"2023-11-11T22:49:43.050704Z","shell.execute_reply.started":"2023-11-11T22:49:43.012077Z","shell.execute_reply":"2023-11-11T22:49:43.049773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile = np.hstack((X_cell_dum, de_train.iloc[:, 18216:], fgp_array_train))\nid_map_smile = np.hstack((id_map_cell_dum, id_map.iloc[:, 4:-1], fgp_array_test))\n\ny_smile = np.array(de_train.iloc[:, 5:18216])","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.052134Z","iopub.execute_input":"2023-11-11T22:49:43.052520Z","iopub.status.idle":"2023-11-11T22:49:43.591516Z","shell.execute_reply.started":"2023-11-11T22:49:43.052488Z","shell.execute_reply":"2023-11-11T22:49:43.590182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile, X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.592845Z","iopub.execute_input":"2023-11-11T22:49:43.593165Z","iopub.status.idle":"2023-11-11T22:49:43.602989Z","shell.execute_reply.started":"2023-11-11T22:49:43.593140Z","shell.execute_reply":"2023-11-11T22:49:43.601655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.604475Z","iopub.execute_input":"2023-11-11T22:49:43.604867Z","iopub.status.idle":"2023-11-11T22:49:43.616429Z","shell.execute_reply.started":"2023-11-11T22:49:43.604837Z","shell.execute_reply":"2023-11-11T22:49:43.614626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_smile","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.617886Z","iopub.execute_input":"2023-11-11T22:49:43.618277Z","iopub.status.idle":"2023-11-11T22:49:43.629565Z","shell.execute_reply.started":"2023-11-11T22:49:43.618247Z","shell.execute_reply":"2023-11-11T22:49:43.628455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.630791Z","iopub.execute_input":"2023-11-11T22:49:43.631151Z","iopub.status.idle":"2023-11-11T22:49:43.643791Z","shell.execute_reply.started":"2023-11-11T22:49:43.631121Z","shell.execute_reply":"2023-11-11T22:49:43.641291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA?","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.645255Z","iopub.execute_input":"2023-11-11T22:49:43.645621Z","iopub.status.idle":"2023-11-11T22:49:43.972170Z","shell.execute_reply.started":"2023-11-11T22:49:43.645594Z","shell.execute_reply":"2023-11-11T22:49:43.970152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Shapes","metadata":{}},{"cell_type":"code","source":"print('X |',np.shape(X))\nprint('y |',np.shape(y))\nprint('X_bl |',np.shape(X_bl))\nprint('y_bl |',np.shape(y_bl))\nprint('X_dummies |',np.shape(X_dummies))\nprint('X_dummies_bl |',np.shape(X_dummies_bl))\n\nprint('id_map_dummies |',np.shape(id_map_dummies))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.973699Z","iopub.execute_input":"2023-11-11T22:49:43.974113Z","iopub.status.idle":"2023-11-11T22:49:43.980965Z","shell.execute_reply.started":"2023-11-11T22:49:43.974064Z","shell.execute_reply":"2023-11-11T22:49:43.979789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SetUp\n### Data Split","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MRRMSE","metadata":{}},{"cell_type":"code","source":"def MRRMSE(y_pred: pd.DataFrame, y_true: pd.DataFrame):\n    return ((y_pred - y_true)**2).mean(axis=1).apply(np.sqrt).mean()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.982071Z","iopub.execute_input":"2023-11-11T22:49:43.982418Z","iopub.status.idle":"2023-11-11T22:49:43.997173Z","shell.execute_reply.started":"2023-11-11T22:49:43.982394Z","shell.execute_reply":"2023-11-11T22:49:43.994831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import LinearSVR\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.model_selection import train_test_split\n\n#model = LinearSVR(max_iter= 1000, epsilon= 0.1)\n#wrapper = MultiOutputRegressor(model)\n#wrapper.fit(X_dummies_train, y_dummies_train)\n\n#predict = wrapper.predict(X_dummies_test) \n#MRRMSE(pd.DataFrame(predict), pd.DataFrame(y_dummies_test.values))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:43.998711Z","iopub.execute_input":"2023-11-11T22:49:43.999235Z","iopub.status.idle":"2023-11-11T22:49:44.011779Z","shell.execute_reply.started":"2023-11-11T22:49:43.999197Z","shell.execute_reply":"2023-11-11T22:49:44.009541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# \nSubmission","metadata":{}},{"cell_type":"markdown","source":"# CustomKFold","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold\n\nkf = KFold(n_splits=5, random_state=10, shuffle=True)\n\nfolds_index_data_AmbrosM = [ ]\ntrain_sm_names = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib', 'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428', 'Porcn Inhibitor III', \n  'Belinostat', 'Foretinib', 'MLN 2238', 'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene', 'Oprozomib (ONX 0912)', 'CHIR-99021']\nlist_fold_ids =  ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells'] \nfor fold_id in list_fold_ids:\n    mask_va = (de_train.cell_type == fold_id) & ~de_train.sm_name.isin(train_sm_names)\n    mask_tr = ~mask_va # 485 or 487 training rows\n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_AmbrosM.append( [IX_train, IX_test ])\n\n# For MT CV scheme:\nfolds_index_data_MT = []\nfold_to_compounds = {0: ['Alvocidib', 'Belinostat', 'Foretinib', 'LDN 193189',  'Linagliptin', 'O-Demethylated Adapalene'],\n 1: ['Dabrafenib', 'Dactolisib', 'Idelalisib', 'MLN 2238', 'Palbociclib', 'Porcn Inhibitor III'],\n 2: ['CHIR-99021', 'Crizotinib', 'Oprozomib (ONX 0912)', 'Penfluridol',  'R428']}\nfor fold_id in [0,1,2]:\n    mask_va = de_train['cell_type'].isin(['Myeloid cells', 'B cells']) & de_train['sm_name'].isin(fold_to_compounds[fold_id])\n    mask_tr = ~mask_va    \n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_MT.append( [IX_train, IX_test ]) \n    \nimport random \nfrom sklearn.model_selection import KFold\n\nclass KFold_custom:\n    '''\n    Class with similar to sklearn \"KFold\" class, interface.\n    Support of CV schemes proposed by AmbrosM, MT, Kishan and standard sklearn KFold\n    Supports options to return IX_train,IX_valid,IX_test - triplet - if valid_size > 0\n    Examples:\n    kf = KFold_custom('AmbrosM'):     \n    for i_fold,(IX_train,IX_test)   in enumerate( kf.split()) :\n        pass\n    kf = KFold_custom('AmbrosM', valid_size = 0.1 )     \n    for i_fold,(IX_train ,IX_valid,IX_test)   in enumerate( kf.split()) :\n        pass\n        \n    kf = KFold_custom(CV_scheme = 'Random') # wrapper for sklearn Kfolds with shuffle = True\n    kf = KFold_custom(CV_scheme = CV_scheme, random_state = 42 ) # fixing random_state ensures reprodicibility of folds \n    '''\n    def __init__(self, CV_scheme,  valid_size = 0, n_splits=5, random_state = 42, verbose = 0 ):\n        self.CV_scheme = CV_scheme\n        self.CV_scheme_inf =  CV_scheme\n        self.valid_size = np.clip(valid_size,0,1)\n        self.random_state = random_state\n\n        self.folds_index_data = [ [np.arange(0), np.arange(0), np.arange(0)] ] # Example data\n        self.n_splits = 1 # Example data\n        if CV_scheme == 'Kishan1':\n            # One fold scheme which contains train, valid, test parts \n            IX_train_Kishan1 = np.arange(429)\n            IX_val_Kishan1 = np.arange(429,521)\n            IX_test_Kishan1 = np.arange(521,614)\n            self.n_splits = 1\n            self.folds_index_data =  [   [IX_train_Kishan1, IX_val_Kishan1, IX_test_Kishan1]  ]\n        elif CV_scheme == 'AmbrosM':\n            self.n_splits = 4\n            self.folds_index_data = folds_index_data_AmbrosM\n        elif CV_scheme == 'MT':\n            self.n_splits = 3\n            self.folds_index_data = folds_index_data_MT\n        elif 'Random'.lower() in CV_scheme.lower():\n            self.n_splits = n_splits\n            self.CV_scheme_inf = 'Random_'+str(n_splits) +'_'+str(random_state )\n            kf = KFold(n_splits=n_splits, random_state = random_state, shuffle=True )#,\n            self.folds_index_data = list( kf.split( np.arange(614) ) )\n        elif 'Full'.lower() in CV_scheme.lower():\n            # Just return the full set as both train and test - it useful to re-train the model on the entire data\n            IX_train_full = np.arange(614)\n            IX_test_full = np.arange(614)\n            self.n_splits = 1\n            self.folds_index_data = [   [IX_train_full, IX_test_full]  ]\n        else:\n            s = 'Uncrecognized CV_scheme ' + str(CV_scheme) \n            raise ValueError(s)\n\n        self.verbose = verbose \n        if verbose >= 10:\n            print(self.CV_scheme, self.n_splits, len(self.folds_index_data) )\n            #print(self.folds_index_data)\n            \n    def split(self, X=None):\n        '''\n        X - NOT used, just for compatibility with sklearn \n        '''\n        for item in self.folds_index_data:\n            if self.CV_scheme in ['Kishan1']:\n                if self.valid_size > 0:\n                    yield item[0],item[1],item[2]\n                else :\n                    yield np.array( list(set(item[0])|set(item[1]))  ) , item[2] # Return train, test only \n            elif self.CV_scheme in ['AmbrosM','MT', 'Random', 'Full']:\n                if self.valid_size == 0:\n                    yield item[0],item[1]\n                else:\n                    # Split \"full-train\"->( real-train, valid ) \n                    index_for_real_train_part = int(  (1-self.valid_size) * len(item[0]) )\n                    if self.random_state is None:\n                        IX_train =  item[0][:index_for_real_train_part]\n                        IX_valid =  item[0][index_for_real_train_part:]\n                    elif self.random_state == -1:\n                        p = np.random.permutation(len(item[0])) \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    else:\n                        np.random.seed(self.random_state) # Set temporary random seed\n                        p = np.random.permutation(len(item[0])) \n                        np.random.seed(random.randint(0,30000)) # Randomize seed again - use Python random, not numpy       \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    yield IX_train, IX_valid, item[1]\n                    \n        \n    def get_list_main_CV_schemes(self):\n        return ['AmbrosM','MT','Kishan1','Random']\n    def get_n_splits(self, X=np.array([])):\n        return self.n_splits    ","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.013460Z","iopub.execute_input":"2023-11-11T22:49:44.013814Z","iopub.status.idle":"2023-11-11T22:49:44.050280Z","shell.execute_reply.started":"2023-11-11T22:49:44.013784Z","shell.execute_reply":"2023-11-11T22:49:44.048606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Try Grid Search","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold\nfrom sklearn.decomposition import TruncatedSVD\n\nkf = KFold_custom('MT', valid_size = 0) ","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.052412Z","iopub.execute_input":"2023-11-11T22:49:44.052822Z","iopub.status.idle":"2023-11-11T22:49:44.068023Z","shell.execute_reply.started":"2023-11-11T22:49:44.052797Z","shell.execute_reply":"2023-11-11T22:49:44.065559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape\nfor i_fold,(train_index ,test_index) in enumerate( kf.split(de_train)) :\n    print(test_index)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.070256Z","iopub.execute_input":"2023-11-11T22:49:44.071164Z","iopub.status.idle":"2023-11-11T22:49:44.079364Z","shell.execute_reply.started":"2023-11-11T22:49:44.071117Z","shell.execute_reply":"2023-11-11T22:49:44.078176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LSVR = LinearSVR(max_iter=max_iter,random_state=0,C=gridsearch.best_params_[\"C\"], epsilon=gridsearch.best_params_[\"epsilon\"])\n\n#C = [2]\n#gamma = [2.5]\n#best_score = 10\n\n#for c in C:\n#    for g in gamma:\n#        predictions_array =[]\n#        CV_score_array    =[]\n#   \n#        for i_fold,(train_index ,test_index) in enumerate( kf.split(de_train)) :\n#            reducer = PCA(n_components=100)\n#            reducer.fit(y_smile)\n#            y_smile_PCA = reducer.transform(y_smile)\n#            \n#            X_train, X_val = X_smile[train_index], X_smile[test_index]\n#            y_train, y_val = y_smile_PCA[train_index], y_smile_PCA[test_index]\n#            \n#            model = LinearSVR(epsilon=g, C=c)\n#            wrapper = MultiOutputRegressor(model)\n#            wrapper.fit(X_train, y_train)\n#            \n#            y_PCA_pred = wrapper.predict(X_val)\n#            y_pred = reducer.inverse_transform(y_PCA_pred)\n#            y_val = reducer.inverse_transform(y_val)\n#        \n#            CV_score_array.append(MRRMSE(y_val, pd.DataFrame(y_pred)))\n##            \n#        gs_score = np.mean(CV_score_array)\n#        if best_score > gs_score:\n#            best_score = gs_score\n#            best_param = {'epsilon':g, 'C':c}\n#    print('g_done')\n#print('best_param: ',best_param)\n#print('best_score: ',best_score)\n###model = LinearSVR(max_iter= 1000, epsilon= 0.1)\n##wrapper = MultiOutputRegressor(model)\n#wrapper.fit(X_dummies_train, y_dummies_train)\n\n#predict = wrapper.predict(X_dummies_test) \n#MRRMSE(pd.DataFrame(predict), pd.DataFrame(y_dummies_test.values))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.080707Z","iopub.execute_input":"2023-11-11T22:49:44.081680Z","iopub.status.idle":"2023-11-11T22:49:44.090668Z","shell.execute_reply.started":"2023-11-11T22:49:44.081645Z","shell.execute_reply":"2023-11-11T22:49:44.089541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LSVR = LinearSVR(max_iter=max_iter,random_state=0,C=gridsearch.best_params_[\"C\"], epsilon=gridsearch.best_params_[\"epsilon\"])\n\n#C = [1.5]\n#gamma = [3,3.5,4,4.5]\n#best_score = 10\n\n#for c in C:\n#    for g in gamma:\n#        predictions_array =[]\n#        CV_score_array    =[]\n   \n#        for i_fold,(train_index ,test_index) in enumerate( kf.split(de_train)) :\n#            reducer = TruncatedSVD(n_components=100)\n#            reducer.fit(y_smile)\n#            y_smile_PCA = reducer.transform(y_smile)\n#            \n#            X_train, X_val = X_smile[train_index], X_smile[test_index]\n#            y_train, y_val = y_smile_PCA[train_index], y_smile_PCA[test_index]\n#            \n#            model = LinearSVR(epsilon=g, C=c)\n#            wrapper = MultiOutputRegressor(model)\n#            wrapper.fit(X_train, y_train)\n            \n#            y_PCA_pred = wrapper.predict(X_val)\n#            y_pred = reducer.inverse_transform(y_PCA_pred)\n#            y_val = reducer.inverse_transform(y_val)\n#        \n#            CV_score_array.append(MRRMSE(y_val, pd.DataFrame(y_pred)))\n#            \n#        gs_score = np.mean(CV_score_array)\n#        if best_score > gs_score:\n#            best_score = gs_score\n#            best_param = {'epsilon':g, 'C':c}\n#    print('g_done')\n#print('best_param: ',best_param)\n#print('best_score: ',best_score)\n##model = LinearSVR(max_iter= 1000, epsilon= 0.1)\n#wrapper = MultiOutputRegressor(model)\n#wrapper.fit(X_dummies_train, y_dummies_train)\n\n#predict = wrapper.predict(X_dummies_test) \n#MRRMSE(pd.DataFrame(predict), pd.DataFrame(y_dummies_test.values))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.091657Z","iopub.execute_input":"2023-11-11T22:49:44.092005Z","iopub.status.idle":"2023-11-11T22:49:44.107918Z","shell.execute_reply.started":"2023-11-11T22:49:44.091972Z","shell.execute_reply":"2023-11-11T22:49:44.106450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"best_param:  {'epsilon': 2.35, 'C': 6.2}","metadata":{}},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"#w/out PCA\n#model = LinearSVR(max_iter= 2000, epsilon= 0.1)\n#wrapper = MultiOutputRegressor(model)\n#wrapper.fit(X_smile, y_smile)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.109274Z","iopub.execute_input":"2023-11-11T22:49:44.109641Z","iopub.status.idle":"2023-11-11T22:49:44.123902Z","shell.execute_reply.started":"2023-11-11T22:49:44.109616Z","shell.execute_reply":"2023-11-11T22:49:44.123037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predict = wrapper.predict(id_map_smile) ","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.125232Z","iopub.execute_input":"2023-11-11T22:49:44.125718Z","iopub.status.idle":"2023-11-11T22:49:44.135737Z","shell.execute_reply.started":"2023-11-11T22:49:44.125690Z","shell.execute_reply":"2023-11-11T22:49:44.134851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission2 = pd.DataFrame(predict, columns= de_train.columns[5:-1])\n#submission2.index.name = 'id'\n\n\n#submission2.to_csv('submission_SMILES_PCA.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.137055Z","iopub.execute_input":"2023-11-11T22:49:44.137592Z","iopub.status.idle":"2023-11-11T22:49:44.146308Z","shell.execute_reply.started":"2023-11-11T22:49:44.137560Z","shell.execute_reply":"2023-11-11T22:49:44.145200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Y_submit = reducer.inverse_transform(Y_submit_PCA)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.147726Z","iopub.execute_input":"2023-11-11T22:49:44.148378Z","iopub.status.idle":"2023-11-11T22:49:44.157543Z","shell.execute_reply.started":"2023-11-11T22:49:44.148344Z","shell.execute_reply":"2023-11-11T22:49:44.156040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n#print(Y_submit)\n\n#df_submit = pd.DataFrame(Y_submit, columns = de_train.columns[5:])\n#df_submit.index.name = 'id'\n#print( df_submit.shape )\n#display(df_submit)\n#df_submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.159182Z","iopub.execute_input":"2023-11-11T22:49:44.159685Z","iopub.status.idle":"2023-11-11T22:49:44.169347Z","shell.execute_reply.started":"2023-11-11T22:49:44.159657Z","shell.execute_reply":"2023-11-11T22:49:44.168195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from scipy.special import logit, expit","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.171232Z","iopub.execute_input":"2023-11-11T22:49:44.171891Z","iopub.status.idle":"2023-11-11T22:49:44.184084Z","shell.execute_reply.started":"2023-11-11T22:49:44.171854Z","shell.execute_reply":"2023-11-11T22:49:44.183172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y2 = logit(y)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.185476Z","iopub.execute_input":"2023-11-11T22:49:44.186500Z","iopub.status.idle":"2023-11-11T22:49:44.198679Z","shell.execute_reply.started":"2023-11-11T22:49:44.186465Z","shell.execute_reply":"2023-11-11T22:49:44.197163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_li = []\nfor i in np.arange(0, id_map_dummies.shape[0], 10 ):\n    t = id_map_smile[i:i+10]\n    test_li.append(t)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.202707Z","iopub.execute_input":"2023-11-11T22:49:44.203069Z","iopub.status.idle":"2023-11-11T22:49:44.212719Z","shell.execute_reply.started":"2023-11-11T22:49:44.203043Z","shell.execute_reply":"2023-11-11T22:49:44.211793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reducer = PCA(n_components=100)\nreducer.fit(y_smile)\ny_smile_PCA = reducer.transform(y_smile)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:44.213817Z","iopub.execute_input":"2023-11-11T22:49:44.215517Z","iopub.status.idle":"2023-11-11T22:49:45.787062Z","shell.execute_reply.started":"2023-11-11T22:49:44.215436Z","shell.execute_reply":"2023-11-11T22:49:45.786158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_smile_PCA","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:45.788452Z","iopub.execute_input":"2023-11-11T22:49:45.788731Z","iopub.status.idle":"2023-11-11T22:49:45.794977Z","shell.execute_reply.started":"2023-11-11T22:49:45.788706Z","shell.execute_reply":"2023-11-11T22:49:45.794130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map_smile.shape, X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:45.796047Z","iopub.execute_input":"2023-11-11T22:49:45.796605Z","iopub.status.idle":"2023-11-11T22:49:45.805404Z","shell.execute_reply.started":"2023-11-11T22:49:45.796580Z","shell.execute_reply":"2023-11-11T22:49:45.804523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model = LinearSVR(max_iter=2000, epsilon=best_pram['epsilon'], C=best_param['C'])\n#wrapper = MultiOutputRegressor(model)\n#wrapper.fit(X_smile, y_smile_PCA)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:29.966532Z","iopub.execute_input":"2023-11-11T22:52:29.967296Z","iopub.status.idle":"2023-11-11T22:52:29.972915Z","shell.execute_reply.started":"2023-11-11T22:52:29.967258Z","shell.execute_reply":"2023-11-11T22:52:29.972118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_dummies_bl, y_dummies_PCA=X_smile, y_smile_PCA","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:30.556738Z","iopub.execute_input":"2023-11-11T22:52:30.557398Z","iopub.status.idle":"2023-11-11T22:52:30.563000Z","shell.execute_reply.started":"2023-11-11T22:52:30.557358Z","shell.execute_reply":"2023-11-11T22:52:30.561256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#count=0\n#pred_PCA = []\n#for test in test_li:\n#    model = LinearSVR(max_iter=2000, epsilon=best_param['epsilon'], C=best_param['C'])\n#    wrapper = MultiOutputRegressor(model)\n#    wrapper.fit(X_dummies_bl, y_dummies_PCA)\n#    pr = wrapper.predict(test)\n    \n#    y_dummies_PCA = np.vstack([y_dummies_PCA,pr])\n#    X_dummies_bl = np.vstack([X_dummies_bl, test])\n    \n#    count +=1\n#    print(count)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:31.742244Z","iopub.execute_input":"2023-11-11T22:52:31.742686Z","iopub.status.idle":"2023-11-11T22:52:31.746932Z","shell.execute_reply.started":"2023-11-11T22:52:31.742655Z","shell.execute_reply":"2023-11-11T22:52:31.746057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inv_predict = y_dummies_PCA[-255:]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:31.928286Z","iopub.execute_input":"2023-11-11T22:52:31.928780Z","iopub.status.idle":"2023-11-11T22:52:31.935018Z","shell.execute_reply.started":"2023-11-11T22:52:31.928747Z","shell.execute_reply":"2023-11-11T22:52:31.933189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = reducer.inverse_transform(inv_predict)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:37.778565Z","iopub.execute_input":"2023-11-11T22:52:37.779067Z","iopub.status.idle":"2023-11-11T22:52:37.821228Z","shell.execute_reply.started":"2023-11-11T22:52:37.779030Z","shell.execute_reply":"2023-11-11T22:52:37.819730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2 = pd.DataFrame(predict, columns= de_train.columns[5:18216])\nsubmission2.index.name = 'id'\n\n\nsubmission2.to_csv('submission_FgPrint+Descriptors_multiOR.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:38.610065Z","iopub.execute_input":"2023-11-11T22:52:38.610735Z","iopub.status.idle":"2023-11-11T22:52:44.815319Z","shell.execute_reply.started":"2023-11-11T22:52:38.610689Z","shell.execute_reply":"2023-11-11T22:52:44.812949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:44.818885Z","iopub.execute_input":"2023-11-11T22:52:44.819435Z","iopub.status.idle":"2023-11-11T22:52:44.859078Z","shell.execute_reply.started":"2023-11-11T22:52:44.819392Z","shell.execute_reply":"2023-11-11T22:52:44.857187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import TruncatedSVD\nreducer = TruncatedSVD(n_components=100)\ny_dummies_PCA = reducer.fit_transform(y_smile)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:44.861220Z","iopub.execute_input":"2023-11-11T22:52:44.861970Z","iopub.status.idle":"2023-11-11T22:52:56.748297Z","shell.execute_reply.started":"2023-11-11T22:52:44.861931Z","shell.execute_reply":"2023-11-11T22:52:56.747160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:56.750396Z","iopub.execute_input":"2023-11-11T22:52:56.750965Z","iopub.status.idle":"2023-11-11T22:52:56.758851Z","shell.execute_reply.started":"2023-11-11T22:52:56.750913Z","shell.execute_reply":"2023-11-11T22:52:56.757694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_dummies_PCA.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:52:56.760528Z","iopub.execute_input":"2023-11-11T22:52:56.760849Z","iopub.status.idle":"2023-11-11T22:52:56.772955Z","shell.execute_reply.started":"2023-11-11T22:52:56.760821Z","shell.execute_reply":"2023-11-11T22:52:56.771536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb#\n\nmodel = lgb.LGBMRegressor()#\nX_dummies_bl = X_smile\npred = []\ncount=0\npred_PCA = []\ntest_li = []\nfor i in np.arange(0, id_map_dummies.shape[0], 10 ):\n    t = id_map_smile[i:i+10]\n    test_li.append(t)\nfor test in test_li:\n    wrapper = MultiOutputRegressor(model)\n    wrapper.fit(X_dummies_bl, y_dummies_PCA)\n    pr = wrapper.predict(test)\n    \n    y_dummies_PCA = np.vstack([y_dummies_PCA,pr])\n    X_dummies_bl = np.vstack([X_dummies_bl, test])\n    \n    count +=1\n    print(count)\n\n#wrapper = MultiOutut\n#for i in range(y_smile_PCA.shape[1]):\n#    \n#    yi = y_smile_PCA[:, i]\n#    Xi = X_smile\n#    id_map_smile\n    \n#    model.fit(Xi, yi)\n#    pred.append(model.predict(id_map_smile))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:46.208290Z","iopub.status.idle":"2023-11-11T22:49:46.208917Z","shell.execute_reply.started":"2023-11-11T22:49:46.208731Z","shell.execute_reply":"2023-11-11T22:49:46.208752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inv_predict = y_dummies_PCA[-255:]\npred =  reducer.inverse_transform(inv_predict)\nprediction =  pd.DataFrame(pred, columns= de_train.columns[5:18216])\nprediction.index.name = 'id'\nprediction","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:46.210543Z","iopub.status.idle":"2023-11-11T22:49:46.211991Z","shell.execute_reply.started":"2023-11-11T22:49:46.211666Z","shell.execute_reply":"2023-11-11T22:49:46.211705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.to_csv('submission_lgb_FgPrint+Descriptors.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:49:46.213449Z","iopub.status.idle":"2023-11-11T22:49:46.213865Z","shell.execute_reply.started":"2023-11-11T22:49:46.213675Z","shell.execute_reply":"2023-11-11T22:49:46.213693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}