{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:36:46.492683Z","iopub.execute_input":"2023-11-11T20:36:46.493084Z","iopub.status.idle":"2023-11-11T20:36:46.502705Z","shell.execute_reply.started":"2023-11-11T20:36:46.493053Z","shell.execute_reply":"2023-11-11T20:36:46.501691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\nimport warnings # suppress warnings\nwarnings.filterwarnings('ignore')","metadata":{"papermill":{"duration":0.387254,"end_time":"2023-10-19T18:20:50.862073","exception":false,"start_time":"2023-10-19T18:20:50.474819","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:36:46.504780Z","iopub.execute_input":"2023-11-11T20:36:46.505208Z","iopub.status.idle":"2023-11-11T20:36:46.539502Z","shell.execute_reply.started":"2023-11-11T20:36:46.505171Z","shell.execute_reply":"2023-11-11T20:36:46.538509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv')\nde_train = reduce_mem_usage(pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'))\n#adata_train = reduce_mem_usage(pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet'))#heavy\n#adata_obs_meta = reduce_mem_usage(pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/adata_obs_meta.csv', usecols=['obs_id', 'cell_type', 'sm_name']))","metadata":{"papermill":{"duration":172.488789,"end_time":"2023-10-19T18:23:43.364991","exception":false,"start_time":"2023-10-19T18:20:50.876202","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:36:46.541096Z","iopub.execute_input":"2023-11-11T20:36:46.541444Z","iopub.status.idle":"2023-11-11T20:37:03.627148Z","shell.execute_reply.started":"2023-11-11T20:36:46.541416Z","shell.execute_reply":"2023-11-11T20:37:03.626160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport torch\nimport matplotlib.pyplot as plt\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:03.631763Z","iopub.execute_input":"2023-11-11T20:37:03.632147Z","iopub.status.idle":"2023-11-11T20:37:03.636457Z","shell.execute_reply.started":"2023-11-11T20:37:03.632115Z","shell.execute_reply":"2023-11-11T20:37:03.635572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\n\nX=de_train.iloc[:, :2]\ny=de_train.iloc[:, 5:]","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:37:03.637540Z","iopub.execute_input":"2023-11-11T20:37:03.637802Z","iopub.status.idle":"2023-11-11T20:37:04.680541Z","shell.execute_reply.started":"2023-11-11T20:37:03.637779Z","shell.execute_reply":"2023-11-11T20:37:04.679483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"control = de_train[de_train['control']==1]\ntrain = de_train[de_train['control']==0]\ncontrol","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:04.681801Z","iopub.execute_input":"2023-11-11T20:37:04.682124Z","iopub.status.idle":"2023-11-11T20:37:05.828821Z","shell.execute_reply.started":"2023-11-11T20:37:04.682097Z","shell.execute_reply":"2023-11-11T20:37:05.827703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=train.iloc[:, :2]\ny=train.iloc[:, 5:]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:05.830081Z","iopub.execute_input":"2023-11-11T20:37:05.830401Z","iopub.status.idle":"2023-11-11T20:37:06.834504Z","shell.execute_reply.started":"2023-11-11T20:37:05.830370Z","shell.execute_reply":"2023-11-11T20:37:06.833684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def MRRMSE_torch(y_true, y_pred):\n    mse = torch.mean((y_pred - y_true)**2,1, True)\n    rmse = torch.sqrt(mse)\n    return rmse.mean()\n\ndef MRRMSE(y_pred, y_true):\n    y_pred = pd.DataFrame(y_pred)\n    y_true = pd.DataFrame(y_true)\n    return ((y_pred - y_true)**2).mean(axis=1).apply(np.sqrt).mean()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:37:06.837677Z","iopub.execute_input":"2023-11-11T20:37:06.838083Z","iopub.status.idle":"2023-11-11T20:37:06.844579Z","shell.execute_reply.started":"2023-11-11T20:37:06.838045Z","shell.execute_reply":"2023-11-11T20:37:06.843542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:06.845957Z","iopub.execute_input":"2023-11-11T20:37:06.846378Z","iopub.status.idle":"2023-11-11T20:37:06.864025Z","shell.execute_reply.started":"2023-11-11T20:37:06.846345Z","shell.execute_reply":"2023-11-11T20:37:06.863034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SMILES fingerprint","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:06.867829Z","iopub.execute_input":"2023-11-11T20:37:06.868231Z","iopub.status.idle":"2023-11-11T20:37:18.674059Z","shell.execute_reply.started":"2023-11-11T20:37:06.868198Z","shell.execute_reply":"2023-11-11T20:37:18.672758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:18.675800Z","iopub.execute_input":"2023-11-11T20:37:18.676161Z","iopub.status.idle":"2023-11-11T20:37:18.681452Z","shell.execute_reply.started":"2023-11-11T20:37:18.676128Z","shell.execute_reply":"2023-11-11T20:37:18.680304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sm2smiles = dict(zip(train['sm_name'], train['SMILES']))\nid_map['SMILES'] = id_map['sm_name'].map(sm2smiles)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:18.682761Z","iopub.execute_input":"2023-11-11T20:37:18.683142Z","iopub.status.idle":"2023-11-11T20:37:18.696483Z","shell.execute_reply.started":"2023-11-11T20:37:18.683104Z","shell.execute_reply":"2023-11-11T20:37:18.695466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['MF'] = train['SMILES'].apply(lambda x: AllChem.GetMorganFingerprintAsBitVect(Chem.MolFromSmiles(x), 3, nBits=2048))\nid_map['MF'] = id_map['SMILES'].apply(lambda x: AllChem.GetMorganFingerprintAsBitVect(Chem.MolFromSmiles(x), 3, nBits=2048))","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:18.697857Z","iopub.execute_input":"2023-11-11T20:37:18.698521Z","iopub.status.idle":"2023-11-11T20:37:18.904816Z","shell.execute_reply.started":"2023-11-11T20:37:18.698482Z","shell.execute_reply":"2023-11-11T20:37:18.903947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fgp_array_train = np.array(train['MF'].tolist())\nfgp_array_test = np.array(id_map['MF'].tolist())","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:18.905989Z","iopub.execute_input":"2023-11-11T20:37:18.906288Z","iopub.status.idle":"2023-11-11T20:37:20.732456Z","shell.execute_reply.started":"2023-11-11T20:37:18.906261Z","shell.execute_reply":"2023-11-11T20:37:20.731361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:20.734047Z","iopub.execute_input":"2023-11-11T20:37:20.734416Z","iopub.status.idle":"2023-11-11T20:37:20.744858Z","shell.execute_reply.started":"2023-11-11T20:37:20.734381Z","shell.execute_reply":"2023-11-11T20:37:20.743904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encoder","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nenc = OrdinalEncoder()\nX_enc = enc.fit_transform(X)\nprint( len(np.unique(X_enc[:,0])),  len(np.unique(X_enc[:,1]) )) \nprint( str(enc.categories_)[:30],'   ', str(enc.categories_)[-30:]  )\nprint('X_enc.shape',X_enc.shape)\nprint()\nid_map_enc = enc.transform(id_map.iloc[:, 1:3])\nprint('id_map_enc.shape', id_map_enc.shape)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:37:20.746158Z","iopub.execute_input":"2023-11-11T20:37:20.746428Z","iopub.status.idle":"2023-11-11T20:37:20.760621Z","shell.execute_reply.started":"2023-11-11T20:37:20.746405Z","shell.execute_reply":"2023-11-11T20:37:20.759637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(columns='MF', inplace=True)\nid_map.drop(columns='MF', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:20.762023Z","iopub.execute_input":"2023-11-11T20:37:20.762450Z","iopub.status.idle":"2023-11-11T20:37:21.352628Z","shell.execute_reply.started":"2023-11-11T20:37:20.762400Z","shell.execute_reply":"2023-11-11T20:37:21.351723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fgp_array_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:21.353984Z","iopub.execute_input":"2023-11-11T20:37:21.354292Z","iopub.status.idle":"2023-11-11T20:37:21.360486Z","shell.execute_reply.started":"2023-11-11T20:37:21.354265Z","shell.execute_reply":"2023-11-11T20:37:21.359529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit.Chem import Descriptors\nfor tup in Descriptors._descList:\n    train[tup[0]] = train['SMILES'].apply(lambda x: tup[1](Chem.MolFromSmiles(x))).astype(float)\n    id_map[tup[0]] = id_map['SMILES'].apply(lambda x: tup[1](Chem.MolFromSmiles(x))).astype(float)\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ntrain.iloc[:, 18216:]= scaler.fit_transform(train.iloc[:, 18216:])","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:37:21.361778Z","iopub.execute_input":"2023-11-11T20:37:21.362075Z","iopub.status.idle":"2023-11-11T20:38:26.393083Z","shell.execute_reply.started":"2023-11-11T20:37:21.362050Z","shell.execute_reply":"2023-11-11T20:38:26.392226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain.iloc[:, 18215:]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:26.394308Z","iopub.execute_input":"2023-11-11T20:38:26.394605Z","iopub.status.idle":"2023-11-11T20:38:26.441165Z","shell.execute_reply.started":"2023-11-11T20:38:26.394580Z","shell.execute_reply":"2023-11-11T20:38:26.440199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map_enc.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:26.442615Z","iopub.execute_input":"2023-11-11T20:38:26.442905Z","iopub.status.idle":"2023-11-11T20:38:26.448927Z","shell.execute_reply.started":"2023-11-11T20:38:26.442879Z","shell.execute_reply":"2023-11-11T20:38:26.447996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile = np.hstack((X_enc, fgp_array_train, train.iloc[:, 18216:]))\nid_map_enc = np.hstack((id_map_enc, fgp_array_test, scaler.transform(id_map.iloc[:, 4:])))\n\ny_smile = np.array(train.iloc[:, 5:18216])\ny_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:26.450098Z","iopub.execute_input":"2023-11-11T20:38:26.450367Z","iopub.status.idle":"2023-11-11T20:38:27.638331Z","shell.execute_reply.started":"2023-11-11T20:38:26.450344Z","shell.execute_reply":"2023-11-11T20:38:27.637381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.639538Z","iopub.execute_input":"2023-11-11T20:38:27.639832Z","iopub.status.idle":"2023-11-11T20:38:27.645893Z","shell.execute_reply.started":"2023-11-11T20:38:27.639806Z","shell.execute_reply":"2023-11-11T20:38:27.645019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"markdown","source":"# Split X and Y","metadata":{}},{"cell_type":"code","source":"np.array(y).mean(axis=0).shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.647046Z","iopub.execute_input":"2023-11-11T20:38:27.647357Z","iopub.status.idle":"2023-11-11T20:38:27.772917Z","shell.execute_reply.started":"2023-11-11T20:38:27.647333Z","shell.execute_reply":"2023-11-11T20:38:27.771913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.774166Z","iopub.execute_input":"2023-11-11T20:38:27.774462Z","iopub.status.idle":"2023-11-11T20:38:27.780458Z","shell.execute_reply.started":"2023-11-11T20:38:27.774437Z","shell.execute_reply":"2023-11-11T20:38:27.779532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.781811Z","iopub.execute_input":"2023-11-11T20:38:27.782184Z","iopub.status.idle":"2023-11-11T20:38:27.790121Z","shell.execute_reply.started":"2023-11-11T20:38:27.782159Z","shell.execute_reply":"2023-11-11T20:38:27.789235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X_smile, y_smile, test_size=0.10, random_state=42)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:38:27.791322Z","iopub.execute_input":"2023-11-11T20:38:27.791636Z","iopub.status.idle":"2023-11-11T20:38:27.839692Z","shell.execute_reply.started":"2023-11-11T20:38:27.791606Z","shell.execute_reply":"2023-11-11T20:38:27.838885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nX_train_t = torch.tensor(X_train, dtype=torch.float)\nX_val_t = torch.tensor(X_val, dtype=torch.float)\ny_train_t = torch.tensor(y_train, dtype=torch.float)\ny_val_t = torch.tensor(y_val, dtype=torch.float)\n\nid_map_t = torch.tensor(id_map_enc, dtype=torch.float)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.847151Z","iopub.execute_input":"2023-11-11T20:38:27.847448Z","iopub.status.idle":"2023-11-11T20:38:27.877755Z","shell.execute_reply.started":"2023-11-11T20:38:27.847424Z","shell.execute_reply":"2023-11-11T20:38:27.876857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset\n\nclass trainvalData(Dataset):\n    def __init__(self, X_data, y_data):\n        self.X_data = X_data\n        self.y_data = y_data\n    def __getitem__(self, index):\n        return self.X_data[index], self.y_data[index]\n    def __len__(self):\n        return len(self.X_data)\n    \nclass testData(Dataset):\n    def __init__(self, X_data):\n        self.X_data = X_data\n    def __getitem__(self, index):\n        return self.X_data[index]\n    def __len__(self):\n        return len(self.X_data)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.878905Z","iopub.execute_input":"2023-11-11T20:38:27.879205Z","iopub.status.idle":"2023-11-11T20:38:27.886180Z","shell.execute_reply.started":"2023-11-11T20:38:27.879179Z","shell.execute_reply":"2023-11-11T20:38:27.885448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = trainvalData(X_train_t, y_train_t)\nval_data = trainvalData(X_val_t, y_val_t)\ntest_data = testData(id_map_t)\n\nBatch_size = 16\n\ntrain_loader = DataLoader(dataset=train_data, batch_size = Batch_size, drop_last=True, num_workers=4, pin_memory=True)\nval_loader = DataLoader(dataset=val_data, batch_size = Batch_size,drop_last=True, num_workers=4, pin_memory=True)\ntest_loader =  DataLoader(dataset=test_data, batch_size = Batch_size,drop_last=True, num_workers=4, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.886963Z","iopub.execute_input":"2023-11-11T20:38:27.887231Z","iopub.status.idle":"2023-11-11T20:38:27.904199Z","shell.execute_reply.started":"2023-11-11T20:38:27.887207Z","shell.execute_reply":"2023-11-11T20:38:27.903204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.905461Z","iopub.execute_input":"2023-11-11T20:38:27.905774Z","iopub.status.idle":"2023-11-11T20:38:27.912125Z","shell.execute_reply.started":"2023-11-11T20:38:27.905739Z","shell.execute_reply":"2023-11-11T20:38:27.911278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import DataLoader\n\nBatch_size = 64\n","metadata":{"execution":{"iopub.status.busy":"2023-11-11T20:38:27.913541Z","iopub.execute_input":"2023-11-11T20:38:27.913906Z","iopub.status.idle":"2023-11-11T20:38:27.921137Z","shell.execute_reply.started":"2023-11-11T20:38:27.913872Z","shell.execute_reply":"2023-11-11T20:38:27.920121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.optim as optim","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T20:38:27.922497Z","iopub.execute_input":"2023-11-11T20:38:27.922878Z","iopub.status.idle":"2023-11-11T20:38:27.930347Z","shell.execute_reply.started":"2023-11-11T20:38:27.922844Z","shell.execute_reply":"2023-11-11T20:38:27.929487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EmbNN(nn.Module):\n    def __init__(self, num_categories1=6, num_categories2=257,\n                 emb1_size=2, emb2_size=5, emb3_size=2258, hidden_layer=[2048,1024,512,1024, 2048]):\n        super(EmbNN, self).__init__()\n        self.embedding1 = nn.Embedding(num_categories1, emb1_size)  # Embedding layer for categorical feature 1\n        self.embedding2 = nn.Embedding(num_categories2, emb2_size)   # Embedding layer for categorical feature 2\n        \n        self.fc1 = nn.Sequential(\n            nn.Linear(emb1_size + emb2_size + emb3_size , hidden_layer[0]),\n            nn.BatchNorm1d(hidden_layer[0]), \n            nn.ELU()) # Input: 10 (from embedding 1) + 5 (from embedding 2) + 1 (continuous), Output: 32 features\n        self.fc2 = nn.Sequential(\n            nn.Linear(hidden_layer[0], hidden_layer[1]),\n            #nn.BatchNorm1d(hidden_layer[1]),\n            nn.Dropout(0.1, False),\n            nn.ELU())#Input: 32 features, Output: 64 features\n        self.fc3 = nn.Sequential(\n            nn.Linear(hidden_layer[1], hidden_layer[2]),\n            #nn.BatchNorm1d(hidden_layer[2]), \n            nn.Dropout(0.1, False),\n            nn.ELU())#Input: 32 features, Output: 64 features\n        self.fc4 = nn.Sequential(\n            nn.Linear(hidden_layer[2], hidden_layer[3]),\n            nn.BatchNorm1d(hidden_layer[3]), \n            nn.Dropout(0.1, False),\n            nn.ELU())#Input: 32 features, Output: 64 features\n        self.fc5 = nn.Sequential(\n            nn.Linear(hidden_layer[3], hidden_layer[4]),\n            #nn.BatchNorm1d(hidden_layer[4]), \n            nn.Dropout(0.1, False),\n            nn.ELU())#Input: 32 features, Output: 64 features\n        #self.fc6 = nn.Sequential(\n        #    nn.Linear(hidden_layer[4], hidden_layer[5]),\n        #    nn.BatchNorm1d(hidden_layer[5]), \n        #    nn.Dropout(0.2, False),\n        #    nn.ELU())#Input: 32 features, Output: 64 features\n        self.fc7 = nn.Sequential(\n            nn.Linear(hidden_layer[4], y_smile.shape[1]))\n\n\n    def forward(self, cat_input1, cat_input2, cat_input3):\n        cat_embed1 = self.embedding1(cat_input1.long() )  # Embed categorical feature 1\n        cat_embed2 = self.embedding2(cat_input2.long())\n        cat_embed3 = torch.flatten(cat_input3.long() )\n        cat_embed3 = cat_embed3 + (0.009)*torch.randn(cat_embed3.size()).to(device)\n        x = torch.cat((cat_embed1.view(cat_input1.size(0), -1), cat_embed2.view(cat_input2.size(0), -1), cat_embed3.view(cat_input3.size(0), -1)), dim=1)# Concatenate with continuous input\n        x = self.fc1(x)\n        x = self.fc2(x)\n        x = self.fc3(x)\n        x = self.fc4(x)\n        x = self.fc5(x)\n        #x = self.fc6(x) \n        x = self.fc7(x)\n        return x\n\n    def emb1(self, cat_input):\n        cat_embed1 = self.embedding1(cat_input.long() )  # Embed cell_type\n        return cat_embed1\n    def emb2(self, cat_input):\n        cat_embed2 = self.embedding2(cat_input.long() )  # Embed molecule\n        return cat_embed2\n    def emb3(self, cat_input):\n        cat_embed3 = torch.flatten(cat_input3.long() )\n        cat_embed3 = cat_embed3 + (0.09)*torch.randn(emb3_size)\n        return cat_embed3 #SMILES embedding","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:20.118158Z","iopub.execute_input":"2023-11-11T21:53:20.118517Z","iopub.status.idle":"2023-11-11T21:53:20.134714Z","shell.execute_reply.started":"2023-11-11T21:53:20.118486Z","shell.execute_reply":"2023-11-11T21:53:20.133779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 最適化試す","metadata":{}},{"cell_type":"code","source":"#class optnn(nn.Module):\n#    def __init__(self, trial, num_layers, num_features):\n#        super(optnn, self).__init__()\n#        self.embedding1 = nn.Embedding(num_categories1, emb1_size)  # Embedding layer for categorical feature 1\n#        self.embedding2 = nn.Embedding(num_categories2, emb2_size)   # Embedding layer for categorical feature 2\n#        self.layers = nn.ModuleList(\n#            [nn.Linear(emb1_size + emb2_size , num_features[0])])\n#        self.layers.append(nn.BatchNorm1d(num_features[0])) \n#        self.layers.append(nn.ReLU())\n#        \n#        for i in range(1, num_layers):\n#            self.layers.append(\n#                nn.Linear(in_features=num_features[i-1], out_features=num_features[i])\n#            )\n#            self.layers.append(nn.BatchNorm1d(num_features[i]))\n#            self.layers.append(nn.ReLU())\n##        self.layers_out = nn.Linear(in_features=num_features[i], out_features=1)\n #   def forward(self, cat_input1, cat_input2):\n #       cat_embed1 = self.embedding1(cat_input1.long() )  # Embed categorical feature 1\n #       cat_embed2 = self.embedding2(cat_input2.long())  # Embed categorical feature 2\n #       x = torch.cat((cat_embed1.view(cat_input1.size(0), -1), cat_embed2.view(cat_input2.size(0), -1)), dim=1) \n ##       for _, l in enumerate(self.layers):\n  #          x = l(x)\n  #      x = self.layers_out(x)\n  #      return x","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:21.947364Z","iopub.execute_input":"2023-11-11T21:53:21.948017Z","iopub.status.idle":"2023-11-11T21:53:21.952883Z","shell.execute_reply.started":"2023-11-11T21:53:21.947960Z","shell.execute_reply":"2023-11-11T21:53:21.951962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_categories1 = len( set(X_enc[:,0]) )\nnum_categories2 = len( set(X_enc[:,1]) )\nemb1_size = 2\nemb2_size = 5\nhidden_layer1 = 1024\nhidden_layer2 = 2258\nmodel = EmbNN(num_categories1=num_categories1, num_categories2=num_categories2,  emb1_size=emb1_size, emb2_size=emb2_size) \n\nprint(model)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T21:53:22.197553Z","iopub.execute_input":"2023-11-11T21:53:22.197853Z","iopub.status.idle":"2023-11-11T21:53:22.580749Z","shell.execute_reply.started":"2023-11-11T21:53:22.197826Z","shell.execute_reply":"2023-11-11T21:53:22.579778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EmbNN(\n  (embedding1): Embedding(6, 2)\n  (embedding2): Embedding(144, 5)\n  (fc1): Sequential(\n    (0): Linear(in_features=2055, out_features=1024, bias=True)\n    (1): BatchNorm1d(1024, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)\n    (2): ReLU()\n  )\n  (fc2): Sequential(\n    (0): Linear(in_features=1024, out_features=2048, bias=True)\n    (1): BatchNorm1d(2048, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)\n    (2): ReLU()\n  )\n  (fc3): Sequential(\n    (0): Linear(in_features=2048, out_features=4096, bias=True)\n    (1): BatchNorm1d(4096, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)\n    (2): ReLU()\n  )\n  (fc4): Sequential(\n    (0): Linear(in_features=4096, out_features=9192, bias=True)\n    (1): BatchNorm1d(9192, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)\n    (2): ReLU()\n  )\n  (fc7): Sequential(\n    (0): Linear(in_features=9192, out_features=18211, bias=True)\n  )","metadata":{}},{"cell_type":"code","source":"device = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\nprint(device)\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:22.768857Z","iopub.execute_input":"2023-11-11T21:53:22.769224Z","iopub.status.idle":"2023-11-11T21:53:22.829347Z","shell.execute_reply.started":"2023-11-11T21:53:22.769194Z","shell.execute_reply":"2023-11-11T21:53:22.828448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 参考: https://github.com/Bjarten/early-stopping-pytorch/blob/master/pytorchtools.py\n\nclass EarlyStopping:\n    '''Early stopはval_lossが規定回数を経ても改善しなかった場合にそこで終了させる関数である'''\n\n    def __init__(\n            self, patience=100, verbose=False, delta=0, path='checkpoint.pt', trace_func=print\n    ):\n\n        '''\n        Args:\n            patience (int): val_lossが最後に改善されてからpatienceで指定された回まで改善なければ終了\n                            Default: 7\n            verbose (bool): Trueなら, val_lossの改善状況を示す\n                            Default: False\n            delta (float):  改善と認定するための最小変化量. delta以下の改善は改善と認めない\n                            Default: 0\n            path (str): PyTorchモデルを保存するパス\n                            Default: 'checkpoint.pt'\n            trace_func (function): val_lossが更新されないとき，それが何回目を示す\n                            Default: print\n        '''\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n        self.delta = delta\n        self.path = path\n        self.trace_func = trace_func\n\n    def __call__(self, val_loss, model):\n        score = val_loss\n\n        # best_scoreがなければscoreを当面のbest_scoreにする\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        # scoreがbest_score+deltaに満たなければ棄却\n        elif score > self.best_score + self.delta:\n            # 棄却カウント1追加\n            self.counter += 1\n            if self.counter % 10 ==0:\n                self.trace_func(\n                    f'EarlyStopping counter: {self.counter} out of {self.patience}'\n                )\n            if self.counter % (self.patience/3) ==0:\n                model.load_state_dict(torch.load(early_stopping.path))\n            # 棄却カウントの合計がpatienceを超えたとき\n            if self.counter >= self.patience:\n                # Early stopで中止\n                self.early_stop = True\n                self.counter=0\n        # scoreがそれまでのbest_scoreより大きいとき\n        else:\n            # scoreを更新\n            self.best_score = score\n            # modelとval_lossを保存\n            self.save_checkpoint(val_loss, model)\n            # 棄却カウントを0に戻す\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        torch.save(model.state_dict(), self.path)\n        self.val_loss_min = val_loss\n        self.counter=0","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:22.997482Z","iopub.execute_input":"2023-11-11T21:53:22.998074Z","iopub.status.idle":"2023-11-11T21:53:23.009181Z","shell.execute_reply.started":"2023-11-11T21:53:22.998045Z","shell.execute_reply":"2023-11-11T21:53:23.008174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_t.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:23.201479Z","iopub.execute_input":"2023-11-11T21:53:23.202046Z","iopub.status.idle":"2023-11-11T21:53:23.207419Z","shell.execute_reply.started":"2023-11-11T21:53:23.202019Z","shell.execute_reply":"2023-11-11T21:53:23.206508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimizer Selection","metadata":{}},{"cell_type":"code","source":"import torch.optim as optim\n\ndef get_optimizer(trial, model):\n    optimizer_names = ['Adam']\n    optimier_name = trial.suggest_categorical('optimizer',optimizer_names)\n    weight_decay = trial.suggest_float('weight_decay', 1e-8, 1e-2, log=True)\n    Adam_lr = trial.suggest_float('Adam_lr', 1e-5, 1e-1, log=True)\n    optimizer = optim.Adam(\n    model.parameters(), lr=Adam_lr, weight_decay = weight_decay)\n    return optimizer","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:24.285850Z","iopub.execute_input":"2023-11-11T21:53:24.286238Z","iopub.status.idle":"2023-11-11T21:53:24.292208Z","shell.execute_reply.started":"2023-11-11T21:53:24.286204Z","shell.execute_reply":"2023-11-11T21:53:24.291153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS =20000\ncriterion = nn.MSELoss()\nearly_stopping = EarlyStopping(patience=5000)\n#def objective_nn(trial):\n#    num_layers = trial.suggest_int('num_layer', 2,8)\n#    num_features = [\n#        int(trial.suggest_float(f'num_features_{i}',128,2048, step=128))\n#        for i in range(num_layers)\n#    ]\n#    \n##    model = optnn(trial, num_layers, num_features).to(device)\n#    optimizer = get_optimizer(trial, model)\n#    \n#    for step in range(EPOCHS):\n#        #train\n#        X_train_t1 = X_train_t.to(device)\n#        X_val_t1 = X_val_t.to(device)\n#        y_train_t1 = y_train_t.to(device)\n#        y_val_t1 = y_val_t.to(device)\n#        #1\n#        model.train()\n#        categorical_feature1 = X_train_t1[:,0]\n#        categorical_feature2 = X_train_t1[:,1]\n#        preds = model(categorical_feature1, categorical_feature2)\n##        #2\n #       loss = criterion(preds, y_train_t1)# .float() )\n#        #3, 4, 5\n#        optimizer.zero_grad()\n#        loss.backward()\n#        optimizer.step()\n#        #test\n#        model.eval()\n#        with torch.no_grad():\n#            categorical_feature3 = X_val_t1[:,0]\n#            categorical_feature4 = X_val_t1[:,1]\n#            preds = model(categorical_feature3, categorical_feature4)\n#            loss_test = criterion(preds, y_val_t1)# .float() )\n#        #early_stopping(loss_test,model)\n        #if early_stopping.early_stop:\n        #    break\n#    return loss_test","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:24.699862Z","iopub.execute_input":"2023-11-11T21:53:24.700723Z","iopub.status.idle":"2023-11-11T21:53:24.706557Z","shell.execute_reply.started":"2023-11-11T21:53:24.700690Z","shell.execute_reply":"2023-11-11T21:53:24.705609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from functools import partial\n#import optuna\n\n#obj_nn = partial(objective_nn)\n#sampler = optuna.samplers.TPESampler(seed=0)\n#study_nn = optuna.create_study(sampler=sampler, direction='minimize')\n#study_nn.optimize(obj_nn, n_trials=10000)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:24.951091Z","iopub.execute_input":"2023-11-11T21:53:24.951708Z","iopub.status.idle":"2023-11-11T21:53:24.955796Z","shell.execute_reply.started":"2023-11-11T21:53:24.951674Z","shell.execute_reply":"2023-11-11T21:53:24.954822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train_t[:, 3:].type","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:25.349852Z","iopub.execute_input":"2023-11-11T21:53:25.350201Z","iopub.status.idle":"2023-11-11T21:53:25.354233Z","shell.execute_reply.started":"2023-11-11T21:53:25.350170Z","shell.execute_reply":"2023-11-11T21:53:25.353217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train_t[:, 2].type","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:25.787758Z","iopub.execute_input":"2023-11-11T21:53:25.788514Z","iopub.status.idle":"2023-11-11T21:53:25.792757Z","shell.execute_reply.started":"2023-11-11T21:53:25.788481Z","shell.execute_reply":"2023-11-11T21:53:25.791840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold\n\nkf = KFold(n_splits=5, random_state=10, shuffle=True)\n\nfolds_index_data_AmbrosM = [ ]\ntrain_sm_names = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib', 'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428', 'Porcn Inhibitor III', \n  'Belinostat', 'Foretinib', 'MLN 2238', 'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene', 'Oprozomib (ONX 0912)', 'CHIR-99021']\nlist_fold_ids =  ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells'] \nfor fold_id in list_fold_ids:\n    mask_va = (train.cell_type == fold_id) & ~train.sm_name.isin(train_sm_names)\n    mask_tr = ~mask_va # 485 or 487 training rows\n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_AmbrosM.append( [IX_train, IX_test ])\n\n# For MT CV scheme:\nfolds_index_data_MT = []\nfold_to_compounds = {0: ['Alvocidib', 'Belinostat', 'Foretinib', 'LDN 193189',  'Linagliptin', 'O-Demethylated Adapalene'],\n 1: ['Dabrafenib', 'Dactolisib', 'Idelalisib', 'MLN 2238', 'Palbociclib', 'Porcn Inhibitor III'],\n 2: ['CHIR-99021', 'Crizotinib', 'Oprozomib (ONX 0912)', 'Penfluridol',  'R428']}\nfor fold_id in [0,1,2]:\n    mask_va = train['cell_type'].isin(['Myeloid cells', 'B cells']) & train['sm_name'].isin(fold_to_compounds[fold_id])\n    mask_tr = ~mask_va    \n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_MT.append( [IX_train, IX_test ]) \n    \nimport random \nfrom sklearn.model_selection import KFold\n\nclass KFold_custom:\n    '''\n    Class with similar to sklearn \"KFold\" class, interface.\n    Support of CV schemes proposed by AmbrosM, MT, Kishan and standard sklearn KFold\n    Supports options to return IX_train,IX_valid,IX_test - triplet - if valid_size > 0\n    Examples:\n    kf = KFold_custom('AmbrosM'):     \n    for i_fold,(IX_train,IX_test)   in enumerate( kf.split()) :\n        pass\n    kf = KFold_custom('AmbrosM', valid_size = 0.1 )     \n    for i_fold,(IX_train ,IX_valid,IX_test)   in enumerate( kf.split()) :\n        pass\n        \n    kf = KFold_custom(CV_scheme = 'Random') # wrapper for sklearn Kfolds with shuffle = True\n    kf = KFold_custom(CV_scheme = CV_scheme, random_state = 42 ) # fixing random_state ensures reprodicibility of folds \n    '''\n    def __init__(self, CV_scheme,  valid_size = 0, n_splits=5, random_state = 42, verbose = 0 ):\n        self.CV_scheme = CV_scheme\n        self.CV_scheme_inf =  CV_scheme\n        self.valid_size = np.clip(valid_size,0,1)\n        self.random_state = random_state\n\n        self.folds_index_data = [ [np.arange(0), np.arange(0), np.arange(0)] ] # Example data\n        self.n_splits = 1 # Example data\n        if CV_scheme == 'Kishan1':\n            # One fold scheme which contains train, valid, test parts \n            IX_train_Kishan1 = np.arange(429)\n            IX_val_Kishan1 = np.arange(429,521)\n            IX_test_Kishan1 = np.arange(521,614)\n            self.n_splits = 1\n            self.folds_index_data =  [   [IX_train_Kishan1, IX_val_Kishan1, IX_test_Kishan1]  ]\n        elif CV_scheme == 'AmbrosM':\n            self.n_splits = 4\n            self.folds_index_data = folds_index_data_AmbrosM\n        elif CV_scheme == 'MT':\n            self.n_splits = 3\n            self.folds_index_data = folds_index_data_MT\n        elif 'Random'.lower() in CV_scheme.lower():\n            self.n_splits = n_splits\n            self.CV_scheme_inf = 'Random_'+str(n_splits) +'_'+str(random_state )\n            kf = KFold(n_splits=n_splits, random_state = random_state, shuffle=True )#,\n            self.folds_index_data = list( kf.split( np.arange(614) ) )\n        elif 'Full'.lower() in CV_scheme.lower():\n            # Just return the full set as both train and test - it useful to re-train the model on the entire data\n            IX_train_full = np.arange(614)\n            IX_test_full = np.arange(614)\n            self.n_splits = 1\n            self.folds_index_data = [   [IX_train_full, IX_test_full]  ]\n        else:\n            s = 'Uncrecognized CV_scheme ' + str(CV_scheme) \n            raise ValueError(s)\n\n        self.verbose = verbose \n        if verbose >= 10:\n            print(self.CV_scheme, self.n_splits, len(self.folds_index_data) )\n            #print(self.folds_index_data)\n            \n    def split(self, X=None):\n        '''\n        X - NOT used, just for compatibility with sklearn \n        '''\n        for item in self.folds_index_data:\n            if self.CV_scheme in ['Kishan1']:\n                if self.valid_size > 0:\n                    yield item[0],item[1],item[2]\n                else :\n                    yield np.array( list(set(item[0])|set(item[1]))  ) , item[2] # Return train, test only \n            elif self.CV_scheme in ['AmbrosM','MT', 'Random', 'Full']:\n                if self.valid_size == 0:\n                    yield item[0],item[1]\n                else:\n                    # Split \"full-train\"->( real-train, valid ) \n                    index_for_real_train_part = int(  (1-self.valid_size) * len(item[0]) )\n                    if self.random_state is None:\n                        IX_train =  item[0][:index_for_real_train_part]\n                        IX_valid =  item[0][index_for_real_train_part:]\n                    elif self.random_state == -1:\n                        p = np.random.permutation(len(item[0])) \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    else:\n                        np.random.seed(self.random_state) # Set temporary random seed\n                        p = np.random.permutation(len(item[0])) \n                        np.random.seed(random.randint(0,30000)) # Randomize seed again - use Python random, not numpy       \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    yield IX_train, IX_valid, item[1]\n                    \n        \n    def get_list_main_CV_schemes(self):\n        return ['AmbrosM','MT','Kishan1','Random']\n    def get_n_splits(self, X=np.array([])):\n        return self.n_splits    ","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-11T21:53:26.207752Z","iopub.execute_input":"2023-11-11T21:53:26.208557Z","iopub.status.idle":"2023-11-11T21:53:26.247191Z","shell.execute_reply.started":"2023-11-11T21:53:26.208521Z","shell.execute_reply":"2023-11-11T21:53:26.246222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_smile.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:27.084011Z","iopub.execute_input":"2023-11-11T21:53:27.084368Z","iopub.status.idle":"2023-11-11T21:53:27.090407Z","shell.execute_reply.started":"2023-11-11T21:53:27.084337Z","shell.execute_reply":"2023-11-11T21:53:27.089502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map_t.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:27.745333Z","iopub.execute_input":"2023-11-11T21:53:27.745689Z","iopub.status.idle":"2023-11-11T21:53:27.751717Z","shell.execute_reply.started":"2023-11-11T21:53:27.745659Z","shell.execute_reply":"2023-11-11T21:53:27.750825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_array =[]\nCV_score_array    =[]\nl = nn.MSELoss()\nbest_val_loss = float('inf')  # 最良の検証損失\nBatch_size = 64\nid_map_t =id_map_t.to(device)\ntorch.backends.cudnn.benchmark = True\n\npredictions_array =[]\nCV_score_array    =[]\nl = nn.MSELoss()\nbest_val_loss = float('inf')  # 最良の検証損失\nbad_epochs = 0 \ni=0\n\n'AmbrosM'\nkf = KFold_custom('MT', valid_size = 0)     \nfor i_fold,(train_index ,test_index) in enumerate( kf.split(None)) :\n    X_train, X_val = X_smile[train_index], X_smile[test_index]\n    y_train, y_val = y_smile[train_index], y_smile[test_index]\n    X_train_t = torch.tensor(X_train, dtype=torch.float, requires_grad=True)\n    X_val_t = torch.tensor(X_val, dtype=torch.float,requires_grad=True)\n    y_train_t = torch.tensor(y_train, dtype=torch.float,requires_grad=True)\n    y_val_t = torch.tensor(y_val, dtype=torch.float,requires_grad=True)\n    \n    X_train_t = X_train_t.to(device)\n    X_val_t = X_val_t.to(device)\n    y_train_t = y_train_t.to(device)\n    y_val_t = y_val_t.to(device)\n    \n    num_categories1 = len( set(X_enc[:,0]) )\n    num_categories2 = len( set(X_enc[:,1]) )\n    emb1_size = 2\n    emb2_size = 5\n    emb3_size = 2258\n    model = EmbNN(num_categories1=num_categories1, num_categories2=num_categories2,  emb1_size=emb1_size, emb2_size=emb2_size, emb3_size=2258) \n    model.to(device)\n\n    criterion = MRRMSE_torch  # Mean Squared Error loss\n    optimizer = torch.optim.SGD(model.parameters(), lr=1)\n    pt = 50\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, 'min', factor=0.1, patience=pt, verbose=True, min_lr=1e-8)\n    \n    best_val_loss = float('inf')\n    flag = False\n    \n    num_epochs = 3000\n    early_stopping = EarlyStopping(patience=400, path=f'checkpoint{i}.pt')\n    i+=1\n    \n    val_loss_li = []\n    val_MRRMSE_li = []\n    for epoch in range(num_epochs):\n        model.train()\n        # Forward pass\n        #     print(epoch)\n        categorical_feature1 = X_train_t[:,0]\n        categorical_feature2 = X_train_t[:,1]\n        cat_feature3 = X_train_t[:, 2:]\n        preds = model(categorical_feature1, categorical_feature2, cat_feature3)\n        #     print(outputs.shape)\n        loss = criterion(preds, y_train_t)# .float() )\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        model.eval()\n        \n        with torch.no_grad():\n            categorical_feature1 = X_val_t[:,0]\n            categorical_feature2 = X_val_t[:,1]\n            cat_feature3 = X_val_t[:, 2:]\n            preds = model(categorical_feature1, categorical_feature2, cat_feature3)\n            s = criterion(preds, y_val_t).item()# .float() )\n            loss_test = l(preds, y_val_t) \n            val_loss_li.append(loss_test.item())\n            val_MRRMSE_li.append(s)\n            if (val_MRRMSE_li[0]>s):\n                flag = True\n                \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{num_epochs}], Loss: {loss.item():.4f}')\n            print('loss_val:',loss_test.item())\n            print('MRRMSE_val:',s)\n        \n\n        if flag:\n            early_stopping(s,model)\n            scheduler.step(s)\n            if s < best_val_loss:\n                best_val_loss = s\n                bad_epochs = 0\n            else:\n                bad_epochs += 1\n            if bad_epochs > pt:\n                model.load_state_dict(torch.load(early_stopping.path))\n                print('reset model')\n                bad_epochs = 0  # カウンタをリセット\n\n        if early_stopping.early_stop:\n            break\n        \n    print(len(val_loss_li))\n    fig = plt.figure()\n\n    ax1 = fig.add_subplot(121)\n    ax1.set_xlabel('epochs', size=16)\n    ax1.set_ylabel('loss',size=16)\n    ax1.plot(val_loss_li, label='loss')\n    ax1.legend()\n\n    ax2 = fig.add_subplot(122)\n    ax2.set_xlabel('epochs', size=16)\n    ax2.set_ylabel('MRRMSE',size=16)\n    ax2.plot(val_MRRMSE_li, label='MRRMSE')\n    ax2.legend()\n\n    plt.show()\n\n    model.eval()\n    with torch.no_grad():  \n        model.load_state_dict(torch.load(early_stopping.path))\n        categorical_feature1 = torch.tensor(id_map_t[:,0]).to(device)\n        categorical_feature2 = torch.tensor(id_map_t[:,1]).to(device)\n        categorical_feature3 = torch.tensor(id_map_t[:,2:]).to(device)\n        y_preds = model(categorical_feature1, categorical_feature2, categorical_feature3)\n\n    if flag:\n        CV_score_array.append(np.min(val_loss_li))\n        predictions_array.append(y_preds)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T21:53:28.339920Z","iopub.execute_input":"2023-11-11T21:53:28.340758Z","iopub.status.idle":"2023-11-11T22:04:34.700063Z","shell.execute_reply.started":"2023-11-11T21:53:28.340727Z","shell.execute_reply":"2023-11-11T22:04:34.699035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:05:19.105270Z","iopub.execute_input":"2023-11-11T22:05:19.105885Z","iopub.status.idle":"2023-11-11T22:05:19.110665Z","shell.execute_reply.started":"2023-11-11T22:05:19.105847Z","shell.execute_reply":"2023-11-11T22:05:19.109724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure()\nax1 = fig.add_subplot(121)\nax1.set_xlabel('epochs')\nax1.set_ylabel('loss')\nax1.plot(val_loss_li, label='loss')\nax1.legend()\nax2 = fig.add_subplot(122)\nax2.set_xlabel('epochs')\nax2.set_ylabel('MRRMSE')\nax2.plot(val_loss_li, label='MRRMSE')\nax2.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:05:21.719125Z","iopub.execute_input":"2023-11-11T22:05:21.719762Z","iopub.status.idle":"2023-11-11T22:05:22.115214Z","shell.execute_reply.started":"2023-11-11T22:05:21.719726Z","shell.execute_reply":"2023-11-11T22:05:22.114273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_array","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:05:22.234513Z","iopub.execute_input":"2023-11-11T22:05:22.234869Z","iopub.status.idle":"2023-11-11T22:05:22.250689Z","shell.execute_reply.started":"2023-11-11T22:05:22.234839Z","shell.execute_reply":"2023-11-11T22:05:22.249583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nt_avg = predictions_array[0]\nfor t in predictions_array[1:]:\n    t_avg += t\nt_avg = t_avg/len(predictions_array)","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:05:25.622708Z","iopub.execute_input":"2023-11-11T22:05:25.623571Z","iopub.status.idle":"2023-11-11T22:05:25.628445Z","shell.execute_reply.started":"2023-11-11T22:05:25.623539Z","shell.execute_reply":"2023-11-11T22:05:25.627422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_avg.shape","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T22:05:27.378414Z","iopub.execute_input":"2023-11-11T22:05:27.379026Z","iopub.status.idle":"2023-11-11T22:05:27.385000Z","shell.execute_reply.started":"2023-11-11T22:05:27.378990Z","shell.execute_reply":"2023-11-11T22:05:27.384020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(t_avg.cpu(), columns= de_train.columns[5:])\nsubmission.index.name = 'id'\n\nsubmission","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T22:05:30.235802Z","iopub.execute_input":"2023-11-11T22:05:30.236531Z","iopub.status.idle":"2023-11-11T22:05:30.273948Z","shell.execute_reply.started":"2023-11-11T22:05:30.236498Z","shell.execute_reply":"2023-11-11T22:05:30.273044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.iloc[153, :]","metadata":{"execution":{"iopub.status.busy":"2023-11-11T22:05:39.956122Z","iopub.execute_input":"2023-11-11T22:05:39.956880Z","iopub.status.idle":"2023-11-11T22:05:39.965104Z","shell.execute_reply.started":"2023-11-11T22:05:39.956830Z","shell.execute_reply":"2023-11-11T22:05:39.964149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('sub_nn5_CV_6layers_MRRMSE_Fgprint+Descriptor_MT_ELU.csv')","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-11-11T22:05:43.423358Z","iopub.execute_input":"2023-11-11T22:05:43.423722Z","iopub.status.idle":"2023-11-11T22:05:51.696740Z","shell.execute_reply.started":"2023-11-11T22:05:43.423691Z","shell.execute_reply":"2023-11-11T22:05:51.695937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]}]}