{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Stacking with all targets \n\n* This is a sample notebook for the training process of stacking with torch nn/cnn models, catboost models, LGBM models(very slow) and sklearn models. All targets are trained in the single model\n\n* Due to memory limit, the training process can hardly performed on kaggle notebooks. you may try it with other computers or servers. \n\n* In this notebook, the training data is sampled and only one model can be trained once. You need to train the second model by restarting.\n\n### PS: \n* This models may not be the best method. I only used nn.Linear and nn.Dropout as the model of the third layer without any activation functions, which may be treated as a linear model. If you have any other better model please feel free to tell me.\n\n* The stacking model class can be found here `../input/cite-stacking/stacking_model.py`. You may use `fit_last` method replace `fit` to only train the last layer of the model.\n\n* Wappers can be found here `../input/cite-stacking/wappers.py`.You can theoretically use the `SklearnWrapper`,`MultiOutputSklearnWrapper` to use all of the sklearn regression models\n\n* NN/CNN models, Catboost models and LGBM models can be found here `../input/cite-stacking/models.py`","metadata":{}},{"cell_type":"markdown","source":"### Example of the results of training with all the targets\n``` json\n{'layer_1': {'0': {'KNN': 0.8763735658091053,\n   'CNN': 0.8930353496340678,\n   'ridge': 0.8874581374664077,\n   'rf': 0.875366821886147,\n   'catboost': 0.8900694588016426,\n   'torch': 0.8929359028294755},\n  '1': {'KNN': 0.8767452887472104,\n   'CNN': 0.8934511875739827,\n   'ridge': 0.8879966040048188,\n   'rf': 0.8759859359116621,\n   'catboost': 0.8904868082524582,\n   'torch': 0.8935968531631666},\n  '2': {'KNN': 0.8764993364275501,\n   'CNN': 0.8927719420168505,\n   'ridge': 0.8875339798562072,\n   'rf': 0.8755747886715507,\n   'catboost': 0.8900819544523466,\n   'torch': 0.8932289327068298}},\n 'layer_2': {'0': {'CNN': 0.8934846399719779,\n   'catboost': 0.8954075191676761,\n   'torch': 0.8935958665569982},\n  '1': {'CNN': 0.893632382038459,\n   'catboost': 0.8956694876875458,\n   'torch': 0.893871631204817},\n  '2': {'CNN': 0.8934526331609398,\n   'catboost': 0.8955541480612449,\n   'torch': 0.8936154654530928}},\n 'layer_3': {'0': {'mlp': 0.8946680330034871},\n  '1': {'mlp': 0.8951168758727184},\n  '2': {'mlp': 0.8949404122982575}}}\n```","metadata":{}},{"cell_type":"code","source":"import sys\nsys.path.append(\"../input/cite-stacking\")\nimport gc\nimport numpy as np\nimport pandas as pd\nimport random\nfrom stacking_model import ModelStacking\nfrom tqdm.notebook import tqdm\nfrom sklearn.preprocessing import LabelEncoder,OneHotEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-19T14:00:14.358090Z","iopub.execute_input":"2022-11-19T14:00:14.359418Z","iopub.status.idle":"2022-11-19T14:00:19.473120Z","shell.execute_reply.started":"2022-11-19T14:00:14.359312Z","shell.execute_reply":"2022-11-19T14:00:19.472120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_path = '/kaggle/input/single-cell-features/'\n# train\ntrain = np.load(feature_path+'train_cite_X.npy')\ntrain_index = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/train_cite_inputs_idxcol.npz\",allow_pickle=True)\nmeta = pd.read_csv(\"../input/open-problems-multimodal/metadata.csv\",index_col = \"cell_id\")\nmeta = meta[meta.technology==\"citeseq\"]\nlbe = LabelEncoder()\nmeta[\"cell_type\"] = lbe.fit_transform(meta[\"cell_type\"])\nmeta[\"gender\"] = meta.apply(lambda x:0 if x[\"donor\"]==13176 else 1,axis =1)\nmeta_train = meta.reindex(train_index[\"index\"])\ntrain_meta = meta_train[\"cell_type\"].values.reshape(-1, 1)\nohe = OneHotEncoder(sparse=False)\ntrain_meta = ohe.fit_transform(train_meta)\ntrain = np.concatenate([train,train_meta],axis= -1)\n\n# target\ntarget = np.load(feature_path+'train_cite_targets.npy') \ntarget -= target.mean(axis=1).reshape(-1, 1)\ntarget /= target.std(axis=1).reshape(-1, 1)\n\n# test\ntest = np.load(feature_path+'test_cite_X.npy')\n\ntest_index = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/test_cite_inputs_idxcol.npz\",allow_pickle=True)\nmeta_test = meta.reindex(test_index[\"index\"])\ntest_meta = meta_test[\"cell_type\"].values.reshape(-1, 1)\ntest_meta = ohe.transform(test_meta)\ntest = np.concatenate([test,test_meta],axis= -1)\n\n# all\nfea_columns = [f\"fea_{i}\" for i in range(train.shape[1])]\nlab_columns = [f\"lab_{i}\" for i in range(target.shape[1])]\ntrain = pd.DataFrame(train,columns=fea_columns)\ntest = pd.DataFrame(test,columns=fea_columns)\ntarget = pd.DataFrame(target,columns=lab_columns)\nall = pd.concat([train,target],axis=1)\ndel train,target\ngc.collect()\nprint(all.shape,test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:19.480139Z","iopub.execute_input":"2022-11-19T14:00:19.481089Z","iopub.status.idle":"2022-11-19T14:00:29.827787Z","shell.execute_reply.started":"2022-11-19T14:00:19.481051Z","shell.execute_reply":"2022-11-19T14:00:29.826593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_config = dict(\n    train_features = fea_columns,   \n    predict_label = lab_columns,\n    random_state = 42,\n    layer_1_model_list = [\"KNN\",\"CNN\",'ridge',\"rf\",\"catboost\",\"torch\"], # [\"KNN\",\"CNN\",'ridge',\"rf\",\"catboost\",\"torch\"],#,\"lgbm\",\"CNN\",\"KernelRidge\",\"ElasticNet\",'ridge',\"rf\",\"et\",\"catboost\",\"torch\",\"KernelRidge\"\n    layer_2_model_list = [\"CNN\",\"catboost\",\"torch\"], # [\"CNN\",\"catboost\",\"torch\"],#,\"lgbm\",,\"catboost\"\"CNN\",\"torch\"\n    layer_3_model = \"mlp\"\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.829513Z","iopub.execute_input":"2022-11-19T14:00:29.830170Z","iopub.status.idle":"2022-11-19T14:00:29.836437Z","shell.execute_reply.started":"2022-11-19T14:00:29.830126Z","shell.execute_reply":"2022-11-19T14:00:29.835248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_folds(lis,folds):\n    random.seed(42)\n    random.shuffle(lis)\n    num_fold = int(len(lis)/folds)\n\n    return [lis[i::folds] for i in range(folds)]\n\ndef sample_folds(fold,k):\n    random.seed(42)\n    return random.choices(fold,k = k)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.839388Z","iopub.execute_input":"2022-11-19T14:00:29.840432Z","iopub.status.idle":"2022-11-19T14:00:29.856198Z","shell.execute_reply.started":"2022-11-19T14:00:29.840392Z","shell.execute_reply":"2022-11-19T14:00:29.855095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta_train[\"id\"] = [i for i in range(meta_train.shape[0])]\n# people_list = [32606,13176,31800]\n# day_list = [2,3,4]\n# fold_list = []\n# num_fold = 3\n\n# for val_people in tqdm([32606,13176,31800]):\n#     train_people = [i for i in people_list if i != val_people]\n#     train_idx = meta_train[meta_train.donor.isin(train_people)].id.to_list()\n#     val_idx = meta_train[meta_train.donor == val_people].id.to_list()\n#     useless_idx = [i for i in meta_train.id.to_list() if i not in train_idx+val_idx]\n#     train_fold_1,train_fold_2,train_fold_3 = get_folds(train_idx,num_fold)\n    \n#     train_fold_1 = sample_folds(train_fold_1,k = 100)\n#     train_fold_2 = sample_folds(train_fold_2,k = 100)\n#     train_fold_3 = sample_folds(train_fold_3,k = 100)\n#     val_fold_1,val_fold_2,val_fold_3 = get_folds(val_idx,num_fold)\n#     val_fold_1 = sample_folds(val_fold_1,k = 100)\n#     val_fold_2 = sample_folds(val_fold_2,k = 100)\n#     val_fold_3 = sample_folds(val_fold_3,k = 100)\n#     use_folds = train_fold_1 + train_fold_2 + train_fold_3 + val_fold_1 + val_fold_2 + val_fold_3\n#     useless_idx = [i for i in meta_train.id.to_list() if i not in use_folds]\n\n#     one_fold = [\n#         [[train_fold_1+train_fold_2,val_fold_1+val_fold_2],train_fold_3+val_fold_3],\n#         [[train_fold_1+train_fold_3,val_fold_1+val_fold_3],train_fold_2+val_fold_2],\n#         [[train_fold_2+train_fold_3,val_fold_2+val_fold_3],train_fold_1+val_fold_1+useless_idx],\n#     ]\n#     fold_list.append(one_fold)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.857757Z","iopub.execute_input":"2022-11-19T14:00:29.859120Z","iopub.status.idle":"2022-11-19T14:00:29.867195Z","shell.execute_reply.started":"2022-11-19T14:00:29.859077Z","shell.execute_reply":"2022-11-19T14:00:29.866091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n# with open(\"fold_list.pkl\",\"wb\") as f:\n#     pickle.dump(fold_list,f)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.900128Z","iopub.execute_input":"2022-11-19T14:00:29.900517Z","iopub.status.idle":"2022-11-19T14:00:29.905569Z","shell.execute_reply.started":"2022-11-19T14:00:29.900482Z","shell.execute_reply":"2022-11-19T14:00:29.904499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The spliting process would take a long time, so reuse the data\nimport pickle\nwith open(\"./fold_list.pkl\",\"rb\") as f:\n    fold_list = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.868799Z","iopub.execute_input":"2022-11-19T14:00:29.869438Z","iopub.status.idle":"2022-11-19T14:00:29.898643Z","shell.execute_reply.started":"2022-11-19T14:00:29.869402Z","shell.execute_reply":"2022-11-19T14:00:29.897750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor id,part in enumerate(tqdm(fold_list)):\n    if id == 0:\n        continue\n    elif id == 1:\n        continue\n    stacking_model = ModelStacking(all,test,meta_train,main_config,part)\n    stacking_model.fit(id)\n    with open(f\"./score_record_{id}.pkl\",\"wb\") as f:\n        pickle.dump(stacking_model.score_record,f)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T14:00:29.906845Z","iopub.execute_input":"2022-11-19T14:00:29.907944Z","iopub.status.idle":"2022-11-19T15:03:22.508120Z","shell.execute_reply.started":"2022-11-19T14:00:29.907907Z","shell.execute_reply":"2022-11-19T15:03:22.506009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f\"score_record_0.pkl\",\"rb\") as f:\n    stacking_model_0 = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T15:32:14.532889Z","iopub.execute_input":"2022-11-19T15:32:14.533309Z","iopub.status.idle":"2022-11-19T15:32:14.541031Z","shell.execute_reply.started":"2022-11-19T15:32:14.533275Z","shell.execute_reply":"2022-11-19T15:32:14.539768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacking_model_0","metadata":{"execution":{"iopub.status.busy":"2022-11-19T15:32:19.681883Z","iopub.execute_input":"2022-11-19T15:32:19.682276Z","iopub.status.idle":"2022-11-19T15:32:19.693563Z","shell.execute_reply.started":"2022-11-19T15:32:19.682233Z","shell.execute_reply":"2022-11-19T15:32:19.692361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}