{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Our highest private score is obtained by integrating 4 MLP models and 2 LGBM models. The method is very simple. If you read it, you will not regret it.**\n\nThis notebook is our LGBM training part. There are two LGBM models in this notebook.\n\nBecause the training time is too long, this code cannot be run directly on kaggle. If you want to run it on kaggle, you need to segment it according to kfold.","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:41:21.710988Z","iopub.execute_input":"2022-11-11T01:41:21.711854Z","iopub.status.idle":"2022-11-11T01:41:35.714644Z","shell.execute_reply.started":"2022-11-11T01:41:21.711802Z","shell.execute_reply":"2022-11-11T01:41:35.713430Z"}}},{"cell_type":"code","source":"!pip install lightgbm\nimport os, gc, pickle\nimport pandas as pd\nimport numpy as np\nfrom colorama import Fore, Back, Style\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.multioutput import MultiOutputRegressor\nimport lightgbm as lgb\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet tables\ndef correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:41:35.717201Z","iopub.execute_input":"2022-11-11T01:41:35.718093Z","iopub.status.idle":"2022-11-11T01:41:48.890060Z","shell.execute_reply.started":"2022-11-11T01:41:35.718041Z","shell.execute_reply":"2022-11-11T01:41:48.888638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.read_csv(FP_CELL_METADATA, index_col='cell_id')\nmetadata_df = metadata_df[metadata_df.technology==\"citeseq\"]\nmetadata_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:41:48.891970Z","iopub.execute_input":"2022-11-11T01:41:48.892397Z","iopub.status.idle":"2022-11-11T01:41:49.420895Z","shell.execute_reply.started":"2022-11-11T01:41:48.892351Z","shell.execute_reply":"2022-11-11T01:41:49.419754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(27678) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(13176) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(31800) & metadata_df['day'].eq(7),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(2),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(3),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(4),\n    metadata_df['donor'].eq(32606) & metadata_df['day'].eq(7)\n    ]\n\n# create a list of the values we want to assign for each condition\nvalues = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16]\n\n# create a new column and use np.select to assign values to it using our lists as arguments\nmetadata_df['comb'] = np.select(conditions, values)","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:41:49.423218Z","iopub.execute_input":"2022-11-11T01:41:49.423809Z","iopub.status.idle":"2022-11-11T01:41:49.450260Z","shell.execute_reply.started":"2022-11-11T01:41:49.423777Z","shell.execute_reply":"2022-11-11T01:41:49.449045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = pd.read_hdf(FP_CITE_TRAIN_INPUTS)\ncell_index = X.index\nmeta = metadata_df.reindex(cell_index)\ndel X\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:42:16.847869Z","iopub.execute_input":"2022-11-11T01:42:16.849001Z","iopub.status.idle":"2022-11-11T01:43:08.286731Z","shell.execute_reply.started":"2022-11-11T01:42:16.848949Z","shell.execute_reply":"2022-11-11T01:43:08.285823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_train_x = pd.read_csv('../input/cite-preprocessing/X_876.csv').values\n\ncite_train_y = pd.read_hdf(FP_CITE_TRAIN_TARGETS).values\n\nprint(cite_train_x.shape)\nprint(cite_train_y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-11T01:43:24.996939Z","iopub.execute_input":"2022-11-11T01:43:24.997409Z","iopub.status.idle":"2022-11-11T01:43:42.280774Z","shell.execute_reply.started":"2022-11-11T01:43:24.997376Z","shell.execute_reply":"2022-11-11T01:43:42.279437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_test_x = pd.read_csv('../input/cite-preprocessing/Xt_876.csv').values","metadata":{"execution":{"iopub.status.busy":"2022-10-12T09:29:31.104432Z","iopub.execute_input":"2022-10-12T09:29:31.104998Z","iopub.status.idle":"2022-10-12T09:29:42.311014Z","shell.execute_reply.started":"2022-10-12T09:29:31.104957Z","shell.execute_reply":"2022-10-12T09:29:42.309899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = 1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if model == 1:\n    params = {\n     'n_estimators': 300, \n     'learning_rate': 0.1, \n     'max_depth': 10, \n     'num_leaves': 200,\n     'min_child_samples': 250,\n     'colsample_bytree': 0.8, \n     'subsample': 0.6, \n     \"seed\": 1,\n        }\nelse:\n    params = {\n        'metric': 'mae', 'random_state': 42, 'n_estimators': 520, \n        'reg_alpha':2.0814479686080976, \n        'reg_lambda': 0.6089890766483735, \n        'colsample_bytree': 0.3, \n        'subsample': 0.6, \n        'learning_rate': 0.044688975142653374, \n        'max_depth': 10, \n        'num_leaves': 245, \n        'min_child_samples': 6, \n         }\n\nfrom sklearn.model_selection import GroupKFold, train_test_split\ntest_pred = 0\nN_SPLITS_ANN = len(meta['comb'].value_counts())\nkf = GroupKFold(n_splits=N_SPLITS_ANN)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(cite_train_x, groups=meta.comb)):\n    model = None\n    gc.collect()\n    \n    X_train = cite_train_x[idx_tr] \n    y_train = cite_train_y[idx_tr]\n    X_val = cite_train_x[idx_va]\n    y_val = cite_train_y[idx_va]\n    \n    model = MultiOutputRegressor(lgb.LGBMRegressor(**params))\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_val)\n    \n    mse = mean_squared_error(y_val, y_pred)\n    corrscore = correlation_score(y_val, y_pred)\n    \n    print(mse)\n    print(corrscore)\n    test_pred = test_pred + model.predict(cite_test_x)        \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/open-problems-multimodal/sample_submission.csv')\nsubmission.loc[:48663*140-1,'target'] = test_pred.reshape(-1)","metadata":{},"execution_count":null,"outputs":[]}]}