{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on notebook \nwww.kaggle.com/code/fabiencrom/msci-multiome-quickstart-w-sparse-matrices,\n\n+ I used TruncatedSVD(n_components=50).\n  However, when n_components is 50 , its **explained_variance_ratio_** is only **0.008763663**.\n  \n+ Training and test data were normalized before input.\n+ A training strategy of randomsampling. First, shuffled the train_norms and split it to 4 parts. Trained four models on the 4 parts. Because of 5 fold cv, I get 20 models.\n+ I will try more models to get a better result.","metadata":{}},{"cell_type":"code","source":"import os, gc, pickle\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\nfrom matplotlib.ticker import MaxNLocator\n\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, scale\nfrom sklearn.decomposition import PCA, TruncatedSVD\nfrom sklearn.dummy import DummyRegressor\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.linear_model import Ridge, LinearRegression, Lasso\nfrom sklearn.metrics import mean_squared_error\n\nimport scipy\nimport scipy.sparse\n\nimport gc\nimport pickle\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-09-06T06:47:58.957358Z","iopub.execute_input":"2022-09-06T06:47:58.957757Z","iopub.status.idle":"2022-09-06T06:48:00.398186Z","shell.execute_reply.started":"2022-09-06T06:47:58.957730Z","shell.execute_reply":"2022-09-06T06:48:00.396304Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating submission\n\nWe load the cells that will have to appear in submission.","metadata":{}},{"cell_type":"code","source":"preds = np.load('../input/msci-ridge/preds.npy').astype('float32',copy=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:49:08.073789Z","iopub.execute_input":"2022-09-06T06:49:08.074149Z","iopub.status.idle":"2022-09-06T06:51:06.022256Z","shell.execute_reply.started":"2022-09-06T06:49:08.074124Z","shell.execute_reply":"2022-09-06T06:51:06.019843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Read the table of rows and columns required for submission\neval_ids = pd.read_parquet(\"../input/multimodal-single-cell-as-sparse-matrix/evaluation.parquet\")\n# Convert the string columns to more efficient categorical types\n#eval_ids.cell_id = eval_ids.cell_id.apply(lambda s: int(s, base=16))\neval_ids.cell_id = eval_ids.cell_id.astype(pd.CategoricalDtype())\neval_ids.gene_id = eval_ids.gene_id.astype(pd.CategoricalDtype())","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:53:30.676191Z","iopub.execute_input":"2022-09-06T06:53:30.676597Z","iopub.status.idle":"2022-09-06T06:53:58.933708Z","shell.execute_reply.started":"2022-09-06T06:53:30.676570Z","shell.execute_reply":"2022-09-06T06:53:58.931722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare an empty series which will be filled with predictions\nsubmission = pd.Series(name='target',\n                       index=pd.MultiIndex.from_frame(eval_ids), \n                       dtype=np.float32)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:53:58.936264Z","iopub.execute_input":"2022-09-06T06:53:58.936642Z","iopub.status.idle":"2022-09-06T06:54:16.652995Z","shell.execute_reply.started":"2022-09-06T06:53:58.936615Z","shell.execute_reply":"2022-09-06T06:54:16.651298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We load the `index`  and `columns` of the original dataframe, as we need them to make the submission.","metadata":{}},{"cell_type":"code","source":"%%time\ny_columns = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_targets_idxcol.npz\",\n                   allow_pickle=True)[\"columns\"]\n\ntest_index = np.load(\"../input/multimodal-single-cell-as-sparse-matrix/test_multi_inputs_idxcol.npz\",\n                    allow_pickle=True)[\"index\"]","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:16.655059Z","iopub.execute_input":"2022-09-06T06:54:16.655455Z","iopub.status.idle":"2022-09-06T06:54:16.768413Z","shell.execute_reply.started":"2022-09-06T06:54:16.655395Z","shell.execute_reply":"2022-09-06T06:54:16.767483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We assign the predicted values to the correct row in the submission file.","metadata":{}},{"cell_type":"code","source":"cell_dict = dict((k,v) for v,k in enumerate(test_index)) \nassert len(cell_dict)  == len(test_index)\n\ngene_dict = dict((k,v) for v,k in enumerate(y_columns))\nassert len(gene_dict) == len(y_columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:16.770782Z","iopub.execute_input":"2022-09-06T06:54:16.771378Z","iopub.status.idle":"2022-09-06T06:54:16.796077Z","shell.execute_reply.started":"2022-09-06T06:54:16.771341Z","shell.execute_reply":"2022-09-06T06:54:16.795046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_ids_cell_num = eval_ids.cell_id.apply(lambda x:cell_dict.get(x, -1))\neval_ids_gene_num = eval_ids.gene_id.apply(lambda x:gene_dict.get(x, -1))\n\nvalid_multi_rows = (eval_ids_gene_num !=-1) & (eval_ids_cell_num!=-1)","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:16.797072Z","iopub.execute_input":"2022-09-06T06:54:16.797805Z","iopub.status.idle":"2022-09-06T06:54:18.199808Z","shell.execute_reply.started":"2022-09-06T06:54:16.797777Z","shell.execute_reply":"2022-09-06T06:54:18.198463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.iloc[valid_multi_rows] = preds[eval_ids_cell_num[valid_multi_rows].to_numpy(),\neval_ids_gene_num[valid_multi_rows].to_numpy()]","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:18.201182Z","iopub.execute_input":"2022-09-06T06:54:18.201548Z","iopub.status.idle":"2022-09-06T06:54:20.180020Z","shell.execute_reply.started":"2022-09-06T06:54:18.201513Z","shell.execute_reply":"2022-09-06T06:54:20.178663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del eval_ids_cell_num, eval_ids_gene_num, valid_multi_rows, eval_ids, test_index, y_columns\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:29.763595Z","iopub.execute_input":"2022-09-06T06:54:29.763959Z","iopub.status.idle":"2022-09-06T06:54:29.964102Z","shell.execute_reply.started":"2022-09-06T06:54:29.763934Z","shell.execute_reply":"2022-09-06T06:54:29.962446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:31.016411Z","iopub.execute_input":"2022-09-06T06:54:31.016817Z","iopub.status.idle":"2022-09-06T06:54:31.030568Z","shell.execute_reply.started":"2022-09-06T06:54:31.016788Z","shell.execute_reply":"2022-09-06T06:54:31.029176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merging with CITEseq predictions\n\nWe use the CITEseq predictions from [this notebook](https://www.kaggle.com/code/vuonglam/lgbm-baseline-optuna-drop-constant-cite-task) by VuongLam.","metadata":{}},{"cell_type":"code","source":"submission.reset_index(drop=True, inplace=True)\nsubmission.index.name = 'row_id'","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:35.377812Z","iopub.execute_input":"2022-09-06T06:54:35.378249Z","iopub.status.idle":"2022-09-06T06:54:35.505824Z","shell.execute_reply.started":"2022-09-06T06:54:35.378221Z","shell.execute_reply":"2022-09-06T06:54:35.504533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_submission = pd.read_csv(\"../input/lgbm-baseline-optuna-drop-constant-cite-task/submission.csv\")\ncite_submission = cite_submission.set_index(\"row_id\")\ncite_submission = cite_submission[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:38.481450Z","iopub.execute_input":"2022-09-06T06:54:38.481923Z","iopub.status.idle":"2022-09-06T06:54:59.416678Z","shell.execute_reply.started":"2022-09-06T06:54:38.481895Z","shell.execute_reply":"2022-09-06T06:54:59.414683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[submission.isnull()] = cite_submission[submission.isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:54:59.419361Z","iopub.execute_input":"2022-09-06T06:54:59.419840Z","iopub.status.idle":"2022-09-06T06:55:00.724810Z","shell.execute_reply.started":"2022-09-06T06:54:59.419802Z","shell.execute_reply":"2022-09-06T06:55:00.723186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:55:00.726108Z","iopub.execute_input":"2022-09-06T06:55:00.726449Z","iopub.status.idle":"2022-09-06T06:55:00.736761Z","shell.execute_reply.started":"2022-09-06T06:55:00.726391Z","shell.execute_reply":"2022-09-06T06:55:00.735032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:55:00.740378Z","iopub.execute_input":"2022-09-06T06:55:00.741663Z","iopub.status.idle":"2022-09-06T06:55:00.784033Z","shell.execute_reply.started":"2022-09-06T06:55:00.741625Z","shell.execute_reply":"2022-09-06T06:55:00.782505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:55:00.786478Z","iopub.execute_input":"2022-09-06T06:55:00.786920Z","iopub.status.idle":"2022-09-06T06:57:02.944697Z","shell.execute_reply.started":"2022-09-06T06:55:00.786893Z","shell.execute_reply":"2022-09-06T06:57:02.942636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-09-06T06:57:02.947032Z","iopub.execute_input":"2022-09-06T06:57:02.948759Z","iopub.status.idle":"2022-09-06T06:57:03.373186Z","shell.execute_reply.started":"2022-09-06T06:57:02.948701Z","shell.execute_reply":"2022-09-06T06:57:03.371923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}