{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **🥈This Notebook now rank 15 in LB!**\n+ I used `TruncatedSVD(n_components=128, random_state=42)` both on train and test data.\n+ CatBoostRegressor is used. To save time， only trained on fold_0. Maybe more fold will help to impore score.\n+ This notebook is based on [FABIEN CROM](https://www.kaggle.com/code/fabiencrom/msci-multiome-quickstart-w-sparse-matrices). Please Upvoted if help !\n+ Version_1 is a quick submission and version_2_3 show the training process.","metadata":{}},{"cell_type":"code","source":"import os\n\n\nlen(os.listdir(\"../input/sparse-target-chuck/output/\"))\nFROM = 9000\nTO = 10000","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:36.579214Z","iopub.execute_input":"2022-10-11T05:25:36.579752Z","iopub.status.idle":"2022-10-11T05:25:36.846046Z","shell.execute_reply.started":"2022-10-11T05:25:36.57965Z","shell.execute_reply":"2022-10-11T05:25:36.844987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, gc, pickle\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\nfrom matplotlib.ticker import MaxNLocator\n\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, scale\nfrom sklearn.decomposition import PCA, TruncatedSVD\nfrom sklearn.dummy import DummyRegressor\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.linear_model import Ridge, LinearRegression, Lasso\nfrom sklearn.metrics import mean_squared_error\n\nimport scipy\nimport scipy.sparse\n\nimport gc\nimport pickle\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-11T05:25:36.847597Z","iopub.execute_input":"2022-10-11T05:25:36.848425Z","iopub.status.idle":"2022-10-11T05:25:37.526534Z","shell.execute_reply.started":"2022-10-11T05:25:36.848389Z","shell.execute_reply":"2022-10-11T05:25:37.525387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    if y_true.shape != y_pred.shape: raise ValueError(\"Shapes are different.\")\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:37.527955Z","iopub.execute_input":"2022-10-11T05:25:37.528357Z","iopub.status.idle":"2022-10-11T05:25:37.536319Z","shell.execute_reply.started":"2022-10-11T05:25:37.528322Z","shell.execute_reply":"2022-10-11T05:25:37.534954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing and cross-validation\n\nWe first load all of the training input data for Multiome. It should take less than a minute.","metadata":{}},{"cell_type":"code","source":"# %%time\n# train_inputs = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_inputs_values.sparse.npz\")","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:37.538891Z","iopub.execute_input":"2022-10-11T05:25:37.539268Z","iopub.status.idle":"2022-10-11T05:25:37.547935Z","shell.execute_reply.started":"2022-10-11T05:25:37.539234Z","shell.execute_reply":"2022-10-11T05:25:37.546805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_inputs = train_inputs.astype('float16', copy=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:37.549558Z","iopub.execute_input":"2022-10-11T05:25:37.550013Z","iopub.status.idle":"2022-10-11T05:25:37.558777Z","shell.execute_reply.started":"2022-10-11T05:25:37.549979Z","shell.execute_reply":"2022-10-11T05:25:37.557466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PCA / TruncatedSVD\nIt is not possible to directly apply PCA to a sparse matrix, because PCA has to first \"center\" the data, which destroys the sparsity. This is why we apply `TruncatedSVD` instead (which is pretty much \"PCA without centering\"). It might be better to normalize the data a bit more here, but we will keep it simple.","metadata":{}},{"cell_type":"code","source":"%%time\nwith open(\"../input/multiome-svd-128/pca.pickle\", \"rb\") as f:\n    pca = pickle.load(f)\n\nwith open(\"../input/multiome-svd-128/multiome_train_x.pickle\", \"rb\") as f:\n    train_inputs = pickle.load(f)\n    \nprint(pca.explained_variance_ratio_.sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:37.560195Z","iopub.execute_input":"2022-10-11T05:25:37.560616Z","iopub.status.idle":"2022-10-11T05:25:39.591314Z","shell.execute_reply.started":"2022-10-11T05:25:37.560583Z","shell.execute_reply":"2022-10-11T05:25:39.59006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# with open(\"../input/multiome-svd-128/output_pca.pickle\", \"rb\") as f:\n#     pca2 = pickle.load(f)\n# with open(\"../input/multiome-svd-128/multiome_train_y.pickle\", \"rb\") as f:\n#     train_targets = pickle.load(f)\n# train_targets = scipy.sparse.load_npz(\n#     \"../input/multimodal-single-cell-as-sparse-matrix/train_multi_targets_values.sparse.npz\"\n# )\n# print(pca2.explained_variance_ratio_.sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.594985Z","iopub.execute_input":"2022-10-11T05:25:39.595939Z","iopub.status.idle":"2022-10-11T05:25:39.602339Z","shell.execute_reply.started":"2022-10-11T05:25:39.5959Z","shell.execute_reply":"2022-10-11T05:25:39.600957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save(name, model):\n    with open(name, 'wb') as f:\n        pickle.dump(model, f)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.603945Z","iopub.execute_input":"2022-10-11T05:25:39.604444Z","iopub.status.idle":"2022-10-11T05:25:39.618509Z","shell.execute_reply.started":"2022-10-11T05:25:39.604397Z","shell.execute_reply":"2022-10-11T05:25:39.617506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.base import RegressorMixin, ClassifierMixin, is_classifier, clone\nfrom sklearn.utils.validation import check_is_fitted, has_fit_parameter, _check_fit_params\nfrom joblib import Parallel\nfrom sklearn.utils.fixes import delayed\nimport os\nimport gc\nimport pickle\nimport numpy as np\nfrom tqdm import tqdm\n\n\nclass MultiOutputRegressorChuck(MultiOutputRegressor):\n    def _fit_estimator(self, estimator, X, y, i: int, sample_weight=None, **fit_params):\n        estimator = clone(estimator)\n        filename = f\"{self.model_dir}/{i}\"\n        if filename not in os.listdir():\n            if sample_weight is not None:\n                estimator.fit(X, y, sample_weight=sample_weight, **fit_params)\n            else:\n                estimator.fit(X, y, **fit_params)\n            with open(f\"{self.model_dir}/{i}\", \"wb\") as f:\n                pickle.dump(estimator, f)\n            del estimator\n            gc.collect()\n        return filename\n\n    def fit(self, X, y, sample_weight=None, **fit_params):\n        \"\"\"Fit the model to data, separately for each output variable.\n        Parameters\n        ----------\n        X : {array-like, sparse matrix} of shape (n_samples, n_features)\n            The input data.\n        y : {array-like, sparse matrix} of shape (n_samples, n_outputs)\n            Multi-output targets. An indicator matrix turns on multilabel\n            estimation.\n        sample_weight : array-like of shape (n_samples,), default=None\n            Sample weights. If `None`, then samples are equally weighted.\n            Only supported if the underlying regressor supports sample\n            weights.\n        **fit_params : dict of string -> object\n            Parameters passed to the ``estimator.fit`` method of each step.\n            .. versionadded:: 0.23\n        Returns\n        -------\n        self : object\n            Returns a fitted instance.\n        \"\"\"\n\n        if not hasattr(self.estimator, \"fit\"):\n            raise ValueError(\"The base estimator should implement a fit method\")\n\n        # y = self._validate_data(X=X, y=y, multi_output=True)\n\n        if is_classifier(self):\n            check_classification_targets(y)\n\n        if y.ndim == 1:\n            raise ValueError(\n                \"y must have at least two dimensions for \"\n                \"multi-output regression but has only one.\"\n            )\n\n        if sample_weight is not None and not has_fit_parameter(\n            self.estimator, \"sample_weight\"\n        ):\n            raise ValueError(\"Underlying estimator does not support sample weights.\")\n\n        fit_params_validated = _check_fit_params(X, fit_params)\n        self.y_shape = len(os.listdir(\"../input/sparse-target-chuck/output/\"))\n        Parallel(n_jobs=self.n_jobs)(\n            delayed(self._fit_estimator)(\n                self.estimator, X, y[:, i].toarray(), i, sample_weight, **fit_params_validated\n            )\n            for i in range(self.y_shape)\n        )\n\n        return self\n    def _predict(self, X, i):\n        with open(f\"{self.model_dir}/{i}\", \"rb\") as f:\n            estimator = pickle.load(f)\n        try:\n            return estimator.predict(X)\n        finally:\n            del estimator\n            gc.collect()\n\n    def predict(self, X):\n        \"\"\"Predict multi-output variable using model for each target variable.\n        Parameters\n        ----------\n        X : {array-like, sparse matrix} of shape (n_samples, n_features)\n            The input data.\n        Returns\n        -------\n        y : {array-like, sparse matrix} of shape (n_samples, n_outputs)\n            Multi-output targets predicted across multiple predictors.\n            Note: Separate models are generated for each predictor.\n        \"\"\"\n\n        y = Parallel(n_jobs=self.n_jobs)(\n            delayed(self._predict)(X, i) for i in tqdm(range(self.y_shape))\n        )\n\n        return np.asarray(y).T\n    \n    def __init__(self, *args, model_dir: str = \"multioutput\", **kwargs):\n        super().__init__(*args, **kwargs)\n        if model_dir not in os.listdir():\n            try:\n                os.mkdir(model_dir)\n            except OSError:\n                pass\n        \n        self.model_dir = model_dir","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.619988Z","iopub.execute_input":"2022-10-11T05:25:39.621312Z","iopub.status.idle":"2022-10-11T05:25:39.643487Z","shell.execute_reply.started":"2022-10-11T05:25:39.621262Z","shell.execute_reply":"2022-10-11T05:25:39.6423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_catboost():\n    from catboost import CatBoostRegressor\n\n\n    params = {\n        'learning_rate': 0.2, \n        'depth': 7, \n        'l2_leaf_reg': 4, \n        'loss_function': 'RMSE', \n        'eval_metric': 'RMSE', \n        'task_type': 'CPU', \n        'iterations': 200,\n        'od_type': 'Iter', \n        'boosting_type': 'Plain', \n        'bootstrap_type': 'Bayesian', \n        'allow_const_label': True, \n        'random_state': 1,\n#         'silent': True,\n        'verbose': 0\n    }\n    return CatBoostRegressor(**params)\n\ndef get_lgbm():\n    from sklearn.multioutput import MultiOutputRegressor\n    import lightgbm as lgb\n\n    params = {\n        'learning_rate': 0.1, \n        'metric': 'mse', \n        \"seed\": 42,\n        'reg_alpha': 0.0014, \n        'reg_lambda': 0.2, \n        'colsample_bytree': 0.8, \n        'subsample': 0.5, \n        'max_depth': 10, \n        'num_leaves': 722, \n        'min_child_samples': 83,\n#         'device':'gpu'\n    }\n\n    return lgb.LGBMRegressor(**params, n_estimators=400)\n\n\ndef get_voting():\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.multioutput import MultiOutputRegressor\n    \n    return MultiOutputRegressorChuck(VotingRegressor([(\"cat\", get_catboost()), (\"lgbm\", get_lgbm())]))\n\nget_a_model = get_lgbm","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.6468Z","iopub.execute_input":"2022-10-11T05:25:39.647253Z","iopub.status.idle":"2022-10-11T05:25:39.658711Z","shell.execute_reply.started":"2022-10-11T05:25:39.647214Z","shell.execute_reply":"2022-10-11T05:25:39.657156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 1","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.660774Z","iopub.execute_input":"2022-10-11T05:25:39.661382Z","iopub.status.idle":"2022-10-11T05:25:39.6767Z","shell.execute_reply.started":"2022-10-11T05:25:39.661332Z","shell.execute_reply":"2022-10-11T05:25:39.67572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir output","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:39.677925Z","iopub.execute_input":"2022-10-11T05:25:39.678532Z","iopub.status.idle":"2022-10-11T05:25:40.799148Z","shell.execute_reply.started":"2022-10-11T05:25:39.678494Z","shell.execute_reply":"2022-10-11T05:25:40.797445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    if y_true.shape != y_pred.shape: raise ValueError(\"Shapes are different.\")\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:40.801307Z","iopub.execute_input":"2022-10-11T05:25:40.801818Z","iopub.status.idle":"2022-10-11T05:25:40.810499Z","shell.execute_reply.started":"2022-10-11T05:25:40.801768Z","shell.execute_reply":"2022-10-11T05:25:40.809602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\nall_row_indices = np.arange(train_inputs.shape[0])\nnp.random.shuffle(all_row_indices)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nindex = 0\nscore = []\n\n# model = Ridge(copy_X=False)\nd = train_inputs.shape[0]//n\nX = train_inputs\nwith open(\"../input/multiome-svd-128/multiome_test_x.pickle\", \"rb\") as f:\n    multi_test_x = pickle.load(f)\n    \nfor i in range(0, n*d, d):\n    print(f'start [{i}:{i+d}]')\n    ind = all_row_indices[i:i+d]    \n    \n    \n    for j in range(FROM, TO):\n        print(f\"output: {j}\")\n        with open(f\"../input/sparse-target-chuck/output/{j}\", \"rb\") as f:\n            Y = pickle.load(f).T\n\n\n        print('Train...')\n        model = get_a_model()\n        model.fit(X, Y)\n\n        pred = model.predict(multi_test_x)\n        save(f\"output/{j}\", pred)\n        del pred, model\n        gc.collect()\n\n    break\n","metadata":{"execution":{"iopub.status.busy":"2022-10-11T05:25:40.811673Z","iopub.execute_input":"2022-10-11T05:25:40.812748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FROM","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}