{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Here is the 5th solution (Multiome part). \n**Our highest private score is obtained by integrating 2 MLP models and 2 1D-CNN models. The method is very simple. If you read it, you will not regret it.**\n\nThis notebook is our preprocessing part.\n\nThe main idea of the preprocessing scheme comes from Novel, one of the top method last year. We have made some changes on his basis to enhance the model effect. \n\nThanks for his sharing!","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:08.789641Z","iopub.execute_input":"2022-11-17T15:39:08.790117Z","iopub.status.idle":"2022-11-17T15:39:10.474868Z","shell.execute_reply.started":"2022-11-17T15:39:08.790003Z","shell.execute_reply":"2022-11-17T15:39:10.473837Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestRegressor\nimport sklearn\nimport copy\nimport os\n\nimport numpy as np\nimport sys\n\nfrom sklearn.decomposition import TruncatedSVD\nfrom scipy.sparse import csc_matrix\nimport logging\n\nimport numpy as np\n\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\nimport scipy\nimport scipy.sparse\n\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:10.476715Z","iopub.execute_input":"2022-11-17T15:39:10.477536Z","iopub.status.idle":"2022-11-17T15:39:10.484149Z","shell.execute_reply.started":"2022-11-17T15:39:10.477495Z","shell.execute_reply":"2022-11-17T15:39:10.482881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class tfidfTransformer():\n    def __init__(self):\n        self.idf = None\n        self.fitted = False\n\n    def fit(self, X):\n        self.idf = X.shape[0] / X.sum(axis=0)\n        self.fitted = True\n\n    def transform(self, X):\n        if not self.fitted:\n            raise RuntimeError('Transformer was not fitted on any data')\n        if scipy.sparse.issparse(X):\n            tf = X.multiply(1 / X.sum(axis=1))\n            return tf.multiply(self.idf)\n        else:\n            tf = X / X.sum(axis=1).reshape(-1,1)\n            return tf * self.idf\n\n    def fit_transform(self, X):\n        self.fit(X)\n        return self.transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:10.485800Z","iopub.execute_input":"2022-11-17T15:39:10.486510Z","iopub.status.idle":"2022-11-17T15:39:10.509107Z","shell.execute_reply.started":"2022-11-17T15:39:10.486460Z","shell.execute_reply":"2022-11-17T15:39:10.507682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_inputs_values.sparse.npz\")\nXt = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/test_multi_inputs_values.sparse.npz\")\nboth = scipy.sparse.vstack([X, Xt])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:10.512172Z","iopub.execute_input":"2022-11-17T15:39:10.512559Z","iopub.status.idle":"2022-11-17T15:39:42.320158Z","shell.execute_reply.started":"2022-11-17T15:39:10.512526Z","shell.execute_reply":"2022-11-17T15:39:42.317835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = sklearn.decomposition.TruncatedSVD(n_components=512, random_state=64)\nboth = pca.fit_transform(both)\nboth -= both.mean(axis=1).reshape(-1,1)\nboth /= both.std(axis=1, ddof=1).reshape(-1,1)\nboth = both[:,:64]\nX = both[:105942]\nXt = both[105942:]\ndel both\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.321261Z","iopub.status.idle":"2022-11-17T15:39:42.321756Z","shell.execute_reply.started":"2022-11-17T15:39:42.321537Z","shell.execute_reply":"2022-11-17T15:39:42.321559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(X).to_csv('X_64.csv', index=False)\npd.DataFrame(Xt).to_csv('Xt_64.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.323246Z","iopub.status.idle":"2022-11-17T15:39:42.323646Z","shell.execute_reply.started":"2022-11-17T15:39:42.323459Z","shell.execute_reply":"2022-11-17T15:39:42.323477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X, Xt\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.325344Z","iopub.status.idle":"2022-11-17T15:39:42.326207Z","shell.execute_reply.started":"2022-11-17T15:39:42.325980Z","shell.execute_reply":"2022-11-17T15:39:42.326001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_inputs_values.sparse.npz\")\nXt = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/test_multi_inputs_values.sparse.npz\")\nboth = scipy.sparse.vstack([X, Xt])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.327529Z","iopub.status.idle":"2022-11-17T15:39:42.328489Z","shell.execute_reply.started":"2022-11-17T15:39:42.328269Z","shell.execute_reply":"2022-11-17T15:39:42.328292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"method = 1","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.329542Z","iopub.status.idle":"2022-11-17T15:39:42.329919Z","shell.execute_reply.started":"2022-11-17T15:39:42.329735Z","shell.execute_reply":"2022-11-17T15:39:42.329753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if method == 1:\n    TfidfTransformer = tfidfTransformer()\n    pca = sklearn.decomposition.TruncatedSVD(n_components=512, random_state=64)\n    normalizer = sklearn.preprocessing.Normalizer(norm=\"l2\")\n    \n    both = TfidfTransformer.fit_transform(both)\n    both = normalizer.fit_transform(both)\n    both = np.log1p(both * 1e4)\n    both = pca.fit_transform(both)\n    both -= both.mean(axis=1).reshape(-1,1)\n    both /= both.std(axis=1, ddof=1).reshape(-1,1)\n    both = both[:,:100]\n    \n    X = both[:105942]\n    Xt = both[105942:]\n    del both\n    gc.collect()\n    \n    X0 = pd.read_csv('X_64.csv').values\n    Xt0 = pd.read_csv('Xt_64.csv').values\n    \n    X = np.hstack([X, X0])\n    Xt = np.hstack([Xt, Xt0])\n    \n    pd.DataFrame(X).to_csv('X_164_l2.csv', index=False)\n    pd.DataFrame(Xt).to_csv('Xt_164_l2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.331913Z","iopub.status.idle":"2022-11-17T15:39:42.332324Z","shell.execute_reply.started":"2022-11-17T15:39:42.332136Z","shell.execute_reply":"2022-11-17T15:39:42.332155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if method == 2:\n    TfidfTransformer = tfidfTransformer()\n    pca = sklearn.decomposition.TruncatedSVD(n_components=512, random_state=64)\n    normalizer = sklearn.preprocessing.Normalizer(norm=\"max\")\n    \n    both = TfidfTransformer.fit_transform(both)\n    both = normalizer.fit_transform(both)\n    both = np.log1p(both * 1e4)\n    both = pca.fit_transform(both)\n    both -= both.mean(axis=1).reshape(-1,1)\n    both /= both.std(axis=1, ddof=1).reshape(-1,1)    \n    both = both[:,:100]\n    \n    X = both[:105942]\n    Xt = both[105942:]\n    del both\n    gc.collect()\n    \n    X0 = pd.read_csv('X_64.csv').values\n    Xt0 = pd.read_csv('Xt_64.csv').values\n    \n    X = np.hstack([X, X0])\n    Xt = np.hstack([Xt, Xt0])\n    \n    pd.DataFrame(X).to_csv('X_164_max.csv', index=False)\n    pd.DataFrame(Xt).to_csv('Xt_164_max.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T15:39:42.333691Z","iopub.status.idle":"2022-11-17T15:39:42.334143Z","shell.execute_reply.started":"2022-11-17T15:39:42.333926Z","shell.execute_reply":"2022-11-17T15:39:42.333945Z"},"trusted":true},"execution_count":null,"outputs":[]}]}