{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc, pickle, datetime, scipy.sparse\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.model_selection import GroupKFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler, scale, MinMaxScaler\n\nimport scipy\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\nTUNE = False\nSUBMIT = True","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-06T14:18:59.477446Z","iopub.execute_input":"2022-10-06T14:18:59.478497Z","iopub.status.idle":"2022-10-06T14:19:00.988167Z","shell.execute_reply.started":"2022-10-06T14:18:59.478394Z","shell.execute_reply":"2022-10-06T14:19:00.986743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# If you see a warning \"Failed to establish a new connection\" running this cell,\n# go to \"Settings\" on the right hand side, \n# and turn on internet. Note, you need to be phone verified.\n# We need this library to read HDF files.\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-06T14:19:00.990081Z","iopub.execute_input":"2022-10-06T14:19:00.990481Z","iopub.status.idle":"2022-10-06T14:19:18.673407Z","shell.execute_reply.started":"2022-10-06T14:19:00.990436Z","shell.execute_reply":"2022-10-06T14:19:18.671935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading and preprocessing\n\nThe metadata is used only for the `GroupKFold`: ","metadata":{}},{"cell_type":"code","source":"metadata_df = pd.read_csv(FP_CELL_METADATA, index_col='cell_id')\nmetadata_df = metadata_df[metadata_df.technology==\"multiome\"]\nmetadata_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:19:18.675322Z","iopub.execute_input":"2022-10-06T14:19:18.676329Z","iopub.status.idle":"2022-10-06T14:19:19.242194Z","shell.execute_reply.started":"2022-10-06T14:19:18.676270Z","shell.execute_reply":"2022-10-06T14:19:19.241091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We read train and test datasets, keep the important columns and convert the rest to sparse matrices.","metadata":{}},{"cell_type":"code","source":"X = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_inputs_values.sparse.npz\")\n# X = X.astype('float16', copy=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:19:19.244798Z","iopub.execute_input":"2022-10-06T14:19:19.245181Z","iopub.status.idle":"2022-10-06T14:20:24.791993Z","shell.execute_reply.started":"2022-10-06T14:19:19.245139Z","shell.execute_reply":"2022-10-06T14:20:24.789443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Original X shape: {str(X.shape):14} {X.size*4/1024/1024/1024:2.3f} GByte\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:20:24.794858Z","iopub.execute_input":"2022-10-06T14:20:24.795368Z","iopub.status.idle":"2022-10-06T14:20:25.074666Z","shell.execute_reply.started":"2022-10-06T14:20:24.795297Z","shell.execute_reply":"2022-10-06T14:20:25.073411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nY = scipy.sparse.load_npz(\"../input/multimodal-single-cell-as-sparse-matrix/train_multi_targets_values.sparse.npz\")\n\nprint(f\"Original Y shape: {str(Y.shape):14} {Y.size*4/1024/1024/1024:2.3f} GByte\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:20:25.076589Z","iopub.execute_input":"2022-10-06T14:20:25.077274Z","iopub.status.idle":"2022-10-06T14:21:18.356108Z","shell.execute_reply.started":"2022-10-06T14:20:25.077224Z","shell.execute_reply":"2022-10-06T14:21:18.355029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_ROWS = 5000\nidxs = np.random.randint(0, X.shape[0], N_ROWS)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:18.357876Z","iopub.execute_input":"2022-10-06T14:21:18.358267Z","iopub.status.idle":"2022-10-06T14:21:18.366215Z","shell.execute_reply.started":"2022-10-06T14:21:18.358234Z","shell.execute_reply":"2022-10-06T14:21:18.364781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X[idxs]\nY = Y[idxs]\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:18.368304Z","iopub.execute_input":"2022-10-06T14:21:18.368667Z","iopub.status.idle":"2022-10-06T14:21:18.674275Z","shell.execute_reply.started":"2022-10-06T14:21:18.368634Z","shell.execute_reply":"2022-10-06T14:21:18.672932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA, SparsePCA","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:18.676005Z","iopub.execute_input":"2022-10-06T14:21:18.676343Z","iopub.status.idle":"2022-10-06T14:21:18.681167Z","shell.execute_reply.started":"2022-10-06T14:21:18.676311Z","shell.execute_reply":"2022-10-06T14:21:18.680205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(256, random_state=32)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:18.684803Z","iopub.execute_input":"2022-10-06T14:21:18.685627Z","iopub.status.idle":"2022-10-06T14:21:18.693594Z","shell.execute_reply.started":"2022-10-06T14:21:18.685585Z","shell.execute_reply":"2022-10-06T14:21:18.692403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca.fit(Y.toarray())","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:18.694946Z","iopub.execute_input":"2022-10-06T14:21:18.695950Z","iopub.status.idle":"2022-10-06T14:21:31.109426Z","shell.execute_reply.started":"2022-10-06T14:21:18.695907Z","shell.execute_reply":"2022-10-06T14:21:31.108265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = pca.transform(Y.toarray())","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:31.110931Z","iopub.execute_input":"2022-10-06T14:21:31.111322Z","iopub.status.idle":"2022-10-06T14:21:32.511777Z","shell.execute_reply.started":"2022-10-06T14:21:31.111289Z","shell.execute_reply":"2022-10-06T14:21:32.510480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Y -= Y.mean(axis=1).reshape(-1, 1)\n# Y /= Y.std(axis=1).reshape(-1, 1)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:21:32.513827Z","iopub.execute_input":"2022-10-06T14:21:32.514654Z","iopub.status.idle":"2022-10-06T14:21:32.658717Z","shell.execute_reply.started":"2022-10-06T14:21:32.514602Z","shell.execute_reply":"2022-10-06T14:21:32.657475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 2022\n\nparams = {\n#     'criterion': 'gini',\n    'min_samples_split': 0.3,\n    'n_estimators': 200,\n    'max_depth': 7,\n    'min_samples_leaf': 20,\n    'max_samples': 0.4,\n    'oob_score': False,\n    'max_features': 0.4,\n    'random_state': seed      \n         }\n\nall_imp = np.zeros(X.shape[1], dtype=\"float64\")\nfor i in range(Y.shape[1]):\n    model = RandomForestRegressor(**params, verbose=1, n_jobs=-1)\n    model.fit(X, Y[:, i])\n    importances = model.feature_importances_\n    all_imp += importances","metadata":{"execution":{"iopub.status.busy":"2022-10-06T14:25:09.903036Z","iopub.execute_input":"2022-10-06T14:25:09.903683Z","iopub.status.idle":"2022-10-06T14:26:35.478140Z","shell.execute_reply.started":"2022-10-06T14:25:09.903639Z","shell.execute_reply":"2022-10-06T14:26:35.476316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('rf_multiome_imps.npy', 'wb') as f:\n    np.save(f, all_imp)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T14:05:51.612785Z","iopub.execute_input":"2022-09-28T14:05:51.613251Z","iopub.status.idle":"2022-09-28T14:05:51.619256Z","shell.execute_reply.started":"2022-09-28T14:05:51.61321Z","shell.execute_reply":"2022-09-28T14:05:51.618418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}