{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tables\n!pip install --user magic-impute\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-09-27T13:44:39.300796Z","iopub.execute_input":"2022-09-27T13:44:39.301424Z","iopub.status.idle":"2022-09-27T13:45:03.729581Z","shell.execute_reply.started":"2022-09-27T13:44:39.301293Z","shell.execute_reply":"2022-09-27T13:45:03.727711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os, gc, pickle\nimport time\nfrom sklearn.decomposition import PCA\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.preprocessing import MinMaxScaler, minmax_scale\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import Ridge, Lasso\nfrom sklearn.model_selection import cross_val_score, train_test_split\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error, log_loss, make_scorer\nimport magic\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-09-27T13:45:03.732116Z","iopub.execute_input":"2022-09-27T13:45:03.733139Z","iopub.status.idle":"2022-09-27T13:45:06.426530Z","shell.execute_reply.started":"2022-09-27T13:45:03.733082Z","shell.execute_reply":"2022-09-27T13:45:06.425299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I try to use MAGIC denoising (https://github.com/KrishnaswamyLab/MAGIC) for feature selection.","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")\n# set index as cell_id\nmeta.set_index('cell_id', inplace=True)\nmeta=meta[meta.technology=='citeseq'].drop('technology', axis=1)\nprint(meta.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:45:06.428136Z","iopub.execute_input":"2022-09-27T13:45:06.428631Z","iopub.status.idle":"2022-09-27T13:45:06.982073Z","shell.execute_reply.started":"2022-09-27T13:45:06.428568Z","shell.execute_reply":"2022-09-27T13:45:06.980770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Count nonzero rows for each column in train and test","metadata":{}},{"cell_type":"code","source":"string = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\",\n    stop=10000)\ncols=(string > 0).sum(axis=0)\ntrain_indexes=list(string.index)\nfor i in range(10000,meta.shape[0],10000):\n    string = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\",\n    start=i, stop=i+10000)\n    cols+=(string > 0).sum(axis=0)\n    train_indexes.extend(list(string.index))\ndel string\ngc.collect()\nlen(train_indexes)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:45:06.986586Z","iopub.execute_input":"2022-09-27T13:45:06.987036Z","iopub.status.idle":"2022-09-27T13:46:02.516141Z","shell.execute_reply.started":"2022-09-27T13:45:06.986995Z","shell.execute_reply":"2022-09-27T13:46:02.514788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"string = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/test_cite_inputs.h5\",\n    stop=10000)\ntest_indexes=list(string.index)\nfor i in range(10000,meta.shape[0],10000):\n    string = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/test_cite_inputs.h5\",\n    start=i, stop=i+10000)\n    cols+=(string > 0).sum(axis=0)\n    test_indexes.extend(list(string.index))\ndel string\ngc.collect()\nlen(test_indexes)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:02.517762Z","iopub.execute_input":"2022-09-27T13:46:02.518357Z","iopub.status.idle":"2022-09-27T13:46:40.711262Z","shell.execute_reply.started":"2022-09-27T13:46:02.518316Z","shell.execute_reply":"2022-09-27T13:46:40.709946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols[cols<200].hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:40.713232Z","iopub.execute_input":"2022-09-27T13:46:40.714041Z","iopub.status.idle":"2022-09-27T13:46:41.149637Z","shell.execute_reply.started":"2022-09-27T13:46:40.713987Z","shell.execute_reply":"2022-09-27T13:46:41.148258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nonzero_columns=cols[cols>50].index\nlen(nonzero_columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:41.153419Z","iopub.execute_input":"2022-09-27T13:46:41.153827Z","iopub.status.idle":"2022-09-27T13:46:41.161762Z","shell.execute_reply.started":"2022-09-27T13:46:41.153790Z","shell.execute_reply":"2022-09-27T13:46:41.160837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metatrain=meta.loc[train_indexes]\nmetatest=meta.loc[test_indexes]\ndel meta\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:41.163421Z","iopub.execute_input":"2022-09-27T13:46:41.163956Z","iopub.status.idle":"2022-09-27T13:46:41.440777Z","shell.execute_reply.started":"2022-09-27T13:46:41.163900Z","shell.execute_reply":"2022-09-27T13:46:41.439578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As MAGIC is time consuming, sample the dataset","metadata":{}},{"cell_type":"code","source":"random.seed(52)\nsamples_train=metatrain.groupby(['day','donor','cell_type'], group_keys=False).apply(lambda x: x.sample(frac=0.1)).index\nsamples_test=metatest.groupby(['day','donor','cell_type'], group_keys=False).apply(lambda x: x.sample(frac=0.1)).index\n\nprint(len(samples_train), len(samples_test))","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:41.442331Z","iopub.execute_input":"2022-09-27T13:46:41.442699Z","iopub.status.idle":"2022-09-27T13:46:41.554562Z","shell.execute_reply.started":"2022-09-27T13:46:41.442666Z","shell.execute_reply":"2022-09-27T13:46:41.553015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_inputs = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")[nonzero_columns]\ntrain_cite_inputs=train_cite_inputs.loc[samples_train]\nprint(train_cite_inputs.shape)\ntest_cite_inputs = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/test_cite_inputs.h5\")[nonzero_columns]\ntest_cite_inputs=test_cite_inputs.loc[samples_test]\nprint(test_cite_inputs.shape)\ncite_inputs=pd.concat([train_cite_inputs, test_cite_inputs])\nprint(cite_inputs.shape)\ndel train_cite_inputs, test_cite_inputs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:46:41.560325Z","iopub.execute_input":"2022-09-27T13:46:41.560736Z","iopub.status.idle":"2022-09-27T13:48:18.848668Z","shell.execute_reply.started":"2022-09-27T13:46:41.560701Z","shell.execute_reply":"2022-09-27T13:48:18.847436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"magic_operator = magic.MAGIC(random_state=32)\ndenoised=magic_operator.fit_transform(cite_inputs)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:48:18.850082Z","iopub.execute_input":"2022-09-27T13:48:18.850443Z","iopub.status.idle":"2022-09-27T13:52:14.140749Z","shell.execute_reply.started":"2022-09-27T13:48:18.850409Z","shell.execute_reply":"2022-09-27T13:52:14.139295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this is a custom function that calculates the correlation between input and target values \n# it splits the data into batches to use RAM more accurately\n\ndef corr_matrix(X_data, y_data):\n    batch_size=256 \n    X=np.array(X_data)\n    y=np.array(y_data)\n    n=X.shape[0]\n    assert n==y.shape[0]\n    xmean=np.mean(X,axis=0)\n    ymean=np.mean(y,axis=0)\n    A=np.sum(X**2, axis=0) - n*xmean**2\n    B=np.sum(y**2, axis=0) - n*ymean**2\n    C=np.zeros((X.shape[1],y.shape[1]))\n    for batch in range(0,n,batch_size):\n        C+=np.sum(X[batch:batch+batch_size,:,None]*y[batch:batch+batch_size,None,:], axis=0)\n    corr=(C-n*xmean[:,None]*ymean[None,:])/ np.sqrt(A[:,None]*B[None,:])\n    return corr","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:52:14.143027Z","iopub.execute_input":"2022-09-27T13:52:14.143882Z","iopub.status.idle":"2022-09-27T13:52:14.159566Z","shell.execute_reply.started":"2022-09-27T13:52:14.143827Z","shell.execute_reply":"2022-09-27T13:52:14.158095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_inputs=cite_inputs.loc[samples_train]\ndenoised=denoised.loc[samples_train]\n\ndel cite_inputs\n\ntrain_cite_targets = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\").loc[samples_train]\n","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:52:14.161428Z","iopub.execute_input":"2022-09-27T13:52:14.162338Z","iopub.status.idle":"2022-09-27T13:52:15.886357Z","shell.execute_reply.started":"2022-09-27T13:52:14.162277Z","shell.execute_reply":"2022-09-27T13:52:15.884895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Show correlations between genes and proteins on the original and denoised data","metadata":{}},{"cell_type":"code","source":"m=corr_matrix(train_cite_inputs,train_cite_targets)\nm=pd.DataFrame(columns=train_cite_targets.columns,\n               index=train_cite_inputs.columns, data=m)\nm=m.loc[m.abs().median(axis=1).sort_values(ascending=False).index]\nm=m[m.abs().median(axis=0).sort_values(ascending=False).index]\n\nplt.figure(figsize=(15,40))\nax = sns.heatmap(m, cmap='hot')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:52:15.887871Z","iopub.execute_input":"2022-09-27T13:52:15.888310Z","iopub.status.idle":"2022-09-27T13:53:05.695123Z","shell.execute_reply.started":"2022-09-27T13:52:15.888270Z","shell.execute_reply":"2022-09-27T13:53:05.693740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For each target protein find 4 most correlated genes","metadata":{}},{"cell_type":"code","source":"corr_features=[]\ncorr_table=pd.DataFrame(columns=['protein','corr1','value1','corr2','value2','corr3','value3','corr4','value4'])\nfor feature in m.columns:\n    c=m[feature].abs().sort_values(ascending=False)\n    i=0\n    corr_table=pd.concat((corr_table,\n                         pd.DataFrame(columns=['protein','corr1','value1','corr2','value2','corr3','value3','corr4','value4'],\n                                     data=[[feature,c.index[0],c[0].round(4),c.index[1],c[1].round(4),\n                                                           c.index[2],c[2].round(4),c.index[3],c[3].round(4)]])))\n    # i try to get more features for lowly correlated proteins, \n    # that is why i look for other correlated genes if most correlated are already found for other proteins\n    for f in c.index:\n        if f not in corr_features:\n            corr_features.append(f)\n            i+=1\n            if i>=4:\n                break\ncorr_table.set_index('protein', inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:53:05.697136Z","iopub.execute_input":"2022-09-27T13:53:05.697887Z","iopub.status.idle":"2022-09-27T13:53:06.655568Z","shell.execute_reply.started":"2022-09-27T13:53:05.697834Z","shell.execute_reply":"2022-09-27T13:53:06.654399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(corr_table.to_string())","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:53:06.656778Z","iopub.execute_input":"2022-09-27T13:53:06.657125Z","iopub.status.idle":"2022-09-27T13:53:06.684717Z","shell.execute_reply.started":"2022-09-27T13:53:06.657092Z","shell.execute_reply":"2022-09-27T13:53:06.683617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,40))\nax = sns.heatmap(m.loc[corr_features], cmap='hot')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:53:06.686346Z","iopub.execute_input":"2022-09-27T13:53:06.686668Z","iopub.status.idle":"2022-09-27T13:53:11.937428Z","shell.execute_reply.started":"2022-09-27T13:53:06.686638Z","shell.execute_reply":"2022-09-27T13:53:11.936247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The same for denoised data","metadata":{}},{"cell_type":"code","source":"m=corr_matrix(denoised,train_cite_targets)\n\nm=pd.DataFrame(columns=train_cite_targets.columns,\n               index=train_cite_inputs.columns, data=m)\nm=m.loc[m.abs().median(axis=1).sort_values(ascending=False).index]\nm=m[m.abs().median(axis=0).sort_values(ascending=False).index]\n\nplt.figure(figsize=(15,40))\nax = sns.heatmap(m, cmap='hot')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:53:11.939065Z","iopub.execute_input":"2022-09-27T13:53:11.939428Z","iopub.status.idle":"2022-09-27T13:54:59.439142Z","shell.execute_reply.started":"2022-09-27T13:53:11.939395Z","shell.execute_reply":"2022-09-27T13:54:59.437813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"magic_corr_features=[]\ncorr_table=pd.DataFrame(columns=['protein','corr1','value1','corr2','value2','corr3','value3','corr4','value4'])\nfor feature in m.columns:\n    c=m[feature].abs().sort_values(ascending=False)\n    i=0\n    corr_table=pd.concat((corr_table,\n                         pd.DataFrame(columns=['protein','corr1','value1','corr2','value2','corr3','value3','corr4','value4'],\n                                     data=[[feature,c.index[0],c[0].round(4),c.index[1],c[1].round(4),\n                                                           c.index[2],c[2].round(4),c.index[3],c[3].round(4)]])))\n    for f in c.index:\n        if f not in magic_corr_features:\n            magic_corr_features.append(f)\n            i+=1\n            if i>=4:\n                break\ncorr_table.set_index('protein', inplace=True)\nprint(corr_table.to_string())","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:54:59.440802Z","iopub.execute_input":"2022-09-27T13:54:59.441237Z","iopub.status.idle":"2022-09-27T13:55:00.416816Z","shell.execute_reply.started":"2022-09-27T13:54:59.441196Z","shell.execute_reply":"2022-09-27T13:55:00.415632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,40))\nax = sns.heatmap(m.loc[magic_corr_features], cmap='hot')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:55:00.418159Z","iopub.execute_input":"2022-09-27T13:55:00.418505Z","iopub.status.idle":"2022-09-27T13:55:05.877691Z","shell.execute_reply.started":"2022-09-27T13:55:00.418472Z","shell.execute_reply":"2022-09-27T13:55:05.876532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del m, magic_operator, metatrain, metatest\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:55:05.879203Z","iopub.execute_input":"2022-09-27T13:55:05.879553Z","iopub.status.idle":"2022-09-27T13:55:06.096246Z","shell.execute_reply.started":"2022-09-27T13:55:05.879519Z","shell.execute_reply":"2022-09-27T13:55:06.094739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pearson(X, y):\n    n=X.shape[1]\n    assert n==y.shape[1]\n    xmean=np.mean(X,axis=1)\n    ymean=np.mean(y,axis=1)\n    A=np.sum(X**2, axis=1) - n*xmean**2\n    B=np.sum(y**2, axis=1) - n*ymean**2\n    C=np.sum(X*y, axis=1)\n    corr=(C-n*xmean*ymean)/ np.sqrt(A*B)\n    return corr","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:55:06.098442Z","iopub.execute_input":"2022-09-27T13:55:06.098828Z","iopub.status.idle":"2022-09-27T13:55:06.107066Z","shell.execute_reply.started":"2022-09-27T13:55:06.098794Z","shell.execute_reply":"2022-09-27T13:55:06.105915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compare predictions on original and denoised data for different feature sets (all features with PCA decomposition, most correlated features on original data, most correlated on denoised data)","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    train_cite_inputs.values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=Pipeline([('decompositor',PCA(n_components=512, copy=False)),\n                ('regressor',MultiOutputRegressor((Ridge())))])\nmodel.fit(X_train, y_train)\nsimple_scores = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:55:06.108529Z","iopub.execute_input":"2022-09-27T13:55:06.108915Z","iopub.status.idle":"2022-09-27T13:55:40.801499Z","shell.execute_reply.started":"2022-09-27T13:55:06.108880Z","shell.execute_reply":"2022-09-27T13:55:40.800035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    denoised.fillna(0).values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=Pipeline([('decompositor',PCA(n_components=512, copy=False)),\n                ('regressor',MultiOutputRegressor(Ridge()))])\nmodel.fit(X_train, y_train)\nmagic_scores = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:55:40.808237Z","iopub.execute_input":"2022-09-27T13:55:40.809448Z","iopub.status.idle":"2022-09-27T13:56:46.901942Z","shell.execute_reply.started":"2022-09-27T13:55:40.809375Z","shell.execute_reply":"2022-09-27T13:56:46.900451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    train_cite_inputs[corr_features].values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=MultiOutputRegressor(Ridge())\nmodel.fit(X_train, y_train)\nsimple_scores_corr = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:56:46.904578Z","iopub.execute_input":"2022-09-27T13:56:46.905626Z","iopub.status.idle":"2022-09-27T13:57:00.236819Z","shell.execute_reply.started":"2022-09-27T13:56:46.905564Z","shell.execute_reply":"2022-09-27T13:57:00.235429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    denoised.fillna(0)[corr_features].values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=MultiOutputRegressor(Ridge())\nmodel.fit(X_train, y_train)\nmagic_scores_corr = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:00.239254Z","iopub.execute_input":"2022-09-27T13:57:00.240224Z","iopub.status.idle":"2022-09-27T13:57:21.359252Z","shell.execute_reply.started":"2022-09-27T13:57:00.240158Z","shell.execute_reply":"2022-09-27T13:57:21.358100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    train_cite_inputs[magic_corr_features].values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=MultiOutputRegressor(Ridge())\nmodel.fit(X_train, y_train)\nsimple_scores_magic_corr = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:21.361163Z","iopub.execute_input":"2022-09-27T13:57:21.361649Z","iopub.status.idle":"2022-09-27T13:57:33.966036Z","shell.execute_reply.started":"2022-09-27T13:57:21.361602Z","shell.execute_reply":"2022-09-27T13:57:33.964599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    denoised[magic_corr_features].values, train_cite_targets.values, \n    test_size=0.1, random_state=42)\nmodel=MultiOutputRegressor(Ridge())\nmodel.fit(X_train, y_train)\nmagic_scores_magic_corr = pearson(model.predict(X_test), y_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:33.980507Z","iopub.execute_input":"2022-09-27T13:57:33.984473Z","iopub.status.idle":"2022-09-27T13:57:52.976208Z","shell.execute_reply.started":"2022-09-27T13:57:33.984394Z","shell.execute_reply":"2022-09-27T13:57:52.974680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nplt.grid()\nplt.boxplot([simple_scores,simple_scores_corr, simple_scores_magic_corr, \n             magic_scores, magic_scores_corr, magic_scores_magic_corr],\n            labels=['pca no denoising', 'corr no denoising','magic corr no denoising',\n                    'pca magic','corr magic', 'magic corr magic' ],\n           showfliers=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:52.978022Z","iopub.execute_input":"2022-09-27T13:57:52.980095Z","iopub.status.idle":"2022-09-27T13:57:53.249512Z","shell.execute_reply.started":"2022-09-27T13:57:52.980037Z","shell.execute_reply":"2022-09-27T13:57:53.248233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_cite_inputs, denoised, X_train, X_test, y_train, y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:53.251188Z","iopub.execute_input":"2022-09-27T13:57:53.252475Z","iopub.status.idle":"2022-09-27T13:57:53.442246Z","shell.execute_reply.started":"2022-09-27T13:57:53.252425Z","shell.execute_reply":"2022-09-27T13:57:53.441045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, predictions on denoised dataset are rather accurate and stable on different feature sets. But i cannot perform denoising for the whole dataset inside the notebook, so, let us get the prediction on original dataset with originally correlated features","metadata":{}},{"cell_type":"code","source":"train_cite_inputs = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")[corr_features]","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:57:53.445126Z","iopub.execute_input":"2022-09-27T13:57:53.445546Z","iopub.status.idle":"2022-09-27T13:58:48.395831Z","shell.execute_reply.started":"2022-09-27T13:57:53.445505Z","shell.execute_reply":"2022-09-27T13:58:48.394719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://www.kaggle.com/code/vuonglam/tune-lgbm-only-final-cite-task\nparams = {\n     'boosting': 'dart',\n     'learning_rate': 0.1, \n     'metric': 'mae', \n     \"seed\": 42,\n    'reg_alpha': 0.0015, \n    'reg_lambda': 0.2, \n    'colsample_bytree': 0.8, \n    'subsample': 0.5, \n    'max_depth': 10, \n    'num_leaves': 700, \n    'min_child_samples': 80, \n    }\n\nmodel = MultiOutputRegressor(lgb.LGBMRegressor(**params, n_estimators=400))\n#model=Pipeline([('scaler',MinMaxScaler()),('regressor',MultiOutputRegressor(Ridge()))])\n","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:58:48.397397Z","iopub.execute_input":"2022-09-27T13:58:48.398384Z","iopub.status.idle":"2022-09-27T13:58:48.405298Z","shell.execute_reply.started":"2022-09-27T13:58:48.398342Z","shell.execute_reply":"2022-09-27T13:58:48.404090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_targets=pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\ntarget_cols=train_cite_targets.columns","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:58:48.406964Z","iopub.execute_input":"2022-09-27T13:58:48.407633Z","iopub.status.idle":"2022-09-27T13:58:49.137238Z","shell.execute_reply.started":"2022-09-27T13:58:48.407574Z","shell.execute_reply":"2022-09-27T13:58:49.136160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time0=time.time()\nmodel.fit(train_cite_inputs.values, train_cite_targets.values)\nprint(time.time()-time0)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T13:58:49.139000Z","iopub.execute_input":"2022-09-27T13:58:49.139870Z","iopub.status.idle":"2022-09-27T14:02:09.714787Z","shell.execute_reply.started":"2022-09-27T13:58:49.139811Z","shell.execute_reply":"2022-09-27T14:02:09.712721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_cite_inputs,train_cite_targets\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.716366Z","iopub.status.idle":"2022-09-27T14:02:09.717701Z","shell.execute_reply.started":"2022-09-27T14:02:09.717369Z","shell.execute_reply":"2022-09-27T14:02:09.717402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_cite_inputs = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/test_cite_inputs.h5\")[corr_features]\ntarget_ids=test_cite_inputs.index\nlen(target_ids)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.719403Z","iopub.status.idle":"2022-09-27T14:02:09.720290Z","shell.execute_reply.started":"2022-09-27T14:02:09.719978Z","shell.execute_reply":"2022-09-27T14:02:09.720010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = model.predict(test_cite_inputs.values)\ndel test_cite_inputs\ntest_pred.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.721911Z","iopub.status.idle":"2022-09-27T14:02:09.722789Z","shell.execute_reply.started":"2022-09-27T14:02:09.722477Z","shell.execute_reply":"2022-09-27T14:02:09.722509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result=pd.DataFrame(data=test_pred,columns=target_cols,index=target_ids)\nresult['cell_id']=result.index\nresult=pd.melt(result, id_vars=['cell_id'])\ndel test_pred\nresult","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.724382Z","iopub.status.idle":"2022-09-27T14:02:09.724806Z","shell.execute_reply.started":"2022-09-27T14:02:09.724599Z","shell.execute_reply":"2022-09-27T14:02:09.724619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_ids = pd.read_csv(\"/kaggle/input/open-problems-multimodal/evaluation_ids.csv\", nrows=result.shape[0])\neval_ids=eval_ids.merge(result, on=['cell_id','gene_id'])\neval_ids=eval_ids.dropna()\neval_ids.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.726107Z","iopub.status.idle":"2022-09-27T14:02:09.726500Z","shell.execute_reply.started":"2022-09-27T14:02:09.726307Z","shell.execute_reply":"2022-09-27T14:02:09.726326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with open('citeseq_pred.pickle', 'wb') as f: \n#    pickle.dump(test_pred, f) # float32 array of shape (48663, 140)\n    \nwith open(\"../input/msci-multiome-quickstart/partial_submission_multi.pickle\", 'rb') as f: \n    submission = pickle.load(f)\n    \nsubmission.iloc[eval_ids.row_id] = eval_ids.value\nassert not submission.isna().any()\nsubmission = submission.round(6) # reduce the size of the csv\nsubmission.to_csv('submission.csv')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-27T14:02:09.727999Z","iopub.status.idle":"2022-09-27T14:02:09.728386Z","shell.execute_reply.started":"2022-09-27T14:02:09.728196Z","shell.execute_reply":"2022-09-27T14:02:09.728214Z"},"trusted":true},"execution_count":null,"outputs":[]}]}