{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:29.375957Z","iopub.execute_input":"2023-09-18T10:10:29.376462Z","iopub.status.idle":"2023-09-18T10:10:29.820017Z","shell.execute_reply.started":"2023-09-18T10:10:29.376419Z","shell.execute_reply":"2023-09-18T10:10:29.818252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:31.324195Z","iopub.execute_input":"2023-09-18T10:10:31.324743Z","iopub.status.idle":"2023-09-18T10:10:32.255631Z","shell.execute_reply.started":"2023-09-18T10:10:31.324708Z","shell.execute_reply":"2023-09-18T10:10:32.254265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE = '/kaggle/input/open-problems-single-cell-perturbations/'","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:33.284545Z","iopub.execute_input":"2023-09-18T10:10:33.285014Z","iopub.status.idle":"2023-09-18T10:10:33.291574Z","shell.execute_reply.started":"2023-09-18T10:10:33.284975Z","shell.execute_reply":"2023-09-18T10:10:33.290151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_parquet(f'{BASE}de_train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:33.454984Z","iopub.execute_input":"2023-09-18T10:10:33.455494Z","iopub.status.idle":"2023-09-18T10:10:36.666793Z","shell.execute_reply.started":"2023-09-18T10:10:33.455454Z","shell.execute_reply":"2023-09-18T10:10:36.665262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:36.669487Z","iopub.execute_input":"2023-09-18T10:10:36.669883Z","iopub.status.idle":"2023-09-18T10:10:36.712570Z","shell.execute_reply.started":"2023-09-18T10:10:36.669850Z","shell.execute_reply":"2023-09-18T10:10:36.711248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv(f'{BASE}id_map.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:36.714482Z","iopub.execute_input":"2023-09-18T10:10:36.714853Z","iopub.status.idle":"2023-09-18T10:10:36.732618Z","shell.execute_reply.started":"2023-09-18T10:10:36.714821Z","shell.execute_reply":"2023-09-18T10:10:36.731066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:38.594641Z","iopub.execute_input":"2023-09-18T10:10:38.595117Z","iopub.status.idle":"2023-09-18T10:10:38.608760Z","shell.execute_reply.started":"2023-09-18T10:10:38.595078Z","shell.execute_reply":"2023-09-18T10:10:38.607017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:10:39.035207Z","iopub.execute_input":"2023-09-18T10:10:39.035624Z","iopub.status.idle":"2023-09-18T10:11:20.505452Z","shell.execute_reply.started":"2023-09-18T10:10:39.035584Z","shell.execute_reply":"2023-09-18T10:11:20.504193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.hist(train_data['A1BG'], bins=50, color='blue', alpha=0.7)\nplt.title(\"Distribution of Gene Expression (A1BG)\")\nplt.xlabel(\"Gene Expression Value\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:11:20.507521Z","iopub.execute_input":"2023-09-18T10:11:20.508774Z","iopub.status.idle":"2023-09-18T10:11:20.927402Z","shell.execute_reply.started":"2023-09-18T10:11:20.508733Z","shell.execute_reply":"2023-09-18T10:11:20.926199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training Data Shape:\", train_data.shape)\nprint(\"Id Map Data Shape:\", id_map.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:11:20.928926Z","iopub.execute_input":"2023-09-18T10:11:20.929975Z","iopub.status.idle":"2023-09-18T10:11:20.935784Z","shell.execute_reply.started":"2023-09-18T10:11:20.929941Z","shell.execute_reply":"2023-09-18T10:11:20.934571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nimport time\nt0start = time.time() \n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:11:20.938486Z","iopub.execute_input":"2023-09-18T10:11:20.939869Z","iopub.status.idle":"2023-09-18T10:11:20.957079Z","shell.execute_reply.started":"2023-09-18T10:11:20.939819Z","shell.execute_reply":"2023-09-18T10:11:20.955411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nprint(df_de_train.shape)\ndf_de_train","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:11:20.959123Z","iopub.execute_input":"2023-09-18T10:11:20.959802Z","iopub.status.idle":"2023-09-18T10:11:22.561621Z","shell.execute_reply.started":"2023-09-18T10:11:20.959750Z","shell.execute_reply":"2023-09-18T10:11:22.560215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn)\nprint(df_id_map.shape)\ndisplay(df_id_map)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf = pd.read_csv(fn, index_col = 0)\nprint(df.shape)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:24.275153Z","iopub.execute_input":"2023-09-18T10:15:24.275862Z","iopub.status.idle":"2023-09-18T10:15:29.356345Z","shell.execute_reply.started":"2023-09-18T10:15:24.275826Z","shell.execute_reply":"2023-09-18T10:15:29.355102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom sklearn.decomposition import PCA\nfrom sklearn.decomposition import FastICA\nfrom sklearn.decomposition import TruncatedSVD\n\nn_components = 35\n\npredict_method = 'train_aggregation_by_compounds_with_denoising_TSVD'\n\nif '_pca' in predict_method:\n    str_inf_target_dimred = 'PCA' \n    reducer = PCA(n_components=n_components )\nelif '_ICA' in predict_method:\n    str_inf_target_dimred = 'ICA' \n    reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\nelif '_TSVD' in predict_method:\n    str_inf_target_dimred = 'TSVD' \n    reducer = TruncatedSVD(n_components=n_components, n_iter=7, random_state=42)\nelse:\n    str_inf_target_dimred = ''\n    \nprint(str_inf_target_dimred, reducer)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:33.544554Z","iopub.execute_input":"2023-09-18T10:15:33.545000Z","iopub.status.idle":"2023-09-18T10:15:33.929755Z","shell.execute_reply.started":"2023-09-18T10:15:33.544965Z","shell.execute_reply":"2023-09-18T10:15:33.928494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = df_de_train.iloc[:,5:].values\nYr = reducer.fit_transform(Y)\npd.DataFrame(Yr).corr(method = 'spearman').round(2)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:39.179576Z","iopub.execute_input":"2023-09-18T10:15:39.180054Z","iopub.status.idle":"2023-09-18T10:15:41.891484Z","shell.execute_reply.started":"2023-09-18T10:15:39.180020Z","shell.execute_reply":"2023-09-18T10:15:41.890528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nn_components_for_compound_encoding = 3\nquantile = 0.54\ndf_tmp = pd.DataFrame(Yr[:, :n_components_for_compound_encoding  ], index = df_de_train.index  )\ndf_tmp['column for aggregation'] = df_de_train['sm_name']\ndf_compound_encoded = df_tmp.groupby('column for aggregation').quantile( quantile )\nprint('df_compound_encoded.shape', df_compound_encoded.shape )\ndisplay( df_compound_encoded )\n\nX = pd.DataFrame(index = df_de_train['sm_name']) # \nX['IX'] = np.arange(len(df_de_train))\nX = X.join( df_compound_encoded , how = 'left').sort_values('IX')\nX = X.iloc[:,1:]\nX = X.values \nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:46.154665Z","iopub.execute_input":"2023-09-18T10:15:46.155078Z","iopub.status.idle":"2023-09-18T10:15:46.205520Z","shell.execute_reply.started":"2023-09-18T10:15:46.155048Z","shell.execute_reply":"2023-09-18T10:15:46.204253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nenc = OrdinalEncoder()\n\nX = enc.fit_transform(df_de_train[ ['cell_type', 'sm_name']])\nX.shape\n\nfrom sklearn.preprocessing import OneHotEncoder\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:49.574859Z","iopub.execute_input":"2023-09-18T10:15:49.575373Z","iopub.status.idle":"2023-09-18T10:15:49.590827Z","shell.execute_reply.started":"2023-09-18T10:15:49.575333Z","shell.execute_reply":"2023-09-18T10:15:49.589401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nn_components_for_compound_encoding = 20\nquantile = 0.54\ndf_tmp = pd.DataFrame(Yr[:, :n_components_for_compound_encoding  ], index = df_de_train.index  )\ndf_tmp['column for aggregation'] = df_de_train['sm_name']\ndf_compound_encoded = df_tmp.groupby('column for aggregation').mean() # quantile( quantile )\nprint('df_compound_encoded.shape', df_compound_encoded.shape )\n\nX = pd.DataFrame(index = df_de_train['sm_name']) # \nX['IX'] = np.arange(len(df_de_train))\nX = X.join( df_compound_encoded , how = 'left').sort_values('IX')\nX = X.iloc[:,1:]\nX = X.values \nprint( X.shape )\n\nX_submit = pd.DataFrame(index = df_id_map['sm_name']) # \nX_submit['IX'] = np.arange(len(X_submit))\nX_submit = X_submit.join( df_compound_encoded , how = 'left').sort_values('IX')\nX_submit = X_submit.iloc[:,1:]\nX_submit = X_submit.values \nprint( X_submit.shape )\n","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:51.239404Z","iopub.execute_input":"2023-09-18T10:15:51.239895Z","iopub.status.idle":"2023-09-18T10:15:51.271231Z","shell.execute_reply.started":"2023-09-18T10:15:51.239860Z","shell.execute_reply":"2023-09-18T10:15:51.269693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import KFold\n\nfrom sklearn.metrics import r2_score\n\nn_splits = 3\nkf = KFold(n_splits=n_splits, random_state = 42, shuffle = True )\nIX = pd.DataFrame(); IX['IX'] = range(len(df_de_train))\n\nalpha_regularization_for_linear_models = 100000\nmodel = Ridge(alpha=alpha_regularization_for_linear_models)\n\nY = df_de_train.iloc[:,5:].values\n\nY_reduced_submit = np.zeros( (len(df_id_map) , Yr.shape[1] )   ); cnt_blend_submit = 0\nY_submit = np.zeros( (len(df_id_map) , 18211 )   ); cnt_blend_submit = 0\nY_oof = np.zeros( (len(df_de_train) , 18211 )   ); cnt_blend_submit = 0","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:52.675046Z","iopub.execute_input":"2023-09-18T10:15:52.675539Z","iopub.status.idle":"2023-09-18T10:15:52.743021Z","shell.execute_reply.started":"2023-09-18T10:15:52.675502Z","shell.execute_reply":"2023-09-18T10:15:52.741360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape)\nfor i, (train_index, test_index) in enumerate(kf.split(IX)):\n    print(f\"Fold {i}:\", len(test_index) )\n    \n    Yr_train = reducer.fit_transform(Y[train_index,:])\n    Yr_test = reducer.transform(Y[test_index,:])\n    X_train = X[train_index,:]\n    X_test = X[test_index,:]\n    model.fit(X_train, Yr_train)\n    Yr_pred = model.predict(X_test) # , Yr_train)\n    r2 = r2_score(Yr_test,  Yr_pred )\n    print('r2 test:', r2)\n    r2_test = r2\n    Y_oof[test_index,: ] = reducer.inverse_transform(Yr_pred)\n    \n    Yr_pred = model.predict(X_train) # , Yr_train)\n    r2 = r2_score(Yr_train,  Yr_pred )\n    print('r2 train', r2)\n    Y_reduced_submit =  (Y_reduced_submit * cnt_blend_submit + model.predict(X_submit) ) / ( cnt_blend_submit + 1)\n    Y_submit =  (Y_submit * cnt_blend_submit + reducer.inverse_transform(model.predict(X_submit) ) ) / ( cnt_blend_submit + 1)\n    cnt_blend_submit += 1\n    \n    \nfrom sklearn.metrics import mean_squared_error\ns = mean_squared_error(Y,Y_oof, squared = False)\nprint(s)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:15:54.235121Z","iopub.execute_input":"2023-09-18T10:15:54.235549Z","iopub.status.idle":"2023-09-18T10:16:01.840277Z","shell.execute_reply.started":"2023-09-18T10:15:54.235517Z","shell.execute_reply":"2023-09-18T10:16:01.838674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import KFold\n\nfrom sklearn.metrics import r2_score\n\nn_splits = 3\nkf = KFold(n_splits=n_splits, random_state = 42, shuffle = True )\nIX = pd.DataFrame(); IX['IX'] = range(len(df_de_train))\n\nalpha_regularization_for_linear_models = 100000\nmodel = Ridge(alpha=alpha_regularization_for_linear_models)\n\n\nimport catboost\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.multioutput import MultiOutputRegressor\ncategorical_features = ['cell_type','sm_name']\nmodel = MultiOutputRegressor( CatBoostRegressor(cat_features=categorical_features, verbose = 0,  # Categorical features ) ) \n                        iterations=300,  # Number of boosting iterations\n                          depth=5,        # Depth of the trees\n                          learning_rate=0.015,  # Learning rate\n                          loss_function='RMSE'))  # Specify your loss function (e.g., RMSE for regression)\n#                           verbose=0) ) # Set verbose to 0 to suppress output\n\n\nY = df_de_train.iloc[:,5:].values","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:16:01.842465Z","iopub.execute_input":"2023-09-18T10:16:01.843704Z","iopub.status.idle":"2023-09-18T10:16:02.332183Z","shell.execute_reply.started":"2023-09-18T10:16:01.843645Z","shell.execute_reply":"2023-09-18T10:16:02.330704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_reduced_submit = np.zeros( (len(df_id_map) , Yr.shape[1] )   ); cnt_blend_submit = 0\nY_submit = np.zeros( (len(df_id_map) , 18211 )   ); cnt_blend_submit = 0\nY_oof = np.zeros( (len(df_de_train) , 18211 )   ); cnt_blend_submit = 0\n\ndf_train = df_de_train[['cell_type','sm_name']]\n\nX_submit = df_id_map[ ['cell_type','sm_name'] ]","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:16:07.324355Z","iopub.execute_input":"2023-09-18T10:16:07.324829Z","iopub.status.idle":"2023-09-18T10:16:07.343271Z","shell.execute_reply.started":"2023-09-18T10:16:07.324793Z","shell.execute_reply":"2023-09-18T10:16:07.341679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape)\nfor i, (train_index, test_index) in enumerate(kf.split(IX)):\n    print(f\"Fold {i}:\", len(test_index) )\n    \n    Yr_train = reducer.fit_transform(Y[train_index,:])\n    Yr_test = reducer.transform(Y[test_index,:])\n    X_train = df_de_train[categorical_features].iloc[train_index,:]\n    X_test = df_de_train[categorical_features].iloc[test_index,:]\n    \n    model.fit(X_train, Yr_train)\n    Yr_pred = model.predict(X_test) # , Yr_train)\n    r2 = r2_score(Yr_test,  Yr_pred )\n    print('r2 test:', r2)\n    r2_test = r2\n    Y_oof[test_index,: ] = reducer.inverse_transform(Yr_pred)\n    \n    Yr_pred = model.predict(X_train) # , Yr_train)\n    r2 = r2_score(Yr_train,  Yr_pred )\n    print('r2 train', r2)\n    Y_reduced_submit =  (Y_reduced_submit * cnt_blend_submit + model.predict(X_submit) ) / ( cnt_blend_submit + 1)\n    Y_submit =  (Y_submit * cnt_blend_submit + reducer.inverse_transform(model.predict(X_submit) ) ) / ( cnt_blend_submit + 1)\n    cnt_blend_submit += 1\n    \n    \nfrom sklearn.metrics import mean_squared_error\ns = mean_squared_error(Y,Y_oof, squared = False)\nprint(s)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:16:09.020940Z","iopub.execute_input":"2023-09-18T10:16:09.021869Z","iopub.status.idle":"2023-09-18T10:16:42.698849Z","shell.execute_reply.started":"2023-09-18T10:16:09.021817Z","shell.execute_reply":"2023-09-18T10:16:42.697415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submit = pd.DataFrame(Y_submit, columns = df_de_train.columns[5:])\ndf_submit.index.name = 'id'\nprint( df_submit.shape )\ndisplay(df_submit)\ndf_submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T10:16:42.701511Z","iopub.execute_input":"2023-09-18T10:16:42.702515Z","iopub.status.idle":"2023-09-18T10:16:55.777892Z","shell.execute_reply.started":"2023-09-18T10:16:42.702468Z","shell.execute_reply":"2023-09-18T10:16:55.776482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}