{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-24T09:10:43.992618Z","iopub.execute_input":"2023-11-24T09:10:43.993231Z","iopub.status.idle":"2023-11-24T09:10:44.443642Z","shell.execute_reply.started":"2023-11-24T09:10:43.993199Z","shell.execute_reply":"2023-11-24T09:10:44.442790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport pandas as pd\nimport numpy as np\nimport time\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport networkx as nx\n\n\n\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import Ridge\nfrom sklearn.manifold import TSNE\nfrom sklearn.decomposition import PCA, FastICA, TruncatedSVD\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import r2_score, mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder, OrdinalEncoder, OneHotEncoder\n\n\nimport umap\nimport shap\nimport scipy.cluster.hierarchy as sch\nimport scipy.stats as stats\nfrom scipy.spatial.distance import squareform\n\n\nimport catboost\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.multioutput import MultiOutputRegressor","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:10:47.310914Z","iopub.execute_input":"2023-11-24T09:10:47.311502Z","iopub.status.idle":"2023-11-24T09:11:27.261074Z","shell.execute_reply.started":"2023-11-24T09:10:47.311467Z","shell.execute_reply":"2023-11-24T09:11:27.259786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE = '/kaggle/input/open-problems-single-cell-perturbations/'","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:11:27.263291Z","iopub.execute_input":"2023-11-24T09:11:27.264119Z","iopub.status.idle":"2023-11-24T09:11:27.269414Z","shell.execute_reply.started":"2023-11-24T09:11:27.264083Z","shell.execute_reply":"2023-11-24T09:11:27.268210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_parquet(f'{BASE}de_train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:11:30.236825Z","iopub.execute_input":"2023-11-24T09:11:30.237314Z","iopub.status.idle":"2023-11-24T09:11:31.575565Z","shell.execute_reply.started":"2023-11-24T09:11:30.237271Z","shell.execute_reply":"2023-11-24T09:11:31.574707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:11:35.168447Z","iopub.execute_input":"2023-11-24T09:11:35.169210Z","iopub.status.idle":"2023-11-24T09:11:36.388196Z","shell.execute_reply.started":"2023-11-24T09:11:35.169170Z","shell.execute_reply":"2023-11-24T09:11:36.387072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install matplotlib","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:15:42.455644Z","iopub.execute_input":"2023-11-24T09:15:42.456113Z","iopub.status.idle":"2023-11-24T09:16:17.640449Z","shell.execute_reply.started":"2023-11-24T09:15:42.456079Z","shell.execute_reply":"2023-11-24T09:16:17.639054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:16:39.271848Z","iopub.execute_input":"2023-11-24T09:16:39.272296Z","iopub.status.idle":"2023-11-24T09:17:13.367130Z","shell.execute_reply.started":"2023-11-24T09:16:39.272262Z","shell.execute_reply":"2023-11-24T09:17:13.365909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(train_data.sample(613)[['A1BG', 'A1BG-AS1', 'A2M', 'cell_type']], hue='cell_type', plot_kws={'alpha':0.5})\nplt.suptitle('Pairwise Relationships among Selected Genes Across Cell Types')\nplt.figure(figsize=(16,7))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:17:24.229368Z","iopub.execute_input":"2023-11-24T09:17:24.230185Z","iopub.status.idle":"2023-11-24T09:17:29.247804Z","shell.execute_reply.started":"2023-11-24T09:17:24.230129Z","shell.execute_reply":"2023-11-24T09:17:29.246702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.countplot(x='sm_name', data=train_data)\n# plt.title('Distribution of Small Molecules')\n# plt.xticks(rotation=90)\n# plt.figure(figsize=(16,7))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:19:18.230866Z","iopub.execute_input":"2023-11-24T09:19:18.231302Z","iopub.status.idle":"2023-11-24T09:19:18.236896Z","shell.execute_reply.started":"2023-11-24T09:19:18.231271Z","shell.execute_reply":"2023-11-24T09:19:18.235564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adjust the sample size to the number of rows in your DataFrame if needed\nsample_size = min(1000, len(train_data))  # Ensuring sample size is not greater than the dataset size\n\n# Perform t-SNE\ntsne = TSNE(n_components=2, perplexity=30, random_state=42)\ntsne_results = tsne.fit_transform(train_data.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], axis=1).sample(sample_size))\n\n# Visualization\ntrain_data_subset = train_data.sample(sample_size)  # Ensure you're sampling the same rows for coloring by cell_type\n\nplt.figure(figsize=(16,10))\nsns.scatterplot(\n    x=tsne_results[:,0], y=tsne_results[:,1],\n    hue=train_data_subset['cell_type'],\n    palette=sns.color_palette(\"hsv\", len(train_data_subset['cell_type'].unique())),\n    legend=\"full\",\n    alpha=0.7\n)\nplt.title('t-SNE Visualization of Gene Expressions')\nplt.figure(figsize=(16,7))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:19:32.941219Z","iopub.execute_input":"2023-11-24T09:19:32.941613Z","iopub.status.idle":"2023-11-24T09:19:38.816114Z","shell.execute_reply.started":"2023-11-24T09:19:32.941584Z","shell.execute_reply":"2023-11-24T09:19:38.814984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# UMAP for dimensionality reduction\nreducer = umap.UMAP(random_state=42)\nembedding = reducer.fit_transform(train_data.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], axis=1))\n\n# Using DBSCAN for clustering UMAP output\nclustering = DBSCAN(eps=0.5, min_samples=10).fit(embedding)\ntrain_data['umap_cluster'] = clustering.labels_\n\n# Visualizing UMAP output with clusters\nplt.figure(figsize=(12, 10))\nsns.scatterplot(\n    x=embedding[:, 0], \n    y=embedding[:, 1], \n    hue=train_data['umap_cluster'], \n    palette=sns.color_palette(\"hsv\", len(set(clustering.labels_))),\n    legend='full'\n)\nplt.title('UMAP projection of the Dataset', fontsize=20)\nplt.xlabel('UMAP Dimension 1')\nplt.ylabel('UMAP Dimension 2')\nplt.figure(figsize=(16,7))\nplt.show()\n\n\n# Prepare data for SHAP analysis\nX = train_data.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control', 'umap_cluster'], axis=1)\ny = train_data['umap_cluster']\n\n# Train a model for SHAP analysis\nmodel = RandomForestClassifier(random_state=0, n_estimators=100)\nmodel.fit(X, y)\n\n# Explain the model's predictions using SHAP\nexplainer = shap.TreeExplainer(model)\nshap_values = explainer.shap_values(X.sample(100))  # Using a sample for speed\n\n# Visualize the first prediction's explanation\nshap.initjs()\nshap.force_plot(explainer.expected_value[0], shap_values[0][0], X.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:19:52.127796Z","iopub.execute_input":"2023-11-24T09:19:52.128211Z","iopub.status.idle":"2023-11-24T09:20:14.248951Z","shell.execute_reply.started":"2023-11-24T09:19:52.128169Z","shell.execute_reply":"2023-11-24T09:20:14.247702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for gene in ['A1BG', 'A1BG-AS1', 'A2M']:  # Replace with genes of interest\n    result = stats.kruskal(*[group[gene].values for name, group in train_data.groupby('cell_type')])\n    print(f\"Kruskal-Wallis test for {gene}: H-statistic={result.statistic}, p-value={result.pvalue}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:15.331474Z","iopub.execute_input":"2023-11-24T09:20:15.331938Z","iopub.status.idle":"2023-11-24T09:20:15.482248Z","shell.execute_reply.started":"2023-11-24T09:20:15.331899Z","shell.execute_reply":"2023-11-24T09:20:15.481127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\ntrain_data['cell_type_encoded'] = le.fit_transform(train_data['cell_type'])\n\n# Define features and target\nX = train_data.drop(['cell_type', 'cell_type_encoded', 'sm_name', 'sm_lincs_id', 'SMILES', 'control'], axis=1)\ny = train_data['cell_type_encoded']\n\n# Split the data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Fit the RandomForest model on the training data\nrf = RandomForestClassifier(n_estimators=100, random_state=42)\nrf.fit(X_train, y_train)\n\n# Extract feature importances from the model\nimportances = rf.feature_importances_\n\n# Plot the top N feature importances\ntop_n = 20  # Specify how many top features you'd like to visualize\nindices = np.argsort(importances)[-top_n:]\nplt.figure(figsize=(15, 7))\nplt.title('Top Feature Importances')\nplt.barh(range(len(indices)), importances[indices], color='b', align='center')\nplt.yticks(range(len(indices)), [X_train.columns[i] for i in indices])\nplt.xlabel('Relative Importance')\nplt.figure(figsize=(16,7))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:22.204365Z","iopub.execute_input":"2023-11-24T09:20:22.204791Z","iopub.status.idle":"2023-11-24T09:20:25.921355Z","shell.execute_reply.started":"2023-11-24T09:20:22.204754Z","shell.execute_reply":"2023-11-24T09:20:25.919993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the shape of the training and ID map datasets\nid_map = pd.read_csv(f'{BASE}id_map.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:32.639980Z","iopub.execute_input":"2023-11-24T09:20:32.640442Z","iopub.status.idle":"2023-11-24T09:20:32.655371Z","shell.execute_reply.started":"2023-11-24T09:20:32.640404Z","shell.execute_reply":"2023-11-24T09:20:32.654198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:35.546185Z","iopub.execute_input":"2023-11-24T09:20:35.546566Z","iopub.status.idle":"2023-11-24T09:20:35.561970Z","shell.execute_reply.started":"2023-11-24T09:20:35.546536Z","shell.execute_reply":"2023-11-24T09:20:35.560707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:38.663095Z","iopub.execute_input":"2023-11-24T09:20:38.663512Z","iopub.status.idle":"2023-11-24T09:20:38.694106Z","shell.execute_reply.started":"2023-11-24T09:20:38.663480Z","shell.execute_reply":"2023-11-24T09:20:38.692936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.describe())","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:20:41.492578Z","iopub.execute_input":"2023-11-24T09:20:41.493359Z","iopub.status.idle":"2023-11-24T09:21:22.691542Z","shell.execute_reply.started":"2023-11-24T09:20:41.493323Z","shell.execute_reply":"2023-11-24T09:21:22.690349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.hist(train_data['A1BG'], bins=50, color='blue', alpha=0.7)\nplt.title(\"Distribution of Gene Expression (A1BG)\")\nplt.xlabel(\"Gene Expression Value\")\nplt.ylabel(\"Frequency\")\nplt.figure(figsize=(16,7))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:21:41.612679Z","iopub.execute_input":"2023-11-24T09:21:41.613103Z","iopub.status.idle":"2023-11-24T09:21:41.993429Z","shell.execute_reply.started":"2023-11-24T09:21:41.613070Z","shell.execute_reply":"2023-11-24T09:21:41.992292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training Data Shape:\", train_data.shape)\nprint(\"Id Map Data Shape:\", id_map.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:21:46.271976Z","iopub.execute_input":"2023-11-24T09:21:46.272415Z","iopub.status.idle":"2023-11-24T09:21:46.278823Z","shell.execute_reply.started":"2023-11-24T09:21:46.272380Z","shell.execute_reply":"2023-11-24T09:21:46.277634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0start = time.time()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:21:48.885925Z","iopub.execute_input":"2023-11-24T09:21:48.886343Z","iopub.status.idle":"2023-11-24T09:21:48.891914Z","shell.execute_reply.started":"2023-11-24T09:21:48.886301Z","shell.execute_reply":"2023-11-24T09:21:48.890604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nprint(df_de_train.shape)\ndf_de_train","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:21:54.092037Z","iopub.execute_input":"2023-11-24T09:21:54.092494Z","iopub.status.idle":"2023-11-24T09:21:55.454825Z","shell.execute_reply.started":"2023-11-24T09:21:54.092454Z","shell.execute_reply":"2023-11-24T09:21:55.453385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn)\nprint(df_id_map.shape)\ndisplay(df_id_map)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf = pd.read_csv(fn, index_col = 0)\nprint(df.shape)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:21:59.900886Z","iopub.execute_input":"2023-11-24T09:21:59.901267Z","iopub.status.idle":"2023-11-24T09:22:04.801515Z","shell.execute_reply.started":"2023-11-24T09:21:59.901231Z","shell.execute_reply":"2023-11-24T09:22:04.800406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nn_components = 35\n\npredict_method = 'train_aggregation_by_compounds_with_denoising_TSVD'\n\nif '_pca' in predict_method:\n    str_inf_target_dimred = 'PCA' \n    reducer = PCA(n_components=n_components )\nelif '_ICA' in predict_method:\n    str_inf_target_dimred = 'ICA' \n    reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\nelif '_TSVD' in predict_method:\n    str_inf_target_dimred = 'TSVD' \n    reducer = TruncatedSVD(n_components=n_components, n_iter=7, random_state=42)\nelse:\n    str_inf_target_dimred = ''\n    \nprint(str_inf_target_dimred, reducer)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:04.803261Z","iopub.execute_input":"2023-11-24T09:22:04.803639Z","iopub.status.idle":"2023-11-24T09:22:04.816796Z","shell.execute_reply.started":"2023-11-24T09:22:04.803602Z","shell.execute_reply":"2023-11-24T09:22:04.815412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = df_de_train.iloc[:,5:].values\nYr = reducer.fit_transform(Y)\npd.DataFrame(Yr).corr(method = 'spearman').round(2)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:07.543454Z","iopub.execute_input":"2023-11-24T09:22:07.544031Z","iopub.status.idle":"2023-11-24T09:22:10.099652Z","shell.execute_reply.started":"2023-11-24T09:22:07.543987Z","shell.execute_reply":"2023-11-24T09:22:10.098434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nn_components_for_compound_encoding = 3\nquantile = 0.54\ndf_tmp = pd.DataFrame(Yr[:, :n_components_for_compound_encoding  ], index = df_de_train.index  )\ndf_tmp['column for aggregation'] = df_de_train['sm_name']\ndf_compound_encoded = df_tmp.groupby('column for aggregation').quantile( quantile )\nprint('df_compound_encoded.shape', df_compound_encoded.shape )\ndisplay( df_compound_encoded )\n\nX = pd.DataFrame(index = df_de_train['sm_name']) # \nX['IX'] = np.arange(len(df_de_train))\nX = X.join( df_compound_encoded , how = 'left').sort_values('IX')\nX = X.iloc[:,1:]\nX = X.values \nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:14.242328Z","iopub.execute_input":"2023-11-24T09:22:14.242866Z","iopub.status.idle":"2023-11-24T09:22:14.287365Z","shell.execute_reply.started":"2023-11-24T09:22:14.242823Z","shell.execute_reply":"2023-11-24T09:22:14.286568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = OrdinalEncoder()\nX = enc.fit_transform(df_de_train[ ['cell_type', 'sm_name']])\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:17.663482Z","iopub.execute_input":"2023-11-24T09:22:17.663896Z","iopub.status.idle":"2023-11-24T09:22:17.677033Z","shell.execute_reply.started":"2023-11-24T09:22:17.663863Z","shell.execute_reply":"2023-11-24T09:22:17.675720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nn_components_for_compound_encoding = 20\nquantile = 0.54\ndf_tmp = pd.DataFrame(Yr[:, :n_components_for_compound_encoding  ], index = df_de_train.index  )\ndf_tmp['column for aggregation'] = df_de_train['sm_name']\ndf_compound_encoded = df_tmp.groupby('column for aggregation').mean() # quantile( quantile )\nprint('df_compound_encoded.shape', df_compound_encoded.shape )\n\nX = pd.DataFrame(index = df_de_train['sm_name']) # \nX['IX'] = np.arange(len(df_de_train))\nX = X.join( df_compound_encoded , how = 'left').sort_values('IX')\nX = X.iloc[:,1:]\nX = X.values \nprint( X.shape )\n\nX_submit = pd.DataFrame(index = df_id_map['sm_name']) # \nX_submit['IX'] = np.arange(len(X_submit))\nX_submit = X_submit.join( df_compound_encoded , how = 'left').sort_values('IX')\nX_submit = X_submit.iloc[:,1:]\nX_submit = X_submit.values \nprint( X_submit.shape )","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:21.459197Z","iopub.execute_input":"2023-11-24T09:22:21.459697Z","iopub.status.idle":"2023-11-24T09:22:21.485397Z","shell.execute_reply.started":"2023-11-24T09:22:21.459594Z","shell.execute_reply":"2023-11-24T09:22:21.484581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nn_splits = 3\nkf = KFold(n_splits=n_splits, random_state = 42, shuffle = True )\nIX = pd.DataFrame(); IX['IX'] = range(len(df_de_train))\n\nalpha_regularization_for_linear_models = 100000\nmodel = Ridge(alpha=alpha_regularization_for_linear_models)\n\nY = df_de_train.iloc[:,5:].values\n\nY_reduced_submit = np.zeros( (len(df_id_map) , Yr.shape[1] )   ); cnt_blend_submit = 0\nY_submit = np.zeros( (len(df_id_map) , 18211 )   ); cnt_blend_submit = 0\nY_oof = np.zeros( (len(df_de_train) , 18211 )   ); cnt_blend_submit = 0","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:26.706204Z","iopub.execute_input":"2023-11-24T09:22:26.706690Z","iopub.status.idle":"2023-11-24T09:22:26.750242Z","shell.execute_reply.started":"2023-11-24T09:22:26.706611Z","shell.execute_reply":"2023-11-24T09:22:26.749106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape)\nfor i, (train_index, test_index) in enumerate(kf.split(IX)):\n    print(f\"Fold {i}:\", len(test_index) )\n    \n    Yr_train = reducer.fit_transform(Y[train_index,:])\n    Yr_test = reducer.transform(Y[test_index,:])\n    X_train = X[train_index,:]\n    X_test = X[test_index,:]\n    model.fit(X_train, Yr_train)\n    Yr_pred = model.predict(X_test) # , Yr_train)\n    r2 = r2_score(Yr_test,  Yr_pred )\n    print('r2 test:', r2)\n    r2_test = r2\n    Y_oof[test_index,: ] = reducer.inverse_transform(Yr_pred)\n    \n    Yr_pred = model.predict(X_train) # , Yr_train)\n    r2 = r2_score(Yr_train,  Yr_pred )\n    print('r2 train', r2)\n    Y_reduced_submit =  (Y_reduced_submit * cnt_blend_submit + model.predict(X_submit) ) / ( cnt_blend_submit + 1)\n    Y_submit =  (Y_submit * cnt_blend_submit + reducer.inverse_transform(model.predict(X_submit) ) ) / ( cnt_blend_submit + 1)\n    cnt_blend_submit += 1\n    \ns = mean_squared_error(Y,Y_oof, squared = False)\nprint(s)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:30.247683Z","iopub.execute_input":"2023-11-24T09:22:30.248083Z","iopub.status.idle":"2023-11-24T09:22:35.661992Z","shell.execute_reply.started":"2023-11-24T09:22:30.248053Z","shell.execute_reply":"2023-11-24T09:22:35.660273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nn_splits = 3\nkf = KFold(n_splits=n_splits, random_state = 42, shuffle = True )\nIX = pd.DataFrame(); IX['IX'] = range(len(df_de_train))\n\nalpha_regularization_for_linear_models = 100000\nmodel = Ridge(alpha=alpha_regularization_for_linear_models)\n\ncategorical_features = ['cell_type','sm_name']\nmodel = MultiOutputRegressor( CatBoostRegressor(cat_features=categorical_features, verbose = 0,  # Categorical features ) ) \n                        iterations=300,  # Number of boosting iterations\n                          depth=5,        # Depth of the trees\n                          learning_rate=0.015,  # Learning rate\n                          loss_function='RMSE'))  # Specify your loss function (e.g., RMSE for regression)\n#                           verbose=0) ) # Set verbose to 0 to suppress output\n\n\nY = df_de_train.iloc[:,5:].values","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:35.663614Z","iopub.execute_input":"2023-11-24T09:22:35.663990Z","iopub.status.idle":"2023-11-24T09:22:35.708485Z","shell.execute_reply.started":"2023-11-24T09:22:35.663960Z","shell.execute_reply":"2023-11-24T09:22:35.707111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_reduced_submit = np.zeros( (len(df_id_map) , Yr.shape[1] )   ); cnt_blend_submit = 0\nY_submit = np.zeros( (len(df_id_map) , 18211 )   ); cnt_blend_submit = 0\nY_oof = np.zeros( (len(df_de_train) , 18211 )   ); cnt_blend_submit = 0\n\ndf_train = df_de_train[['cell_type','sm_name']]\n\nX_submit = df_id_map[ ['cell_type','sm_name'] ]","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:42.564889Z","iopub.execute_input":"2023-11-24T09:22:42.566277Z","iopub.status.idle":"2023-11-24T09:22:42.583005Z","shell.execute_reply.started":"2023-11-24T09:22:42.566224Z","shell.execute_reply":"2023-11-24T09:22:42.581754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape)\nfor i, (train_index, test_index) in enumerate(kf.split(IX)):\n    print(f\"Fold {i}:\", len(test_index) )\n    \n    Yr_train = reducer.fit_transform(Y[train_index,:])\n    Yr_test = reducer.transform(Y[test_index,:])\n    X_train = df_de_train[categorical_features].iloc[train_index,:]\n    X_test = df_de_train[categorical_features].iloc[test_index,:]\n    \n    model.fit(X_train, Yr_train)\n    Yr_pred = model.predict(X_test) # , Yr_train)\n    r2 = r2_score(Yr_test,  Yr_pred )\n    print('r2 test:', r2)\n    r2_test = r2\n    Y_oof[test_index,: ] = reducer.inverse_transform(Yr_pred)\n    \n    Yr_pred = model.predict(X_train) # , Yr_train)\n    r2 = r2_score(Yr_train,  Yr_pred )\n    print('r2 train', r2)\n    Y_reduced_submit =  (Y_reduced_submit * cnt_blend_submit + model.predict(X_submit) ) / ( cnt_blend_submit + 1)\n    Y_submit =  (Y_submit * cnt_blend_submit + reducer.inverse_transform(model.predict(X_submit) ) ) / ( cnt_blend_submit + 1)\n    cnt_blend_submit += 1\n    \n\ns = mean_squared_error(Y,Y_oof, squared = False)\nprint(s)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:22:46.022567Z","iopub.execute_input":"2023-11-24T09:22:46.023005Z","iopub.status.idle":"2023-11-24T09:23:15.371353Z","shell.execute_reply.started":"2023-11-24T09:22:46.022974Z","shell.execute_reply":"2023-11-24T09:23:15.369929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_submit = pd.DataFrame(Y_submit, columns = df_de_train.columns[5:])\ndf_submit.index.name = 'id'\nprint( df_submit.shape )\ndisplay(df_submit)\ndf_submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T09:23:15.373111Z","iopub.execute_input":"2023-11-24T09:23:15.373471Z","iopub.status.idle":"2023-11-24T09:23:28.410597Z","shell.execute_reply.started":"2023-11-24T09:23:15.373439Z","shell.execute_reply":"2023-11-24T09:23:28.409274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}