{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-30T00:29:44.457359Z","iopub.execute_input":"2023-11-30T00:29:44.457791Z","iopub.status.idle":"2023-11-30T00:29:44.798783Z","shell.execute_reply.started":"2023-11-30T00:29:44.457759Z","shell.execute_reply":"2023-11-30T00:29:44.796995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.model_selection import KFold, RandomizedSearchCV\nfrom sklearn.pipeline import make_pipeline\nfrom xgboost import XGBRegressor\nimport category_encoders as ce\nfrom sklearn.metrics import r2_score\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:29:49.548339Z","iopub.execute_input":"2023-11-30T00:29:49.548921Z","iopub.status.idle":"2023-11-30T00:29:50.860252Z","shell.execute_reply.started":"2023-11-30T00:29:49.548880Z","shell.execute_reply":"2023-11-30T00:29:50.858576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:29:59.154188Z","iopub.execute_input":"2023-11-30T00:29:59.154672Z","iopub.status.idle":"2023-11-30T00:30:01.204240Z","shell.execute_reply.started":"2023-11-30T00:29:59.154635Z","shell.execute_reply":"2023-11-30T00:30:01.203306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nid_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:05.048143Z","iopub.execute_input":"2023-11-30T00:30:05.048558Z","iopub.status.idle":"2023-11-30T00:30:05.072172Z","shell.execute_reply.started":"2023-11-30T00:30:05.048526Z","shell.execute_reply":"2023-11-30T00:30:05.070311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:09.257083Z","iopub.execute_input":"2023-11-30T00:30:09.257459Z","iopub.status.idle":"2023-11-30T00:30:09.280112Z","shell.execute_reply.started":"2023-11-30T00:30:09.257430Z","shell.execute_reply":"2023-11-30T00:30:09.279331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.sum().isnull()/data.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:12.025226Z","iopub.execute_input":"2023-11-30T00:30:12.025667Z","iopub.status.idle":"2023-11-30T00:30:12.069572Z","shell.execute_reply.started":"2023-11-30T00:30:12.025632Z","shell.execute_reply":"2023-11-30T00:30:12.067967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:15.558357Z","iopub.execute_input":"2023-11-30T00:30:15.558869Z","iopub.status.idle":"2023-11-30T00:30:43.879494Z","shell.execute_reply.started":"2023-11-30T00:30:15.558830Z","shell.execute_reply":"2023-11-30T00:30:43.878127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:48.118498Z","iopub.execute_input":"2023-11-30T00:30:48.118860Z","iopub.status.idle":"2023-11-30T00:30:48.128223Z","shell.execute_reply.started":"2023-11-30T00:30:48.118833Z","shell.execute_reply":"2023-11-30T00:30:48.126054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:50.652937Z","iopub.execute_input":"2023-11-30T00:30:50.653270Z","iopub.status.idle":"2023-11-30T00:30:50.661127Z","shell.execute_reply.started":"2023-11-30T00:30:50.653248Z","shell.execute_reply":"2023-11-30T00:30:50.659388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_diff = list(set(data['sm_name'])- set(id_map['sm_name']))\ncolumn_diff","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:54.157338Z","iopub.execute_input":"2023-11-30T00:30:54.157740Z","iopub.status.idle":"2023-11-30T00:30:54.165107Z","shell.execute_reply.started":"2023-11-30T00:30:54.157710Z","shell.execute_reply":"2023-11-30T00:30:54.164070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_random_numbers(num_feature_samples,low,high):\n    np.random.seed(0)\n    return {i: np.random.uniform(low, high) for i in range(num_feature_samples)}","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:30:58.327053Z","iopub.execute_input":"2023-11-30T00:30:58.327423Z","iopub.status.idle":"2023-11-30T00:30:58.334207Z","shell.execute_reply.started":"2023-11-30T00:30:58.327395Z","shell.execute_reply":"2023-11-30T00:30:58.332375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_feature_samples = 100\nX_train = data[['cell_type', 'sm_name']]\nX_id_map = id_map[['cell_type', 'sm_name']]\ntsd = TruncatedSVD(num_feature_samples, n_iter=7, random_state=0)\nY = data.iloc[:, 5:].values\nY_tsd = tsd.fit_transform(Y)\n\nkf = KFold(n_splits=8, shuffle=True, random_state=0)\n\nY_pred_tsd = np.zeros(Y_tsd.shape)\nY_id_map_tsd = np.zeros((len(id_map), Y_tsd.shape[1]))\nparams = {'subsample': 0.1,\n          'scale_pos_weight': 7,\n          'reg_lambda': 2,\n          'reg_alpha': 0.8,\n          'n_estimators': 270,\n          'min_split_loss': 100,\n          'min_child_weight': 2,\n          'learning_rate': 0.002,\n          'gamma': 0.8}\nsmooth_value = generate_random_numbers(num_feature_samples,0.0,100000.0)\nxgb = XGBRegressor(**params,n_jobs=-1,random_state=0)\n\n# Define a target encoder pipeline for each target label\nencoder_pipelines = []\n\nfor i in range(Y_tsd.shape[1]):\n    encoder_pipeline = make_pipeline(\n        ce.TargetEncoder(cols=['cell_type', 'sm_name'], smoothing=smooth_value[i]),  \n    )\n    encoder_pipelines.append(encoder_pipeline)\n\n# Perform k-fold cross-validation\nfor train_idx, val_idx in kf.split(X_train):\n    X_train_fold, X_val_fold = X_train.iloc[train_idx], X_train.iloc[val_idx]\n    Y_train_fold, Y_val_fold = Y_tsd[train_idx], Y_tsd[val_idx]\n\n    # Fit and transform on the training data for each target label\n    X_train_encoded = pd.DataFrame(index=X_train_fold.index)\n\n    for i, encoder_pipeline in enumerate(encoder_pipelines):\n        target_col = f'target_{i}'  # Create a unique name for each target column\n\n        # Fit and transform for the cell_type and sm_name\n        target_encoded = encoder_pipeline.fit_transform(X_train_fold, Y_train_fold[:, i])\n        X_train_encoded[target_col] = target_encoded[X_train_fold.columns[0]]\n\n    # Train the model\n    xgb.fit(X_train_encoded, Y_train_fold)\n\n    # Predict on the validation set\n    X_val_encoded = pd.DataFrame(index=X_val_fold.index)\n\n    for i, encoder_pipeline in enumerate(encoder_pipelines):\n        target_col = f'target_{i}'\n        \n        # Transform for the validation set\n        target_encoded = encoder_pipeline.transform(X_val_fold)\n        X_val_encoded[target_col] = target_encoded[X_val_fold.columns[0]]\n\n    y_pred = xgb.predict(X_val_encoded)\n\n    Y_pred_tsd[val_idx] = y_pred\n\n    # Predict on the selected samples (Y_id_map_tsd)\n    X_id_map_encoded = pd.DataFrame(index=X_id_map.index)\n\n    for i, encoder_pipeline in enumerate(encoder_pipelines):\n        target_col = f'target_{i}'\n\n        # Transform for the id_map set\n        target_encoded = encoder_pipeline.transform(X_id_map)\n        X_id_map_encoded[target_col] = target_encoded[X_id_map.columns[0]]\n\n    y_id_map_pred = xgb.predict(X_id_map_encoded.values)\n\n    Y_id_map_tsd += y_id_map_pred\n\n# Average the predictions for the selected samples\nY_id_map_tsd /= kf.get_n_splits()\n\n# Inverse transform the predictions\nY_pred_inverse_tsvd = tsd.inverse_transform(Y_pred_tsd)\nY_submit = tsd.inverse_transform(Y_id_map_tsd)\n\n# Evaluate the model\nmrrmse = np.sqrt(np.square(Y - Y_pred_inverse_tsvd).mean(axis=1)).mean()\nr2 = r2_score(Y,Y_pred_inverse_tsvd)\nprint(\"MRRMSE:\", mrrmse)\nprint(\"R^2:\", r2)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:31:01.919950Z","iopub.execute_input":"2023-11-30T00:31:01.920621Z","iopub.status.idle":"2023-11-30T00:32:37.028877Z","shell.execute_reply.started":"2023-11-30T00:31:01.920587Z","shell.execute_reply":"2023-11-30T00:32:37.027898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(Y_submit,columns=data.columns[5:])\ndf_submit = pd.concat([id_map['id'],df],axis=1)\ndf_submit.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T00:33:07.096107Z","iopub.execute_input":"2023-11-30T00:33:07.096838Z","iopub.status.idle":"2023-11-30T00:33:13.313988Z","shell.execute_reply.started":"2023-11-30T00:33:07.096805Z","shell.execute_reply":"2023-11-30T00:33:13.312588Z"},"trusted":true},"execution_count":null,"outputs":[]}]}