{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Import deps**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:45:53.164838Z","iopub.execute_input":"2023-10-06T18:45:53.165241Z","iopub.status.idle":"2023-10-06T18:45:53.170892Z","shell.execute_reply.started":"2023-10-06T18:45:53.165206Z","shell.execute_reply":"2023-10-06T18:45:53.169758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EDA**","metadata":{}},{"cell_type":"markdown","source":"👇🏽 Understanding about the files","metadata":{}},{"cell_type":"markdown","source":"de_train.parquet: Aggregated differential expression data for gene expression analysis.\n\nid_map.csv: Identifies cell_type / sm_name pairs for prediction in the competition.\n\nsample_submission.csv: A template for submitting predictions in the required format.\n\nadata_train.parquet: Additional gene expression data in COO sparse-array format.\n\nmultiome_train.parquet: Optional multi-modal data for each sample.\n\nmultiome_obs_meta.csv: Metadata for multiome observations.\n\nmultiome_var_meta.csv: Metadata for multiome variables.","metadata":{}},{"cell_type":"code","source":"# Loading the DataSets\ndf_train =   pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\ndf_id_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T14:13:25.028302Z","iopub.execute_input":"2023-10-06T14:13:25.028922Z","iopub.status.idle":"2023-10-06T14:13:30.385270Z","shell.execute_reply.started":"2023-10-06T14:13:25.028894Z","shell.execute_reply":"2023-10-06T14:13:30.384177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understanding the submssion format\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T08:41:18.490555Z","iopub.execute_input":"2023-10-06T08:41:18.490945Z","iopub.status.idle":"2023-10-06T08:41:18.529921Z","shell.execute_reply.started":"2023-10-06T08:41:18.490914Z","shell.execute_reply":"2023-10-06T08:41:18.529162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations on Problem statement:\n\n- Regression Task\n- task is to find which compounds are involved in changing the expression of specific genes.\n- predict the impact of different compounds on the expression levels of genes in our cells","metadata":{}},{"cell_type":"code","source":"# Top 5 of train set\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T08:45:32.720998Z","iopub.execute_input":"2023-10-06T08:45:32.721445Z","iopub.status.idle":"2023-10-06T08:45:32.752991Z","shell.execute_reply.started":"2023-10-06T08:45:32.721412Z","shell.execute_reply":"2023-10-06T08:45:32.751842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# id map head\ndf_id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T08:45:37.570745Z","iopub.execute_input":"2023-10-06T08:45:37.571152Z","iopub.status.idle":"2023-10-06T08:45:37.582704Z","shell.execute_reply.started":"2023-10-06T08:45:37.571124Z","shell.execute_reply":"2023-10-06T08:45:37.581275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_df(df):\n    cols = ['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control']\n    for col in cols:\n        unique_count = df[col].nunique()\n        print(f\"Column '{col}' has {unique_count} unique values.\")\n        \nplot_them = unique_df(df_train)\nplot_them","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:14:48.535558Z","iopub.execute_input":"2023-10-06T09:14:48.535929Z","iopub.status.idle":"2023-10-06T09:14:48.543994Z","shell.execute_reply.started":"2023-10-06T09:14:48.535902Z","shell.execute_reply":"2023-10-06T09:14:48.542883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Analyzing 'cell_type' category\n# df_train['cell_type'].nunique()\nprint(df_train['cell_type'].unique().tolist())","metadata":{"execution":{"iopub.status.busy":"2023-10-06T08:50:20.532928Z","iopub.execute_input":"2023-10-06T08:50:20.533378Z","iopub.status.idle":"2023-10-06T08:50:20.540032Z","shell.execute_reply.started":"2023-10-06T08:50:20.533347Z","shell.execute_reply":"2023-10-06T08:50:20.538812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_cols = ['cell_type', 'control']\n\n# value counts for each column\ncounts_cell_type = df_train['cell_type'].value_counts()\ncounts_control = df_train['control'].value_counts()\n\n# Create subplots\nfig, axes = plt.subplots(1, 2, figsize=(15, 6))\n\n# Plot the frequency distribution of 'cell_type'\nsns.barplot(x=counts_cell_type.index, y=counts_cell_type.values, palette='viridis', ax=axes[0])\naxes[0].set_xticklabels(axes[0].get_xticklabels(), rotation=90)\naxes[0].set_xlabel('Cell Type')\naxes[0].set_ylabel('Frequency')\naxes[0].set_title('Frequency Distribution of Cell Types')\n\n# Plot the frequency distribution of 'control'\nsns.barplot(x=counts_control.index, y=counts_control.values, palette='viridis', ax=axes[1])\naxes[1].set_xticklabels(axes[1].get_xticklabels(), rotation=90)\naxes[1].set_xlabel('Control')\naxes[1].set_ylabel('Frequency')\naxes[1].set_title('Frequency Distribution of Control')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:16:57.975898Z","iopub.execute_input":"2023-10-06T09:16:57.976359Z","iopub.status.idle":"2023-10-06T09:16:58.446811Z","shell.execute_reply.started":"2023-10-06T09:16:57.976328Z","shell.execute_reply":"2023-10-06T09:16:58.445721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for null values in the dataset\ndf_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:00:50.052246Z","iopub.execute_input":"2023-10-06T09:00:50.052617Z","iopub.status.idle":"2023-10-06T09:00:50.088511Z","shell.execute_reply.started":"2023-10-06T09:00:50.052591Z","shell.execute_reply":"2023-10-06T09:00:50.087449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Obs:\n- Seems like no null val's","metadata":{}},{"cell_type":"markdown","source":"**Notice**:\n- will be removing **sm_lincs_id or SMILES** col, doesnt seem to need both of them ->[**SMILES**]\n- Going to built a Prototype baseline model for a check without **SMILES** column","metadata":{}},{"cell_type":"markdown","source":"# Pre Processing","metadata":{}},{"cell_type":"code","source":"# shuffle the train dataset\ndf_train = df_train.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:23.474932Z","iopub.execute_input":"2023-10-06T18:46:23.475322Z","iopub.status.idle":"2023-10-06T18:46:23.638824Z","shell.execute_reply.started":"2023-10-06T18:46:23.475290Z","shell.execute_reply":"2023-10-06T18:46:23.637954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Main Copy of df_train\ndf = df_train.copy()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:25.624778Z","iopub.execute_input":"2023-10-06T18:46:25.625127Z","iopub.status.idle":"2023-10-06T18:46:25.727331Z","shell.execute_reply.started":"2023-10-06T18:46:25.625101Z","shell.execute_reply":"2023-10-06T18:46:25.726256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_columns = [\"cell_type\", \"sm_name\"]\nlabels_columns=[\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"] # Not much concered when creating a baseline model\nlabels = df.drop(columns=labels_columns) \nfeatures = pd.DataFrame(df, columns=features_columns) # creating a df with features col only","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:26.449502Z","iopub.execute_input":"2023-10-06T18:46:26.449874Z","iopub.status.idle":"2023-10-06T18:46:26.505003Z","shell.execute_reply.started":"2023-10-06T18:46:26.449844Z","shell.execute_reply":"2023-10-06T18:46:26.503781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels df\nlabels.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:23:35.621935Z","iopub.execute_input":"2023-10-06T09:23:35.622318Z","iopub.status.idle":"2023-10-06T09:23:35.644835Z","shell.execute_reply.started":"2023-10-06T09:23:35.622291Z","shell.execute_reply":"2023-10-06T09:23:35.644003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature df\nfeatures.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:23:42.528885Z","iopub.execute_input":"2023-10-06T09:23:42.529285Z","iopub.status.idle":"2023-10-06T09:23:42.541145Z","shell.execute_reply.started":"2023-10-06T09:23:42.529244Z","shell.execute_reply":"2023-10-06T09:23:42.539657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Just veryfying how submission looks like again xD\nsample_submission.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:24:06.650746Z","iopub.execute_input":"2023-10-06T09:24:06.651148Z","iopub.status.idle":"2023-10-06T09:24:06.678704Z","shell.execute_reply.started":"2023-10-06T09:24:06.651121Z","shell.execute_reply":"2023-10-06T09:24:06.677639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test data \ntest_data = pd.DataFrame(df_id_map, columns=features_columns)\ntest_data.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:31.884450Z","iopub.execute_input":"2023-10-06T18:46:31.884836Z","iopub.status.idle":"2023-10-06T18:46:31.896604Z","shell.execute_reply.started":"2023-10-06T18:46:31.884804Z","shell.execute_reply":"2023-10-06T18:46:31.895372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_id_map.shape) # involved id col\nprint(test_data.shape) # doesn't involve id col only [features_columns]","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:44:35.202660Z","iopub.execute_input":"2023-10-06T09:44:35.203918Z","iopub.status.idle":"2023-10-06T09:44:35.209899Z","shell.execute_reply.started":"2023-10-06T09:44:35.203871Z","shell.execute_reply":"2023-10-06T09:44:35.209024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Computers are good with numbers than us\n- One Hot encoding categorical variables on our train & test dataset [features]","metadata":{"execution":{"iopub.status.busy":"2023-10-06T09:36:13.958836Z","iopub.execute_input":"2023-10-06T09:36:13.959279Z","iopub.status.idle":"2023-10-06T09:36:13.967141Z","shell.execute_reply.started":"2023-10-06T09:36:13.959250Z","shell.execute_reply":"2023-10-06T09:36:13.965653Z"}}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# Create an instance of the encoder\nencoder = OneHotEncoder()\n\n# Fit the encoder on features\nencoder.fit(features)\n\n# Transform the features into one-hot encoded format\none_hot_train = encoder.transform(features)\n\n# Transform the test data(id_map)\none_hot_test = encoder.transform(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:35.309428Z","iopub.execute_input":"2023-10-06T18:46:35.309831Z","iopub.status.idle":"2023-10-06T18:46:35.320322Z","shell.execute_reply.started":"2023-10-06T18:46:35.309801Z","shell.execute_reply":"2023-10-06T18:46:35.319472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape\none_hot_train.shape, one_hot_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-06T10:36:00.899554Z","iopub.execute_input":"2023-10-06T10:36:00.899936Z","iopub.status.idle":"2023-10-06T10:36:00.907250Z","shell.execute_reply.started":"2023-10-06T10:36:00.899909Z","shell.execute_reply":"2023-10-06T10:36:00.906000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This train set completely will be used for a final model prediction test\n\n# Convert the CSR matrix to a dense array and then to a DataFrame\none_hot_train_df = pd.DataFrame(one_hot_train.toarray())\n# dataframe of csr matrix test data\none_hot_test_df = pd.DataFrame(one_hot_test.toarray())","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:41.324517Z","iopub.execute_input":"2023-10-06T18:46:41.324905Z","iopub.status.idle":"2023-10-06T18:46:41.331582Z","shell.execute_reply.started":"2023-10-06T18:46:41.324876Z","shell.execute_reply":"2023-10-06T18:46:41.330407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one_hot_train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T10:40:34.608203Z","iopub.execute_input":"2023-10-06T10:40:34.608591Z","iopub.status.idle":"2023-10-06T10:40:34.631706Z","shell.execute_reply.started":"2023-10-06T10:40:34.608562Z","shell.execute_reply":"2023-10-06T10:40:34.630413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one_hot_test_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T10:40:43.093363Z","iopub.execute_input":"2023-10-06T10:40:43.093976Z","iopub.status.idle":"2023-10-06T10:40:43.117817Z","shell.execute_reply.started":"2023-10-06T10:40:43.093945Z","shell.execute_reply":"2023-10-06T10:40:43.116865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train Test Split these OHE columns **","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split train set to train and temp and then split temp set into test and val\n# Split the data into 70% training, 15% validation, and 15% testing\nX_train, X_temp, y_train, y_temp = train_test_split(one_hot_train_df, labels.values, test_size=0.3, shuffle=False)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, shuffle=False) # splitting the temp as test and val","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:46:44.354239Z","iopub.execute_input":"2023-10-06T18:46:44.355021Z","iopub.status.idle":"2023-10-06T18:46:44.600069Z","shell.execute_reply.started":"2023-10-06T18:46:44.354985Z","shell.execute_reply":"2023-10-06T18:46:44.599256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the shapes of the data splits\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_val shape:\", X_val.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_val shape:\", y_val.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T13:22:43.349484Z","iopub.execute_input":"2023-10-06T13:22:43.349836Z","iopub.status.idle":"2023-10-06T13:22:43.356798Z","shell.execute_reply.started":"2023-10-06T13:22:43.349811Z","shell.execute_reply":"2023-10-06T13:22:43.355493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to perfrom last test we will try this {all in one}\none_hot_train_df\nfull_labels = labels.values","metadata":{"execution":{"iopub.status.busy":"2023-10-06T14:15:24.249673Z","iopub.execute_input":"2023-10-06T14:15:24.250036Z","iopub.status.idle":"2023-10-06T14:15:24.255124Z","shell.execute_reply.started":"2023-10-06T14:15:24.250011Z","shell.execute_reply":"2023-10-06T14:15:24.253674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML Algo auxiallary FN's","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error, mean_squared_error\n\n# Custom function to calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\ndef custom_mean_rowwise_rmse(y_true, y_pred):\n    # Calculate RMSE for each row\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mrrmse_score = np.mean(rowwise_rmse)\n    return mrrmse_score\n\n# Custom function to calculate Mean Absolute Error (MAE)\ndef custom_mean_absolute_error(y_true, y_pred):\n    mae = mean_absolute_error(y_true, y_pred)\n    return mae","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:48:24.130915Z","iopub.status.idle":"2023-10-06T18:48:24.131303Z","shell.execute_reply.started":"2023-10-06T18:48:24.131130Z","shell.execute_reply":"2023-10-06T18:48:24.131147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RandomForestRegressor , TRY #1","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\n\n# Create a Random Forest Regressor model\nmodel_rf = RandomForestRegressor(n_estimators=100, random_state=42)\n\n# Fit the model on the training data\nmodel_rf.fit(X_train, y_train)\n\n# Predict on the validation data\ny_val_pred = model_rf.predict(X_val)\n\n# Calculate Mean Absolute Error (MAE) for validation predictions\nmae = custom_mean_absolute_error(y_val, y_val_pred)\nprint(f\"Mean Absolute Error (Validation): {mae}\")\n\n# Calculate Mean Rowwise Root Mean Squared Error (MRRMSE) for validation predictions\nmrrse = custom_mean_rowwise_rmse(y_val, y_val_pred)\nprint(f\"Mean Absolute Error (Validation): {mrrse}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:48:31.206147Z","iopub.execute_input":"2023-10-06T18:48:31.206540Z","iopub.status.idle":"2023-10-06T18:52:05.252731Z","shell.execute_reply.started":"2023-10-06T18:48:31.206509Z","shell.execute_reply":"2023-10-06T18:52:05.251754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of labels for the different evaluation metrics\nmetric_labels = ['MAE', 'MRRMSE']\n\n# Create a list of values for the corresponding metrics\nmetric_values = [mae, mrrse]\n\n# Plot the metrics\nplt.figure(figsize=(8, 6))\nplt.bar(metric_labels, metric_values, color=['blue', 'orange'])\nplt.title('Model Evaluation Metrics (Validation)')\nplt.ylabel('Metric Value')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:27:25.250856Z","iopub.execute_input":"2023-10-06T15:27:25.251441Z","iopub.status.idle":"2023-10-06T15:27:25.492100Z","shell.execute_reply.started":"2023-10-06T15:27:25.251412Z","shell.execute_reply":"2023-10-06T15:27:25.491014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Hyper Parameter Tuning for the Above RandomForestRegressor model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# Create a Random Forest Regressor model with specified hyperparameters\nmodel_rf_1 = RandomForestRegressor(\n    n_estimators=100,\n    max_depth=10,              # Set the maximum depth of each tree\n    min_samples_split=2,       # Minimum samples required to split a node\n    min_samples_leaf=1,        # Minimum samples required at a leaf node\n    max_features='sqrt',       # Number of features to consider for splitting\n    bootstrap=True,            # Use bootstrap samples\n    n_jobs=-1,                 # Use all available CPU cores\n    random_state=42\n)\n\n# Fit the model on the training data\nmodel_rf_1.fit(X_train, y_train)\n\n# Predict on the validation data\ny_val_pred_1 = model_rf_1.predict(X_val)\n\n# Calculate Mean Absolute Error (MAE) for validation predictions\nmae_1 = custom_mean_absolute_error(y_val, y_val_pred_1)\nprint(f\"Mean Absolute Error (Validation): {mae_1}\")\n\n# Calculate Mean Rowwise Root Mean Squared Error (MRRMSE) for validation predictions\nmrrse_1 = custom_mean_rowwise_rmse(y_val, y_val_pred_1)\nprint(f\"Mean Rowwise Root Mean Squared Error (Validation): {mrrse_1}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:30:58.375661Z","iopub.execute_input":"2023-10-06T15:30:58.376701Z","iopub.status.idle":"2023-10-06T15:31:07.301950Z","shell.execute_reply.started":"2023-10-06T15:30:58.376659Z","shell.execute_reply":"2023-10-06T15:31:07.300884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of labels for the different evaluation metrics\nmetric_labels = ['MAE', 'MRRMSE']\n\n# Create a list of values for the corresponding metrics\nmetric_values = [mae_1, mrrse_1]\n\n# Plot the metrics\nplt.figure(figsize=(8, 6))\nplt.bar(metric_labels, metric_values, color=['blue', 'orange'])\nplt.title('Model Evaluation Metrics (Validation Test using HyperTuning)')\nplt.ylabel('Metric Value')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:51:10.988791Z","iopub.execute_input":"2023-10-06T16:51:10.989152Z","iopub.status.idle":"2023-10-06T16:51:11.202750Z","shell.execute_reply.started":"2023-10-06T16:51:10.989126Z","shell.execute_reply":"2023-10-06T16:51:11.201700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Doesn't look good with hyper paramter tuning 🤣** 👆🏽","metadata":{}},{"cell_type":"code","source":"# Predict on the Test Data\ny_test_pred = model_rf.predict(X_test)\n\n# Calculate Mean Absolute Error (MAE) for validation predictions\nmae_test = custom_mean_absolute_error(y_test, y_test_pred)\nprint(f\"Mean Absolute Error (Test): {mae_test}\")\n\n# Calculate Mean Rowwise Root Mean Squared Error (MRRMSE) for validation predictions\nmrrse_test = custom_mean_rowwise_rmse(y_test, y_test_pred)\nprint(f\"Mean Rowwise Root Mean Squared Error (Test): {mrrse_test}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:18:41.830969Z","iopub.execute_input":"2023-10-06T16:18:41.831922Z","iopub.status.idle":"2023-10-06T16:18:42.271015Z","shell.execute_reply.started":"2023-10-06T16:18:41.831889Z","shell.execute_reply":"2023-10-06T16:18:42.269913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of labels for the different evaluation metrics\nmetric_labels = ['MAE', 'MRRMSE']\n\n# Create a list of values for the corresponding metrics\nmetric_values = [mae_test, mrrse_test]\n\n# Plot the metrics\nplt.figure(figsize=(8, 6))\nplt.bar(metric_labels, metric_values, color=['blue', 'orange'])\nplt.title('Model Evaluation Metrics (Test)')\nplt.ylabel('Metric Value')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:49:06.599298Z","iopub.execute_input":"2023-10-06T16:49:06.600436Z","iopub.status.idle":"2023-10-06T16:49:06.836933Z","shell.execute_reply.started":"2023-10-06T16:49:06.600398Z","shell.execute_reply":"2023-10-06T16:49:06.835639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict on the full data\ny_full_pred_train = model_rf.predict(one_hot_train_df)\n\n# Calculate Mean Absolute Error (MAE) for test predictions\nmae_full_train = custom_mean_absolute_error(full_labels, y_full_pred_train)\nprint(f\"Mean Absolute Error (Full_train): {mae_full_train}\")\n\n# Calculate Mean Rowwise Root Mean Squared Error (MRRMSE) for test predictions\nmrrse_full_train = custom_mean_rowwise_rmse(full_labels, y_full_pred_train)\nprint(f\"Mean Rowwise Root Mean Squared Error (Full_train): {mrrse_full_train}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:34:37.279223Z","iopub.execute_input":"2023-10-06T16:34:37.279609Z","iopub.status.idle":"2023-10-06T16:34:41.189018Z","shell.execute_reply.started":"2023-10-06T16:34:37.279580Z","shell.execute_reply":"2023-10-06T16:34:41.187901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of labels for the different evaluation metrics\nmetric_labels = ['MAE', 'MRRMSE']\n\n# Create a list of values for the corresponding metrics\nmetric_values = [mae_full_train, mrrse_full_train]\n\n# Plot the metrics\nplt.figure(figsize=(8, 6))\nplt.bar(metric_labels, metric_values, color=['blue', 'orange'])\nplt.title('Model Evaluation Metrics (Full DataSet)')\nplt.ylabel('Metric Value')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:48:52.429536Z","iopub.execute_input":"2023-10-06T16:48:52.430112Z","iopub.status.idle":"2023-10-06T16:48:52.655670Z","shell.execute_reply.started":"2023-10-06T16:48:52.430084Z","shell.execute_reply":"2023-10-06T16:48:52.654849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict on Test Data","metadata":{}},{"cell_type":"code","source":"# Predict on the full test data\ny_full_pred_test = model_rf.predict(one_hot_test_df)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:40:34.010667Z","iopub.execute_input":"2023-10-06T16:40:34.011542Z","iopub.status.idle":"2023-10-06T16:40:35.352984Z","shell.execute_reply.started":"2023-10-06T16:40:34.011489Z","shell.execute_reply":"2023-10-06T16:40:35.351827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a df to store results similar to submission file~\nsample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(y_full_pred_test, columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:43:28.134047Z","iopub.execute_input":"2023-10-06T16:43:28.134424Z","iopub.status.idle":"2023-10-06T16:43:28.140262Z","shell.execute_reply.started":"2023-10-06T16:43:28.134382Z","shell.execute_reply":"2023-10-06T16:43:28.139213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))\nsubmission_df.head() # Model Preds on a df!!","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:44:09.948780Z","iopub.execute_input":"2023-10-06T16:44:09.949130Z","iopub.status.idle":"2023-10-06T16:44:09.978881Z","shell.execute_reply.started":"2023-10-06T16:44:09.949104Z","shell.execute_reply":"2023-10-06T16:44:09.977856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_submission.head() # confirming xD","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:44:52.420991Z","iopub.execute_input":"2023-10-06T16:44:52.421410Z","iopub.status.idle":"2023-10-06T16:44:52.449935Z","shell.execute_reply.started":"2023-10-06T16:44:52.421344Z","shell.execute_reply":"2023-10-06T16:44:52.448707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv(\"submission_df.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:49:40.473803Z","iopub.execute_input":"2023-10-06T16:49:40.474163Z","iopub.status.idle":"2023-10-06T16:49:49.347168Z","shell.execute_reply.started":"2023-10-06T16:49:40.474135Z","shell.execute_reply":"2023-10-06T16:49:49.346429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission_preds.zip /kaggle/working/submission_df.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:54:34.009913Z","iopub.execute_input":"2023-10-06T16:54:34.010406Z","iopub.status.idle":"2023-10-06T16:54:46.893847Z","shell.execute_reply.started":"2023-10-06T16:54:34.010344Z","shell.execute_reply":"2023-10-06T16:54:46.892181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n\n# # Define the source path of your kaggle.json file\n# source_path = '/kaggle/input/kaggle-token/kaggle.json'\n\n# # Define the target directory where Kaggle expects the kaggle.json file\n# target_directory = '/root/.kaggle/'\n\n# # Copy the kaggle.json file to the target directory\n# shutil.copy(source_path, target_directory)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:29:29.141204Z","iopub.execute_input":"2023-10-06T18:29:29.141611Z","iopub.status.idle":"2023-10-06T18:29:29.154947Z","shell.execute_reply.started":"2023-10-06T18:29:29.141573Z","shell.execute_reply":"2023-10-06T18:29:29.153769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!chmod 600 /root/.kaggle/kaggle.json","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:32:08.243002Z","iopub.execute_input":"2023-10-06T18:32:08.243388Z","iopub.status.idle":"2023-10-06T18:32:09.659734Z","shell.execute_reply.started":"2023-10-06T18:32:08.243344Z","shell.execute_reply":"2023-10-06T18:32:09.657784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle competitions submit -c open-problems-single-cell-perturbations -f /kaggle/working/submission_preds.zip -m \"Message\"","metadata":{"execution":{"iopub.status.busy":"2023-10-06T18:30:25.484844Z","iopub.execute_input":"2023-10-06T18:30:25.485272Z","iopub.status.idle":"2023-10-06T18:30:35.538658Z","shell.execute_reply.started":"2023-10-06T18:30:25.485237Z","shell.execute_reply":"2023-10-06T18:30:35.537572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB Boost Regressor with 2 features , TRY #2","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:19.466544Z","iopub.execute_input":"2023-10-06T15:29:19.467344Z","iopub.status.idle":"2023-10-06T15:29:19.471563Z","shell.execute_reply.started":"2023-10-06T15:29:19.467314Z","shell.execute_reply":"2023-10-06T15:29:19.470426Z"}}},{"cell_type":"code","source":"# from sklearn.multioutput import MultiOutputRegressor\n# from sklearn.ensemble import GradientBoostingRegressor\n\n# # Create a Gradient Boosting Regressor model\n# model_gb = GradientBoostingRegressor(n_estimators=100, random_state=42)\n\n# # Wrap the model in MultiOutputRegressor\n# multi_output_model_gb = MultiOutputRegressor(model_gb)\n\n# # Fit the multi-output model on the training data\n# multi_output_model_gb.fit(X_train, y_train)\n\n# # Predict on the validation data\n# y_val_pred_gb = multi_output_model_gb.predict(X_val)\n\n# # Calculate Mean Absolute Error (MAE) for validation predictions\n# mae_2 = custom_mean_absolute_error(y_val, y_val_pred_gb)\n# print(f\"Mean Absolute Error (Validation): {mae_2}\")\n\n# # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE) for validation predictions\n# mrrse_2 = custom_mean_rowwise_rmse(y_val, y_val_pred_gb)\n# print(f\"Mean Rowwise Root Mean Squared Error (Validation): {mrrse_2}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:52:05.914567Z","iopub.execute_input":"2023-10-06T16:52:05.914951Z","iopub.status.idle":"2023-10-06T16:52:05.920252Z","shell.execute_reply.started":"2023-10-06T16:52:05.914919Z","shell.execute_reply":"2023-10-06T16:52:05.919158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a list of labels for the different evaluation metrics\n# metric_labels = ['MAE', 'MRRMSE']\n\n# # Create a list of values for the corresponding metrics\n# metric_values = [mae_2, mrrse_2]\n\n# # Plot the metrics\n# plt.figure(figsize=(8, 6))\n# plt.bar(metric_labels, metric_values, color=['blue', 'orange'])\n# plt.title('Model Evaluation Metrics (Validation)')\n# plt.ylabel('Metric Value')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T16:52:10.198538Z","iopub.execute_input":"2023-10-06T16:52:10.198917Z","iopub.status.idle":"2023-10-06T16:52:10.203437Z","shell.execute_reply.started":"2023-10-06T16:52:10.198886Z","shell.execute_reply":"2023-10-06T16:52:10.202113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generally Auxilary Function for Neural Net # Tensorflow","metadata":{}},{"cell_type":"code","source":"# from tensorflow.keras.callbacks import ModelCheckpoint\n# import matplotlib.pyplot as plt\n# import numpy as np\n# import tensorflow as tf\n# from sklearn.metrics import mean_absolute_error\n\n# # Create a ModelCheckpoint callback\n# def create_model_checkpoint(filepath, monitor='val_mae', save_best_only=True,\n#                             save_weights_only=True, mode='auto', verbose=0):\n#     checkpoint = ModelCheckpoint(\n#         filepath=filepath,\n#         monitor=monitor,\n#         save_best_only=save_best_only,\n#         save_weights_only=save_weights_only,\n#         mode=mode,\n#         verbose=verbose\n#     )\n#     return checkpoint\n\n# # Plot training history (loss and metrics)\n# def plot_training_history(history, metrics):\n#     loss = history.history['loss']\n#     val_loss = history.history['val_loss']\n#     epochs = range(len(loss))\n    \n#     # Plot loss and specified metrics\n#     plt.figure(figsize=(12, 6))\n#     plt.subplot(1, 2, 1)\n#     plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n#     plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n#     plt.title('Loss')\n#     plt.xlabel('Epochs')\n#     plt.legend()\n    \n#     for metric in metrics:\n#         train_metric_name = f'Training {metric.capitalize()}'\n#         val_metric_name = f'Validation {metric.capitalize()}'\n#         train_metric = history.history[metric]\n#         val_metric = history.history['val_' + metric]\n        \n#         plt.subplot(1, 2, 2)\n#         plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n#         plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n    \n#     plt.title('Metrics')\n#     plt.xlabel('Epochs')\n#     plt.legend(loc='upper right')\n#     plt.tight_layout()\n#     plt.show()\n\n# # Calculate MAE and MRRMSE\n# def calculate_mae_and_mrrmse(model, data, y_true):\n#     y_pred_original = model.predict(data, batch_size=1)\n#     mae = mean_absolute_error(y_true , y_pred_original)\n    \n#     # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n#     rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original), axis=1))\n#     mrrmse_score = np.mean(rowwise_rmse)\n    \n#     # Print results\n#     print(f\"Mean Absolute Error (MAE): {mae}\")\n#     print(f\"Mean Rowwise Root Mean Squared Error (MRRMSE): {mrrmse_score}\")\n\n# # Custom loss function for Mean Rowwise RMSE\n# def mean_rowwise_rmse_loss(y_true, y_pred):\n#     rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n#     mean_rmse = tf.reduce_mean(rmse_per_row)\n#     return mean_rmse\n\n# # Custom metric for Mean Rowwise RMSE\n# def custom_mean_rowwise_rmse(y_true, y_pred):\n#     rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n#     mean_rmse = tf.reduce_mean(rmse_per_row)\n#     return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2023-10-06T14:33:37.351481Z","iopub.execute_input":"2023-10-06T14:33:37.351886Z","iopub.status.idle":"2023-10-06T14:33:37.364182Z","shell.execute_reply.started":"2023-10-06T14:33:37.351858Z","shell.execute_reply":"2023-10-06T14:33:37.363343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Set a random seed for reproducibility\n# tf.random.set_seed(42)\n\n# # Define a sequential neural network model\n# model_0 = Sequential([ \n#     # Dense layer with 256 units\n#     Dense(256),\n#     # Batch normalization layer for stabilizing training\n#     BatchNormalization(),\n#     # ReLU activation function for introducing non-linearity\n#     Activation(\"relu\"),\n#     # Dropout layer to prevent overfitting (20% dropout rate)\n#     Dropout(0.2),\n    \n#     # Another dense layer with 128 units and ReLU activation\n#     Dense(128, activation=\"relu\"),\n#     # Dropout layer (20% dropout rate)\n#     Dropout(0.2),\n    \n#     # Dense layer with 64 units and ReLU activation\n#     Dense(64, activation=\"relu\"),\n#     # Batch normalization layer\n#     BatchNormalization(),\n#     # Dropout layer (20% dropout rate)\n#     Dropout(0.2),\n    \n#     # Dense layer with 32 units and ReLU activation\n#     Dense(32, activation=\"relu\"),\n#     # Dropout layer (20% dropout rate)\n#     Dropout(0.2),\n    \n#     # Dense layer with 16 units and ReLU activation\n#     Dense(16, activation=\"relu\"),\n#     # Dropout layer (20% dropout rate)\n#     Dropout(0.2),\n    \n#     # Output layer with 18211 units and linear activation (regression task)\n#     Dense(18211, activation=\"linear\")\n# ])\n\n# # Compile the model with Mean Absolute Error (MAE) as the loss function,\n# # Adam optimizer, and MAE as the evaluation metric\n# model_0.compile(loss=\"mae\", \n#                 optimizer=tf.keras.optimizers.Adam(),\n#                 metrics=[\"mae\"])\n\n# # Train the model on the training data for 30 epochs\n# # with validation data provided, and save the best model checkpoint\n# history_0 = model_2.fit(X_train, y_train,\n#                        epochs=30,\n#                        verbose=0,  # Train in silent mode (no progress bars)\n#                        validation_data=(X_val, y_val),\n#                        callbacks=[create_model_checkpoint(\"model_2\", monitor=\"val_mae\")])\n","metadata":{},"execution_count":null,"outputs":[]}]}