{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Neural Network Regression  \nIn this notebook the goal is to levarge the power of neural networks and get lowest MRRMSE(Mean Rowwise Root Mean Sqared Error)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:21.437578Z","iopub.execute_input":"2024-04-02T07:57:21.438266Z","iopub.status.idle":"2024-04-02T07:57:21.449451Z","shell.execute_reply.started":"2024-04-02T07:57:21.438201Z","shell.execute_reply":"2024-04-02T07:57:21.447583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read data ","metadata":{}},{"cell_type":"code","source":"de_train =   pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:21.452135Z","iopub.execute_input":"2024-04-02T07:57:21.452950Z","iopub.status.idle":"2024-04-02T07:57:27.975571Z","shell.execute_reply.started":"2024-04-02T07:57:21.452911Z","shell.execute_reply":"2024-04-02T07:57:27.973994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Become one with data","metadata":{}},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:27.977640Z","iopub.execute_input":"2024-04-02T07:57:27.978077Z","iopub.status.idle":"2024-04-02T07:57:28.025021Z","shell.execute_reply.started":"2024-04-02T07:57:27.978033Z","shell.execute_reply":"2024-04-02T07:57:28.023560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.027757Z","iopub.execute_input":"2024-04-02T07:57:28.028195Z","iopub.status.idle":"2024-04-02T07:57:28.041954Z","shell.execute_reply.started":"2024-04-02T07:57:28.028161Z","shell.execute_reply":"2024-04-02T07:57:28.040407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.043627Z","iopub.execute_input":"2024-04-02T07:57:28.044120Z","iopub.status.idle":"2024-04-02T07:57:28.101052Z","shell.execute_reply.started":"2024-04-02T07:57:28.044083Z","shell.execute_reply":"2024-04-02T07:57:28.099773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique= de_train['sm_name'].unique()\nunique, print(f\"Count {len(unique)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.102556Z","iopub.execute_input":"2024-04-02T07:57:28.102950Z","iopub.status.idle":"2024-04-02T07:57:28.114770Z","shell.execute_reply.started":"2024-04-02T07:57:28.102917Z","shell.execute_reply":"2024-04-02T07:57:28.113000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.117038Z","iopub.execute_input":"2024-04-02T07:57:28.117590Z","iopub.status.idle":"2024-04-02T07:57:28.128562Z","shell.execute_reply.started":"2024-04-02T07:57:28.117539Z","shell.execute_reply":"2024-04-02T07:57:28.127172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(de_train['cell_type'].unique())\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.130510Z","iopub.execute_input":"2024-04-02T07:57:28.131073Z","iopub.status.idle":"2024-04-02T07:57:28.140900Z","shell.execute_reply.started":"2024-04-02T07:57:28.131024Z","shell.execute_reply":"2024-04-02T07:57:28.139561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train[de_train[\"control\"] == True].count  ","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.142638Z","iopub.execute_input":"2024-04-02T07:57:28.143063Z","iopub.status.idle":"2024-04-02T07:57:28.174737Z","shell.execute_reply.started":"2024-04-02T07:57:28.143029Z","shell.execute_reply":"2024-04-02T07:57:28.173392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing more and staring at data for sometime will be more helpful for understanding but let's try building some simple model.","metadata":{}},{"cell_type":"markdown","source":"# Preprocess data \nWe'll take only 'cell_type' and 'sm_name' as features and 18211 gene values as labels.","metadata":{}},{"cell_type":"code","source":"# shuffle the data\nde_train = de_train.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.181039Z","iopub.execute_input":"2024-04-02T07:57:28.182137Z","iopub.status.idle":"2024-04-02T07:57:28.276845Z","shell.execute_reply.started":"2024-04-02T07:57:28.182084Z","shell.execute_reply":"2024-04-02T07:57:28.275774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.278375Z","iopub.execute_input":"2024-04-02T07:57:28.279037Z","iopub.status.idle":"2024-04-02T07:57:28.324273Z","shell.execute_reply.started":"2024-04-02T07:57:28.278995Z","shell.execute_reply":"2024-04-02T07:57:28.322741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create features and and labels for reverse model 18211 features and 152 labels for true model\nfeatures_columns = [\"cell_type\", \"sm_name\"]\nlabels_columns=[\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"]\nlabels = de_train.drop(columns=labels_columns)\nfeatures = pd.DataFrame(de_train, columns=features_columns)\n\nfeatures","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.326518Z","iopub.execute_input":"2024-04-02T07:57:28.328038Z","iopub.status.idle":"2024-04-02T07:57:28.395176Z","shell.execute_reply.started":"2024-04-02T07:57:28.327983Z","shell.execute_reply":"2024-04-02T07:57:28.393786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.397840Z","iopub.execute_input":"2024-04-02T07:57:28.398750Z","iopub.status.idle":"2024-04-02T07:57:28.440773Z","shell.execute_reply.started":"2024-04-02T07:57:28.398668Z","shell.execute_reply":"2024-04-02T07:57:28.439505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test data \ntest_data = pd.DataFrame(id_map, columns=features_columns)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.442263Z","iopub.execute_input":"2024-04-02T07:57:28.442637Z","iopub.status.idle":"2024-04-02T07:57:28.448897Z","shell.execute_reply.started":"2024-04-02T07:57:28.442603Z","shell.execute_reply":"2024-04-02T07:57:28.447869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process categorical data\nWe'll use OneHotEncode for categorical data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# Create an instance of the encoder\nencoder = OneHotEncoder()\n\n# Fit the encoder on features\nencoder.fit(features)\n\n# Transform the features into one-hot encoded format\none_hot_encode_features = encoder.transform(features)\n\n# Transform the test data(id_map)\none_hot_test = encoder.transform(test_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.450543Z","iopub.execute_input":"2024-04-02T07:57:28.451212Z","iopub.status.idle":"2024-04-02T07:57:28.469437Z","shell.execute_reply.started":"2024-04-02T07:57:28.451176Z","shell.execute_reply":"2024-04-02T07:57:28.467672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check shape\none_hot_encode_features.toarray().shape, one_hot_test.toarray().shape","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.471605Z","iopub.execute_input":"2024-04-02T07:57:28.472184Z","iopub.status.idle":"2024-04-02T07:57:28.485240Z","shell.execute_reply.started":"2024-04-02T07:57:28.472137Z","shell.execute_reply":"2024-04-02T07:57:28.483561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check one sample\none_hot_encode_features.toarray()[112]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.487400Z","iopub.execute_input":"2024-04-02T07:57:28.487916Z","iopub.status.idle":"2024-04-02T07:57:28.497778Z","shell.execute_reply.started":"2024-04-02T07:57:28.487870Z","shell.execute_reply":"2024-04-02T07:57:28.496756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Data into training, validation and test sets ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into 70% training, 15% validation, and 15% testing\nX_train, X_temp, y_train, y_temp = train_test_split(one_hot_encode_features, labels.values, test_size=0.3, shuffle=False)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.499194Z","iopub.execute_input":"2024-04-02T07:57:28.500792Z","iopub.status.idle":"2024-04-02T07:57:28.618777Z","shell.execute_reply.started":"2024-04-02T07:57:28.500746Z","shell.execute_reply":"2024-04-02T07:57:28.617188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the shapes of the data splits\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_val shape:\", X_val.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_val shape:\", y_val.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.621362Z","iopub.execute_input":"2024-04-02T07:57:28.621900Z","iopub.status.idle":"2024-04-02T07:57:28.631036Z","shell.execute_reply.started":"2024-04-02T07:57:28.621851Z","shell.execute_reply":"2024-04-02T07:57:28.629474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We also get full features for final training \nfull_features = one_hot_encode_features.toarray()\nfull_labels = labels.values","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.633315Z","iopub.execute_input":"2024-04-02T07:57:28.633828Z","iopub.status.idle":"2024-04-02T07:57:28.641727Z","shell.execute_reply.started":"2024-04-02T07:57:28.633783Z","shell.execute_reply":"2024-04-02T07:57:28.640268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"full_features shape:\", full_features.shape)\nprint(\"full_labels shape:\", full_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.643929Z","iopub.execute_input":"2024-04-02T07:57:28.644445Z","iopub.status.idle":"2024-04-02T07:57:28.657118Z","shell.execute_reply.started":"2024-04-02T07:57:28.644403Z","shell.execute_reply":"2024-04-02T07:57:28.655340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Auxillary functions ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\ndef create_model_checkpoint(filepath, monitor='val_mae', save_best_only=True,\n                            save_weights_only=True, mode='auto', verbose=0):\n    \"\"\"\n    Create a ModelCheckpoint callback for saving the best model weights during training.\n\n    Args:\n        filepath (str): Filepath to save the best weights.\n        monitor (str): Metric to monitor (e.g., 'val_loss' or 'val_mae').\n        save_best_only (bool): Save only the best weights.\n        save_weights_only (bool): Save only the model's weights, not the entire model.\n        mode (str): One of {'auto', 'min', 'max'}. In 'min' mode, it saves when the monitored metric decreases.\n        verbose (int): Verbosity mode. 0 = silent, 1 = progress bar, 2 = one line per epoch.\n\n    Returns:\n\n        keras.callbacks.ModelCheckpoint: ModelCheckpoint callback.\n    \"\"\"\n    checkpoint = ModelCheckpoint(\n        filepath=filepath,\n        monitor=monitor,\n        save_best_only=save_best_only,\n        save_weights_only=save_weights_only,\n        mode=mode,\n        verbose=verbose\n    )\n    return checkpoint","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.659250Z","iopub.execute_input":"2024-04-02T07:57:28.659826Z","iopub.status.idle":"2024-04-02T07:57:28.669448Z","shell.execute_reply.started":"2024-04-02T07:57:28.659776Z","shell.execute_reply":"2024-04-02T07:57:28.668049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, metrics):\n    \"\"\"\n    Plot training history curves for loss and evaluation metrics on the same line.\n\n    Args:\n        history (keras.callbacks.History): Training history object.\n        metrics (list): List of metric names to plot.\n\n    Returns:\n        None\n    \"\"\"\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n    plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot specified evaluation metrics on the same line\n    for metric in metrics:\n        train_metric_name = f'Training {metric.capitalize()}'\n        val_metric_name = f'Validation {metric.capitalize()}'\n        train_metric = history.history[metric]\n        val_metric = history.history['val_' + metric]\n\n        plt.subplot(1, 2, 2)\n        plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n        plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n\n    plt.title('Metrics')\n    plt.xlabel('Epochs')\n    plt.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:39.236757Z","iopub.execute_input":"2024-04-02T08:01:39.237518Z","iopub.status.idle":"2024-04-02T08:01:39.250842Z","shell.execute_reply.started":"2024-04-02T08:01:39.237483Z","shell.execute_reply":"2024-04-02T08:01:39.249306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\ndef calculate_mae_and_mrrmse(model, data, y_true):\n    \"\"\"\n    Calculate Mean Absolute Error (MAE) and Mean Rowwise Root Mean Squared Error (MRRMSE).\n\n    Parameters:\n    - model: The trained  model.\n    - data: The input data for prediction.\n    - y_true: The true target values.\n    - scaler: The scaler used for data normalization.\n\n    Returns:\n    - None\n    \"\"\"\n    # Predict using the model\n    y_pred_original = model.predict(data, batch_size=1)\n    \n    # Calculate Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_true , y_pred_original)\n    \n    # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    \n    # Print the results\n    print(f\"Mean Absolute Error (MAE): {mae}\")\n    print(f\"Mean Rowwise Root Mean Squared Error (MRRMSE): {mrrmse_score}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.691498Z","iopub.execute_input":"2024-04-02T07:57:28.692549Z","iopub.status.idle":"2024-04-02T07:57:28.707660Z","shell.execute_reply.started":"2024-04-02T07:57:28.692502Z","shell.execute_reply":"2024-04-02T07:57:28.706149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_rowwise_rmse_loss(y_true, y_pred):\n    \"\"\"\n    Custom loss function to calculate the Mean Rowwise Root Mean Squared Error (RMSE) loss.\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE loss as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.709541Z","iopub.execute_input":"2024-04-02T07:57:28.710021Z","iopub.status.idle":"2024-04-02T07:57:28.725545Z","shell.execute_reply.started":"2024-04-02T07:57:28.709971Z","shell.execute_reply":"2024-04-02T07:57:28.724056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_mean_rowwise_rmse(y_true, y_pred):\n    \"\"\"\n    Custom metric to calculate the Mean Rowwise Root Mean Squared Error (RMSE).\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.727572Z","iopub.execute_input":"2024-04-02T07:57:28.728042Z","iopub.status.idle":"2024-04-02T07:57:28.739577Z","shell.execute_reply.started":"2024-04-02T07:57:28.728005Z","shell.execute_reply":"2024-04-02T07:57:28.738158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Beautiful functions let's start experiment with building models","metadata":{}},{"cell_type":"markdown","source":"# Build the model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Activation\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.741597Z","iopub.execute_input":"2024-04-02T07:57:28.742150Z","iopub.status.idle":"2024-04-02T07:57:28.754857Z","shell.execute_reply.started":"2024-04-02T07:57:28.742103Z","shell.execute_reply":"2024-04-02T07:57:28.753502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Model_0:** Let's start with only 2 dense layers ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_0 = Sequential([\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_0.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_0 = model_0.fit(X_train, y_train,\n                       epochs=10,\n                       validation_data=(X_val,y_val),\n                       batch_size=32,\n                       callbacks=[create_model_checkpoint(\"model_0\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:57:28.765606Z","iopub.execute_input":"2024-04-02T07:57:28.766829Z","iopub.status.idle":"2024-04-02T07:58:00.146000Z","shell.execute_reply.started":"2024-04-02T07:57:28.766771Z","shell.execute_reply":"2024-04-02T07:58:00.144422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_0.load_weights(\"model_0\")\ncalculate_mae_and_mrrmse(model=model_0, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:58:00.148406Z","iopub.execute_input":"2024-04-02T07:58:00.148943Z","iopub.status.idle":"2024-04-02T07:58:01.132448Z","shell.execute_reply.started":"2024-04-02T07:58:00.148905Z","shell.execute_reply":"2024-04-02T07:58:01.130750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_0, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:58:01.135153Z","iopub.execute_input":"2024-04-02T07:58:01.137422Z","iopub.status.idle":"2024-04-02T07:58:06.352127Z","shell.execute_reply.started":"2024-04-02T07:58:01.137371Z","shell.execute_reply":"2024-04-02T07:58:06.350786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_0, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:58:06.353898Z","iopub.execute_input":"2024-04-02T07:58:06.355153Z","iopub.status.idle":"2024-04-02T07:58:07.035283Z","shell.execute_reply.started":"2024-04-02T07:58:06.355103Z","shell.execute_reply":"2024-04-02T07:58:07.034075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, it seems the model is not learning much and is underfitting. Let's try something different. ","metadata":{}},{"cell_type":"markdown","source":"> **model_1:** Increase layers and neurons ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_1 = Sequential([\n    Dense(1024, activation=\"tanh\"),\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_1.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_1 = model_1.fit(X_train, y_train,\n                       epochs=30,\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_1\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:58:07.036968Z","iopub.execute_input":"2024-04-02T07:58:07.037607Z","iopub.status.idle":"2024-04-02T07:59:26.972233Z","shell.execute_reply.started":"2024-04-02T07:58:07.037573Z","shell.execute_reply":"2024-04-02T07:59:26.971054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_1.load_weights(\"model_1\")\ncalculate_mae_and_mrrmse(model=model_1, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:26.974577Z","iopub.execute_input":"2024-04-02T07:59:26.975048Z","iopub.status.idle":"2024-04-02T07:59:27.983467Z","shell.execute_reply.started":"2024-04-02T07:59:26.975010Z","shell.execute_reply":"2024-04-02T07:59:27.982049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_1, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:27.985985Z","iopub.execute_input":"2024-04-02T07:59:27.986948Z","iopub.status.idle":"2024-04-02T07:59:33.146748Z","shell.execute_reply.started":"2024-04-02T07:59:27.986895Z","shell.execute_reply":"2024-04-02T07:59:33.145200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_1, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:33.148589Z","iopub.execute_input":"2024-04-02T07:59:33.149077Z","iopub.status.idle":"2024-04-02T07:59:33.876226Z","shell.execute_reply.started":"2024-04-02T07:59:33.149032Z","shell.execute_reply":"2024-04-02T07:59:33.874769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The model is not performing good let's increase complexity and also add Dropout and BatchNormalization layers","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_2 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(18211,activation= \"linear\")\n])\n\n\nmodel_2.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[\"mae\"])\n\nhistory_2 = model_2.fit(X_train, y_train,\n                       epochs=30,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_2\", monitor=\"val_mae\")])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:33.877995Z","iopub.execute_input":"2024-04-02T07:59:33.878475Z","iopub.status.idle":"2024-04-02T07:59:45.021346Z","shell.execute_reply.started":"2024-04-02T07:59:33.878416Z","shell.execute_reply":"2024-04-02T07:59:45.019796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2.load_weights(\"model_2\")\ncalculate_mae_and_mrrmse(model=model_2, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:45.023996Z","iopub.execute_input":"2024-04-02T07:59:45.024528Z","iopub.status.idle":"2024-04-02T07:59:45.572558Z","shell.execute_reply.started":"2024-04-02T07:59:45.024480Z","shell.execute_reply":"2024-04-02T07:59:45.570894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_2, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:45.574362Z","iopub.execute_input":"2024-04-02T07:59:45.574751Z","iopub.status.idle":"2024-04-02T07:59:47.682897Z","shell.execute_reply.started":"2024-04-02T07:59:45.574718Z","shell.execute_reply":"2024-04-02T07:59:47.681576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Visualize the learning from our helper functions\nplot_training_history(history_2, metrics=[\"mae\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:47.684434Z","iopub.execute_input":"2024-04-02T07:59:47.684820Z","iopub.status.idle":"2024-04-02T07:59:48.454533Z","shell.execute_reply.started":"2024-04-02T07:59:47.684786Z","shell.execute_reply":"2024-04-02T07:59:48.453245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Changing metrics and using same model","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n# clone model 2\nmodel_3 = tf.keras.models.clone_model(model_2)\n\nmodel_3.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(learning_rate=0.0027),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_3 = model_3.fit(X_train, y_train,\n                       epochs=25,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_3\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:48.456473Z","iopub.execute_input":"2024-04-02T07:59:48.456918Z","iopub.status.idle":"2024-04-02T07:59:59.122312Z","shell.execute_reply.started":"2024-04-02T07:59:48.456874Z","shell.execute_reply":"2024-04-02T07:59:59.121236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_3.load_weights(\"model_3\")\ncalculate_mae_and_mrrmse(model=model_3, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:59.124189Z","iopub.execute_input":"2024-04-02T07:59:59.125808Z","iopub.status.idle":"2024-04-02T07:59:59.646891Z","shell.execute_reply.started":"2024-04-02T07:59:59.125739Z","shell.execute_reply":"2024-04-02T07:59:59.645469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_3, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T07:59:59.648603Z","iopub.execute_input":"2024-04-02T07:59:59.649012Z","iopub.status.idle":"2024-04-02T08:00:01.710378Z","shell.execute_reply.started":"2024-04-02T07:59:59.648979Z","shell.execute_reply":"2024-04-02T08:00:01.709357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_3, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:00:01.711712Z","iopub.execute_input":"2024-04-02T08:00:01.712627Z","iopub.status.idle":"2024-04-02T08:00:02.415556Z","shell.execute_reply.started":"2024-04-02T08:00:01.712576Z","shell.execute_reply":"2024-04-02T08:00:02.414560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-fold validation ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n# clone model 2\nmodel_4 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.1),\n    Dense(18211,activation= \"linear\")\n])\n\nmodel_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:00:02.416902Z","iopub.execute_input":"2024-04-02T08:00:02.417542Z","iopub.status.idle":"2024-04-02T08:00:02.489391Z","shell.execute_reply.started":"2024-04-02T08:00:02.417507Z","shell.execute_reply":"2024-04-02T08:00:02.488262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n# Define the number of folds (K)\nnum_folds = 5 # You can change this value as needed\n\n# Initialize lists to store the model's performance scores\nmae_scores = []\nmrrmse_scores = []\n\n# Initialize the KFold object\nkf = KFold(n_splits=num_folds, shuffle=True, random_state=51)\n\n# Loop through the K folds\nfor train_index, val_index in kf.split(full_features):\n    # Convert indices to integers and split the data\n    train_index = train_index.astype(int)\n    val_index = val_index.astype(int)\n    X_train_, X_val_ = full_features[train_index], full_features[val_index]\n    y_train_, y_val_ = full_labels[train_index], full_labels[val_index]\n\n    # Train your model on X_train and y_train\n    model_4.fit(X_train_, y_train_, epochs=50, verbose=0)\n\n    # Make predictions on the validation set\n    y_preds = model_4.predict(X_val_)\n\n    # Calculate the Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_val_, y_preds)\n    mae_scores.append(mae)\n\n    # Calculate the Mean Rowwise Root Mean Square Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_val_ - y_preds), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    mrrmse_scores.append(mrrmse_score)\n\n# Calculate the mean and standard deviation of MAE and MRRMSE scores\nmean_mae = np.mean(mae_scores)\nmean_mrrmse = np.mean(mrrmse_scores)\n\n# Print the results\nprint(f'Average MAE across {num_folds} folds: {mean_mae:.4f} ')\nprint(f'Average MRRMSE across {num_folds} folds: {mean_mrrmse:.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:00:02.490864Z","iopub.execute_input":"2024-04-02T08:00:02.491267Z","iopub.status.idle":"2024-04-02T08:00:56.503793Z","shell.execute_reply.started":"2024-04-02T08:00:02.491231Z","shell.execute_reply":"2024-04-02T08:00:56.500669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train the model on full data","metadata":{}},{"cell_type":"code","source":"model_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_4 = model_4.fit(full_features, full_labels,\n                       epochs=50,\n                       verbose=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:00:56.507648Z","iopub.execute_input":"2024-04-02T08:00:56.508266Z","iopub.status.idle":"2024-04-02T08:01:19.341880Z","shell.execute_reply.started":"2024-04-02T08:00:56.508207Z","shell.execute_reply":"2024-04-02T08:01:19.340378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_4, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:19.344153Z","iopub.execute_input":"2024-04-02T08:01:19.344774Z","iopub.status.idle":"2024-04-02T08:01:21.206359Z","shell.execute_reply.started":"2024-04-02T08:01:19.344726Z","shell.execute_reply":"2024-04-02T08:01:21.204852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting on test data ","metadata":{}},{"cell_type":"code","source":"preds = model_4.predict(one_hot_test.toarray(), batch_size=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:21.208336Z","iopub.execute_input":"2024-04-02T08:01:21.209725Z","iopub.status.idle":"2024-04-02T08:01:21.909721Z","shell.execute_reply.started":"2024-04-02T08:01:21.209640Z","shell.execute_reply":"2024-04-02T08:01:21.908296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape ","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:21.911909Z","iopub.execute_input":"2024-04-02T08:01:21.913315Z","iopub.status.idle":"2024-04-02T08:01:21.922626Z","shell.execute_reply.started":"2024-04-02T08:01:21.913274Z","shell.execute_reply":"2024-04-02T08:01:21.920860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(preds, columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:21.924526Z","iopub.execute_input":"2024-04-02T08:01:21.925719Z","iopub.status.idle":"2024-04-02T08:01:21.933367Z","shell.execute_reply.started":"2024-04-02T08:01:21.925663Z","shell.execute_reply":"2024-04-02T08:01:21.931792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:21.934911Z","iopub.execute_input":"2024-04-02T08:01:21.935958Z","iopub.status.idle":"2024-04-02T08:01:21.954286Z","shell.execute_reply.started":"2024-04-02T08:01:21.935868Z","shell.execute_reply":"2024-04-02T08:01:21.952896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:21.956234Z","iopub.execute_input":"2024-04-02T08:01:21.956697Z","iopub.status.idle":"2024-04-02T08:01:22.014840Z","shell.execute_reply.started":"2024-04-02T08:01:21.956634Z","shell.execute_reply":"2024-04-02T08:01:22.013570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:22.018021Z","iopub.execute_input":"2024-04-02T08:01:22.018435Z","iopub.status.idle":"2024-04-02T08:01:22.062635Z","shell.execute_reply.started":"2024-04-02T08:01:22.018401Z","shell.execute_reply":"2024-04-02T08:01:22.061239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission_df.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:22.064471Z","iopub.execute_input":"2024-04-02T08:01:22.064864Z","iopub.status.idle":"2024-04-02T08:01:31.749449Z","shell.execute_reply.started":"2024-04-02T08:01:22.064831Z","shell.execute_reply":"2024-04-02T08:01:31.747818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission_preds.zip /kaggle/working/submission_df.csv","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:01:31.752208Z","iopub.execute_input":"2024-04-02T08:01:31.752738Z","iopub.status.idle":"2024-04-02T08:01:39.206379Z","shell.execute_reply.started":"2024-04-02T08:01:31.752669Z","shell.execute_reply":"2024-04-02T08:01:39.204918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Improving model**\n\n* Try out different layers and number of neurons in layer and see how it goes.\n* Experiment with different activation functions.\n* Experiment with different architecture, try deep or wide architecture.\n* Find the best learning rate for the model.\n* Normalize labels and try optimizing.\n* Validate model using different techniques like K-fold validation.\n* Experiment.. Experiment.. Experiment.. Visualize.. Visualize.. Visualize..\n\nGood Luck :)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:14:36.993246Z","iopub.execute_input":"2023-10-05T09:14:36.993577Z","iopub.status.idle":"2023-10-05T09:14:37.000183Z","shell.execute_reply.started":"2023-10-05T09:14:36.99355Z","shell.execute_reply":"2023-10-05T09:14:36.999213Z"}}}]}