{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Neural Network Regression  \nIn this notebook the goal is to levarge the power of neural networks and get lowest MRRMSE(Mean Rowwise Root Mean Sqared Error)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:25:50.619052Z","iopub.execute_input":"2023-10-05T08:25:50.619553Z","iopub.status.idle":"2023-10-05T08:25:59.267832Z","shell.execute_reply.started":"2023-10-05T08:25:50.619513Z","shell.execute_reply":"2023-10-05T08:25:59.266760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read data ","metadata":{}},{"cell_type":"code","source":"de_train =   pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:25:59.269518Z","iopub.execute_input":"2023-10-05T08:25:59.270233Z","iopub.status.idle":"2023-10-05T08:26:04.621474Z","shell.execute_reply.started":"2023-10-05T08:25:59.270205Z","shell.execute_reply":"2023-10-05T08:26:04.620216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Become one with data","metadata":{}},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.622759Z","iopub.execute_input":"2023-10-05T08:26:04.623059Z","iopub.status.idle":"2023-10-05T08:26:04.661436Z","shell.execute_reply.started":"2023-10-05T08:26:04.623033Z","shell.execute_reply":"2023-10-05T08:26:04.660162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.663985Z","iopub.execute_input":"2023-10-05T08:26:04.664285Z","iopub.status.idle":"2023-10-05T08:26:04.673255Z","shell.execute_reply.started":"2023-10-05T08:26:04.664259Z","shell.execute_reply":"2023-10-05T08:26:04.672158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.674467Z","iopub.execute_input":"2023-10-05T08:26:04.674949Z","iopub.status.idle":"2023-10-05T08:26:04.714906Z","shell.execute_reply.started":"2023-10-05T08:26:04.674923Z","shell.execute_reply":"2023-10-05T08:26:04.713938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique= de_train['sm_name'].unique()\nunique, print(f\"Count {len(unique)}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.716099Z","iopub.execute_input":"2023-10-05T08:26:04.716349Z","iopub.status.idle":"2023-10-05T08:26:04.725375Z","shell.execute_reply.started":"2023-10-05T08:26:04.716328Z","shell.execute_reply":"2023-10-05T08:26:04.724174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.726501Z","iopub.execute_input":"2023-10-05T08:26:04.726797Z","iopub.status.idle":"2023-10-05T08:26:04.739701Z","shell.execute_reply.started":"2023-10-05T08:26:04.726768Z","shell.execute_reply":"2023-10-05T08:26:04.738672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train[de_train[\"control\"] == True].count  ","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.740843Z","iopub.execute_input":"2023-10-05T08:26:04.741677Z","iopub.status.idle":"2023-10-05T08:26:04.766415Z","shell.execute_reply.started":"2023-10-05T08:26:04.741641Z","shell.execute_reply":"2023-10-05T08:26:04.765724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing more and staring at data for sometime will be more helpful for understanding but let's try building some simple model.","metadata":{}},{"cell_type":"markdown","source":"# Preprocess data \nWe'll take only 'cell_type' and 'sm_name' as features and 18211 gene values as labels.","metadata":{}},{"cell_type":"code","source":"# shuffle the data\nde_train = de_train.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.767560Z","iopub.execute_input":"2023-10-05T08:26:04.768072Z","iopub.status.idle":"2023-10-05T08:26:04.824274Z","shell.execute_reply.started":"2023-10-05T08:26:04.768045Z","shell.execute_reply":"2023-10-05T08:26:04.823184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.828007Z","iopub.execute_input":"2023-10-05T08:26:04.828331Z","iopub.status.idle":"2023-10-05T08:26:04.857189Z","shell.execute_reply.started":"2023-10-05T08:26:04.828304Z","shell.execute_reply":"2023-10-05T08:26:04.856176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create features and and labels for reverse model 18211 features and 152 labels for true model\nfeatures_columns = [\"cell_type\", \"sm_name\"]\nlabels_columns=[\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"]\nlabels = de_train.drop(columns=labels_columns)\nfeatures = pd.DataFrame(de_train, columns=features_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.858342Z","iopub.execute_input":"2023-10-05T08:26:04.858626Z","iopub.status.idle":"2023-10-05T08:26:04.905075Z","shell.execute_reply.started":"2023-10-05T08:26:04.858601Z","shell.execute_reply":"2023-10-05T08:26:04.903974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.906576Z","iopub.execute_input":"2023-10-05T08:26:04.907371Z","iopub.status.idle":"2023-10-05T08:26:04.924497Z","shell.execute_reply.started":"2023-10-05T08:26:04.907336Z","shell.execute_reply":"2023-10-05T08:26:04.923533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.926029Z","iopub.execute_input":"2023-10-05T08:26:04.926451Z","iopub.status.idle":"2023-10-05T08:26:04.957028Z","shell.execute_reply.started":"2023-10-05T08:26:04.926411Z","shell.execute_reply":"2023-10-05T08:26:04.956008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test data \ntest_data = pd.DataFrame(id_map, columns=features_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.958425Z","iopub.execute_input":"2023-10-05T08:26:04.959411Z","iopub.status.idle":"2023-10-05T08:26:04.964053Z","shell.execute_reply.started":"2023-10-05T08:26:04.959383Z","shell.execute_reply":"2023-10-05T08:26:04.963120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process categorical data\nWe'll use OneHotEncode for categorical data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# Create an instance of the encoder\nencoder = OneHotEncoder()\n\n# Fit the encoder on features\nencoder.fit(features)\n\n# Transform the features into one-hot encoded format\none_hot_encode_features = encoder.transform(features)\n\n# Transform the test data(id_map)\none_hot_test = encoder.transform(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:04.964936Z","iopub.execute_input":"2023-10-05T08:26:04.965661Z","iopub.status.idle":"2023-10-05T08:26:05.280162Z","shell.execute_reply.started":"2023-10-05T08:26:04.965632Z","shell.execute_reply":"2023-10-05T08:26:05.279262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check shape\none_hot_encode_features.toarray().shape, one_hot_test.toarray().shape","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.281568Z","iopub.execute_input":"2023-10-05T08:26:05.282141Z","iopub.status.idle":"2023-10-05T08:26:05.291238Z","shell.execute_reply.started":"2023-10-05T08:26:05.282104Z","shell.execute_reply":"2023-10-05T08:26:05.290151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check one sample\none_hot_encode_features.toarray()[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.293163Z","iopub.execute_input":"2023-10-05T08:26:05.293924Z","iopub.status.idle":"2023-10-05T08:26:05.303872Z","shell.execute_reply.started":"2023-10-05T08:26:05.293886Z","shell.execute_reply":"2023-10-05T08:26:05.303063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Data into training, validation and test sets ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into 70% training, 15% validation, and 15% testing\nX_train, X_temp, y_train, y_temp = train_test_split(one_hot_encode_features, labels.values, test_size=0.3, shuffle=False)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.304771Z","iopub.execute_input":"2023-10-05T08:26:05.305451Z","iopub.status.idle":"2023-10-05T08:26:05.496936Z","shell.execute_reply.started":"2023-10-05T08:26:05.305425Z","shell.execute_reply":"2023-10-05T08:26:05.495678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the shapes of the data splits\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_val shape:\", X_val.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_val shape:\", y_val.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.498229Z","iopub.execute_input":"2023-10-05T08:26:05.498528Z","iopub.status.idle":"2023-10-05T08:26:05.504349Z","shell.execute_reply.started":"2023-10-05T08:26:05.498502Z","shell.execute_reply":"2023-10-05T08:26:05.503456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We also get full features for final training \nfull_features = one_hot_encode_features.toarray()\nfull_labels = labels.values","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.505374Z","iopub.execute_input":"2023-10-05T08:26:05.505834Z","iopub.status.idle":"2023-10-05T08:26:05.521842Z","shell.execute_reply.started":"2023-10-05T08:26:05.505809Z","shell.execute_reply":"2023-10-05T08:26:05.520799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"full_features shape:\", full_features.shape)\nprint(\"full_labels shape:\", full_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.523247Z","iopub.execute_input":"2023-10-05T08:26:05.523529Z","iopub.status.idle":"2023-10-05T08:26:05.535099Z","shell.execute_reply.started":"2023-10-05T08:26:05.523505Z","shell.execute_reply":"2023-10-05T08:26:05.534297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Auxillary functions ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\ndef create_model_checkpoint(filepath, monitor='val_mae', save_best_only=True,\n                            save_weights_only=True, mode='auto', verbose=0):\n    \"\"\"\n    Create a ModelCheckpoint callback for saving the best model weights during training.\n\n    Args:\n        filepath (str): Filepath to save the best weights.\n        monitor (str): Metric to monitor (e.g., 'val_loss' or 'val_mae').\n        save_best_only (bool): Save only the best weights.\n        save_weights_only (bool): Save only the model's weights, not the entire model.\n        mode (str): One of {'auto', 'min', 'max'}. In 'min' mode, it saves when the monitored metric decreases.\n        verbose (int): Verbosity mode. 0 = silent, 1 = progress bar, 2 = one line per epoch.\n\n    Returns:\n\n        keras.callbacks.ModelCheckpoint: ModelCheckpoint callback.\n    \"\"\"\n    checkpoint = ModelCheckpoint(\n        filepath=filepath,\n        monitor=monitor,\n        save_best_only=save_best_only,\n        save_weights_only=save_weights_only,\n        mode=mode,\n        verbose=verbose\n    )\n    return checkpoint","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.536408Z","iopub.execute_input":"2023-10-05T08:26:05.537313Z","iopub.status.idle":"2023-10-05T08:26:05.547438Z","shell.execute_reply.started":"2023-10-05T08:26:05.537277Z","shell.execute_reply":"2023-10-05T08:26:05.546805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, metrics):\n    \"\"\"\n    Plot training history curves for loss and evaluation metrics on the same line.\n\n    Args:\n        history (keras.callbacks.History): Training history object.\n        metrics (list): List of metric names to plot.\n\n    Returns:\n        None\n    \"\"\"\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n    plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot specified evaluation metrics on the same line\n    for metric in metrics:\n        train_metric_name = f'Training {metric.capitalize()}'\n        val_metric_name = f'Validation {metric.capitalize()}'\n        train_metric = history.history[metric]\n        val_metric = history.history['val_' + metric]\n\n        plt.subplot(1, 2, 2)\n        plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n        plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n\n    plt.title('Metrics')\n    plt.xlabel('Epochs')\n    plt.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.548303Z","iopub.execute_input":"2023-10-05T08:26:05.549187Z","iopub.status.idle":"2023-10-05T08:26:05.560528Z","shell.execute_reply.started":"2023-10-05T08:26:05.549152Z","shell.execute_reply":"2023-10-05T08:26:05.559870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\ndef calculate_mae_and_mrrmse(model, data, y_true):\n    \"\"\"\n    Calculate Mean Absolute Error (MAE) and Mean Rowwise Root Mean Squared Error (MRRMSE).\n\n    Parameters:\n    - model: The trained  model.\n    - data: The input data for prediction.\n    - y_true: The true target values.\n    - scaler: The scaler used for data normalization.\n\n    Returns:\n    - None\n    \"\"\"\n    # Predict using the model\n    y_pred_original = model.predict(data, batch_size=1)\n    \n    # Calculate Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_true , y_pred_original)\n    \n    # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    \n    # Print the results\n    print(f\"Mean Absolute Error (MAE): {mae}\")\n    print(f\"Mean Rowwise Root Mean Squared Error (MRRMSE): {mrrmse_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.561654Z","iopub.execute_input":"2023-10-05T08:26:05.562187Z","iopub.status.idle":"2023-10-05T08:26:05.578271Z","shell.execute_reply.started":"2023-10-05T08:26:05.562162Z","shell.execute_reply":"2023-10-05T08:26:05.577491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_rowwise_rmse_loss(y_true, y_pred):\n    \"\"\"\n    Custom loss function to calculate the Mean Rowwise Root Mean Squared Error (RMSE) loss.\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE loss as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.579596Z","iopub.execute_input":"2023-10-05T08:26:05.580003Z","iopub.status.idle":"2023-10-05T08:26:05.594783Z","shell.execute_reply.started":"2023-10-05T08:26:05.579975Z","shell.execute_reply":"2023-10-05T08:26:05.593947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_mean_rowwise_rmse(y_true, y_pred):\n    \"\"\"\n    Custom metric to calculate the Mean Rowwise Root Mean Squared Error (RMSE).\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.595846Z","iopub.execute_input":"2023-10-05T08:26:05.596826Z","iopub.status.idle":"2023-10-05T08:26:05.607070Z","shell.execute_reply.started":"2023-10-05T08:26:05.596797Z","shell.execute_reply":"2023-10-05T08:26:05.606023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Beautiful functions let's start experiment with building models","metadata":{}},{"cell_type":"markdown","source":"# Build the model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Activation\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.608113Z","iopub.execute_input":"2023-10-05T08:26:05.608777Z","iopub.status.idle":"2023-10-05T08:26:05.621809Z","shell.execute_reply.started":"2023-10-05T08:26:05.608750Z","shell.execute_reply":"2023-10-05T08:26:05.620790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Model_0:** Let's start with only 2 dense layers ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_0 = Sequential([\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_0.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_0 = model_0.fit(X_train, y_train,\n                       epochs=10,\n                       validation_data=(X_val,y_val),\n                       batch_size=32,\n                       callbacks=[create_model_checkpoint(\"model_0\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:05.629109Z","iopub.execute_input":"2023-10-05T08:26:05.629423Z","iopub.status.idle":"2023-10-05T08:26:34.508004Z","shell.execute_reply.started":"2023-10-05T08:26:05.629396Z","shell.execute_reply":"2023-10-05T08:26:34.506893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_0.load_weights(\"model_0\")\ncalculate_mae_and_mrrmse(model=model_0, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:34.509576Z","iopub.execute_input":"2023-10-05T08:26:34.510104Z","iopub.status.idle":"2023-10-05T08:26:35.231181Z","shell.execute_reply.started":"2023-10-05T08:26:34.510075Z","shell.execute_reply":"2023-10-05T08:26:35.229769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_0, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:35.232545Z","iopub.execute_input":"2023-10-05T08:26:35.232947Z","iopub.status.idle":"2023-10-05T08:26:40.707319Z","shell.execute_reply.started":"2023-10-05T08:26:35.232909Z","shell.execute_reply":"2023-10-05T08:26:40.706452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_0, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:40.708464Z","iopub.execute_input":"2023-10-05T08:26:40.709495Z","iopub.status.idle":"2023-10-05T08:26:41.280137Z","shell.execute_reply.started":"2023-10-05T08:26:40.709465Z","shell.execute_reply":"2023-10-05T08:26:41.279024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, it seems the model is not learning much and is underfitting. Let's try something different. ","metadata":{}},{"cell_type":"markdown","source":"> **model_1:** Increase layers and neurons ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_1 = Sequential([\n    Dense(1024, activation=\"tanh\"),\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_1.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_1 = model_1.fit(X_train, y_train,\n                       epochs=30,\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_1\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:26:41.281247Z","iopub.execute_input":"2023-10-05T08:26:41.281515Z","iopub.status.idle":"2023-10-05T08:27:54.940041Z","shell.execute_reply.started":"2023-10-05T08:26:41.281491Z","shell.execute_reply":"2023-10-05T08:27:54.938929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_1.load_weights(\"model_1\")\ncalculate_mae_and_mrrmse(model=model_1, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:27:55.593297Z","iopub.execute_input":"2023-10-05T08:27:55.594242Z","iopub.status.idle":"2023-10-05T08:27:56.378036Z","shell.execute_reply.started":"2023-10-05T08:27:55.594206Z","shell.execute_reply":"2023-10-05T08:27:56.377122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_1, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:27:56.379359Z","iopub.execute_input":"2023-10-05T08:27:56.380173Z","iopub.status.idle":"2023-10-05T08:28:01.845014Z","shell.execute_reply.started":"2023-10-05T08:27:56.380144Z","shell.execute_reply":"2023-10-05T08:28:01.843783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_1, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:28:01.846674Z","iopub.execute_input":"2023-10-05T08:28:01.847339Z","iopub.status.idle":"2023-10-05T08:28:02.281889Z","shell.execute_reply.started":"2023-10-05T08:28:01.847301Z","shell.execute_reply":"2023-10-05T08:28:02.281104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The model is not performing good let's increase complexity and also add Dropout and BatchNormalization layers","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_2 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(18211,activation= \"linear\")\n])\n\n\nmodel_2.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[\"mae\"])\n\nhistory_2 = model_2.fit(X_train, y_train,\n                       epochs=30,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_2\", monitor=\"val_mae\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:28:02.283158Z","iopub.execute_input":"2023-10-05T08:28:02.283640Z","iopub.status.idle":"2023-10-05T08:28:10.372152Z","shell.execute_reply.started":"2023-10-05T08:28:02.283613Z","shell.execute_reply":"2023-10-05T08:28:10.371101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2.load_weights(\"model_2\")\ncalculate_mae_and_mrrmse(model=model_2, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:28:10.701170Z","iopub.execute_input":"2023-10-05T08:28:10.701678Z","iopub.status.idle":"2023-10-05T08:28:10.928975Z","shell.execute_reply.started":"2023-10-05T08:28:10.701649Z","shell.execute_reply":"2023-10-05T08:28:10.927837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_2, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:28:10.930340Z","iopub.execute_input":"2023-10-05T08:28:10.930762Z","iopub.status.idle":"2023-10-05T08:28:12.363197Z","shell.execute_reply.started":"2023-10-05T08:28:10.930702Z","shell.execute_reply":"2023-10-05T08:28:12.362035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Visualize the learning from our helper functions\nplot_training_history(history_2, metrics=[\"mae\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:28:12.364690Z","iopub.execute_input":"2023-10-05T08:28:12.365058Z","iopub.status.idle":"2023-10-05T08:28:12.830260Z","shell.execute_reply.started":"2023-10-05T08:28:12.365033Z","shell.execute_reply":"2023-10-05T08:28:12.829231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Changing metrics and using same model","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n# clone model 2\nmodel_3 = tf.keras.models.clone_model(model_2)\n\nmodel_3.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(learning_rate=0.0027),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_3 = model_3.fit(X_train, y_train,\n                       epochs=25,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_3\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:48:27.845103Z","iopub.execute_input":"2023-10-05T08:48:27.845472Z","iopub.status.idle":"2023-10-05T08:48:35.381614Z","shell.execute_reply.started":"2023-10-05T08:48:27.845442Z","shell.execute_reply":"2023-10-05T08:48:35.380304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_3.load_weights(\"model_3\")\ncalculate_mae_and_mrrmse(model=model_3, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:48:00.444789Z","iopub.execute_input":"2023-10-05T08:48:00.445195Z","iopub.status.idle":"2023-10-05T08:48:00.937376Z","shell.execute_reply.started":"2023-10-05T08:48:00.445158Z","shell.execute_reply":"2023-10-05T08:48:00.936342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_3, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:48:03.436918Z","iopub.execute_input":"2023-10-05T08:48:03.437257Z","iopub.status.idle":"2023-10-05T08:48:05.126573Z","shell.execute_reply.started":"2023-10-05T08:48:03.437230Z","shell.execute_reply":"2023-10-05T08:48:05.125442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_3, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:48:07.006448Z","iopub.execute_input":"2023-10-05T08:48:07.006857Z","iopub.status.idle":"2023-10-05T08:48:07.454792Z","shell.execute_reply.started":"2023-10-05T08:48:07.006825Z","shell.execute_reply":"2023-10-05T08:48:07.453752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-fold validation ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n# clone model 2\nmodel_4 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.1),\n    Dense(18211,activation= \"linear\")\n])\n\nmodel_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:53:00.355564Z","iopub.execute_input":"2023-10-05T08:53:00.356005Z","iopub.status.idle":"2023-10-05T08:53:00.409230Z","shell.execute_reply.started":"2023-10-05T08:53:00.355959Z","shell.execute_reply":"2023-10-05T08:53:00.408186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n# Define the number of folds (K)\nnum_folds = 5 # You can change this value as needed\n\n# Initialize lists to store the model's performance scores\nmae_scores = []\nmrrmse_scores = []\n\n# Initialize the KFold object\nkf = KFold(n_splits=num_folds, shuffle=True, random_state=51)\n\n# Loop through the K folds\nfor train_index, val_index in kf.split(full_features):\n    # Convert indices to integers and split the data\n    train_index = train_index.astype(int)\n    val_index = val_index.astype(int)\n    X_train_, X_val_ = full_features[train_index], full_features[val_index]\n    y_train_, y_val_ = full_labels[train_index], full_labels[val_index]\n\n    # Train your model on X_train and y_train\n    model_4.fit(X_train_, y_train_, epochs=50, verbose=0)\n\n    # Make predictions on the validation set\n    y_preds = model_4.predict(X_val_)\n\n    # Calculate the Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_val_, y_preds)\n    mae_scores.append(mae)\n\n    # Calculate the Mean Rowwise Root Mean Square Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_val_ - y_preds), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    mrrmse_scores.append(mrrmse_score)\n\n# Calculate the mean and standard deviation of MAE and MRRMSE scores\nmean_mae = np.mean(mae_scores)\nmean_mrrmse = np.mean(mrrmse_scores)\n\n# Print the results\nprint(f'Average MAE across {num_folds} folds: {mean_mae:.4f} ')\nprint(f'Average MRRMSE across {num_folds} folds: {mean_mrrmse:.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:49:57.422546Z","iopub.execute_input":"2023-10-05T08:49:57.423226Z","iopub.status.idle":"2023-10-05T08:50:33.841544Z","shell.execute_reply.started":"2023-10-05T08:49:57.423192Z","shell.execute_reply":"2023-10-05T08:50:33.840519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train the model on full data","metadata":{}},{"cell_type":"code","source":"model_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_4 = model_4.fit(full_features, full_labels,\n                       epochs=50,\n                       verbose=0)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:56:31.868772Z","iopub.execute_input":"2023-10-05T08:56:31.869458Z","iopub.status.idle":"2023-10-05T08:56:54.205310Z","shell.execute_reply.started":"2023-10-05T08:56:31.869424Z","shell.execute_reply":"2023-10-05T08:56:54.204379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_4, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:56:57.447232Z","iopub.execute_input":"2023-10-05T08:56:57.447580Z","iopub.status.idle":"2023-10-05T08:56:58.845021Z","shell.execute_reply.started":"2023-10-05T08:56:57.447551Z","shell.execute_reply":"2023-10-05T08:56:58.843816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting on test data ","metadata":{}},{"cell_type":"code","source":"preds = model_4.predict(one_hot_test.toarray(), batch_size=1)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:59:43.623337Z","iopub.execute_input":"2023-10-05T08:59:43.624472Z","iopub.status.idle":"2023-10-05T08:59:44.095361Z","shell.execute_reply.started":"2023-10-05T08:59:43.624423Z","shell.execute_reply":"2023-10-05T08:59:44.094483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape ","metadata":{"execution":{"iopub.status.busy":"2023-10-05T08:59:52.622284Z","iopub.execute_input":"2023-10-05T08:59:52.622636Z","iopub.status.idle":"2023-10-05T08:59:52.628991Z","shell.execute_reply.started":"2023-10-05T08:59:52.622609Z","shell.execute_reply":"2023-10-05T08:59:52.627810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(preds, columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:01:26.874645Z","iopub.execute_input":"2023-10-05T09:01:26.875005Z","iopub.status.idle":"2023-10-05T09:01:26.880016Z","shell.execute_reply.started":"2023-10-05T09:01:26.874979Z","shell.execute_reply":"2023-10-05T09:01:26.878795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:01:33.709582Z","iopub.execute_input":"2023-10-05T09:01:33.710903Z","iopub.status.idle":"2023-10-05T09:01:33.719478Z","shell.execute_reply.started":"2023-10-05T09:01:33.710855Z","shell.execute_reply":"2023-10-05T09:01:33.718584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:01:43.917166Z","iopub.execute_input":"2023-10-05T09:01:43.917565Z","iopub.status.idle":"2023-10-05T09:01:43.956865Z","shell.execute_reply.started":"2023-10-05T09:01:43.917534Z","shell.execute_reply":"2023-10-05T09:01:43.955816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:01:50.746577Z","iopub.execute_input":"2023-10-05T09:01:50.747687Z","iopub.status.idle":"2023-10-05T09:01:50.774061Z","shell.execute_reply.started":"2023-10-05T09:01:50.747649Z","shell.execute_reply":"2023-10-05T09:01:50.773008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission_df.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:03:29.331673Z","iopub.execute_input":"2023-10-05T09:03:29.332187Z","iopub.status.idle":"2023-10-05T09:03:36.082868Z","shell.execute_reply.started":"2023-10-05T09:03:29.332138Z","shell.execute_reply":"2023-10-05T09:03:36.082041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission_preds.zip /kaggle/working/submission_df.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:21:15.755210Z","iopub.execute_input":"2023-10-05T09:21:15.755619Z","iopub.status.idle":"2023-10-05T09:21:23.231086Z","shell.execute_reply.started":"2023-10-05T09:21:15.755583Z","shell.execute_reply":"2023-10-05T09:21:23.229647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Improving model**\n\n* Try out different layers and number of neurons in layer and see how it goes.\n* Experiment with different activation functions.\n* Experiment with different architecture, try deep or wide architecture.\n* Find the best learning rate for the model.\n* Normalize labels and try optimizing.\n* Validate model using different techniques like K-fold validation.\n* Experiment.. Experiment.. Experiment.. Visualize.. Visualize.. Visualize..\n\nGood Luck :)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:14:36.993246Z","iopub.execute_input":"2023-10-05T09:14:36.993577Z","iopub.status.idle":"2023-10-05T09:14:37.000183Z","shell.execute_reply.started":"2023-10-05T09:14:36.993550Z","shell.execute_reply":"2023-10-05T09:14:36.999213Z"}}}]}