{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Neural Network Regression  \nВ этом блокноте цель состоит в том, чтобы использовать возможности нейронных сетей и получить наименьший MRRMSE(Mean Rowwise Root Mean Sqared Error)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:13.167574Z","iopub.execute_input":"2023-10-27T10:50:13.168080Z","iopub.status.idle":"2023-10-27T10:50:26.313157Z","shell.execute_reply.started":"2023-10-27T10:50:13.168030Z","shell.execute_reply":"2023-10-27T10:50:26.311636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read data ","metadata":{}},{"cell_type":"code","source":"de_train =   pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:26.315533Z","iopub.execute_input":"2023-10-27T10:50:26.316230Z","iopub.status.idle":"2023-10-27T10:50:34.004614Z","shell.execute_reply.started":"2023-10-27T10:50:26.316192Z","shell.execute_reply":"2023-10-27T10:50:34.003075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Become one with data","metadata":{}},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.006763Z","iopub.execute_input":"2023-10-27T10:50:34.008339Z","iopub.status.idle":"2023-10-27T10:50:34.067103Z","shell.execute_reply.started":"2023-10-27T10:50:34.008033Z","shell.execute_reply":"2023-10-27T10:50:34.065688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.070117Z","iopub.execute_input":"2023-10-27T10:50:34.070477Z","iopub.status.idle":"2023-10-27T10:50:34.082298Z","shell.execute_reply.started":"2023-10-27T10:50:34.070444Z","shell.execute_reply":"2023-10-27T10:50:34.081173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.084049Z","iopub.execute_input":"2023-10-27T10:50:34.084493Z","iopub.status.idle":"2023-10-27T10:50:34.138407Z","shell.execute_reply.started":"2023-10-27T10:50:34.084459Z","shell.execute_reply":"2023-10-27T10:50:34.137082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique= de_train['sm_name'].unique()\nunique, print(f\"Count {len(unique)}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.139695Z","iopub.execute_input":"2023-10-27T10:50:34.140029Z","iopub.status.idle":"2023-10-27T10:50:34.154740Z","shell.execute_reply.started":"2023-10-27T10:50:34.139999Z","shell.execute_reply":"2023-10-27T10:50:34.153381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.156026Z","iopub.execute_input":"2023-10-27T10:50:34.156580Z","iopub.status.idle":"2023-10-27T10:50:34.165356Z","shell.execute_reply.started":"2023-10-27T10:50:34.156537Z","shell.execute_reply":"2023-10-27T10:50:34.164102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train[de_train[\"control\"] == True].count  ","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.166806Z","iopub.execute_input":"2023-10-27T10:50:34.167176Z","iopub.status.idle":"2023-10-27T10:50:34.195563Z","shell.execute_reply.started":"2023-10-27T10:50:34.167135Z","shell.execute_reply":"2023-10-27T10:50:34.194343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Больше визуализации и пристального изучения данных в течение некоторого времени будет более полезно для понимания, но давайте попробуем построить какую-нибудь простую модель.","metadata":{}},{"cell_type":"markdown","source":"# Preprocess data \nМы возьмем только 'cell_type' и 'sm_name' в качестве признаков и значения гена 18211 в качестве меток.","metadata":{}},{"cell_type":"code","source":"# перетасуем данные\nde_train = de_train.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.197414Z","iopub.execute_input":"2023-10-27T10:50:34.197840Z","iopub.status.idle":"2023-10-27T10:50:34.249640Z","shell.execute_reply.started":"2023-10-27T10:50:34.197806Z","shell.execute_reply":"2023-10-27T10:50:34.248305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.253986Z","iopub.execute_input":"2023-10-27T10:50:34.254387Z","iopub.status.idle":"2023-10-27T10:50:34.295601Z","shell.execute_reply.started":"2023-10-27T10:50:34.254353Z","shell.execute_reply":"2023-10-27T10:50:34.294364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Создаем элементы и метки для обратной модели 18211 элементов и 152 метки для истинной модели\nfeatures_columns = [\"cell_type\", \"sm_name\"]\nlabels_columns=[\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"]\nlabels = de_train.drop(columns=labels_columns)\nfeatures = pd.DataFrame(de_train, columns=features_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.297358Z","iopub.execute_input":"2023-10-27T10:50:34.297731Z","iopub.status.idle":"2023-10-27T10:50:34.339433Z","shell.execute_reply.started":"2023-10-27T10:50:34.297698Z","shell.execute_reply":"2023-10-27T10:50:34.338100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.341357Z","iopub.execute_input":"2023-10-27T10:50:34.341855Z","iopub.status.idle":"2023-10-27T10:50:34.358076Z","shell.execute_reply.started":"2023-10-27T10:50:34.341806Z","shell.execute_reply":"2023-10-27T10:50:34.356706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.360190Z","iopub.execute_input":"2023-10-27T10:50:34.360674Z","iopub.status.idle":"2023-10-27T10:50:34.401590Z","shell.execute_reply.started":"2023-10-27T10:50:34.360631Z","shell.execute_reply":"2023-10-27T10:50:34.400350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get test data \ntest_data = pd.DataFrame(id_map, columns=features_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.403238Z","iopub.execute_input":"2023-10-27T10:50:34.404019Z","iopub.status.idle":"2023-10-27T10:50:34.410682Z","shell.execute_reply.started":"2023-10-27T10:50:34.403977Z","shell.execute_reply":"2023-10-27T10:50:34.409335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Обработка категориальных данных\nМы будем использовать OneHotEncoder для категориальных данных","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# Create an instance of the encoder\nencoder = OneHotEncoder()\n\n# Fit the encoder on features\nencoder.fit(features)\n\n# Transform the features into one-hot encoded format\none_hot_encode_features = encoder.transform(features)\n\n# Transform the test data(id_map)\none_hot_test = encoder.transform(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.412158Z","iopub.execute_input":"2023-10-27T10:50:34.412525Z","iopub.status.idle":"2023-10-27T10:50:34.813724Z","shell.execute_reply.started":"2023-10-27T10:50:34.412486Z","shell.execute_reply":"2023-10-27T10:50:34.812598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check shape\none_hot_encode_features.toarray().shape, one_hot_test.toarray().shape","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.815390Z","iopub.execute_input":"2023-10-27T10:50:34.815745Z","iopub.status.idle":"2023-10-27T10:50:34.826691Z","shell.execute_reply.started":"2023-10-27T10:50:34.815714Z","shell.execute_reply":"2023-10-27T10:50:34.825131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check one sample\none_hot_encode_features.toarray()[0]","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.828623Z","iopub.execute_input":"2023-10-27T10:50:34.829088Z","iopub.status.idle":"2023-10-27T10:50:34.840341Z","shell.execute_reply.started":"2023-10-27T10:50:34.829052Z","shell.execute_reply":"2023-10-27T10:50:34.839359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Разделение данных на training, validation и test сеты","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into 60% training, 20% validation, and 20% testing\nX_train, X_temp, y_train, y_temp = train_test_split(one_hot_encode_features, labels.values, test_size=0.5, shuffle=False)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:34.841473Z","iopub.execute_input":"2023-10-27T10:50:34.841827Z","iopub.status.idle":"2023-10-27T10:50:35.076836Z","shell.execute_reply.started":"2023-10-27T10:50:34.841796Z","shell.execute_reply":"2023-10-27T10:50:35.075572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the shapes of the data splits\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_val shape:\", X_val.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_val shape:\", y_val.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:35.078689Z","iopub.execute_input":"2023-10-27T10:50:35.079059Z","iopub.status.idle":"2023-10-27T10:50:35.087215Z","shell.execute_reply.started":"2023-10-27T10:50:35.079025Z","shell.execute_reply":"2023-10-27T10:50:35.086303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We also get full features for final training \nfull_features = one_hot_encode_features.toarray()\nfull_labels = labels.values","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:35.088935Z","iopub.execute_input":"2023-10-27T10:50:35.089313Z","iopub.status.idle":"2023-10-27T10:50:35.102293Z","shell.execute_reply.started":"2023-10-27T10:50:35.089268Z","shell.execute_reply":"2023-10-27T10:50:35.100678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"full_features shape:\", full_features.shape)\nprint(\"full_labels shape:\", full_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:35.104082Z","iopub.execute_input":"2023-10-27T10:50:35.104738Z","iopub.status.idle":"2023-10-27T10:50:35.117536Z","shell.execute_reply.started":"2023-10-27T10:50:35.104700Z","shell.execute_reply":"2023-10-27T10:50:35.116064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Модели ML","metadata":{}},{"cell_type":"code","source":"# Linear Regression\nfrom sklearn.linear_model import LinearRegression\nlinear_reg_model = LinearRegression()\n\n# Logistic Regression\nfrom sklearn.linear_model import LogisticRegression\nlogistic_reg_model = LogisticRegression()\n\n# Decision Tree\nfrom sklearn.tree import DecisionTreeClassifier\ndecision_tree_model = DecisionTreeClassifier()\n\n# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nrandom_forest_model = RandomForestClassifier()\n\n# Support Vector Machine (SVM)\nfrom sklearn.svm import SVC\nsvm_model = SVC()\n\n# K-Nearest Neighbors (KNN)\nfrom sklearn.neighbors import KNeighborsClassifier\nknn_model = KNeighborsClassifier()\n\n# K-Means Clustering\nfrom sklearn.cluster import KMeans\nkmeans_model = KMeans(n_clusters=2)  # Assuming 2 clusters\n\n# Naive Bayes\nfrom sklearn.naive_bayes import GaussianNB\nnaive_bayes_model = GaussianNB()\n\n# Neural Network (Multi-layer Perceptron)\nfrom sklearn.neural_network import MLPClassifier\nnn_model = MLPClassifier(max_iter=1000)  # Assuming 1000 iterations","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:35.119501Z","iopub.execute_input":"2023-10-27T10:50:35.119986Z","iopub.status.idle":"2023-10-27T10:50:35.756602Z","shell.execute_reply.started":"2023-10-27T10:50:35.119942Z","shell.execute_reply":"2023-10-27T10:50:35.755090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"linear_reg_model.fit(X_train, y_train)\n#logistic_reg_model.fit(X_train, y_train)\n#decision_tree_model.fit(X_train, y_train)\n#random_forest_model.fit(X_train, y_train)\n#svm_model.fit(X_train, y_train)\n#knn_model.fit(X_train, y_train)\n#kmeans_model.fit(X_train)\n#naive_bayes_model.fit(X_train, y_train)\n#nn_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:50:35.758392Z","iopub.execute_input":"2023-10-27T10:50:35.759336Z","iopub.status.idle":"2023-10-27T10:51:52.461563Z","shell.execute_reply.started":"2023-10-27T10:50:35.759271Z","shell.execute_reply":"2023-10-27T10:51:52.459652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"linear_reg_predictions = linear_reg_model.predict(X_val)\n#logistic_reg_predictions = logistic_reg_model.predict(X_val)\n#decision_tree_predictions = decision_tree_model.predict(X_val)\n#random_forest_predictions = random_forest_model.predict(X_val)\n#svm_predictions = svm_model.predict(X_val)\n#knn_predictions = knn_model.predict(X_val)\n#kmeans_predictions = kmeans_model.predict(X_est)\n#naive_bayes_predictions = naive_bayes_model.predict(X_test)\n#nn_predictions = nn_model.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.463993Z","iopub.execute_input":"2023-10-27T10:51:52.465110Z","iopub.status.idle":"2023-10-27T10:51:52.569295Z","shell.execute_reply.started":"2023-10-27T10:51:52.465042Z","shell.execute_reply":"2023-10-27T10:51:52.567881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, mean_absolute_error, mean_squared_error, r2_score\n\n# Define a function to print results\ndef print_results(model_name, y_true, y_pred):\n    print(f\"Results for {model_name}:\")\n    \n    if isinstance(y_pred[0], int):  # Classification\n        accuracy = accuracy_score(y_true, y_pred)\n        precision = precision_score(y_true, y_pred)\n        recall = recall_score(y_true, y_pred)\n        f1 = f1_score(y_true, y_pred)\n        \n        print(f\"Accuracy: {accuracy}\")\n        print(f\"Precision: {precision}\")\n        print(f\"Recall: {recall}\")\n        print(f\"F1 Score: {f1}\")\n    else:  # Regression\n        mae = mean_absolute_error(y_true, y_pred)\n        mse = mean_squared_error(y_true, y_pred)\n        rmse = mse*0.5\n        r2 = r2_score(y_true, y_pred)\n        \n        print(f\"Mean Absolute Error: {mae}\")\n        print(f\"Mean Squared Error: {mse}\")\n        print(f\"RMSE: {rmse}\")\n        print(f\"R-squared (R2): {r2}\")\n\n# Define model names and predictions\n#model_names = [\"Linear Regression\", \"Logistic Regression\", \"Decision Tree\", \n               #\"Random Forest\", \"SVM\", \"KNN\", \"K-Means\", \n               #\"Naive Bayes\", \"Neural Network\"]\n#predictions = [linear_reg_predictions, logistic_reg_predictions.astype(int), \n               #decision_tree_predictions.astype(int), random_forest_predictions.astype(int), \n               #svm_predictions.astype(int), knn_predictions.astype(int), \n               #kmeans_predictions.astype(int), naive_bayes_predictions.astype(int), \n               #nn_predictions.astype(int)]\n\n# Iterate through models and print results\n#for model_name, y_pred in zip(model_names, predictions):\n#    print_results(model_name, y_val, y_pred)\n#    print(\"\\n\" + \"=\"*30 + \"\\n\")  # Add a separator for better visibility\n    \nprint_results(\"Linear Regression\", y_val, linear_reg_predictions)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.570872Z","iopub.execute_input":"2023-10-27T10:51:52.571293Z","iopub.status.idle":"2023-10-27T10:51:52.665261Z","shell.execute_reply.started":"2023-10-27T10:51:52.571241Z","shell.execute_reply":"2023-10-27T10:51:52.663934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Auxillary functions ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\ndef create_model_checkpoint(filepath, monitor='val_mae', save_best_only=True,\n                            save_weights_only=True, mode='auto', verbose=0):\n    \"\"\"\n    Create a ModelCheckpoint callback for saving the best model weights during training.\n\n    Args:\n        filepath (str): Filepath to save the best weights.\n        monitor (str): Metric to monitor (e.g., 'val_loss' or 'val_mae').\n        save_best_only (bool): Save only the best weights.\n        save_weights_only (bool): Save only the model's weights, not the entire model.\n        mode (str): One of {'auto', 'min', 'max'}. In 'min' mode, it saves when the monitored metric decreases.\n        verbose (int): Verbosity mode. 0 = silent, 1 = progress bar, 2 = one line per epoch.\n\n    Returns:\n\n        keras.callbacks.ModelCheckpoint: ModelCheckpoint callback.\n    \"\"\"\n    checkpoint = ModelCheckpoint(\n        filepath=filepath,\n        monitor=monitor,\n        save_best_only=save_best_only,\n        save_weights_only=save_weights_only,\n        mode=mode,\n        verbose=verbose\n    )\n    return checkpoint","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.667429Z","iopub.execute_input":"2023-10-27T10:51:52.667945Z","iopub.status.idle":"2023-10-27T10:51:52.678487Z","shell.execute_reply.started":"2023-10-27T10:51:52.667898Z","shell.execute_reply":"2023-10-27T10:51:52.677060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, metrics):\n    \"\"\"\n    Plot training history curves for loss and evaluation metrics on the same line.\n\n    Args:\n        history (keras.callbacks.History): Training history object.\n        metrics (list): List of metric names to plot.\n\n    Returns:\n        None\n    \"\"\"\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='Training Loss', color=\"blue\")\n    plt.plot(epochs, val_loss, label='Validation Loss', color=\"red\")\n    plt.title('Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot specified evaluation metrics on the same line\n    for metric in metrics:\n        train_metric_name = f'Training {metric.capitalize()}'\n        val_metric_name = f'Validation {metric.capitalize()}'\n        train_metric = history.history[metric]\n        val_metric = history.history['val_' + metric]\n\n        plt.subplot(1, 2, 2)\n        plt.plot(epochs, train_metric, label=train_metric_name, color=\"green\")\n        plt.plot(epochs, val_metric, label=val_metric_name, color=\"orange\")\n\n    plt.title('Metrics')\n    plt.xlabel('Epochs')\n    plt.legend(loc='upper right')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.680492Z","iopub.execute_input":"2023-10-27T10:51:52.680911Z","iopub.status.idle":"2023-10-27T10:51:52.698618Z","shell.execute_reply.started":"2023-10-27T10:51:52.680875Z","shell.execute_reply":"2023-10-27T10:51:52.697177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\ndef calculate_mae_and_mrrmse(model, data, y_true):\n    \"\"\"\n    Calculate Mean Absolute Error (MAE) and Mean Rowwise Root Mean Squared Error (MRRMSE).\n\n    Parameters:\n    - model: The trained  model.\n    - data: The input data for prediction.\n    - y_true: The true target values.\n    - scaler: The scaler used for data normalization.\n\n    Returns:\n    - None\n    \"\"\"\n    # Predict using the model\n    y_pred_original = model.predict(data, batch_size=1)\n    \n    # Calculate Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_true , y_pred_original)\n    \n    # Calculate Mean Rowwise Root Mean Squared Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_true - y_pred_original), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    \n    # Print the results\n    print(f\"Mean Absolute Error (MAE): {mae}\")\n    print(f\"Mean Rowwise Root Mean Squared Error (MRRMSE): {mrrmse_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.707782Z","iopub.execute_input":"2023-10-27T10:51:52.708197Z","iopub.status.idle":"2023-10-27T10:51:52.719521Z","shell.execute_reply.started":"2023-10-27T10:51:52.708163Z","shell.execute_reply":"2023-10-27T10:51:52.717870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_rowwise_rmse_loss(y_true, y_pred):\n    \"\"\"\n    Custom loss function to calculate the Mean Rowwise Root Mean Squared Error (RMSE) loss.\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE loss as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.721366Z","iopub.execute_input":"2023-10-27T10:51:52.721842Z","iopub.status.idle":"2023-10-27T10:51:52.741428Z","shell.execute_reply.started":"2023-10-27T10:51:52.721806Z","shell.execute_reply":"2023-10-27T10:51:52.740099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_mean_rowwise_rmse(y_true, y_pred):\n    \"\"\"\n    Custom metric to calculate the Mean Rowwise Root Mean Squared Error (RMSE).\n\n    Parameters:\n    - y_true: The true target values.\n    - y_pred: The predicted values.\n\n    Returns:\n    - Mean Rowwise RMSE as a scalar tensor.\n    \"\"\"\n    # Calculate RMSE for each row\n    rmse_per_row = tf.sqrt(tf.reduce_mean(tf.square(y_true - y_pred), axis=1))\n    # Calculate the mean of RMSE values across all rows\n    mean_rmse = tf.reduce_mean(rmse_per_row)\n    \n    return mean_rmse","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.743086Z","iopub.execute_input":"2023-10-27T10:51:52.744094Z","iopub.status.idle":"2023-10-27T10:51:52.754650Z","shell.execute_reply.started":"2023-10-27T10:51:52.744033Z","shell.execute_reply":"2023-10-27T10:51:52.753187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Beautiful functions let's start experiment with building models","metadata":{}},{"cell_type":"markdown","source":"# Build the model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Activation\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.756212Z","iopub.execute_input":"2023-10-27T10:51:52.757358Z","iopub.status.idle":"2023-10-27T10:51:52.770794Z","shell.execute_reply.started":"2023-10-27T10:51:52.757305Z","shell.execute_reply":"2023-10-27T10:51:52.769730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Model_0:** Let's start with only 2 dense layers ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_0 = Sequential([\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_0.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_0 = model_0.fit(X_train, y_train,\n                       epochs=10,\n                       validation_data=(X_val,y_val),\n                       batch_size=32,\n                       callbacks=[create_model_checkpoint(\"model_0\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:51:52.772250Z","iopub.execute_input":"2023-10-27T10:51:52.772867Z","iopub.status.idle":"2023-10-27T10:52:17.628201Z","shell.execute_reply.started":"2023-10-27T10:51:52.772833Z","shell.execute_reply":"2023-10-27T10:52:17.626908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_0.load_weights(\"model_0\")\ncalculate_mae_and_mrrmse(model=model_0, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:52:17.630245Z","iopub.execute_input":"2023-10-27T10:52:17.630668Z","iopub.status.idle":"2023-10-27T10:52:19.195408Z","shell.execute_reply.started":"2023-10-27T10:52:17.630632Z","shell.execute_reply":"2023-10-27T10:52:19.194021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_0, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T10:52:19.196914Z","iopub.execute_input":"2023-10-27T10:52:19.197413Z","iopub.status.idle":"2023-10-27T10:52:24.699744Z","shell.execute_reply.started":"2023-10-27T10:52:19.197374Z","shell.execute_reply":"2023-10-27T10:52:24.698466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nkeras.utils.plot_model(model_0, 'model_0_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:04:10.930881Z","iopub.execute_input":"2023-10-27T11:04:10.931345Z","iopub.status.idle":"2023-10-27T11:04:11.191022Z","shell.execute_reply.started":"2023-10-27T11:04:10.931301Z","shell.execute_reply":"2023-10-27T11:04:11.189872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_0, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:04:30.044460Z","iopub.execute_input":"2023-10-27T11:04:30.044913Z","iopub.status.idle":"2023-10-27T11:04:30.718974Z","shell.execute_reply.started":"2023-10-27T11:04:30.044868Z","shell.execute_reply":"2023-10-27T11:04:30.718003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, it seems the model is not learning much and is underfitting. Let's try something different. ","metadata":{}},{"cell_type":"markdown","source":"> **model_1:** Increase layers and neurons ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_1 = Sequential([\n    Dense(1024, activation=\"tanh\"),\n    Dense(512, activation=\"tanh\"),\n    Dense(18211, activation=\"linear\")\n])\n\nmodel_1.compile(loss=mean_rowwise_rmse_loss, \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_1 = model_1.fit(X_train, y_train,\n                       epochs=30,\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_1\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:04:38.525618Z","iopub.execute_input":"2023-10-27T11:04:38.526390Z","iopub.status.idle":"2023-10-27T11:05:44.062812Z","shell.execute_reply.started":"2023-10-27T11:04:38.526349Z","shell.execute_reply":"2023-10-27T11:05:44.061395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading weights \nmodel_1.load_weights(\"model_1\")\ncalculate_mae_and_mrrmse(model=model_1, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:05:45.411563Z","iopub.execute_input":"2023-10-27T11:05:45.412201Z","iopub.status.idle":"2023-10-27T11:05:46.863171Z","shell.execute_reply.started":"2023-10-27T11:05:45.412167Z","shell.execute_reply":"2023-10-27T11:05:46.862158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model performance on full data \ncalculate_mae_and_mrrmse(model=model_1, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:05:46.865627Z","iopub.execute_input":"2023-10-27T11:05:46.866984Z","iopub.status.idle":"2023-10-27T11:05:51.316002Z","shell.execute_reply.started":"2023-10-27T11:05:46.866905Z","shell.execute_reply":"2023-10-27T11:05:51.314678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model_1, 'model_1_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:13:06.251152Z","iopub.execute_input":"2023-10-27T11:13:06.251589Z","iopub.status.idle":"2023-10-27T11:13:06.317561Z","shell.execute_reply.started":"2023-10-27T11:13:06.251556Z","shell.execute_reply":"2023-10-27T11:13:06.316611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the learning from our helper functions\nplot_training_history(history_1, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:13:06.319686Z","iopub.execute_input":"2023-10-27T11:13:06.320023Z","iopub.status.idle":"2023-10-27T11:13:07.008804Z","shell.execute_reply.started":"2023-10-27T11:13:06.319995Z","shell.execute_reply":"2023-10-27T11:13:07.007178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The model is not performing good let's increase complexity and also add Dropout and BatchNormalization layers","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_2 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(18211,activation= \"linear\")\n])\n\n\nmodel_2.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[\"mae\"])\n\nhistory_2 = model_2.fit(X_train, y_train,\n                       epochs=30,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_2\", monitor=\"val_mae\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:13:07.010952Z","iopub.execute_input":"2023-10-27T11:13:07.011859Z","iopub.status.idle":"2023-10-27T11:13:16.179224Z","shell.execute_reply.started":"2023-10-27T11:13:07.011809Z","shell.execute_reply":"2023-10-27T11:13:16.177809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2.load_weights(\"model_2\")\ncalculate_mae_and_mrrmse(model=model_2, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:13:16.183082Z","iopub.execute_input":"2023-10-27T11:13:16.183647Z","iopub.status.idle":"2023-10-27T11:13:16.765712Z","shell.execute_reply.started":"2023-10-27T11:13:16.183598Z","shell.execute_reply":"2023-10-27T11:13:16.764464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_2, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:13:16.766888Z","iopub.execute_input":"2023-10-27T11:13:16.767211Z","iopub.status.idle":"2023-10-27T11:13:18.788328Z","shell.execute_reply.started":"2023-10-27T11:13:16.767183Z","shell.execute_reply":"2023-10-27T11:13:18.787268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model_2, 'model_2_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:22.752868Z","iopub.execute_input":"2023-10-27T11:14:22.753433Z","iopub.status.idle":"2023-10-27T11:14:22.896564Z","shell.execute_reply.started":"2023-10-27T11:14:22.753359Z","shell.execute_reply":"2023-10-27T11:14:22.894814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Visualize the learning from our helper functions\nplot_training_history(history_2, metrics=[\"mae\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:22.899735Z","iopub.execute_input":"2023-10-27T11:14:22.900256Z","iopub.status.idle":"2023-10-27T11:14:23.607641Z","shell.execute_reply.started":"2023-10-27T11:14:22.900209Z","shell.execute_reply":"2023-10-27T11:14:23.606198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Changing metrics and using same model","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n# clone model 2\nmodel_3 = tf.keras.models.clone_model(model_2)\n\nmodel_3.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(learning_rate=0.0027),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_3 = model_3.fit(X_train, y_train,\n                       epochs=25,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_3\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:23.609638Z","iopub.execute_input":"2023-10-27T11:14:23.610361Z","iopub.status.idle":"2023-10-27T11:14:32.889990Z","shell.execute_reply.started":"2023-10-27T11:14:23.610312Z","shell.execute_reply":"2023-10-27T11:14:32.888384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_3.load_weights(\"model_3\")\ncalculate_mae_and_mrrmse(model=model_3, data=X_test, y_true=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:32.895616Z","iopub.execute_input":"2023-10-27T11:14:32.896041Z","iopub.status.idle":"2023-10-27T11:14:33.552179Z","shell.execute_reply.started":"2023-10-27T11:14:32.896003Z","shell.execute_reply":"2023-10-27T11:14:33.550841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_3, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:33.555179Z","iopub.execute_input":"2023-10-27T11:14:33.555583Z","iopub.status.idle":"2023-10-27T11:14:36.671136Z","shell.execute_reply.started":"2023-10-27T11:14:33.555547Z","shell.execute_reply":"2023-10-27T11:14:36.669985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model_3, 'model_3_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:36.672672Z","iopub.execute_input":"2023-10-27T11:14:36.673573Z","iopub.status.idle":"2023-10-27T11:14:36.816843Z","shell.execute_reply.started":"2023-10-27T11:14:36.673523Z","shell.execute_reply":"2023-10-27T11:14:36.815542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_3, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:36.818092Z","iopub.execute_input":"2023-10-27T11:14:36.818827Z","iopub.status.idle":"2023-10-27T11:14:37.555986Z","shell.execute_reply.started":"2023-10-27T11:14:36.818792Z","shell.execute_reply":"2023-10-27T11:14:37.554757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-fold validation ","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\n#copy model_2\n\nmodel_4 = Sequential([ \n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.2),\n    Dense(128, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.2),\n    Dense(18211,activation= \"linear\")\n])\n\n\nmodel_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[\"mae\"])\n\nhistory_4 = model_4.fit(X_train, y_train,\n                       epochs=30,\n                       verbose=0, #train in silent mode\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_4\", monitor=\"val_mae\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:37.557776Z","iopub.execute_input":"2023-10-27T11:14:37.558609Z","iopub.status.idle":"2023-10-27T11:14:46.496049Z","shell.execute_reply.started":"2023-10-27T11:14:37.558568Z","shell.execute_reply":"2023-10-27T11:14:46.494755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n# Define the number of folds (K)\nnum_folds = 15 # You can change this value as needed\n\n# Initialize lists to store the model's performance scores\nmae_scores = []\nmrrmse_scores = []\n\n# Initialize the KFold object\nkf = KFold(n_splits=num_folds, shuffle=True, random_state=51)\n\n# Loop through the K folds\nfor train_index, val_index in kf.split(full_features):\n    # Convert indices to integers and split the data\n    train_index = train_index.astype(int)\n    val_index = val_index.astype(int)\n    X_train_, X_val_ = full_features[train_index], full_features[val_index]\n    y_train_, y_val_ = full_labels[train_index], full_labels[val_index]\n\n    # Train your model on X_train and y_train\n    model_4.fit(X_train_, y_train_, epochs=50, verbose=0)\n\n    # Make predictions on the validation set\n    y_preds = model_4.predict(X_val_)\n\n    # Calculate the Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_val_, y_preds)\n    mae_scores.append(mae)\n\n    # Calculate the Mean Rowwise Root Mean Square Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_val_ - y_preds), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    mrrmse_scores.append(mrrmse_score)\n\n# Calculate the mean and standard deviation of MAE and MRRMSE scores\nmean_mae = np.mean(mae_scores)\nmean_mrrmse = np.mean(mrrmse_scores)\n\n# Print the results\nprint(f'Average MAE across {num_folds} folds: {mean_mae:.4f} ')\nprint(f'Average MRRMSE across {num_folds} folds: {mean_mrrmse:.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:14:46.497732Z","iopub.execute_input":"2023-10-27T11:14:46.498085Z","iopub.status.idle":"2023-10-27T11:17:02.112016Z","shell.execute_reply.started":"2023-10-27T11:14:46.498053Z","shell.execute_reply":"2023-10-27T11:17:02.110787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_4.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_4 = model_4.fit(full_features, full_labels,\n                       epochs=50,\n                       verbose=0,\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_4\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:02.113843Z","iopub.execute_input":"2023-10-27T11:17:02.114193Z","iopub.status.idle":"2023-10-27T11:17:15.744269Z","shell.execute_reply.started":"2023-10-27T11:17:02.114162Z","shell.execute_reply":"2023-10-27T11:17:15.743061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_4, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:15.746802Z","iopub.execute_input":"2023-10-27T11:17:15.747252Z","iopub.status.idle":"2023-10-27T11:17:17.429884Z","shell.execute_reply.started":"2023-10-27T11:17:15.747209Z","shell.execute_reply":"2023-10-27T11:17:17.428662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model_4, 'model_4_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:17.431811Z","iopub.execute_input":"2023-10-27T11:17:17.432311Z","iopub.status.idle":"2023-10-27T11:17:17.576915Z","shell.execute_reply.started":"2023-10-27T11:17:17.432233Z","shell.execute_reply":"2023-10-27T11:17:17.575578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_4, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:17.578604Z","iopub.execute_input":"2023-10-27T11:17:17.579091Z","iopub.status.idle":"2023-10-27T11:17:18.310561Z","shell.execute_reply.started":"2023-10-27T11:17:17.579045Z","shell.execute_reply":"2023-10-27T11:17:18.309432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-fold validation for changed model","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(42)\n\nmodel_5 = Sequential([ \n    Dense(512),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.1),\n    Dense(256),\n    BatchNormalization(),\n    Activation(\"relu\"),\n    Dropout(0.3),\n    Dense(128, activation=\"elu\"),\n    Dropout(0.2),\n    Dense(64, activation=\"elu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(32, activation=\"relu\"),\n    Dropout(0.1),\n    Dense(16, activation=\"relu\"),\n    Dropout(0.05),\n    Dense(18211,activation= \"linear\")\n])\n\nmodel_5.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:18.315144Z","iopub.execute_input":"2023-10-27T11:17:18.315537Z","iopub.status.idle":"2023-10-27T11:17:18.407499Z","shell.execute_reply.started":"2023-10-27T11:17:18.315504Z","shell.execute_reply":"2023-10-27T11:17:18.406289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n# Define the number of folds (K)\nnum_folds = 42 # You can change this value as needed\n\n# Initialize lists to store the model's performance scores\nmae_scores = []\nmrrmse_scores = []\n\n# Initialize the KFold object\nkf = KFold(n_splits=num_folds, shuffle=True, random_state=51)\n\n# Loop through the K folds\nfor train_index, val_index in kf.split(full_features):\n    # Convert indices to integers and split the data\n    train_index = train_index.astype(int)\n    val_index = val_index.astype(int)\n    X_train_, X_val_ = full_features[train_index], full_features[val_index]\n    y_train_, y_val_ = full_labels[train_index], full_labels[val_index]\n\n    # Train your model on X_train and y_train\n    model_5.fit(X_train_, y_train_, epochs=50, verbose=0)\n\n    # Make predictions on the validation set\n    y_preds = model_5.predict(X_val_)\n\n    # Calculate the Mean Absolute Error (MAE)\n    mae = mean_absolute_error(y_val_, y_preds)\n    mae_scores.append(mae)\n\n    # Calculate the Mean Rowwise Root Mean Square Error (MRRMSE)\n    rowwise_rmse = np.sqrt(np.mean(np.square(y_val_ - y_preds), axis=1))\n    mrrmse_score = np.mean(rowwise_rmse)\n    mrrmse_scores.append(mrrmse_score)\n\n# Calculate the mean and standard deviation of MAE and MRRMSE scores\nmean_mae = np.mean(mae_scores)\nmean_mrrmse = np.mean(mrrmse_scores)\n\n# Print the results\nprint(f'Average MAE across {num_folds} folds: {mean_mae:.4f} ')\nprint(f'Average MRRMSE across {num_folds} folds: {mean_mrrmse:.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:17:18.411383Z","iopub.execute_input":"2023-10-27T11:17:18.411748Z","iopub.status.idle":"2023-10-27T11:25:58.050073Z","shell.execute_reply.started":"2023-10-27T11:17:18.411717Z","shell.execute_reply":"2023-10-27T11:25:58.048642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train the model on full data","metadata":{}},{"cell_type":"code","source":"model_5.compile(loss=\"mae\", \n                optimizer=tf.keras.optimizers.Adam(),\n                metrics=[custom_mean_rowwise_rmse])\n\nhistory_5 = model_5.fit(full_features, full_labels,\n                       epochs=50,\n                       verbose=0,\n                       validation_data=(X_val,y_val),\n                       callbacks=[create_model_checkpoint(\"model_5\", monitor=\"val_custom_mean_rowwise_rmse\")])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:25:58.052757Z","iopub.execute_input":"2023-10-27T11:25:58.053755Z","iopub.status.idle":"2023-10-27T11:26:16.445765Z","shell.execute_reply.started":"2023-10-27T11:25:58.053708Z","shell.execute_reply":"2023-10-27T11:26:16.444239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"calculate_mae_and_mrrmse(model=model_5, data=full_features, y_true=full_labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:16.447920Z","iopub.execute_input":"2023-10-27T11:26:16.448384Z","iopub.status.idle":"2023-10-27T11:26:19.477643Z","shell.execute_reply.started":"2023-10-27T11:26:16.448336Z","shell.execute_reply":"2023-10-27T11:26:19.476154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nmodels = ['Модель №0', 'Модель №1', 'Модель №2', 'Модель №3', 'Модель №4', 'Модель №5']\nMRRMSE = [1.2077, 1.2142, 1.1729, 1.1776, 0.9053, 0.8414]\nMAE = [0.7934, 0.8034, 0.7525, 0.7608, 0.5368, 0.5003]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(x=models, y=MRRMSE, name='Mean Rowwise RMSE', marker_color='green'))\nfig.add_trace(go.Bar(x=models, y=MAE, name='MAE', marker_color='blue'))","metadata":{"execution":{"iopub.status.busy":"2023-10-27T12:11:07.360924Z","iopub.execute_input":"2023-10-27T12:11:07.361514Z","iopub.status.idle":"2023-10-27T12:11:07.382298Z","shell.execute_reply.started":"2023-10-27T12:11:07.361471Z","shell.execute_reply":"2023-10-27T12:11:07.381004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model_5, 'model_5_layers.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:19.478969Z","iopub.execute_input":"2023-10-27T11:26:19.479318Z","iopub.status.idle":"2023-10-27T11:26:19.652751Z","shell.execute_reply.started":"2023-10-27T11:26:19.479267Z","shell.execute_reply":"2023-10-27T11:26:19.651550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_4, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:19.654159Z","iopub.execute_input":"2023-10-27T11:26:19.654545Z","iopub.status.idle":"2023-10-27T11:26:20.393647Z","shell.execute_reply.started":"2023-10-27T11:26:19.654505Z","shell.execute_reply":"2023-10-27T11:26:20.392370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history_5, metrics=[\"custom_mean_rowwise_rmse\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:20.395529Z","iopub.execute_input":"2023-10-27T11:26:20.396622Z","iopub.status.idle":"2023-10-27T11:26:21.135725Z","shell.execute_reply.started":"2023-10-27T11:26:20.396570Z","shell.execute_reply":"2023-10-27T11:26:21.134484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting on test data ","metadata":{}},{"cell_type":"code","source":"preds = model_5.predict(one_hot_test.toarray(), batch_size=1)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.137179Z","iopub.execute_input":"2023-10-27T11:26:21.137555Z","iopub.status.idle":"2023-10-27T11:26:21.825221Z","shell.execute_reply.started":"2023-10-27T11:26:21.137523Z","shell.execute_reply":"2023-10-27T11:26:21.824168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape ","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.826851Z","iopub.execute_input":"2023-10-27T11:26:21.827235Z","iopub.status.idle":"2023-10-27T11:26:21.834664Z","shell.execute_reply.started":"2023-10-27T11:26:21.827204Z","shell.execute_reply":"2023-10-27T11:26:21.833479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_columns = sample_submission.columns\nsample_columns= sample_columns[1:]\nsubmission_df = pd.DataFrame(preds, columns=sample_columns)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.836305Z","iopub.execute_input":"2023-10-27T11:26:21.836622Z","iopub.status.idle":"2023-10-27T11:26:21.848255Z","shell.execute_reply.started":"2023-10-27T11:26:21.836596Z","shell.execute_reply":"2023-10-27T11:26:21.847342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.insert(0, 'id', range(255))","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.849963Z","iopub.execute_input":"2023-10-27T11:26:21.851101Z","iopub.status.idle":"2023-10-27T11:26:21.865037Z","shell.execute_reply.started":"2023-10-27T11:26:21.851059Z","shell.execute_reply":"2023-10-27T11:26:21.863958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.866706Z","iopub.execute_input":"2023-10-27T11:26:21.867712Z","iopub.status.idle":"2023-10-27T11:26:21.928412Z","shell.execute_reply.started":"2023-10-27T11:26:21.867677Z","shell.execute_reply":"2023-10-27T11:26:21.927092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.930020Z","iopub.execute_input":"2023-10-27T11:26:21.930410Z","iopub.status.idle":"2023-10-27T11:26:21.969049Z","shell.execute_reply.started":"2023-10-27T11:26:21.930375Z","shell.execute_reply":"2023-10-27T11:26:21.967842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission_df.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:21.970569Z","iopub.execute_input":"2023-10-27T11:26:21.971222Z","iopub.status.idle":"2023-10-27T11:26:31.509782Z","shell.execute_reply.started":"2023-10-27T11:26:21.971184Z","shell.execute_reply":"2023-10-27T11:26:31.508621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission_preds.zip /kaggle/working/submission_df.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-27T11:26:31.511436Z","iopub.execute_input":"2023-10-27T11:26:31.511786Z","iopub.status.idle":"2023-10-27T11:26:38.973146Z","shell.execute_reply.started":"2023-10-27T11:26:31.511756Z","shell.execute_reply":"2023-10-27T11:26:38.971517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Improving model**\n\n* Try out different layers and number of neurons in layer and see how it goes.\n* Experiment with different activation functions.\n* Experiment with different architecture, try deep or wide architecture.\n* Find the best learning rate for the model.\n* Normalize labels and try optimizing.\n* Validate model using different techniques like K-fold validation.\n* Experiment.. Experiment.. Experiment.. Visualize.. Visualize.. Visualize..\n\nGood Luck :)","metadata":{"execution":{"iopub.status.busy":"2023-10-05T09:14:36.993246Z","iopub.execute_input":"2023-10-05T09:14:36.993577Z","iopub.status.idle":"2023-10-05T09:14:37.000183Z","shell.execute_reply.started":"2023-10-05T09:14:36.99355Z","shell.execute_reply":"2023-10-05T09:14:36.999213Z"}}}]}