{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport joblib\nimport h5py\nimport librosa\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom IPython.display import Audio\nfrom sklearn.metrics import make_scorer\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import accuracy_score, roc_auc_score, roc_curve\nfrom imblearn.over_sampling import RandomOverSampler, SMOTE\nimport lightgbm as lgb\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nimport xgboost as xgb\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\nfrom imblearn.over_sampling import RandomOverSampler\nimport matplotlib.pyplot as plt\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix, roc_curve, roc_auc_score\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:22:58.815535Z","iopub.execute_input":"2024-06-04T14:22:58.816221Z","iopub.status.idle":"2024-06-04T14:23:02.426544Z","shell.execute_reply.started":"2024-06-04T14:22:58.816185Z","shell.execute_reply":"2024-06-04T14:23:02.425675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_metadata = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ndf_ebird_taxonomy = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\nsample_submission = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')\naudio_dir_path = '/kaggle/input/birdclef-2024/train_audio/'\ndf_metadata['file_path'] = audio_dir_path + df_metadata['filename']","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:23:02.428143Z","iopub.execute_input":"2024-06-04T14:23:02.428726Z","iopub.status.idle":"2024-06-04T14:23:02.702330Z","shell.execute_reply.started":"2024-06-04T14:23:02.428687Z","shell.execute_reply":"2024-06-04T14:23:02.701415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_coefficients(input_path):\n    \n    waveform, sample_rate = librosa.load(input_path)\n    mfccs = librosa.feature.mfcc(y=waveform, sr=sample_rate, n_mfcc=30)\n    scaled_mfccs = np.mean(mfccs.T, axis=0)\n    return scaled_mfccs\ndef feature_engineer(df):\n\n    new_features = []\n    for i, file in df.iterrows():\n        input_path = file['file_path']\n        coefficients = get_coefficients(input_path)\n        new_features.append(coefficients)\n    \n    train_columns = np.arange(0, 30)\n    train_columns_str = [str(name) for name in train_columns]\n    \n    df_train = pd.DataFrame(new_features, columns=train_columns_str)\n    df_train = pd.concat([df_train, df['primary_label']], axis=1)\n    \n    return df_train","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:23:15.598699Z","iopub.execute_input":"2024-06-04T14:23:15.599503Z","iopub.status.idle":"2024-06-04T14:23:15.607199Z","shell.execute_reply.started":"2024-06-04T14:23:15.599467Z","shell.execute_reply":"2024-06-04T14:23:15.606196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_engineer(df_metadata)\noriginal_labels = df_metadata['primary_label'].unique()\nlabel_encoder = LabelEncoder()\nlabel_encoder.fit(original_labels)\nencoded_labels = label_encoder.transform(df_train['primary_label'])\ndf_train['encoded_label'] = encoded_labels\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T14:23:29.686589Z","iopub.execute_input":"2024-06-04T14:23:29.687618Z","iopub.status.idle":"2024-06-04T15:39:13.341573Z","shell.execute_reply.started":"2024-06-04T14:23:29.687574Z","shell.execute_reply":"2024-06-04T15:39:13.339769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.to_csv('/kaggle/working/df_train.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T15:39:35.314872Z","iopub.execute_input":"2024-06-04T15:39:35.315625Z","iopub.status.idle":"2024-06-04T15:39:36.401586Z","shell.execute_reply.started":"2024-06-04T15:39:35.315584Z","shell.execute_reply":"2024-06-04T15:39:36.400635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T15:39:39.268282Z","iopub.execute_input":"2024-06-04T15:39:39.268765Z","iopub.status.idle":"2024-06-04T15:39:39.303023Z","shell.execute_reply.started":"2024-06-04T15:39:39.268729Z","shell.execute_reply":"2024-06-04T15:39:39.302048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the primary_label column from df_train\ndf_train_pd = df_train.drop(columns=['primary_label'])","metadata":{"execution":{"iopub.status.busy":"2024-06-04T15:39:42.490689Z","iopub.execute_input":"2024-06-04T15:39:42.491069Z","iopub.status.idle":"2024-06-04T15:39:42.500802Z","shell.execute_reply.started":"2024-06-04T15:39:42.491038Z","shell.execute_reply":"2024-06-04T15:39:42.499693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate features and label\nX = df_train_pd.drop(\"encoded_label\", axis=1)\ny = df_train_pd[\"encoded_label\"]\n\n# Initialize the oversampler\nsampler = RandomOverSampler(random_state=0)\n\n# Fit oversampler and resample\nX_resampled, y_resampled = sampler.fit_resample(X, y)\n\ndf_train_resampled = pd.DataFrame(X_resampled)\ndf_train_resampled['encoded_label'] = y_resampled\ndisplay(df_train_resampled.head(2))\ndisplay(df_train_resampled.info())\n\n# Splitting the data into training, validation, and testing sets\nX_train, X_temp, y_train, y_temp = train_test_split(X_resampled, y_resampled, test_size=0.3, random_state=12)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, random_state=12)\n\nclf = XGBClassifier(objective='multi:softmax', num_class=182, eval_metric='mlogloss', use_label_encoder=False)\n\n# Training the model and recording training and validation loss\neval_set = [(X_train, y_train), (X_val, y_val)]\nclf.fit(X_train, y_train, eval_set=eval_set, verbose=True)\n\n# Making predictions on the test set\ny_predictions = clf.predict(X_test)\n\n# Accuracy score of the model\naccuracy = accuracy_score(y_predictions, y_test)\nprint('Accuracy score: ', accuracy)\n\n# Extracting training and validation loss\nresults = clf.evals_result()\nepochs = len(results['validation_0']['mlogloss'])\nx_axis = range(0, epochs)\n\n# Plotting training and validation loss\nplt.figure(figsize=(10, 6))\nplt.plot(x_axis, results['validation_0']['mlogloss'], label='Train')\nplt.plot(x_axis, results['validation_1']['mlogloss'], label='Validation')\nplt.legend()\nplt.xlabel('Epochs')\nplt.ylabel('Log Loss')\nplt.title('XGBoost Training and Validation Loss')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T15:39:59.242081Z","iopub.execute_input":"2024-06-04T15:39:59.242881Z","iopub.status.idle":"2024-06-04T15:43:13.694626Z","shell.execute_reply.started":"2024-06-04T15:39:59.242841Z","shell.execute_reply":"2024-06-04T15:43:13.693393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving/exporting the model\njoblib.dump(clf, 'clf.joblib')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T15:43:13.696884Z","iopub.execute_input":"2024-06-04T15:43:13.697364Z","iopub.status.idle":"2024-06-04T15:43:13.995627Z","shell.execute_reply.started":"2024-06-04T15:43:13.697268Z","shell.execute_reply":"2024-06-04T15:43:13.994582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# import seaborn as sns\n# from sklearn.metrics import confusion_matrix, roc_curve, roc_auc_score\n\n# # Confusion Matrix\n# def plot_confusion_matrix(y_test, y_pred):\n#     cm = confusion_matrix(y_test, y_pred)\n#     plt.figure(figsize=(10, 7))\n#     sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\n#     plt.xlabel('Predicted')\n#     plt.ylabel('True')\n#     plt.title('Confusion Matrix')\n#     plt.show()\n\n# # Plotting the confusion matrix\n# plot_confusion_matrix(y_test, y_predictions)\n\n# # AUC-ROC Plot\n# def plot_roc_curve(y_test, y_proba, num_classes):\n#     fpr = {}\n#     tpr = {}\n#     roc_auc = {}\n#     for i in range(num_classes):\n#         fpr[i], tpr[i], _ = roc_curve(y_test, y_proba[:, i], pos_label=i)\n#         roc_auc[i] = roc_auc_score(y_test == i, y_proba[:, i])\n\n#     plt.figure(figsize=(10, 7))\n#     for i in range(num_classes):\n#         plt.plot(fpr[i], tpr[i], label=f'Class {i} (AUC = {roc_auc[i]:.2f})')\n    \n#     plt.plot([0, 1], [0, 1], 'k--')\n#     plt.xlabel('False Positive Rate')\n#     plt.ylabel('True Positive Rate')\n#     plt.title('Receiver Operating Characteristic (ROC) Curve')\n#     plt.legend(loc='best')\n#     plt.show()\n\n# # Get probability predictions for AUC-ROC\n# y_proba = clf.predict_proba(X_test)\n\n# # Plotting the AUC-ROC curve\n# plot_roc_curve(y_test, y_proba, num_classes=182)\n\n# # # Evaluate model using cross-validation\n# # from sklearn.model_selection import cross_val_score\n\n# # cross_val_scores = cross_val_score(clf, X_resampled, y_resampled, cv=5, scoring='accuracy')\n# # print(\"Cross-Validation Scores: \", cross_val_scores)\n# # print(\"Mean Cross-Validation Score: \", np.mean(cross_val_scores))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Important\n# # Load the saved model\n# model_path = \"/kaggle/working/xgb.model\"\n# model = XGBClassifier()\n# model.load_model(model_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:08.676532Z","iopub.execute_input":"2024-06-04T13:33:08.676930Z","iopub.status.idle":"2024-06-04T13:33:08.697663Z","shell.execute_reply.started":"2024-06-04T13:33:08.676899Z","shell.execute_reply":"2024-06-04T13:33:08.696508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Important\n# from pathlib import Path\n# import pandas as pd\n\n# # Define the path to the directory containing the .ogg files\n# test_soundscapes = '/kaggle/input/birdclef-2024/unlabeled_soundscapes'\n\n# # Get all paths of .ogg files\n# paths = list(Path(test_soundscapes).glob(\"*.ogg\"))\n\n# # Create the DataFrame with filename, id, and path columns\n# test = pd.DataFrame(\n#     [(path.stem, idx, str(path)) for idx, path in enumerate(paths)],\n#     columns=[\"filename\", \"id\", \"path\"]\n# )\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:12.376397Z","iopub.execute_input":"2024-06-04T13:33:12.376856Z","iopub.status.idle":"2024-06-04T13:33:12.773943Z","shell.execute_reply.started":"2024-06-04T13:33:12.376823Z","shell.execute_reply":"2024-06-04T13:33:12.772306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Important\n# filenames = test.filename.values.tolist()\n\n# bird_cols = list(pd.get_dummies(df_metadata['primary_label']).columns)\n# submission_df = pd.DataFrame(columns=['row_id']+bird_cols)\n# submission_df","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:16.789705Z","iopub.execute_input":"2024-06-04T13:33:16.790259Z","iopub.status.idle":"2024-06-04T13:33:16.810820Z","shell.execute_reply.started":"2024-06-04T13:33:16.790218Z","shell.execute_reply":"2024-06-04T13:33:16.809162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Important\n# # Add 'row_id' based on filenames\n# submission_df['row_id'] = filenames\n\n# # Merge the submission_df and test DataFrames on 'filename'\n# merged_df = pd.merge(submission_df, test, left_on='row_id', right_on='filename')\n\n# # Drop the now-redundant 'row_id' column\n# merged_df.drop(columns=['row_id'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:20.558459Z","iopub.execute_input":"2024-06-04T13:33:20.558966Z","iopub.status.idle":"2024-06-04T13:33:20.585566Z","shell.execute_reply.started":"2024-06-04T13:33:20.558923Z","shell.execute_reply":"2024-06-04T13:33:20.584286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Important\n# df1=test","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:24.090688Z","iopub.execute_input":"2024-06-04T13:33:24.091225Z","iopub.status.idle":"2024-06-04T13:33:24.097331Z","shell.execute_reply.started":"2024-06-04T13:33:24.091184Z","shell.execute_reply":"2024-06-04T13:33:24.095823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:26.423423Z","iopub.execute_input":"2024-06-04T13:33:26.423919Z","iopub.status.idle":"2024-06-04T13:33:26.438108Z","shell.execute_reply.started":"2024-06-04T13:33:26.423879Z","shell.execute_reply":"2024-06-04T13:33:26.436487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import time\n# import numpy as np\n\n# def predict_audio_files(df, model, file_path_column='path'):\n#     # Initialize an empty list to store prediction times and cumulative end times\n#     prediction_times = []\n#     cumulative_end_times = []\n    \n#     # Initialize the cumulative time counter\n#     cumulative_time = 0\n    \n#     # Initialize an empty list to store probabilities\n#     probabilities_list = []\n    \n#     # Initialize an empty list to store MFCC features\n#     mfcc_features_list = []\n    \n#     # Extract MFCC coefficients for each audio file and make predictions\n#     for idx, row in df.iterrows():\n#         mfcc_features = get_coefficients(row[file_path_column])\n#         mfcc_features_list.append(mfcc_features)\n        \n#         start_time = time.time()\n#         probabilities = model.predict_proba([mfcc_features])[0]  # predict_proba expects a 2D array\n#         end_time = time.time()\n        \n#         prediction_time = end_time - start_time\n#         prediction_times.append(prediction_time)\n        \n#         # Update the cumulative time\n#         cumulative_time += prediction_time\n#         cumulative_end_times.append(cumulative_time)\n        \n#         probabilities_list.append(probabilities)\n    \n#     # Convert the list of probabilities to a numpy array\n#     probabilities_array = np.array(probabilities_list)\n    \n#     # Add probabilities to the DataFrame\n#     for i, class_name in enumerate(model.classes_):\n#         df[class_name] = probabilities_array[:, i]\n    \n#     # Determine the class with the highest probability\n#     df['predicted_class'] = df[model.classes_].idxmax(axis=1)\n    \n#     # Add prediction times and cumulative end times to the DataFrame\n#     df['prediction_time'] = prediction_times\n#     df['cumulative_end_time'] = cumulative_end_times\n    \n#     return df\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Important\n# import time\n# import numpy as np\n# import math\n\n# def predict_audio_files(df, model, file_path_column='path'):\n#     # Initialize an empty list to store prediction times and cumulative end times\n#     prediction_times = []\n#     cumulative_end_times = []\n    \n#     # Initialize the cumulative time counter\n#     cumulative_time = 0\n    \n#     # Initialize an empty list to store probabilities\n#     probabilities_list = []\n    \n#     # Initialize an empty list to store MFCC features\n#     mfcc_features_list = []\n    \n#     # Extract MFCC coefficients for each audio file and make predictions\n#     for idx, row in df.iterrows():\n#         mfcc_features = get_coefficients(row[file_path_column])\n#         mfcc_features_list.append(mfcc_features)\n        \n#         start_time = time.time()\n#         probabilities = model.predict_proba([mfcc_features])[0]  # predict_proba expects a 2D array\n#         end_time = time.time()\n        \n#         prediction_time = end_time - start_time\n#         prediction_times.append(prediction_time)\n        \n#         # Update the cumulative time\n#         cumulative_time += prediction_time\n#         cumulative_end_times.append(math.ceil(cumulative_time))\n        \n#         probabilities_list.append(probabilities)\n    \n#     # Convert the list of probabilities to a numpy array\n#     probabilities_array = np.array(probabilities_list)\n    \n#     # Add probabilities to the DataFrame\n#     for i, class_name in enumerate(model.classes_):\n#         df[class_name] = probabilities_array[:, i]\n    \n#     # Determine the class with the highest probability\n#     df['predicted_class'] = df[model.classes_].idxmax(axis=1)\n    \n#     # Add prediction times and cumulative end times to the DataFrame\n#     df['prediction_time'] = prediction_times\n#     df['cumulative_end_time'] = cumulative_end_times\n    \n#     return df\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:32.396750Z","iopub.execute_input":"2024-06-04T13:33:32.397268Z","iopub.status.idle":"2024-06-04T13:33:32.410545Z","shell.execute_reply.started":"2024-06-04T13:33:32.397230Z","shell.execute_reply":"2024-06-04T13:33:32.408998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #important\n# df_train=pd.read_csv(\"/kaggle/working/df_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:56.708554Z","iopub.execute_input":"2024-06-04T13:33:56.709018Z","iopub.status.idle":"2024-06-04T13:33:56.720827Z","shell.execute_reply.started":"2024-06-04T13:33:56.708980Z","shell.execute_reply":"2024-06-04T13:33:56.719473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Important\n# # Dropping duplicates based on the combination of 'primary_label' and 'encoded_label'\n# labels = df_train[['primary_label', 'encoded_label']].drop_duplicates()\n\n# labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:33:58.846556Z","iopub.execute_input":"2024-06-04T13:33:58.846979Z","iopub.status.idle":"2024-06-04T13:33:58.862890Z","shell.execute_reply.started":"2024-06-04T13:33:58.846945Z","shell.execute_reply":"2024-06-04T13:33:58.861176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Important\n# df2=df1","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:34:02.246901Z","iopub.execute_input":"2024-06-04T13:34:02.247451Z","iopub.status.idle":"2024-06-04T13:34:02.252863Z","shell.execute_reply.started":"2024-06-04T13:34:02.247412Z","shell.execute_reply":"2024-06-04T13:34:02.251563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del df1\n# del test\n# del merged_df\n# del submission_df","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:34:05.156222Z","iopub.execute_input":"2024-06-04T13:34:05.156683Z","iopub.status.idle":"2024-06-04T13:34:05.163212Z","shell.execute_reply.started":"2024-06-04T13:34:05.156650Z","shell.execute_reply":"2024-06-04T13:34:05.161809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2=df2.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:34:46.562186Z","iopub.execute_input":"2024-06-04T13:34:46.562620Z","iopub.status.idle":"2024-06-04T13:34:46.569024Z","shell.execute_reply.started":"2024-06-04T13:34:46.562587Z","shell.execute_reply":"2024-06-04T13:34:46.567488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_with_predictions = predict_audio_files(df2, model)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:34:49.294337Z","iopub.execute_input":"2024-06-04T13:34:49.294793Z","iopub.status.idle":"2024-06-04T13:35:05.674563Z","shell.execute_reply.started":"2024-06-04T13:34:49.294761Z","shell.execute_reply":"2024-06-04T13:35:05.673474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# df_with_predictions['row_id'] = (\n#     \"soundscape_\" + df_with_predictions['filename'].astype(str) + \"_\" + df_with_predictions['cumulative_end_time'].astype(str)\n# )\n# # Drop the specified columns\n# df_with_predictions = df_with_predictions.drop(columns=[\"cumulative_end_time\",\"prediction_time\",\"predicted_class\",\"filename\",\"id\",\"path\"])\n\n# columns = ['row_id'] + [col for col in df_with_predictions.columns if col != 'row_id']\n# df_with_predictions = df_with_predictions[columns]\n# mapping = dict(zip(labels['encoded_label'], labels['primary_label']))\n\n# # Rename the columns in df_with_predictions using the mapping\n# df_with_predictions.rename(columns=mapping, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:35:10.626330Z","iopub.execute_input":"2024-06-04T13:35:10.626816Z","iopub.status.idle":"2024-06-04T13:35:10.657970Z","shell.execute_reply.started":"2024-06-04T13:35:10.626776Z","shell.execute_reply":"2024-06-04T13:35:10.656699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_with_predictions.tail(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:35:13.566695Z","iopub.execute_input":"2024-06-04T13:35:13.567146Z","iopub.status.idle":"2024-06-04T13:35:13.596917Z","shell.execute_reply.started":"2024-06-04T13:35:13.567106Z","shell.execute_reply":"2024-06-04T13:35:13.595902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission=df_with_predictions\n# # Define the full output path including the filename\n# output_path = '/kaggle/working/submission.csv'\n\n# # Save the DataFrame as a CSV file\n# submission.to_csv(output_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T13:35:17.228813Z","iopub.execute_input":"2024-06-04T13:35:17.229338Z","iopub.status.idle":"2024-06-04T13:35:17.252609Z","shell.execute_reply.started":"2024-06-04T13:35:17.229298Z","shell.execute_reply":"2024-06-04T13:35:17.251182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}