{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11442528,"sourceType":"datasetVersion","datasetId":7168075}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\ndata = pd.read_csv('/kaggle/input/test111/trained_soundscape_bird_clef_2025.csv')\n\nsample_submission = pd.read_csv('/kaggle/input/birdclef-2025/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:51:58.728382Z","iopub.execute_input":"2025-04-17T03:51:58.728705Z","iopub.status.idle":"2025-04-17T03:51:59.041196Z","shell.execute_reply.started":"2025-04-17T03:51:58.728684Z","shell.execute_reply":"2025-04-17T03:51:59.040280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:00.891467Z","iopub.execute_input":"2025-04-17T03:52:00.892072Z","iopub.status.idle":"2025-04-17T03:52:00.919952Z","shell.execute_reply.started":"2025-04-17T03:52:00.892051Z","shell.execute_reply":"2025-04-17T03:52:00.919255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Encode the 'prediction' column\nlabel_encoder = LabelEncoder()\ndata['encoded_prediction'] = label_encoder.fit_transform(data['prediction'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:05.602288Z","iopub.execute_input":"2025-04-17T03:52:05.602995Z","iopub.status.idle":"2025-04-17T03:52:06.180568Z","shell.execute_reply.started":"2025-04-17T03:52:05.602968Z","shell.execute_reply":"2025-04-17T03:52:06.179771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Setup X and y\n\nX = data[['confidence']].values  # Features\ny = data['encoded_prediction'].values  # Target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:07.718649Z","iopub.execute_input":"2025-04-17T03:52:07.719363Z","iopub.status.idle":"2025-04-17T03:52:07.731792Z","shell.execute_reply.started":"2025-04-17T03:52:07.719335Z","shell.execute_reply":"2025-04-17T03:52:07.730984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42)\n\nX_train.shape, X_test.shape, y_train.shape, y_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:10.004490Z","iopub.execute_input":"2025-04-17T03:52:10.004864Z","iopub.status.idle":"2025-04-17T03:52:10.106682Z","shell.execute_reply.started":"2025-04-17T03:52:10.004837Z","shell.execute_reply":"2025-04-17T03:52:10.105848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Scale the features\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:17.155752Z","iopub.execute_input":"2025-04-17T03:52:17.156310Z","iopub.status.idle":"2025-04-17T03:52:17.161079Z","shell.execute_reply.started":"2025-04-17T03:52:17.156286Z","shell.execute_reply":"2025-04-17T03:52:17.160350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import label_binarize\nimport numpy as np\n\n# Binarize y for multi-class ROC AUC\ny_bin = label_binarize(y, classes=np.unique(y))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:20.735117Z","iopub.execute_input":"2025-04-17T03:52:20.735382Z","iopub.status.idle":"2025-04-17T03:52:20.743334Z","shell.execute_reply.started":"2025-04-17T03:52:20.735363Z","shell.execute_reply":"2025-04-17T03:52:20.742461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Define K-Fold cross-validation\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Initialize arrays to store all predictions and true labels\nall_predictions = []\nall_true_labels = []\n\n# Perform K-Fold cross-validation\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X[train_index], X[test_index]\n    y_train, y_test = y_bin[train_index], y_bin[test_index]\n    \n    # Standardize the features\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_test_scaled = scaler.transform(X_test)\n    \n    # Define the neural network model architecture\n    def build_model(input_dim, num_classes):\n        model = Sequential()\n        model.add(Dense(64, input_dim=input_dim, activation='relu'))\n        model.add(Dropout(0.3))\n        model.add(Dense(32, activation='relu'))\n        model.add(Dense(num_classes, activation='softmax'))  # Output layer for multi-class classification\n        model.compile(optimizer='adam', loss='categorical_crossentropy')\n        return model\n\n    # Build and train the model\n    model = build_model(input_dim=X_train_scaled.shape[1], num_classes=y_train.shape[1])\n    \n    # Early stopping callback\n    early_stop = EarlyStopping(monitor='val_loss', patience=1, restore_best_weights=True)\n    \n    # Train the model\n    model.fit(X_train_scaled, y_train, epochs=5, batch_size=32, validation_data=(X_test_scaled, y_test), callbacks=[early_stop], verbose=1)\n    \n    # Predict probabilities on the test set\n    y_pred_proba = model.predict(X_test_scaled)\n    \n    # Store the predictions and true labels\n    all_predictions.append(y_pred_proba)\n    all_true_labels.append(y_test)\n\n# Combine all predictions and true labels\ny_pred_proba_all = np.vstack(all_predictions)\ny_true_all = np.vstack(all_true_labels)\n\n# Calculate the ROC AUC score\nroc_auc = roc_auc_score(y_true_all, y_pred_proba_all, multi_class='ovr')\nroc_auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:21.310503Z","iopub.execute_input":"2025-04-17T03:52:21.311287Z","iopub.status.idle":"2025-04-17T03:52:52.049564Z","shell.execute_reply.started":"2025-04-17T03:52:21.311262Z","shell.execute_reply":"2025-04-17T03:52:52.048982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_proba_corrected = y_pred_proba[:len(sample_submission)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:54.911045Z","iopub.execute_input":"2025-04-17T03:52:54.911596Z","iopub.status.idle":"2025-04-17T03:52:54.915916Z","shell.execute_reply.started":"2025-04-17T03:52:54.911573Z","shell.execute_reply":"2025-04-17T03:52:54.914982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:56.359287Z","iopub.execute_input":"2025-04-17T03:52:56.359536Z","iopub.status.idle":"2025-04-17T03:52:56.378008Z","shell.execute_reply.started":"2025-04-17T03:52:56.359519Z","shell.execute_reply":"2025-04-17T03:52:56.377251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the submission to CSV\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:52:59.211055Z","iopub.execute_input":"2025-04-17T03:52:59.211751Z","iopub.status.idle":"2025-04-17T03:52:59.222719Z","shell.execute_reply.started":"2025-04-17T03:52:59.211723Z","shell.execute_reply":"2025-04-17T03:52:59.221974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"If you like this solution, please upvoted ! TQ","metadata":{}}]}