{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport folium \nfrom folium import Marker , GeoJson , Choropleth , Circle\nfrom folium.plugins import HeatMap , MarkerCluster\nimport librosa.display \nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-30T08:22:51.754443Z","iopub.execute_input":"2023-11-30T08:22:51.754842Z","iopub.status.idle":"2023-11-30T08:22:55.458803Z","shell.execute_reply.started":"2023-11-30T08:22:51.754808Z","shell.execute_reply":"2023-11-30T08:22:55.457857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.read_csv(\"/kaggle/input/birdsong-recognition/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:55.461087Z","iopub.execute_input":"2023-11-30T08:22:55.462467Z","iopub.status.idle":"2023-11-30T08:22:56.140671Z","shell.execute_reply.started":"2023-11-30T08:22:55.462418Z","shell.execute_reply":"2023-11-30T08:22:56.139431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.142708Z","iopub.execute_input":"2023-11-30T08:22:56.143568Z","iopub.status.idle":"2023-11-30T08:22:56.192161Z","shell.execute_reply.started":"2023-11-30T08:22:56.143520Z","shell.execute_reply":"2023-11-30T08:22:56.190969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets check the class distribution of the data","metadata":{}},{"cell_type":"code","source":"df[\"species\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.195520Z","iopub.execute_input":"2023-11-30T08:22:56.196037Z","iopub.status.idle":"2023-11-30T08:22:56.214365Z","shell.execute_reply.started":"2023-11-30T08:22:56.195990Z","shell.execute_reply":"2023-11-30T08:22:56.213103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data is highly imbalanced as we can see some of the speicies audio files are less compared to some other ","metadata":{"execution":{"iopub.status.busy":"2023-11-29T09:39:24.808211Z","iopub.execute_input":"2023-11-29T09:39:24.808648Z","iopub.status.idle":"2023-11-29T09:39:24.816543Z","shell.execute_reply.started":"2023-11-29T09:39:24.808616Z","shell.execute_reply":"2023-11-29T09:39:24.814736Z"}}},{"cell_type":"code","source":"# visualize the data ","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.215850Z","iopub.execute_input":"2023-11-30T08:22:56.216214Z","iopub.status.idle":"2023-11-30T08:22:56.224777Z","shell.execute_reply.started":"2023-11-30T08:22:56.216181Z","shell.execute_reply":"2023-11-30T08:22:56.223340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"year\"] = df[\"date\"].apply(lambda x : x.split(\"-\")[0])\ndf[\"month\"] = df[\"date\"].apply(lambda x : x.split(\"-\")[1])\ngroup_year = df.groupby([\"year\"]).size().reset_index(name = \"counts\")\ngroup_year = group_year.iloc[3:]\ngroup_month = df.groupby([\"month\"]).size().reset_index(name = \"counts\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.226390Z","iopub.execute_input":"2023-11-30T08:22:56.226740Z","iopub.status.idle":"2023-11-30T08:22:56.279025Z","shell.execute_reply.started":"2023-11-30T08:22:56.226711Z","shell.execute_reply":"2023-11-30T08:22:56.277689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.280695Z","iopub.execute_input":"2023-11-30T08:22:56.281471Z","iopub.status.idle":"2023-11-30T08:22:56.343058Z","shell.execute_reply.started":"2023-11-30T08:22:56.281428Z","shell.execute_reply":"2023-11-30T08:22:56.341787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(rows=2, cols=1, subplot_titles = ('Number of recordings accdn year', 'Number of recordings accdn month'))\n\nfig.append_trace(go.Bar(\n    x=group_year['year'],\n    y=group_year['counts'],\n    #tickmode='linear'\n), row=1, col=1)\n\nfig.append_trace(go.Bar(\n    x=group_month['month'],\n    y=group_month['counts'],\n), row=2, col=1)\n\n\n\nfig.update_layout(height=1000, width=700, showlegend=False,  xaxis = dict(\n        tickmode = 'linear',\n    ), xaxis2 = dict(tickmode='linear'))\nfig.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:56.344750Z","iopub.execute_input":"2023-11-30T08:22:56.345260Z","iopub.status.idle":"2023-11-30T08:22:57.083846Z","shell.execute_reply.started":"2023-11-30T08:22:56.345216Z","shell.execute_reply":"2023-11-30T08:22:57.082665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see how the data distributed","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=1, cols=2, specs=[[{\"type\": \"pie\"}, {\"type\": \"pie\"}]], subplot_titles=('Distribution of Channels', 'Distribution of Sampling rate'))\n\ngroup_ch = df.groupby([\"channels\"]).size().reset_index(name = \"counts\")\nfig.append_trace(go.Pie(labels = group_ch[\"channels\"] , \n                       values = group_ch[\"counts\"],),\n                row = 1 , col =1)\ngroup_sr = df.groupby([\"sampling_rate\"]).size().reset_index(name = \"counts\")\nfig.append_trace(go.Pie(labels = group_sr[\"sampling_rate\"] , \n                       values = group_sr[\"counts\"],),\n                row = 1 , col =2)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:57.085690Z","iopub.execute_input":"2023-11-30T08:22:57.086123Z","iopub.status.idle":"2023-11-30T08:22:57.162663Z","shell.execute_reply.started":"2023-11-30T08:22:57.086087Z","shell.execute_reply":"2023-11-30T08:22:57.161366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:57.166942Z","iopub.execute_input":"2023-11-30T08:22:57.167303Z","iopub.status.idle":"2023-11-30T08:22:57.176045Z","shell.execute_reply.started":"2023-11-30T08:22:57.167274Z","shell.execute_reply":"2023-11-30T08:22:57.174813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_path = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\nx, sr = librosa.load(audio_path)\nAudio(x, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:22:57.177731Z","iopub.execute_input":"2023-11-30T08:22:57.178353Z","iopub.status.idle":"2023-11-30T08:23:10.098260Z","shell.execute_reply.started":"2023-11-30T08:22:57.178179Z","shell.execute_reply":"2023-11-30T08:23:10.096975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see how this audio looks likes","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade librosa\n\nfig, ax = plt.subplots(4, figsize = (20, 9))\nfig.suptitle('Waveplots', fontsize=16)\naudio_path1 = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\naudio_path2 = '../input/birdsong-recognition/train_audio/amepip/XC111040.mp3'\naudio_path3 = '../input/birdsong-recognition/train_audio/banswa/XC138517.mp3'\naudio_path4 = '../input/birdsong-recognition/train_audio/bkhgro/XC109305.mp3'\n\ny1, sr1 = librosa.load(audio_path1)\ny2, sr2 = librosa.load(audio_path2)\ny3, sr3 = librosa.load(audio_path3)\ny4, sr4 = librosa.load(audio_path4)\n\nlibrosa.display.waveshow(y=y1, sr=sr1, color = \"#3371FF\", ax=ax[0])\nlibrosa.display.waveshow(y=y2 , sr=sr2, color = \"#F7A81E\", ax=ax[1])\nlibrosa.display.waveshow(y=y3 , sr=sr3, color = \"#2BF71E\", ax=ax[2])\nlibrosa.display.waveshow(y=y4 , sr=sr4, color = \"#F71E6D\", ax=ax[3])","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:10.099529Z","iopub.execute_input":"2023-11-30T08:23:10.100178Z","iopub.status.idle":"2023-11-30T08:23:29.310129Z","shell.execute_reply.started":"2023-11-30T08:23:10.100144Z","shell.execute_reply":"2023-11-30T08:23:29.308680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Extraction\nThe rate at which the signal changes from positive to zero to negative or from negative to zero to positive","metadata":{}},{"cell_type":"code","source":"# Visualize an STFT power spectrum\n\naudio_path = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\ny, sr = librosa.load(audio_path)\nplt.figure(figsize=(12, 8))\nD = librosa.amplitude_to_db(librosa.stft(y))\nplt.subplot(4, 2, 1)\nlibrosa.display.specshow(D, y_axis='linear')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Linear-frequency power spectrogram')\n\n# logarithmic scale\n\nplt.subplot(4, 2, 2)\nlibrosa.display.specshow(D, y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Log-frequency power spectrogram')\n\n#CQT scale\n\nCQT = librosa.amplitude_to_db(librosa.cqt(y, sr=sr), ref=np.max)\nplt.subplot(4, 2, 3)\nlibrosa.display.specshow(CQT, y_axis='cqt_hz')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Constant-Q power spectrogram (Hz)')\n\nCQT = librosa.amplitude_to_db(librosa.cqt(y, sr=sr), ref=np.max)\nplt.subplot(4, 2, 4)\nlibrosa.display.specshow(CQT, y_axis='cqt_note')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Constant-Q power spectrogram (note)')\n\n#Chromagram\nC = librosa.feature.chroma_cqt(y=y, sr=sr)\nplt.subplot(4, 2, 5)\nlibrosa.display.specshow(C, y_axis='chroma')\nplt.colorbar()\nplt.title('Chromagram')\n\n# Log power spectrogram\nplt.subplot(4, 2, 6)\nlibrosa.display.specshow(D, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Log power spectrogram')","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:29.311716Z","iopub.execute_input":"2023-11-30T08:23:29.312066Z","iopub.status.idle":"2023-11-30T08:23:37.806892Z","shell.execute_reply.started":"2023-11-30T08:23:29.312024Z","shell.execute_reply":"2023-11-30T08:23:37.805738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's zoom in \nn0 = 7000\nn1 = 7100\nplt.figure(figsize=(14, 5))\nplt.plot(y[n0:n1])","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:37.808440Z","iopub.execute_input":"2023-11-30T08:23:37.808881Z","iopub.status.idle":"2023-11-30T08:23:38.110391Z","shell.execute_reply.started":"2023-11-30T08:23:37.808841Z","shell.execute_reply":"2023-11-30T08:23:38.109595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_crossings = librosa.zero_crossings(y[n0:n1], pad=False)\nzero_crossings.shape\nzcrs = librosa.feature.zero_crossing_rate(y)\nprint(zcrs.shape)\nplt.figure(figsize=(14, 5))\nplt.plot(zcrs[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.111510Z","iopub.execute_input":"2023-11-30T08:23:38.112201Z","iopub.status.idle":"2023-11-30T08:23:38.436396Z","shell.execute_reply.started":"2023-11-30T08:23:38.112170Z","shell.execute_reply":"2023-11-30T08:23:38.435620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.437499Z","iopub.execute_input":"2023-11-30T08:23:38.438381Z","iopub.status.idle":"2023-11-30T08:23:38.446788Z","shell.execute_reply.started":"2023-11-30T08:23:38.438349Z","shell.execute_reply":"2023-11-30T08:23:38.445502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = (df.dtypes == \"object\")\nlist1= list(s[s].index)\nprint(list1)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.448237Z","iopub.execute_input":"2023-11-30T08:23:38.448640Z","iopub.status.idle":"2023-11-30T08:23:38.460036Z","shell.execute_reply.started":"2023-11-30T08:23:38.448603Z","shell.execute_reply":"2023-11-30T08:23:38.458928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop([\"filename\"  ,\"url\" ,\"ebird_code\", \"author\",\"sci_name\" ,\"secondary_labels\" ,\"xc_id\",\"file_type\" ,\"description\",\"date\"] , axis =1)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.461845Z","iopub.execute_input":"2023-11-30T08:23:38.462707Z","iopub.status.idle":"2023-11-30T08:23:38.481337Z","shell.execute_reply.started":"2023-11-30T08:23:38.462664Z","shell.execute_reply":"2023-11-30T08:23:38.480088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for cols in df.columns:\n    unique = df[cols].unique()\n    print(f\"The values are {cols} : {unique}\")\n# Define the desired order\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.483124Z","iopub.execute_input":"2023-11-30T08:23:38.483520Z","iopub.status.idle":"2023-11-30T08:23:38.569046Z","shell.execute_reply.started":"2023-11-30T08:23:38.483485Z","shell.execute_reply":"2023-11-30T08:23:38.567936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.570756Z","iopub.execute_input":"2023-11-30T08:23:38.571202Z","iopub.status.idle":"2023-11-30T08:23:38.637432Z","shell.execute_reply.started":"2023-11-30T08:23:38.571162Z","shell.execute_reply":"2023-11-30T08:23:38.636283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(axis=0)\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.638820Z","iopub.execute_input":"2023-11-30T08:23:38.639179Z","iopub.status.idle":"2023-11-30T08:23:38.730993Z","shell.execute_reply.started":"2023-11-30T08:23:38.639150Z","shell.execute_reply":"2023-11-30T08:23:38.729863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = (df.dtypes == \"object\")\nlist1= list(s[s].index)\nprint(list1)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.732477Z","iopub.execute_input":"2023-11-30T08:23:38.732858Z","iopub.status.idle":"2023-11-30T08:23:38.738764Z","shell.execute_reply.started":"2023-11-30T08:23:38.732827Z","shell.execute_reply":"2023-11-30T08:23:38.737668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Assuming df is your DataFrame\nlabel_encoder = LabelEncoder()\nmapping_dict = {}\n\nfor column in df.columns:\n    df[column] = label_encoder.fit_transform(df[column])\n    mapping_dict[column] = dict(zip(label_encoder.classes_, label_encoder.transform(label_encoder.classes_)))\n\n# Now df is transformed, and mapping_dict contains the mapping information\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.740187Z","iopub.execute_input":"2023-11-30T08:23:38.740610Z","iopub.status.idle":"2023-11-30T08:23:38.985899Z","shell.execute_reply.started":"2023-11-30T08:23:38.740553Z","shell.execute_reply":"2023-11-30T08:23:38.984640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:38.987318Z","iopub.execute_input":"2023-11-30T08:23:38.987671Z","iopub.status.idle":"2023-11-30T08:23:39.019148Z","shell.execute_reply.started":"2023-11-30T08:23:38.987642Z","shell.execute_reply":"2023-11-30T08:23:39.018044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mapping_dict","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:39.020526Z","iopub.execute_input":"2023-11-30T08:23:39.020865Z","iopub.status.idle":"2023-11-30T08:23:39.400907Z","shell.execute_reply.started":"2023-11-30T08:23:39.020837Z","shell.execute_reply":"2023-11-30T08:23:39.399768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:23:39.402446Z","iopub.execute_input":"2023-11-30T08:23:39.402806Z","iopub.status.idle":"2023-11-30T08:23:39.434644Z","shell.execute_reply.started":"2023-11-30T08:23:39.402775Z","shell.execute_reply":"2023-11-30T08:23:39.433613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df.columns)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:25:23.782277Z","iopub.execute_input":"2023-11-30T08:25:23.782851Z","iopub.status.idle":"2023-11-30T08:25:23.790686Z","shell.execute_reply.started":"2023-11-30T08:25:23.782803Z","shell.execute_reply":"2023-11-30T08:25:23.789511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2, f_regression\nfrom sklearn.metrics import classification_report\n\n\n# Split the data into features (X) and target variable (y)\nX = df.drop('species', axis=1)\ny = df['species']\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nk_best = SelectKBest(chi2, k=9)\n\nX_train = k_best.fit_transform(X_train, y_train)\n\nX_test= k_best.transform(X_test)\n\n# Create a RandomForestClassifier\nmodel = RandomForestClassifier(n_estimators=100 , random_state=42)\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nscore = classification_report(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy * 100:.2f}%\")\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:32:31.881304Z","iopub.execute_input":"2023-11-30T08:32:31.881760Z","iopub.status.idle":"2023-11-30T08:32:40.703649Z","shell.execute_reply.started":"2023-11-30T08:32:31.881725Z","shell.execute_reply":"2023-11-30T08:32:40.702427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = X.columns[k_best.get_support()]\n\n# Display the selected features\nprint(\"Selected Features:\", selected_features)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:25:33.937235Z","iopub.execute_input":"2023-11-30T08:25:33.937729Z","iopub.status.idle":"2023-11-30T08:25:33.944851Z","shell.execute_reply.started":"2023-11-30T08:25:33.937687Z","shell.execute_reply":"2023-11-30T08:25:33.943679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nxgb = XGBClassifier(n_estimators=100)\nxgb.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = xgb.predict(X_test)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy * 100:.2f}%\")\nscore = classification_report(y_test, y_pred)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:33:07.344056Z","iopub.execute_input":"2023-11-30T08:33:07.344453Z","iopub.status.idle":"2023-11-30T08:33:18.930292Z","shell.execute_reply.started":"2023-11-30T08:33:07.344414Z","shell.execute_reply":"2023-11-30T08:33:18.929448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndt = DecisionTreeClassifier()\ndt.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = dt.predict(X_test)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy * 100:.2f}%\")\nscore = classification_report(y_test, y_pred)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:33:28.629981Z","iopub.execute_input":"2023-11-30T08:33:28.630381Z","iopub.status.idle":"2023-11-30T08:33:28.915922Z","shell.execute_reply.started":"2023-11-30T08:33:28.630350Z","shell.execute_reply":"2023-11-30T08:33:28.914690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nxgb_model = XGBClassifier(n_estimators=100)\n\n# Perform 5-fold cross-validation\ncv_scores = cross_val_score(xgb_model, X_train , y_train, cv=5, scoring='accuracy')\n\n# Display the cross-validation scores\nprint(\"Cross-Validation Scores:\", cv_scores)\nprint(\"Mean Accuracy:\", cv_scores.mean())","metadata":{"execution":{"iopub.status.busy":"2023-11-30T08:25:48.700520Z","iopub.execute_input":"2023-11-30T08:25:48.700959Z","iopub.status.idle":"2023-11-30T08:26:45.380481Z","shell.execute_reply.started":"2023-11-30T08:25:48.700917Z","shell.execute_reply":"2023-11-30T08:26:45.379606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}