{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Skimpy\n\n**Skimpy is an open-source python library that is used to generate a statistical summary of the quantitative datasets and can be used in Juptyer Notebook as well as console also.**","metadata":{}},{"cell_type":"code","source":"!pip -q install skimpy","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:09.369806Z","iopub.execute_input":"2022-03-24T06:16:09.370121Z","iopub.status.idle":"2022-03-24T06:16:16.824424Z","shell.execute_reply.started":"2022-03-24T06:16:09.370084Z","shell.execute_reply":"2022-03-24T06:16:16.823617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom plotly.offline import init_notebook_mode,iplot\ninit_notebook_mode(connected=True)\nimport numpy as np \nimport pandas as pd \nimport skimpy \nimport plotly.express as px\nimport plotly.offline as py\nimport plotly.graph_objs as go\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint('Execution success')","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:16.827343Z","iopub.execute_input":"2022-03-24T06:16:16.827778Z","iopub.status.idle":"2022-03-24T06:16:16.838133Z","shell.execute_reply.started":"2022-03-24T06:16:16.827728Z","shell.execute_reply":"2022-03-24T06:16:16.837139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxo = pd.read_csv(\"../input/birdclef-2022/eBird_Taxonomy_v2021.csv\")\ntrain = pd.read_csv(\"../input/birdclef-2022/train_metadata.csv\")\ntest = pd.read_csv(\"../input/birdclef-2022/test.csv\")\nprint('Execution success')","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:16.839442Z","iopub.execute_input":"2022-03-24T06:16:16.839659Z","iopub.status.idle":"2022-03-24T06:16:16.967258Z","shell.execute_reply.started":"2022-03-24T06:16:16.839634Z","shell.execute_reply":"2022-03-24T06:16:16.966512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skimpy.skim(taxo)\nprint(\"\"\"Here we can clearly see the analysis generated which contains all the data points and \nsummary related to it. It contains Data Types, categories, Missing data, etc.\"\"\")","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:16.968682Z","iopub.execute_input":"2022-03-24T06:16:16.969466Z","iopub.status.idle":"2022-03-24T06:16:17.049736Z","shell.execute_reply.started":"2022-03-24T06:16:16.969427Z","shell.execute_reply":"2022-03-24T06:16:17.048816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skimpy.skim(train)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.050839Z","iopub.execute_input":"2022-03-24T06:16:17.051154Z","iopub.status.idle":"2022-03-24T06:16:17.122912Z","shell.execute_reply.started":"2022-03-24T06:16:17.051125Z","shell.execute_reply":"2022-03-24T06:16:17.122045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skimpy.skim(test)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.124360Z","iopub.execute_input":"2022-03-24T06:16:17.124574Z","iopub.status.idle":"2022-03-24T06:16:17.182735Z","shell.execute_reply.started":"2022-03-24T06:16:17.124547Z","shell.execute_reply":"2022-03-24T06:16:17.181828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting Graph based on DataSet","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.patches as mpl_patches\nimport matplotlib.pylab as pylab\nparams = {'legend.fontsize': 'x-large',\n          'figure.figsize': (15, 8),\n         'axes.labelsize': 'x-large',\n         'axes.titlesize':'x-large',\n         'xtick.labelsize':'x-large',\n         'ytick.labelsize':'x-large',\n         'text.color' : \"green\",\n          \n         }\npylab.rcParams.update(params)\n\nimport torch\nimport torchaudio\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data.dataloader import DataLoader\nfrom torchvision.models import resnet34\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport librosa\nfrom tqdm.notebook import tqdm\n\nfrom colorama import Fore, Back, Style\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('fivethirtyeight')\nimport IPython.display as ipd\nfrom IPython.display import display, HTML\nimport plotly.express as px\nprint('Execution success')","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.185913Z","iopub.execute_input":"2022-03-24T06:16:17.186662Z","iopub.status.idle":"2022-03-24T06:16:17.201789Z","shell.execute_reply.started":"2022-03-24T06:16:17.186629Z","shell.execute_reply":"2022-03-24T06:16:17.200950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = Path(\"../input/birdclef-2022/train_audio\")\ntrain_recs = list(train_path.glob(\"**/*.ogg\"))\nprint( ' No. of audio files:', len( train_recs))","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.202869Z","iopub.execute_input":"2022-03-24T06:16:17.203111Z","iopub.status.idle":"2022-03-24T06:16:17.491028Z","shell.execute_reply.started":"2022-03-24T06:16:17.203081Z","shell.execute_reply":"2022-03-24T06:16:17.490139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/birdclef-2022/train_metadata.csv\")\ntest = pd.read_csv(\"../input/birdclef-2022/test.csv\")\nsub = pd.read_csv(\"../input/birdclef-2022/sample_submission.csv\")\ntaxonomy = pd.read_csv( '../input/birdclef-2022/eBird_Taxonomy_v2021.csv')\n\nprint(\"{0}Number of rows in train data: {1}{2}\\n{0}Number of columns in train data: {1}{3}\"\n      .format(y_,r_,train.shape[0],train.shape[1]))\nprint(\"{0}Number of rows in test data: {1}{2}\\n{0}Number of columns in train data: {1}{3}\"\n      .format(m_,r_,test.shape[0],test.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.492549Z","iopub.execute_input":"2022-03-24T06:16:17.493032Z","iopub.status.idle":"2022-03-24T06:16:17.609042Z","shell.execute_reply.started":"2022-03-24T06:16:17.492991Z","shell.execute_reply":"2022-03-24T06:16:17.608134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{m_}number of unique primary label: {r_}{train.primary_label.nunique()}\")\nprint(f\"{c_}number of unique scientific name: {r_}{train.scientific_name.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.611580Z","iopub.execute_input":"2022-03-24T06:16:17.611807Z","iopub.status.idle":"2022-03-24T06:16:17.620418Z","shell.execute_reply.started":"2022-03-24T06:16:17.611779Z","shell.execute_reply":"2022-03-24T06:16:17.619500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.countplot(data=taxonomy, x = \"CATEGORY\")\nplt.title(\"Record count per Category\")\nfor p in g.patches:\n    g.annotate(format(p.get_height(), '.0f'),\n               (p.get_x() + p.get_width() / 2., p.get_height()),\n               ha = 'center', va = 'center', xytext = (0, 10), textcoords = 'offset points')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.621757Z","iopub.execute_input":"2022-03-24T06:16:17.622266Z","iopub.status.idle":"2022-03-24T06:16:17.892001Z","shell.execute_reply.started":"2022-03-24T06:16:17.622232Z","shell.execute_reply":"2022-03-24T06:16:17.891159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec_dict = {}\nlabels_list = train[\"primary_label\"].unique().tolist()[:10]\nfor i,spid in enumerate(labels_list):\n    rec_id = train.loc[np.where(train[\"primary_label\"] == spid)][\"primary_label\"].tolist()[0]\n    audio_file = str(list(Path(os.path.join('../input/birdclef-2022/train_audio', labels_list[0]) ).glob('*.ogg'))[0])\n\n#     print(f\"{spid} : {rec_id}\")\n    rec_dict[spid] = rec_id\n    fig, ax = plt.subplots()\n    rec_path = audio_file #Path(train_path) / f\"{rec_id}.flac\"\n    wf, sr = torchaudio.load(rec_path)\n    labels = []\n    handles = [mpl_patches.Rectangle((0, 0), 1, 1, fc=\"white\", ec=\"white\", lw=0, alpha=0)] * 4 \n    labels.append(f\"label: {spid}\")\n    labels.append(f\"min of waveform: {np.around(wf.min(), 5)}\")\n    labels.append(f\"max of waveform: {np.around(wf.max(), 5)}\")\n    labels.append(f\"mean of waveform: {np.around(wf.mean(), 5)}\")\n    ax.legend(handles, labels, loc='best', fontsize='large', \n          fancybox=True, framealpha=0.7, \n          handlelength=0, handletextpad=0)\n    \n    plt.plot(wf.t().numpy())\n    plt.title(\"Species wave form plot for 10 Category\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:16:17.893386Z","iopub.execute_input":"2022-03-24T06:16:17.893802Z","iopub.status.idle":"2022-03-24T06:17:04.465724Z","shell.execute_reply.started":"2022-03-24T06:16:17.893770Z","shell.execute_reply":"2022-03-24T06:17:04.464560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec_dict = {}\ndisplay(HTML(f'<span style=\"color:green\"> <h3>Sound of each Category </h3></span>'))\n\ndef get_species_audio():\n    \"\"\"\n    get audio image for each of the species\n    \"\"\"\n    for i,spid in enumerate(labels_list):\n        rec_id = train.loc[np.where(train[\"primary_label\"] == spid)][\"primary_label\"].tolist()[0]\n    #     print(f\"{spid} : {rec_id}\")\n        rec_dict[spid] = rec_id\n        audio_file = str(list(Path(os.path.join('../input/birdclef-2022/train_audio', labels_list[0]) ).glob('*.ogg'))[0])\n        wf, sr = torchaudio.load(audio_file)\n        display(HTML(f'<span style=\"color:green\"> Category id: {spid} </span>'))\n        ipd.display(ipd.Audio(data=wf, rate=sr))\nget_species_audio()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:17:04.467114Z","iopub.execute_input":"2022-03-24T06:17:04.467486Z","iopub.status.idle":"2022-03-24T06:17:05.886094Z","shell.execute_reply.started":"2022-03-24T06:17:04.467454Z","shell.execute_reply":"2022-03-24T06:17:05.884193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter_geo(\n    train,\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    width=1_000,\n    height=500,\n    title=\"BirdCLEF 2022 Training Data Location Plot\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:17:05.888326Z","iopub.execute_input":"2022-03-24T06:17:05.888593Z","iopub.status.idle":"2022-03-24T06:17:06.500357Z","shell.execute_reply.started":"2022-03-24T06:17:05.888563Z","shell.execute_reply":"2022-03-24T06:17:06.499141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec_dict = {}\ndisplay(HTML(f'<span style=\"color:green\"> <h3>spectrogram of each species </h3></span>'))\ndef get_species_spectrogram():\n    \"\"\"\n    get audio image for each of the species\n    \"\"\"\n    for i,spid in enumerate(labels_list):\n        rec_id = train.loc[np.where(train[\"primary_label\"] == spid)][\"primary_label\"].tolist()[0]\n        rec_dict[spid] = rec_id\n        audio_file = str(list(Path(os.path.join('../input/birdclef-2022/train_audio', labels_list[0]) ).glob('*.ogg'))[0])\n        wf, sr = torchaudio.load(audio_file)\n        specgram = torchaudio.transforms.Spectrogram()(wf)\n\n        fig, ax = plt.subplots(nrows=1, ncols=1, figsize=(20, 4))\n        ax.tick_params(axis='x', labelsize=10)\n        ax.tick_params(axis='y', labelsize=10)\n\n        cax = ax.matshow(\n            specgram.log2()[0,:,:].numpy(),\n            interpolation=\"nearest\",\n            aspect=\"auto\",\n            cmap=plt.cm.afmhot,\n            origin=\"lower\",\n        )\n        fig.colorbar(cax)\n\n        plt.title(f\"Spectrogram for Category id: {spid}\")\n        plt.show()\nget_species_spectrogram()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:17:06.502456Z","iopub.execute_input":"2022-03-24T06:17:06.502847Z","iopub.status.idle":"2022-03-24T06:17:11.397733Z","shell.execute_reply.started":"2022-03-24T06:17:06.502811Z","shell.execute_reply":"2022-03-24T06:17:11.397062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/birdclef-2022/sample_submission.csv')\nsubmission['target'] = True\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-24T06:17:11.398709Z","iopub.execute_input":"2022-03-24T06:17:11.399297Z","iopub.status.idle":"2022-03-24T06:17:11.415029Z","shell.execute_reply.started":"2022-03-24T06:17:11.399253Z","shell.execute_reply":"2022-03-24T06:17:11.414366Z"},"trusted":true},"execution_count":null,"outputs":[]}]}