{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![Bengali.AI Speech Recognition](https://storage.googleapis.com/kaggle-competitions/kaggle/52324/logos/header.png?t=2023-07-10-22-33-39)","metadata":{"_uuid":"94b3ce4a-0de0-44ef-8d29-e0831556c7fe","_cell_guid":"79dd7ccd-035e-4d83-95fc-3c7feb1be768","trusted":true}},{"cell_type":"code","source":"! pip install -qq wordcloud\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom textblob import TextBlob\nfrom IPython.core.display import HTML\nfrom random import randint\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\ndef render_elem(values, style):\n    if style == 'section':\n        display(HTML(f'<div class=\"section\">{values}</div>'))\n    html = \"\"\n    flag = False\n    for value in values:\n        if style == 'insight':\n            html += f'<ul class=\"bullet\"><li class=\"insight\"> - {value}</li></ul>'\n        if style == 'data-insight':\n            if ':' in value:\n                flag = True\n                html += f'<ul class=\"bullet\"><li class=\"data-insight\"><span>{value.split(\":\")[0]}</span>{value.split(\":\")[1]}</li></ul>'\n            else:\n                html += f'<ul class=\"bullet\"><li class=\"data-insight\">{value}</li></ul>'\n    \n    if flag == True:\n        html += \"<hr>\"\n    display(HTML(html))\ndef css_styling():\n    styles = \"\"\"\n    <style>\n        .section {\n            color:#33A1C9;\n            font-size:20px;\n            font-weight: bold;\n            padding-top:15px;\n        }\n        .sub_section {\n            color:#f1c40f;\n            font-size:16px;\n            font-weight: bold;\n            padding-top:10px;\n        }\n        .insight {\n            color:#C70039;\n            font-size:14px;\n            font-weight: bold;\n        }\n        .data-insight {\n            background-color: #f9f9f9;\n            border-left: 4px solid #C70039;\n            padding: 8px;\n            margin-top: 5px;\n            font-weight: normal;\n        }\n        .data-insight span{\n            font-weight: bold;\n            display: block;\n        }\n        .bullet {\n            list-style-type: circle;\n        }\n        ul{\n            margin:0px;\n            display: contents;\n            list-style-type: square;\n        }\n    </style>\n    \"\"\"\n    return HTML(styles)\n\ncss_styling()","metadata":{"_uuid":"8b90f927-96f8-4522-b97e-91e22d7b213c","_cell_guid":"342366b5-bc62-4b1f-8f15-1aea5f4cb1da","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:36.833606Z","iopub.execute_input":"2023-07-17T22:44:36.834075Z","iopub.status.idle":"2023-07-17T22:44:50.967669Z","shell.execute_reply.started":"2023-07-17T22:44:36.834041Z","shell.execute_reply":"2023-07-17T22:44:50.965921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Introduction\", \"section\")\nrender_elem([\"It's speach to text competition for the Bengali language\"], \"data-insight\")","metadata":{"_uuid":"47e7f9de-4e3f-4c58-9ff3-7bc510de1d43","_cell_guid":"4f739364-62d7-4958-9cbe-184b02c36467","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:50.971298Z","iopub.execute_input":"2023-07-17T22:44:50.972368Z","iopub.status.idle":"2023-07-17T22:44:50.985550Z","shell.execute_reply.started":"2023-07-17T22:44:50.972290Z","shell.execute_reply":"2023-07-17T22:44:50.984305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Loading the Dataset\", \"section\")\ndf = pd.read_csv('/kaggle/input/bengaliai-speech/train.csv')\ndf.head()","metadata":{"_uuid":"3c762ed1-e89d-4057-8bfa-a3a956c1679e","_cell_guid":"ccfa8baa-4680-4302-a5b0-f4c29d36f384","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:50.986914Z","iopub.execute_input":"2023-07-17T22:44:50.987340Z","iopub.status.idle":"2023-07-17T22:44:55.574496Z","shell.execute_reply.started":"2023-07-17T22:44:50.987308Z","shell.execute_reply":"2023-07-17T22:44:55.573190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem([\"ID: ID of the recording\", \"sentence: transcribe of the recording\", \"split: train/valid\"], \"data-insight\")","metadata":{"_uuid":"036d53e8-348c-49ed-a5e7-7b27159d5867","_cell_guid":"14cb6b94-cbff-4496-b4bf-a5d1d922beeb","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:55.577275Z","iopub.execute_input":"2023-07-17T22:44:55.577976Z","iopub.status.idle":"2023-07-17T22:44:55.586465Z","shell.execute_reply.started":"2023-07-17T22:44:55.577939Z","shell.execute_reply":"2023-07-17T22:44:55.585124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Train vs Valid\", \"section\")\nsplit_distribution = df['split'].value_counts()\nplt.figure(figsize=(6,6))\nplt.pie(split_distribution, labels=split_distribution.index, autopct='%1.1f%%')\nplt.title('Distribution of the Train vs Valid')\nplt.show()\nrender_elem([\"Validation data is more quality then train data from data page, better to use same as cv stategy\"], \"insight\")","metadata":{"_uuid":"9ed3f15f-3aa9-47e9-81f0-0ab228ac5743","_cell_guid":"7000363c-eafb-4ef6-8621-fa5f134be2f9","collapsed":false,"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-17T22:44:55.588338Z","iopub.execute_input":"2023-07-17T22:44:55.588829Z","iopub.status.idle":"2023-07-17T22:44:55.926040Z","shell.execute_reply.started":"2023-07-17T22:44:55.588789Z","shell.execute_reply":"2023-07-17T22:44:55.924831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Train Samples\", \"section\")\nfor _ in range(0,3):\n    index = randint(0,len(data))\n    sample = [f\"{k}: {v[0]}\" for k,v in data[index:index+1].reset_index(drop=True).to_dict().items()]\n    render_elem(sample, \"data-insight\")","metadata":{"_uuid":"8409151d-b07b-4bcc-bdf0-86f4fe945886","_cell_guid":"88a40783-a713-4d53-8379-cc60cc649adb","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:55.927763Z","iopub.execute_input":"2023-07-17T22:44:55.929488Z","iopub.status.idle":"2023-07-17T22:44:55.951913Z","shell.execute_reply.started":"2023-07-17T22:44:55.929439Z","shell.execute_reply":"2023-07-17T22:44:55.950626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Setence length distribution\", \"section\")\nimport matplotlib.pyplot as plt\n\ndf['sentence_length'] = df['sentence'].apply(len)\n\nplt.figure(figsize=(10,6))\nplt.hist(df['sentence_length'], bins=50, alpha=0.5, color='g')\nplt.title('Distribution of Sentence Lengths')\nplt.xlabel('Sentence Length')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()\nrender_elem([\"Most of the sentences are relatively short, with the majority having a length of fewer than 50 characters\"], \"insight\")","metadata":{"_uuid":"21510017-c4a3-4616-aa7d-9ef9b3e8eaa0","_cell_guid":"d509bb9a-6d6b-4560-b6a1-4b9fbfca5da9","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:44:55.953901Z","iopub.execute_input":"2023-07-17T22:44:55.954812Z","iopub.status.idle":"2023-07-17T22:44:57.038246Z","shell.execute_reply.started":"2023-07-17T22:44:55.954766Z","shell.execute_reply":"2023-07-17T22:44:57.037090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"#Word distribution\", \"section\")\nimport matplotlib.pyplot as plt\n\ndf['sentence_words'] = df['sentence'].apply(lambda x: len(x.split(\" \")))\n\nplt.figure(figsize=(10,6))\nplt.hist(df['sentence_words'], bins=50, alpha=0.5, color='g')\nplt.title('Distribution of Words')\nplt.xlabel('Words')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()\nrender_elem([\"Maximum 20 words\"], \"insight\")","metadata":{"_uuid":"2b1ac19c-fbeb-47d0-a8ed-6a1e7c3bacde","_cell_guid":"4f4e8c77-d073-4d17-a7ba-2dce692f52cc","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-07-17T22:47:05.307306Z","iopub.execute_input":"2023-07-17T22:47:05.307828Z","iopub.status.idle":"2023-07-17T22:47:07.330015Z","shell.execute_reply.started":"2023-07-17T22:47:05.307791Z","shell.execute_reply":"2023-07-17T22:47:07.328558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"Top 20 Common Words\", \"section\")\nfrom collections import Counter\nwords = ' '.join(df['sentence']).split()\nword_freq = Counter(words)\ncommon_words = word_freq.most_common(20)\nprint(common_words)\nwords, counts = zip(*common_words)\nplt.figure(figsize=(12,8))\nplt.barh(words, counts, color='skyblue')\nplt.xlabel('Frequency')\nplt.ylabel('Words')\nplt.title('Top 20 Most Common Words')\nplt.gca().invert_yaxis()\nplt.show()\nrender_elem([\"Top 20 Most common words repeat in 10% of the recordings\"], \"insight\")","metadata":{"_uuid":"e92e621d-0d7f-46d8-a051-7fc53c96e129","_cell_guid":"a72c46e5-6bc3-4435-ba65-d0da60c7dd59","collapsed":false,"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-17T22:44:59.101304Z","iopub.execute_input":"2023-07-17T22:44:59.101756Z","iopub.status.idle":"2023-07-17T22:45:05.013261Z","shell.execute_reply.started":"2023-07-17T22:44:59.101722Z","shell.execute_reply":"2023-07-17T22:45:05.011847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"WordCloud\", \"section\")\nfrom wordcloud import WordCloud\ndf_sample = df.sample(n=10000, random_state=1)\ntext_sample = ' '.join(sentence for sentence in df_sample['sentence'])\nwordcloud_sample = WordCloud(width = 800, height = 400, \n                             background_color ='white', \n                             stopwords = None,\n                             min_font_size = 10).generate(text_sample)\nplt.figure(figsize = (8, 8), facecolor = None) \nplt.imshow(wordcloud_sample)\nplt.axis(\"off\")\nplt.tight_layout(pad = 0)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-17T22:45:05.017351Z","iopub.execute_input":"2023-07-17T22:45:05.017766Z","iopub.status.idle":"2023-07-17T22:45:07.784752Z","shell.execute_reply.started":"2023-07-17T22:45:05.017735Z","shell.execute_reply":"2023-07-17T22:45:07.783167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"render_elem(\"To Be Continue...\", \"section\")\nrender_elem([\"Upvote: If you like this notebook\"], \"data-insight\")","metadata":{"_uuid":"9a0010c0-6a08-4ab4-90dd-7674a9a81dcb","_cell_guid":"a1172dbe-222c-47cb-9261-5a8d356379cf","collapsed":false,"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-17T22:45:07.787025Z","iopub.execute_input":"2023-07-17T22:45:07.788314Z","iopub.status.idle":"2023-07-17T22:45:07.798793Z","shell.execute_reply.started":"2023-07-17T22:45:07.788278Z","shell.execute_reply":"2023-07-17T22:45:07.797504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"8dd42367-3804-46fb-859f-82220e3307a1","_cell_guid":"9af5057a-b1d9-4f24-a48f-c08e3a8d6372","collapsed":false,"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"d55a48c1-2bea-4be8-b16f-42e1a41bf86a","_cell_guid":"5c59f253-9421-49ed-ad22-3aff73dc970a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}