{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install bnunicodenormalizer\n!pip install indicparser\n!pip install multiprocesspandas","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:13.665577Z","iopub.execute_input":"2022-07-01T23:56:13.666027Z","iopub.status.idle":"2022-07-01T23:56:46.644655Z","shell.execute_reply.started":"2022-07-01T23:56:13.665994Z","shell.execute_reply":"2022-07-01T23:56:46.643341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport math\nfrom bnunicodenormalizer import Normalizer \nfrom pprint import pprint\nfrom tqdm import tqdm\nfrom datasets import load_dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:46.648069Z","iopub.execute_input":"2022-07-01T23:56:46.648478Z","iopub.status.idle":"2022-07-01T23:56:46.654997Z","shell.execute_reply.started":"2022-07-01T23:56:46.648444Z","shell.execute_reply":"2022-07-01T23:56:46.653731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dftrain = pd.read_csv('../input/dlsprint/train.csv')\ndfval = pd.read_csv('../input/dlsprint/validation.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:46.657147Z","iopub.execute_input":"2022-07-01T23:56:46.65753Z","iopub.status.idle":"2022-07-01T23:56:48.065949Z","shell.execute_reply.started":"2022-07-01T23:56:46.657496Z","shell.execute_reply":"2022-07-01T23:56:48.064735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dftrain), len(dfval)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.069353Z","iopub.execute_input":"2022-07-01T23:56:48.069728Z","iopub.status.idle":"2022-07-01T23:56:48.078558Z","shell.execute_reply.started":"2022-07-01T23:56:48.069696Z","shell.execute_reply":"2022-07-01T23:56:48.077634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking the dataframes","metadata":{}},{"cell_type":"code","source":"dftrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.079882Z","iopub.execute_input":"2022-07-01T23:56:48.08073Z","iopub.status.idle":"2022-07-01T23:56:48.102694Z","shell.execute_reply.started":"2022-07-01T23:56:48.080691Z","shell.execute_reply":"2022-07-01T23:56:48.101451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfval.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.104341Z","iopub.execute_input":"2022-07-01T23:56:48.10473Z","iopub.status.idle":"2022-07-01T23:56:48.123229Z","shell.execute_reply.started":"2022-07-01T23:56:48.104697Z","shell.execute_reply":"2022-07-01T23:56:48.122291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Converting the splits into pandas dataframes","metadata":{}},{"cell_type":"markdown","source":"#### Dropping null sentence rows","metadata":{}},{"cell_type":"code","source":"dftrain = dftrain[dftrain['sentence'].notna()]\ndfval = dfval[dfval['sentence'].notna()]","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.125399Z","iopub.execute_input":"2022-07-01T23:56:48.126388Z","iopub.status.idle":"2022-07-01T23:56:48.186053Z","shell.execute_reply.started":"2022-07-01T23:56:48.12635Z","shell.execute_reply":"2022-07-01T23:56:48.18507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Merging the splits into one","metadata":{}},{"cell_type":"code","source":"df = pd.concat([dftrain, dfval], ignore_index=True)\ndf.dropna(inplace = True, subset = ['sentence'])\nlen(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.187869Z","iopub.execute_input":"2022-07-01T23:56:48.188685Z","iopub.status.idle":"2022-07-01T23:56:48.36251Z","shell.execute_reply.started":"2022-07-01T23:56:48.188639Z","shell.execute_reply":"2022-07-01T23:56:48.36144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.363951Z","iopub.execute_input":"2022-07-01T23:56:48.364281Z","iopub.status.idle":"2022-07-01T23:56:48.382607Z","shell.execute_reply.started":"2022-07-01T23:56:48.364252Z","shell.execute_reply":"2022-07-01T23:56:48.381228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Normalizing the sentences","metadata":{}},{"cell_type":"code","source":"def normalizer(row, bnorm=Normalizer()):\n    normalized = []\n    \n    for word in row.split():\n        if bnorm(word)['normalized'] is not None:\n            normalized.append(bnorm(word)['normalized'])\n        else:\n            error = word\n            normalized.append(word)\n    try:\n        return \" \".join(normalized)\n\n    except:\n        print(f'The sentence is: {row}, word: {error}')\n        return row\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.394048Z","iopub.execute_input":"2022-07-01T23:56:48.394777Z","iopub.status.idle":"2022-07-01T23:56:48.404845Z","shell.execute_reply.started":"2022-07-01T23:56:48.394738Z","shell.execute_reply":"2022-07-01T23:56:48.40362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['normalized'] = df['sentence'].apply_parallel(normalizer, num_processes = 4)\ndf['normalized']","metadata":{"execution":{"iopub.status.busy":"2022-07-01T23:56:48.406272Z","iopub.execute_input":"2022-07-01T23:56:48.407218Z","iopub.status.idle":"2022-07-02T00:10:58.742675Z","shell.execute_reply.started":"2022-07-01T23:56:48.407184Z","shell.execute_reply":"2022-07-02T00:10:58.74134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from indicparser import graphemeParser","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:13:32.985036Z","iopub.execute_input":"2022-07-02T00:13:32.985685Z","iopub.status.idle":"2022-07-02T00:13:32.994254Z","shell.execute_reply.started":"2022-07-02T00:13:32.985638Z","shell.execute_reply":"2022-07-02T00:13:32.992349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Parsing the graphemes","metadata":{}},{"cell_type":"code","source":"def gparser(dataframe, column_name: str):\n    graphemes = []\n    \n    for i in tqdm(dataframe.index):\n        for word in dataframe[column_name][i].split():\n            try:\n                grapheme = graphemeParser(\"bangla\").process(word)\n                for g in grapheme:\n                    graphemes.append(g)\n\n            except Exception as e:\n                print(word)\n\n    return graphemes","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:17:44.058977Z","iopub.execute_input":"2022-07-02T00:17:44.059446Z","iopub.status.idle":"2022-07-02T00:17:44.069515Z","shell.execute_reply.started":"2022-07-02T00:17:44.059394Z","shell.execute_reply":"2022-07-02T00:17:44.06795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"graphemes = gparser(df, column_name = 'normalized')\ngraphemes_unique = list(set(graphemes))","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:17:44.580566Z","iopub.execute_input":"2022-07-02T00:17:44.581235Z","iopub.status.idle":"2022-07-02T00:18:25.53824Z","shell.execute_reply.started":"2022-07-02T00:17:44.581199Z","shell.execute_reply":"2022-07-02T00:18:25.536759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(graphemes_unique)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:18:38.175969Z","iopub.execute_input":"2022-07-02T00:18:38.177323Z","iopub.status.idle":"2022-07-02T00:18:38.18789Z","shell.execute_reply.started":"2022-07-02T00:18:38.177265Z","shell.execute_reply":"2022-07-02T00:18:38.186119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Finding out the 50 most common words","metadata":{}},{"cell_type":"code","source":"from collections import Counter\ncount = Counter(graphemes)\n\ncommon_graphemes = count.most_common(50)\ncommon_graphemes","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:18:38.965321Z","iopub.execute_input":"2022-07-02T00:18:38.965832Z","iopub.status.idle":"2022-07-02T00:18:39.702753Z","shell.execute_reply.started":"2022-07-02T00:18:38.965786Z","shell.execute_reply":"2022-07-02T00:18:39.701431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Word counter","metadata":{}},{"cell_type":"markdown","source":"#### Finding out the 50 most common graphemes","metadata":{}},{"cell_type":"code","source":"def wordcount(df):\n    words = []\n    for i in tqdm(df.index):\n        for word in df['normalized'][i].split():\n            words.append(word)\n    count = Counter(words)\n    return count","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:20:50.795855Z","iopub.execute_input":"2022-07-02T00:20:50.796548Z","iopub.status.idle":"2022-07-02T00:20:50.804522Z","shell.execute_reply.started":"2022-07-02T00:20:50.796502Z","shell.execute_reply":"2022-07-02T00:20:50.80327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words = wordcount(df)\ncommon_words = words.most_common(50)\ncommon_words","metadata":{"execution":{"iopub.status.busy":"2022-07-02T01:12:38.252854Z","iopub.execute_input":"2022-07-02T01:12:38.253303Z","iopub.status.idle":"2022-07-02T01:12:42.313838Z","shell.execute_reply.started":"2022-07-02T01:12:38.253268Z","shell.execute_reply":"2022-07-02T01:12:42.312956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def counterToSortedDF(counter):\n    df = pd.DataFrame.from_dict(counter, orient = 'index')\n    df = df.rename(columns={'index':'word', 0:'count'})\n    df= df.sort_values(by = ['count'], ascending=False).reset_index()\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-02T02:23:18.513477Z","iopub.execute_input":"2022-07-02T02:23:18.514006Z","iopub.status.idle":"2022-07-02T02:23:18.525503Z","shell.execute_reply.started":"2022-07-02T02:23:18.513968Z","shell.execute_reply":"2022-07-02T02:23:18.523935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordsDF = counterToSortedDF(words)\nwordsDF","metadata":{"execution":{"iopub.status.busy":"2022-07-02T02:23:19.200227Z","iopub.execute_input":"2022-07-02T02:23:19.200691Z","iopub.status.idle":"2022-07-02T02:23:19.302877Z","shell.execute_reply.started":"2022-07-02T02:23:19.200654Z","shell.execute_reply":"2022-07-02T02:23:19.301462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['gender'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:34:15.933211Z","iopub.execute_input":"2022-07-02T00:34:15.933889Z","iopub.status.idle":"2022-07-02T00:34:15.978436Z","shell.execute_reply.started":"2022-07-02T00:34:15.93384Z","shell.execute_reply":"2022-07-02T00:34:15.977018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Plotting functions","metadata":{}},{"cell_type":"code","source":"import plotly.offline as pyo\nimport plotly.graph_objs as go\nimport numpy as np\nimport matplotlib.pyplot as plt \nfrom matplotlib import cm \nimport plotly.express as px\nimport math\n\npyo.init_notebook_mode()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:39:20.805898Z","iopub.execute_input":"2022-07-02T00:39:20.806558Z","iopub.status.idle":"2022-07-02T00:39:35.679271Z","shell.execute_reply.started":"2022-07-02T00:39:20.806509Z","shell.execute_reply":"2022-07-02T00:39:35.677513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def valueCounter(columnname: str, df):\n    vals = df[columnname].unique()\n    count = {}\n    for val in vals:\n        if isinstance(val, float):\n            if math.isnan(val):\n                count['unspecified'] = df[columnname].isna().sum()\n        else:\n            count[val] = df[df[columnname]==val].shape[0]\n\n    return count","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:34:24.282712Z","iopub.execute_input":"2022-07-02T00:34:24.283457Z","iopub.status.idle":"2022-07-02T00:34:24.2954Z","shell.execute_reply.started":"2022-07-02T00:34:24.283368Z","shell.execute_reply":"2022-07-02T00:34:24.293469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plotpiePlotly(criterion: str, split: str, Dict, colorscale = 'Viridis', legend= False):\n    values = [float(v) for v in Dict.values()]\n    labels = [k for k in Dict.keys()]\n    # cmap = plt.get_cmap('Spectral')\n    # viridis = cm.get_cmap('viridis', 12)\n    # colors = cmap(np.linspace(0,1,3))\n    n_colors = len(Dict.keys())\n    colors = px.colors.sample_colorscale(colorscale, [n/(n_colors -1) for n in range(n_colors)])\n\n    title = criterion + ' ratio among contributors in '+ split + ' set'\n\n    piechart = go.Pie(labels= labels, values= values, marker=dict(colors=colors, line=dict(color='#FFF', width=2)),\n                                                                 showlegend= legend, textinfo='label+percent')\n\n    layout = go.Layout(height = 800,\n                    width = 800,\n                    autosize = False,\n                    font = dict(family='Arial', size=12, color='rgb(150,150,150)'), \n                    plot_bgcolor= 'rgba(0,0,0,0)',\n                    paper_bgcolor= 'rgba(0,0,0,0)')\n    fig = go.Figure(data = piechart, layout= layout)\n\n\n    pyo.iplot(fig, filename='pie_chart')\n#     fig.write_image(f\"Images/{title}.pdf\", scale = 1, height = 1*600, width = 1*600)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:40:15.718595Z","iopub.execute_input":"2022-07-02T00:40:15.719801Z","iopub.status.idle":"2022-07-02T00:40:15.731086Z","shell.execute_reply.started":"2022-07-02T00:40:15.719752Z","shell.execute_reply":"2022-07-02T00:40:15.73002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Gender ratio","metadata":{}},{"cell_type":"code","source":"genratio = valueCounter('gender', df)\ngenratio","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:54:43.328377Z","iopub.execute_input":"2022-07-02T00:54:43.328884Z","iopub.status.idle":"2022-07-02T00:54:43.438712Z","shell.execute_reply.started":"2022-07-02T00:54:43.328845Z","shell.execute_reply":"2022-07-02T00:54:43.437285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"genratiotrain = valueCounter('gender', dftrain)\ngenratiotrain","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:35:28.082773Z","iopub.execute_input":"2022-07-02T00:35:28.083224Z","iopub.status.idle":"2022-07-02T00:35:28.323298Z","shell.execute_reply.started":"2022-07-02T00:35:28.08318Z","shell.execute_reply":"2022-07-02T00:35:28.322335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"genratioval = valueCounter('gender', dfval)\ngenratioval","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:35:34.77621Z","iopub.execute_input":"2022-07-02T00:35:34.77717Z","iopub.status.idle":"2022-07-02T00:35:34.801267Z","shell.execute_reply.started":"2022-07-02T00:35:34.777095Z","shell.execute_reply":"2022-07-02T00:35:34.799636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plotpiePlotly('Gender', 'All', genratio)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T00:55:50.245982Z","iopub.execute_input":"2022-07-02T00:55:50.246451Z","iopub.status.idle":"2022-07-02T00:55:50.295054Z","shell.execute_reply.started":"2022-07-02T00:55:50.246397Z","shell.execute_reply":"2022-07-02T00:55:50.293969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Different accents ratio","metadata":{}},{"cell_type":"code","source":"accent_ratio = valueCounter('accents', dfval)\naccent_ratio_train, accent_ratio_val, accent_ratio","metadata":{"execution":{"iopub.status.busy":"2022-07-02T02:46:03.399461Z","iopub.execute_input":"2022-07-02T02:46:03.399914Z","iopub.status.idle":"2022-07-02T02:46:03.47285Z","shell.execute_reply.started":"2022-07-02T02:46:03.39988Z","shell.execute_reply":"2022-07-02T02:46:03.471387Z"},"trusted":true},"execution_count":null,"outputs":[]}]}