{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport math\nfrom textwrap import wrap\nwarnings.filterwarnings('ignore')\nsns.set_palette('Set2')\nsns.set_style('darkgrid')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T22:44:27.683656Z","iopub.execute_input":"2022-08-13T22:44:27.684124Z","iopub.status.idle":"2022-08-13T22:44:33.596318Z","shell.execute_reply.started":"2022-08-13T22:44:27.684033Z","shell.execute_reply":"2022-08-13T22:44:33.594884Z"},"_kg_hide-input":true,"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#EMOJI\n\n\"Emoji are ideograms and smileys used in electronic messages and web pages. Some examples of emoji are 😃, 🧘🏻‍♂️, 🌍, 🍞, 🚗, 📞, 🎉, ♥️, and 🏁. Emoji exist in various genres, including facial expressions, common objects, places and types of weather, and animals. They are much like emoticons, but emoji are pictures rather than typographic approximations; the term \"emoji\" in the strict sense refers to such pictures which can be represented as encoded characters, but it is sometimes applied to messaging stickers by extension.\"\n\nhttps://en.wikipedia.org/wiki/Emoji","metadata":{}},{"cell_type":"markdown","source":"![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcTOuC-bvw2CEvxTV76VThnoWV_EZmsHvn36ig&usqp=CAU)twinkl.com.br)","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/nca-emoji-challenge/train.csv', encoding='utf8')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:52:23.537556Z","iopub.execute_input":"2022-08-13T22:52:23.537959Z","iopub.status.idle":"2022-08-13T22:52:23.559519Z","shell.execute_reply.started":"2022-08-13T22:52:23.537927Z","shell.execute_reply":"2022-08-13T22:52:23.558150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/nca-emoji-challenge/test.csv', encoding='utf8')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:50:24.473958Z","iopub.execute_input":"2022-08-13T22:50:24.474362Z","iopub.status.idle":"2022-08-13T22:50:24.498781Z","shell.execute_reply.started":"2022-08-13T22:50:24.474330Z","shell.execute_reply":"2022-08-13T22:50:24.497880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"emo = pd.read_csv('../input/nca-emoji-challenge/emojis_reference.csv', encoding='utf8')\nemo.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:48:18.885512Z","iopub.execute_input":"2022-08-13T22:48:18.885937Z","iopub.status.idle":"2022-08-13T22:48:18.939146Z","shell.execute_reply.started":"2022-08-13T22:48:18.885901Z","shell.execute_reply":"2022-08-13T22:48:18.938247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Mohammad Imran Shaikh https://www.kaggle.com/shikhnu/covid19-tweets-eda-visualization-wordcloud\n\nunique_df = pd.DataFrame()\nunique_df['Features'] = df.columns\nunique=[]\nfor i in df.columns:\n    unique.append(df[i].nunique())\nunique_df['Uniques'] = unique\n\nf, ax = plt.subplots(1,1, figsize=(15,7))\n\nsplot = sns.barplot(x=unique_df['Features'], y=unique_df['Uniques'], alpha=0.8)\nfor p in splot.patches:\n    splot.annotate(format(p.get_height(), '.0f'), (p.get_x() + p.get_width() / 2., p.get_height()), ha = 'center',\n                   va = 'center', xytext = (0, 9), textcoords = 'offset points')\nplt.title('Bar plot for number of unique values in each column',weight='bold', size=15)\nplt.ylabel('#Unique values', size=12, weight='bold')\nplt.xlabel('Features', size=12, weight='bold')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:52:48.939245Z","iopub.execute_input":"2022-08-13T22:52:48.939672Z","iopub.status.idle":"2022-08-13T22:52:49.263947Z","shell.execute_reply.started":"2022-08-13T22:52:48.939637Z","shell.execute_reply":"2022-08-13T22:52:49.262577Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#word cloud\nfrom wordcloud import WordCloud, ImageColorGenerator\ntext = \" \".join(str(each) for each in df.type)\n# Create and generate a word cloud image:\nwordcloud = WordCloud(max_words=200,colormap='Set1', background_color=\"purple\").generate(text)\nplt.figure(figsize=(10,6))\nplt.figure(figsize=(15,10))\n# Display the generated image:\nplt.imshow(wordcloud, interpolation='Bilinear')\nplt.axis(\"off\")\nplt.figure(1,figsize=(12, 12))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:54:35.314615Z","iopub.execute_input":"2022-08-13T22:54:35.315084Z","iopub.status.idle":"2022-08-13T22:54:35.708848Z","shell.execute_reply.started":"2022-08-13T22:54:35.315039Z","shell.execute_reply":"2022-08-13T22:54:35.707496Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum=df['clues'].str.len()\nprint(sum)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:56:07.066996Z","iopub.execute_input":"2022-08-13T22:56:07.067440Z","iopub.status.idle":"2022-08-13T22:56:07.075780Z","shell.execute_reply.started":"2022-08-13T22:56:07.067402Z","shell.execute_reply":"2022-08-13T22:56:07.074643Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's check the length of name, the average length is 20 characters.\ndf['clues length'] = df['clues'].apply(len)\ndf['clues length'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:56:56.654150Z","iopub.execute_input":"2022-08-13T22:56:56.654601Z","iopub.status.idle":"2022-08-13T22:56:56.669930Z","shell.execute_reply.started":"2022-08-13T22:56:56.654563Z","shell.execute_reply":"2022-08-13T22:56:56.668789Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x='type', y=df['clues length'], data=df);","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:58:00.952267Z","iopub.execute_input":"2022-08-13T22:58:00.952648Z","iopub.status.idle":"2022-08-13T22:58:01.193546Z","shell.execute_reply.started":"2022-08-13T22:58:00.952617Z","shell.execute_reply":"2022-08-13T22:58:01.192165Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum=df['target_emoji'].str.len()\nprint(sum)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:00:30.156222Z","iopub.execute_input":"2022-08-13T23:00:30.156666Z","iopub.status.idle":"2022-08-13T23:00:30.163826Z","shell.execute_reply.started":"2022-08-13T23:00:30.156619Z","shell.execute_reply":"2022-08-13T23:00:30.162820Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's check the length of emojis, the average length is 20 characters.\ndf['target_emoji length'] = df['target_emoji'].apply(len)\ndf['target_emoji length'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:03:12.793909Z","iopub.execute_input":"2022-08-13T23:03:12.794607Z","iopub.status.idle":"2022-08-13T23:03:12.809629Z","shell.execute_reply.started":"2022-08-13T23:03:12.794565Z","shell.execute_reply":"2022-08-13T23:03:12.808235Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x='type', y=df['target_emoji length'], data=df);","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:07:29.349445Z","iopub.execute_input":"2022-08-13T23:07:29.349904Z","iopub.status.idle":"2022-08-13T23:07:29.526755Z","shell.execute_reply.started":"2022-08-13T23:07:29.349855Z","shell.execute_reply":"2022-08-13T23:07:29.525686Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = (df.isna().sum()/df.shape[0]*100)<50\ndf_modified = df[s.index[s].tolist()]\nprint (df_modified.shape)\ndf_modified.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:08:10.394780Z","iopub.execute_input":"2022-08-13T23:08:10.396086Z","iopub.status.idle":"2022-08-13T23:08:10.420979Z","shell.execute_reply.started":"2022-08-13T23:08:10.396042Z","shell.execute_reply":"2022-08-13T23:08:10.419762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['font.size'] = 14\nfig, ax = plt.subplots(3, 2, figsize=(20,20))\nfor col, ax in zip(['type','clues','emojis_comp','ucodes_comp','target_emoji', 'target_nca'], ax.flat):\n    dict_ = df_modified[col].value_counts().head(10).to_dict()\n    if ('Not Available' in dict_.keys()):\n        dict_.pop('Not Available')\n    labels = []\n    for i in dict_.keys():\n        i = i.split(' ')\n        if (len(i) > 6):\n            i[math.ceil(len(i)/2)-1] += '\\n'\n            labels.append(' '.join(i))\n        else:\n            labels.append(' '.join(i))\n    ax.pie(x=list(dict_.values()), labels=labels, shadow=True, startangle=0)\n    \n    col = (' '.join(col.split('_'))).upper()\n    ax.set_title(col, weight='bold', fontsize=18)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:10:53.807918Z","iopub.execute_input":"2022-08-13T23:10:53.808317Z","iopub.status.idle":"2022-08-13T23:10:55.454335Z","shell.execute_reply.started":"2022-08-13T23:10:53.808285Z","shell.execute_reply":"2022-08-13T23:10:55.453104Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Savita Nair https://www.kaggle.com/savitanair/hr-analytics\n\nprint(f'Dataset has {len(df.type.unique())} unique groups')\nprint('*'*20)\nprint(f'And the top 10 counts are :')\nprint(df.type.value_counts().head(10))\nprint('*'*20)\n\nc = df.type.value_counts().head(10)\nfig, ax = plt.subplots(1,1,figsize=(12,6))\nax.bar(c.index, c.values, width=0.8, color='y')\nplt.xticks(rotation=45)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:13:05.304945Z","iopub.execute_input":"2022-08-13T23:13:05.305367Z","iopub.status.idle":"2022-08-13T23:13:05.676079Z","shell.execute_reply.started":"2022-08-13T23:13:05.305331Z","shell.execute_reply":"2022-08-13T23:13:05.674873Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Savita Nair https://www.kaggle.com/savitanair/hr-analytics\n\nprint(f'Dataset has {len(df.target_emoji.unique())} unique emojis')\nprint('*'*20)\nprint(f'And the top 10 counts are :')\nprint(df.target_emoji.value_counts().head(10))\nprint('*'*20)\n\nc = df.target_emoji.value_counts().head(10)\nfig, ax = plt.subplots(1,1,figsize=(12,6))\nax.bar(c.index, c.values, width=0.8, color='r')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:14:37.745938Z","iopub.execute_input":"2022-08-13T23:14:37.746408Z","iopub.status.idle":"2022-08-13T23:14:38.019154Z","shell.execute_reply.started":"2022-08-13T23:14:37.746369Z","shell.execute_reply":"2022-08-13T23:14:38.017932Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Savita Nair https://www.kaggle.com/savitanair/hr-analytics\n\nprint(f'Dataset has {len(df.clues.unique())} unique names')\nprint('*'*20)\nprint(f'And the top 10 counts are :')\nprint(df.clues.value_counts().head(10))\nprint('*'*20)\n\nc = df.clues.value_counts().head(10)\nfig, ax = plt.subplots(1,1,figsize=(12,6))\nax.bar(c.index, c.values, width=0.8, color='b')\nplt.xticks(rotation=45)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:16:35.919921Z","iopub.execute_input":"2022-08-13T23:16:35.920331Z","iopub.status.idle":"2022-08-13T23:16:36.310705Z","shell.execute_reply.started":"2022-08-13T23:16:35.920297Z","shell.execute_reply":"2022-08-13T23:16:36.309222Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Savita Nair https://www.kaggle.com/savitanair/hr-analytics\n\nprint(f'Dataset has {len(df.emojis_comp.unique())} unique names')\nprint('*'*20)\nprint(f'And the top 10 counts are :')\nprint(df.emojis_comp.value_counts().head(10))\nprint('*'*20)\n\nc = df.emojis_comp.value_counts().head(10)\nfig, ax = plt.subplots(1,1,figsize=(12,6))\nax.bar(c.index, c.values, width=0.8, color='b')\nplt.xticks(rotation=45)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:17:53.189314Z","iopub.execute_input":"2022-08-13T23:17:53.191047Z","iopub.status.idle":"2022-08-13T23:17:53.527141Z","shell.execute_reply.started":"2022-08-13T23:17:53.190986Z","shell.execute_reply":"2022-08-13T23:17:53.525427Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colunas = ['type','clues','emojis_comp','ucodes_comp','target_emoji', 'target_nca']\nfor i in colunas:\n  fig, ax = plt.subplots(1,1, figsize=(15, 6))\n  sns.countplot(y = df[i][1:],data=df.iloc[1:], order=df[i][1:].head(10).value_counts().index, palette='Blues_r')\n  fig.text(0.1, 0.95, f'{df[i][0].split(\"(\")[0]}', fontsize=16, fontweight='bold', fontfamily='serif')\n  plt.xlabel(' ', fontsize=20)\n  plt.ylabel('')\n  plt.yticks(fontsize=13)\n  plt.box(False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:20:36.406772Z","iopub.execute_input":"2022-08-13T23:20:36.407187Z","iopub.status.idle":"2022-08-13T23:20:38.173615Z","shell.execute_reply.started":"2022-08-13T23:20:36.407156Z","shell.execute_reply":"2022-08-13T23:20:38.172060Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install emot","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:33:52.198430Z","iopub.execute_input":"2022-08-13T23:33:52.199035Z","iopub.status.idle":"2022-08-13T23:34:07.588586Z","shell.execute_reply.started":"2022-08-13T23:33:52.198988Z","shell.execute_reply":"2022-08-13T23:34:07.587063Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import emot","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:34:18.033375Z","iopub.execute_input":"2022-08-13T23:34:18.034196Z","iopub.status.idle":"2022-08-13T23:34:18.049048Z","shell.execute_reply.started":"2022-08-13T23:34:18.034152Z","shell.execute_reply":"2022-08-13T23:34:18.047844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Use yours for Kaggle Competitions","metadata":{}},{"cell_type":"code","source":"#3rd row, 3rd column\n\ndf.iloc[2,2]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:47:21.407177Z","iopub.execute_input":"2022-08-13T23:47:21.407571Z","iopub.status.idle":"2022-08-13T23:47:21.417231Z","shell.execute_reply.started":"2022-08-13T23:47:21.407540Z","shell.execute_reply":"2022-08-13T23:47:21.415264Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Respective Emojis for 'Use yours for Kaggle competitions' above\n\n#Use your Emojination since this means: bread, backhand_index_pointing_right, red_question_mark","metadata":{}},{"cell_type":"code","source":"text = \"🍞, 👉, ❓\"\n\n#3rd row, 3rd column 'use yours for Kaggle competitions'  df.iloc [2,2]\n\nemot_obj = emot.core.emot()\n\nemot_obj.emoji(text)\n\n#ans = emot.emoji(text)\n#ans","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:42:55.903229Z","iopub.execute_input":"2022-08-13T23:42:55.903666Z","iopub.status.idle":"2022-08-13T23:42:56.491264Z","shell.execute_reply.started":"2022-08-13T23:42:55.903628Z","shell.execute_reply":"2022-08-13T23:42:56.490412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Acknowledgements: \n\nNeel Shah / @NeelShah18  Emot\n\nShubham Rohilla / @kakashubham\n\n@NeelShah18: https://github.com/NeelShah18 @kakashubham: https://github.com/kakashubham\n\nhttps://emot.readthedocs.io/en/latest/\n\nMohammad Imran Shaikh https://www.kaggle.com/shikhnu/covid19-tweets-eda-visualization-wordcloud\n\nSavita Nair https://www.kaggle.com/savitanair/hr-analytics","metadata":{}},{"cell_type":"markdown","source":"#I see.\n\n![](https://media2.giphy.com/media/3o7buirYcmV5nSwIRW/200w.webp?cid=ecf05e470nhdl4kgjcphjutm33vtne8lpjzyzv2fikr6h4gn&rid=200w.webp&ct=g)","metadata":{}}]}