{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:05.076596Z","iopub.execute_input":"2021-11-27T18:55:05.077013Z","iopub.status.idle":"2021-11-27T18:55:05.112293Z","shell.execute_reply.started":"2021-11-27T18:55:05.076915Z","shell.execute_reply":"2021-11-27T18:55:05.11144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responses_data = pd.read_csv('/kaggle/input/kaggle-survey-2021/kaggle_survey_2021_responses.csv')\nresponses_data.describe(include='all')\nresponses_data=responses_data.iloc[1:]","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:05.113902Z","iopub.execute_input":"2021-11-27T18:55:05.114129Z","iopub.status.idle":"2021-11-27T18:55:08.034086Z","shell.execute_reply.started":"2021-11-27T18:55:05.114101Z","shell.execute_reply":"2021-11-27T18:55:08.033397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def function(row):\n    if (row['Q3'] == 'Turkey'): \n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Greece':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Poland':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Ukraine':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Romania':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Belarus':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Czech Republic':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'Russia':\n        return \"Eastern_Europe\"\n    elif row['Q3'] == 'United States of America':\n        return \"USA\"\n    elif row['Q3'] == 'Belgium':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Italy':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Spain':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'United Kingdom of Great Britain and Northern Ireland':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'France':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Switzerland':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Sweden':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Austria':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Ireland':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Netherlands':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Portugal':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Denmark':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Germany':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'Norway':\n        return \"Rest_of_Europe\"\n    elif row['Q3'] == 'India':\n        return \"India\"\n    else:\n        return \"Other_countries\"\n\nresponses_data['Location'] = responses_data.apply(function, axis=1)\n\nEXPERIENCE_ORDER = ['I have never written code', '< 1 years', '1-3 years', '3-5 years', '5-10 years', '10-20 years', '20+ years']\nGENDER_ORDER = ['Man', 'Woman', 'Nonbinary', 'Prefer not to say', 'Prefer to self-describe']\nEARNINGS_ORDER = ['$0-999','1,000-1,999','2,000-2,999','3,000-3,999','4,000-4,999','5,000-7,499','7,500-9,999','10,000-14,999','15,000-19,999','20,000-24,999','25,000-29,999','30,000-39,999','40,000-49,999','50,000-59,999','60,000-69,999','70,000-79,999','80,000-89,999','90,000-99,999','100,000-124,999','125,000-149,999','150,000-199,999','200,000-249,999','250,000-299,999','300,000-499,999','$500,000-999,999','>$1,000,000']\nAGE = ['18-21', '22-24', '25-29', '30-34', '35-39', '40-44', '45-49', '50-54', '55-59', '60-69', '70+']\n\n#responses_data = responses_data[responses_data['Q25'].notna()] ##students!!!!\nresponses_data.head(20)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:08.035109Z","iopub.execute_input":"2021-11-27T18:55:08.035791Z","iopub.status.idle":"2021-11-27T18:55:10.862821Z","shell.execute_reply.started":"2021-11-27T18:55:08.03576Z","shell.execute_reply":"2021-11-27T18:55:10.861881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (responses_data['Q25'] == '0-999'),\n    (responses_data['Q25'] == '1,000-1,999'),\n    (responses_data['Q25'] == '2,000-2,999'),\n    (responses_data['Q25'] == '3,000-3,999'),\n    (responses_data['Q25'] == '4,000-4,999'),\n    (responses_data['Q25'] == '5,000-7,499'),\n    (responses_data['Q25'] == '7,500-9,999'),\n    (responses_data['Q25'] == '10,000-14,999'),\n    (responses_data['Q25'] == '15,000-19,999'),\n    (responses_data['Q25'] == '20,000-24,999'),\n    (responses_data['Q25'] == '25,000-29,999'),\n    (responses_data['Q25'] == '30,000-39,999'),\n    (responses_data['Q25'] == '40,000-49,999'),\n    (responses_data['Q25'] == '50,000-59,999'),\n    (responses_data['Q25'] == '60,000-69,999'),\n    (responses_data['Q25'] == '70,000-79,999'),\n    (responses_data['Q25'] == '80,000-89,999'),\n    (responses_data['Q25'] == '90,000-99,999'),\n    (responses_data['Q25'] == '100,000-124,999'),\n    (responses_data['Q25'] == '125,000-149,999'),\n    (responses_data['Q25'] == '150,000-199,999'),\n    (responses_data['Q25'] == '200,000-249,999'),\n    (responses_data['Q25'] == '250,000-299,999'),\n    (responses_data['Q25'] == '300,000-499,999'),\n    (responses_data['Q25'] == '500,000-999,999'),\n    (responses_data['Q25'] == '>1,000,000')]\n    \n\n# create a list of the values we want to assign for each condition\nvalues = ['$0-999','1,000-1,999','2,000-2,999','3,000-3,999','4,000-4,999','5,000-7,499','7,500-9,999','10,000-14,999','15,000-19,999','20,000-24,999','25,000-29,999','30,000-39,999','40,000-49,999','50,000-59,999','60,000-69,999','70,000-79,999','80,000-89,999','90,000-99,999','100,000-124,999','125,000-149,999','150,000-199,999','200,000-249,999','250,000-299,999','300,000-499,999','$500,000-999,999','>$1,000,000']\n#values = np.array(values, dtype=int)\n\n# create a new column and use np.select to assign values to it using our lists as arguments\nresponses_data['Salary'] = np.select(conditions, values)\n#responses_data[responses_data['Q3']==\"USA\"].head(40)\n#responses_data","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:10.864718Z","iopub.execute_input":"2021-11-27T18:55:10.865166Z","iopub.status.idle":"2021-11-27T18:55:10.961169Z","shell.execute_reply.started":"2021-11-27T18:55:10.865131Z","shell.execute_reply":"2021-11-27T18:55:10.960131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Introduction\n\nKaggle 2021 survey gives us an another opportunity to investigate trends in data world. In this notebook we will focus on the kagglers from the eastern european countries and we will make comparison with the kagglers from India, USA, rest of Europe and rest of the world. \nCountries that we have included as eastern european are Russia, Turkey, Poland, Ukraine, Greece, Czech Republic, Romania and Belarus. Some of them have gone through similar processes of getting independence and economical transition in the late years of the 20th century. Opening to global markets and access to world wide web have been an opportunity for them to catch up with the west.","metadata":{}},{"cell_type":"markdown","source":"## Part I","metadata":{}},{"cell_type":"markdown","source":"In the first part of the notebook we will look into the structure of kaggle users in east Europe. \n* The **highest** number of kagglers are coming from **Russia** and the **lowest** number from **Belarus**.\n* **19% of kagglers** are in the age group **between 25 and 29 years**, making it the most represented age group\n* There are **over 82% men** and **almost 16% of women**\n* **85,3% are highly educated**, having at least bachelor's degree\n* The most represented **job role** is the one of **data scientist**, followed with students\n","metadata":{}},{"cell_type":"markdown","source":" ","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 6))\n#exp_age_count = responses_data[\"Location\"]#.sort_values('Location')\nCountries_order = ['Other_countries', 'India', 'Rest_of_Europe', 'USA', 'Eastern_Europe']\nax = sns.countplot(x=\"Location\", data=responses_data, order=Countries_order, zorder=3, palette='Blues_r')#, hue=\"Q6\", hue_order=EXPERIENCE_ORDER)\n\nsum = len(responses_data['Q3'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\n\n\n#plt.legend(loc='upper right')\nplt.xlabel('Location')\nplt.title('Distribution of kagglers by location')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:10.962552Z","iopub.execute_input":"2021-11-27T18:55:10.962871Z","iopub.status.idle":"2021-11-27T18:55:12.290862Z","shell.execute_reply.started":"2021-11-27T18:55:10.962837Z","shell.execute_reply":"2021-11-27T18:55:12.289979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nEast_Europe = responses_data[responses_data['Location']=='Eastern_Europe']\ncountries_list = ['Russia', 'Turkey', 'Poland', 'Ukraine', 'Greece', 'Czech Republic', 'Romania', 'Belarus']\nax = sns.countplot(x=\"Q3\", data=East_Europe, order=countries_list, zorder=3, palette = 'Blues_r')#, hue=\"Q6\", hue_order=EXPERIENCE_ORDER)\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\n\nsum = len(East_Europe['Q3'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\n\n#plt.legend(loc='upper right')\nplt.xlabel('Countries')\nplt.title('Distribution of kagglers by eastern european countries')\nplt.grid(axis='y', zorder=0)\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:12.292544Z","iopub.execute_input":"2021-11-27T18:55:12.292874Z","iopub.status.idle":"2021-11-27T18:55:12.902214Z","shell.execute_reply.started":"2021-11-27T18:55:12.292816Z","shell.execute_reply":"2021-11-27T18:55:12.901253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\n\nax = sns.countplot(x=\"Q1\", data=East_Europe, order=AGE, palette='Blues_r', zorder = 3)\n#ax.set_xticklabels(age_groups,rotation=40,ha=\"right\")\nplt.title(\"Distribution by age for eastern Europe\")\nsum = len(East_Europe[\"Q1\"])\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\nplt.xlabel('Age')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:12.903825Z","iopub.execute_input":"2021-11-27T18:55:12.904477Z","iopub.status.idle":"2021-11-27T18:55:13.259422Z","shell.execute_reply.started":"2021-11-27T18:55:12.904434Z","shell.execute_reply":"2021-11-27T18:55:13.25852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\ngender = (East_Europe['Q2']\n    .value_counts()\n    .to_frame()\n    .reset_index()\n    .rename(columns={'index':'Gender', 'Q2':'Count'})\n    .replace(['Man','Woman'], ['Male', 'Female']) \n    .groupby('Gender')\n    .sum()\n    .reset_index())  \n\nfig = go.Figure(data=[go.Pie(labels=gender['Gender'], \n                             values=gender['Count'])])\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:13.260636Z","iopub.execute_input":"2021-11-27T18:55:13.260886Z","iopub.status.idle":"2021-11-27T18:55:13.406612Z","shell.execute_reply.started":"2021-11-27T18:55:13.260858Z","shell.execute_reply":"2021-11-27T18:55:13.405691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\nEDUCATION = [\"Master’s degree\", \"Bachelor’s degree\", \"Doctoral degree\", \"Some college/university study without earning a bachelor’s degree\", \"I prefer not to answer\", \"No formal education past high school\", 'Professional doctorate']\n\nax = sns.countplot(x=\"Q4\", data=East_Europe, zorder=3, order=EDUCATION, palette ='Blues_r')\n\nsum = len(East_Europe['Q4'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.title('Education distribution in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Level of education')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:13.407832Z","iopub.execute_input":"2021-11-27T18:55:13.408153Z","iopub.status.idle":"2021-11-27T18:55:13.756129Z","shell.execute_reply.started":"2021-11-27T18:55:13.40812Z","shell.execute_reply":"2021-11-27T18:55:13.755501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\nEDU_ORDER = [\"Data Scientist\", \"Student\", \"Software Engineer\", \"Research Scientist\", \"Data Analyst\", \"Machine Learning Engineer\", \"Currently not employed\", \"Other\", \"Business Analyst\"]\n\nax = sns.countplot(x=\"Q5\", data=East_Europe, zorder=3, order=EDU_ORDER, palette ='Blues_r')\n\nsum = len(East_Europe['Q5'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.title('Top job roles of kagglers in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Job role')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:13.758516Z","iopub.execute_input":"2021-11-27T18:55:13.759148Z","iopub.status.idle":"2021-11-27T18:55:14.101103Z","shell.execute_reply.started":"2021-11-27T18:55:13.759117Z","shell.execute_reply":"2021-11-27T18:55:14.100548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#top_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\nEast_Europe_salary = East_Europe[East_Europe['Salary']!='0']\nax = sns.countplot(x=\"Salary\", data=East_Europe_salary, order = EARNINGS_ORDER, zorder=3, palette ='Blues_r')\n\nplt.title('Distibution of earnings in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Earnings')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:14.102381Z","iopub.execute_input":"2021-11-27T18:55:14.102613Z","iopub.status.idle":"2021-11-27T18:55:14.569Z","shell.execute_reply.started":"2021-11-27T18:55:14.102586Z","shell.execute_reply":"2021-11-27T18:55:14.568244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part II\n\nFrom the distribution of earnings we notice that the highest number of kagglers is earning between 10,000 USD and 14,999 USD. However, let's see who are top earners - the one with annual salary over 70,000 USD.\n\n* 25% of top earners have **between 45 and 49 years**\n* 50% of top earners are coming either from **Russia** or **Turkey**\n* The highest number of top earners doesn't have doctorate degree but **master's degree**\n* Also, among the top earners there are **no kagglers** who do not have at least bachelor's degree\n* Most of them are **data scientists**\n* **Over 50%** of them are having **more than 10 years of experience**\n* **Over 30%** of them are having **more than 20 years of experience**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\ntop_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\n\nax = sns.countplot(x=\"Q1\", data=top_east_earn, order=AGE, zorder=3, palette ='Blues_r')\n\nsum = len(top_east_earn['Q1'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.title('Age distibution of top earners in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Age')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:14.570198Z","iopub.execute_input":"2021-11-27T18:55:14.570401Z","iopub.status.idle":"2021-11-27T18:55:14.908417Z","shell.execute_reply.started":"2021-11-27T18:55:14.570375Z","shell.execute_reply":"2021-11-27T18:55:14.907628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#top_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\n\nax = sns.countplot(x=\"Q3\", data=top_east_earn, order=top_east_earn.Q3.value_counts().index , zorder=3, palette = 'Blues_r')\n\nsum = len(top_east_earn['Q3'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.title('Country distibution of top earners in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Country')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:14.909799Z","iopub.execute_input":"2021-11-27T18:55:14.910036Z","iopub.status.idle":"2021-11-27T18:55:15.21355Z","shell.execute_reply.started":"2021-11-27T18:55:14.910008Z","shell.execute_reply":"2021-11-27T18:55:15.212756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#top_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\n\nsns.countplot(x=\"Q4\", data=top_east_earn, order=top_east_earn.Q4.value_counts().index , zorder=3, palette = 'Blues_r')\n\nplt.title('Education of top earners in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Education')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:15.214919Z","iopub.execute_input":"2021-11-27T18:55:15.215344Z","iopub.status.idle":"2021-11-27T18:55:15.439945Z","shell.execute_reply.started":"2021-11-27T18:55:15.215314Z","shell.execute_reply":"2021-11-27T18:55:15.439165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#top_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\n\nax = sns.countplot(x=\"Q5\", data=top_east_earn, order=top_east_earn.Q5.value_counts().index , zorder=3, palette = 'Blues_r')\n\nsum = len(top_east_earn['Q5'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.title('Job roles of top earners in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Job role')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:15.441236Z","iopub.execute_input":"2021-11-27T18:55:15.441442Z","iopub.status.idle":"2021-11-27T18:55:15.814723Z","shell.execute_reply.started":"2021-11-27T18:55:15.441416Z","shell.execute_reply":"2021-11-27T18:55:15.813809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#top_east_earn = East_Europe[(East_Europe['Salary']=='90,000-99,999') | (East_Europe['Salary']=='100,000-124,999') | (East_Europe['Salary']=='125,000-149,999') | (East_Europe['Salary']=='150,000-199,999') | (East_Europe['Salary']=='200,000-249,999') | (East_Europe['Salary']=='250,000-299,999') | (East_Europe['Salary']=='300,000-499,999') | (East_Europe['Salary']=='>$1,000,000')]\n\nax = sns.countplot(x=\"Q6\", data=top_east_earn, order=top_east_earn.Q6.value_counts().index , zorder=3, palette = 'Blues_r')\n\nsum = len(top_east_earn['Q6'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\n\nplt.title('Years of experience of top earners in east Europe')\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Years of experience')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:15.815948Z","iopub.execute_input":"2021-11-27T18:55:15.816172Z","iopub.status.idle":"2021-11-27T18:55:16.132787Z","shell.execute_reply.started":"2021-11-27T18:55:15.816144Z","shell.execute_reply":"2021-11-27T18:55:16.131982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part III\n\nNow when we now who are kagglers from eastern Europe and who are top earners, let's make a comparison with the two most represented nations on Kaggle - India and USA, but also with the rest of european countries. All the other countries we will call \"other countries\".\n\n*  Comparison of distribution of experience is showing an interesting trend - **eastern europeans are more similar to kagglers from India** then to kagglers from rest of Europe\n* Not only that, but we see **two trends** - **East Europe, India and other countries** are having the same distribution of experience, while the **rest of Europe is having distribution more similar to the USA** (more kagglers with 5-10 years and 20+ years of experience)!\n* The highest number of kagglers from **eastern Europe is between 25-29 years old (19%)** with the distribution similar to rest of Europe and rest of the world. In **India 18-21** age group is dominating, while in the **USA 30-34** (with **higher number of 60-69** comparing to the others)\n* Men are dominating in each teritory group, with the ratios female to male: **USA - 1:3.2**, **India - 1:3.4**, **Rest of Europe - 1:5.1**, **East Europe - 1:5.2**\n* Comparison of earnings is revealing that the **highest salaries are in the USA**, followed by other european countries, and then **similar distribution in east Europe and India**\n* However, in each of this teritories, most of the **kagglers are paid more then the average** \n* Comparison of **job roles** is once again showing **similarities between \"non-east\" european countries and USA**, where job role \"other\" is on 3rd place. In the appendix, we'll look into this.\n* While **data scientist and students are dominating** everywhere, most of the **unemployed** kagglers are from **India**, a the least from \"non-east\" european countries","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\n#exp_age_count = responses_data[\"Location\"]#.sort_values('Location')\nCountries_order = ['Other_countries', 'India', 'Rest_of_Europe', 'USA', 'Eastern_Europe']\nsns.countplot(x=\"Location\", data=responses_data, order=Countries_order, hue=\"Q6\", hue_order=EXPERIENCE_ORDER, zorder=3, palette = 'magma')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\n\n\n#plt.legend(loc='upper right')\nplt.xlabel('Location')\nplt.title('Distribution of experience by locations')\nplt.grid(axis='y', zorder=0)\nplt.legend(title='Years of experience')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:16.13397Z","iopub.execute_input":"2021-11-27T18:55:16.134185Z","iopub.status.idle":"2021-11-27T18:55:16.767206Z","shell.execute_reply.started":"2021-11-27T18:55:16.134158Z","shell.execute_reply":"2021-11-27T18:55:16.766312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\n#exp_age_count = responses_data[\"Location\"]#.sort_values('Location')\nCountries_order = ['Other_countries', 'India', 'Rest_of_Europe', 'USA', 'Eastern_Europe']\ng = sns.countplot(x=\"Location\", data=responses_data, order=Countries_order, hue=\"Q1\", hue_order=AGE, zorder=3, palette='magma')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\n\n\n#plt.legend(loc='upper right')\nplt.xlabel('Location')\nplt.title('Distribution of age by locations')\nplt.grid(axis='y', zorder=0)\nplt.legend(title='Age')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:16.768345Z","iopub.execute_input":"2021-11-27T18:55:16.768562Z","iopub.status.idle":"2021-11-27T18:55:17.341813Z","shell.execute_reply.started":"2021-11-27T18:55:16.768536Z","shell.execute_reply":"2021-11-27T18:55:17.340946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n#exp_age_count = responses_data[\"Location\"]#.sort_values('Location')\nCountries_order = ['Other_countries', 'India', 'Rest_of_Europe', 'USA', 'Eastern_Europe']\nax=sns.countplot(x=\"Location\", data=responses_data, order=Countries_order, hue=\"Q2\", hue_order=GENDER_ORDER, zorder=3, palette='magma') \n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\n\n\nplt.legend(title='Gender')\nplt.xlabel('Location')\nplt.title('Distribution of gender by locations')\nplt.grid(axis='y', zorder=0)\n\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:17.342848Z","iopub.execute_input":"2021-11-27T18:55:17.343055Z","iopub.status.idle":"2021-11-27T18:55:17.743796Z","shell.execute_reply.started":"2021-11-27T18:55:17.343029Z","shell.execute_reply":"2021-11-27T18:55:17.743037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(15, 6))\nEast_Europe = responses_data[responses_data['Location']=='Eastern_Europe']\nUSA = responses_data[responses_data['Location']=='USA']\nIndia = responses_data[responses_data['Location']=='India']\nEurope = responses_data[responses_data['Location']=='Rest_of_Europe']\n\nfig, axs = plt.subplots(nrows= 2,ncols=2, figsize=(15, 6))\n\nfig.suptitle('Comparison of pay')\n\nsns.countplot(x='Salary',data=East_Europe, order=EARNINGS_ORDER, ax=axs[0,0], palette='magma').set_title('East Europe')\naxs[0,0].arrow(6, 125, 0, -35, head_width = 1, head_length = 8, linewidth = 1.0, color = 'black', length_includes_head = True)\n\nsns.countplot(x='Salary',data=USA, order=EARNINGS_ORDER, ax=axs[0,1], palette='magma').set_title('USA')\naxs[0,1].arrow(12, 155, 0, -50, head_width = 1, head_length = 8, linewidth = 1.0, color = 'black', length_includes_head = True)\n\nsns.countplot(x='Salary',data=India, order=EARNINGS_ORDER, ax=axs[1,0], palette='magma').set_title('India')\naxs[1,0].arrow(5, 400, 0, -80, head_width = 1, head_length = 8, linewidth = 1.0, color = 'black', length_includes_head = True)\n\nsns.countplot(x='Salary',data=Europe, order=EARNINGS_ORDER, ax=axs[1,1], palette='magma').set_title('Rest of Europe')\n#axs[1,1].arrow(6, 125, 0, -35, head_width = 1, head_length = 8, linewidth = 1.0, color = 'black', length_includes_head = True)\n\n\nfor ax in axs.flat:\n    ax.tick_params(labelrotation=90)\n\nfor ax in axs.flat:\n    ax.label_outer()\n    \nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:17.744919Z","iopub.execute_input":"2021-11-27T18:55:17.745132Z","iopub.status.idle":"2021-11-27T18:55:19.063316Z","shell.execute_reply.started":"2021-11-27T18:55:17.745106Z","shell.execute_reply":"2021-11-27T18:55:19.062274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Arrows are indicating average annual salary*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 10))\nx,y = 'Location', 'Q5'\n\nnorm_data = responses_data.groupby(x)[y].value_counts(normalize=True)\nnorm_data = norm_data.mul(100)\nnorm_data = norm_data.rename(\"Percent\").reset_index()\n\n\ng = sns.catplot(x='Location', y=\"Percent\", kind='bar', data=norm_data, hue='Q5', height=6, aspect=2.5, zorder=3)\ng.ax.set_ylim(0,40)\n#sns.set(rc={'figure.figsize':(30,10)})\n#for p in g.ax.patches:\n#    txt = str(p.get_height().round(1)) + '%'\n#    txt_x = p.get_x() \n#    txt_y = p.get_height()\n#    g.ax.text(txt_x,txt_y,txt)\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.grid(axis='y', zorder=0)\nplt.title('Type of job')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T19:45:54.959433Z","iopub.execute_input":"2021-11-27T19:45:54.96027Z","iopub.status.idle":"2021-11-27T19:45:55.997608Z","shell.execute_reply.started":"2021-11-27T19:45:54.960235Z","shell.execute_reply":"2021-11-27T19:45:55.996593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\nunempl_Data = norm_data[norm_data['Q5']=='Currently not employed']\nax = sns.barplot(x=\"Location\", y=\"Percent\",data=unempl_Data, order=unempl_Data.sort_values('Percent', ascending=False).Location, zorder=3, palette = 'magma')#, order=unempl_Data.Location.value_counts().iloc[:10].index, zorder=3, palette='Blues_r')\n#ax.bar_label(ax.containers[0])\n\n#unempl_Data.Percent.sum()/5\n\nplt.xlabel('Location')\nplt.ylabel('Percentage (%)')\nplt.title('Unemployed Kaggle users by location')\nplt.grid(axis='y', zorder=0)\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:20.105569Z","iopub.execute_input":"2021-11-27T18:55:20.105917Z","iopub.status.idle":"2021-11-27T18:55:20.365223Z","shell.execute_reply.started":"2021-11-27T18:55:20.105887Z","shell.execute_reply":"2021-11-27T18:55:20.364472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Appendix - Job role: Other","metadata":{}},{"cell_type":"markdown","source":"I have been surprised with the high number of those who answered for job role - \"other\", especially since a lot of job roles has been listed as options. So, let's have a look into their profile:\n\n* Most of them are coming from **India or USA**\n* They have **less than 3 years of experience**\n* They are using **basic statistical softwares, Microsoft Power BI or Tableau**\n* They **don't have experience with ML methods**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nresponses_data['Q3'] = responses_data['Q3'].str.replace('United States of America', 'USA')\nresponses_data['Q3'] = responses_data['Q3'].str.replace('United Kingdom of Great Britain and Northern Ireland', 'UK')\n\nresponses_other = responses_data[responses_data['Q5'] == 'Other']\nax = sns.countplot(x='Q3',data=responses_other, order=responses_other.Q3.value_counts().iloc[:10].index, zorder=3, palette='flare')\n\nsum = len(responses_other['Q3'])\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / sum:.1f}%\\n'\n    x = p.get_x() + p.get_width() / 2\n    y = p.get_height()\n    ax.annotate(percentage, (x, y), ha='center', va='center')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Country')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - Countries')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T19:55:25.01883Z","iopub.execute_input":"2021-11-27T19:55:25.019308Z","iopub.status.idle":"2021-11-27T19:55:25.415348Z","shell.execute_reply.started":"2021-11-27T19:55:25.019258Z","shell.execute_reply":"2021-11-27T19:55:25.414706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='Q6',data=responses_other, order=responses_other.Q6.value_counts().iloc[:20].index, zorder=3, palette='flare')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Years of experience')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - Years of experience')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:20.757811Z","iopub.execute_input":"2021-11-27T18:55:20.758591Z","iopub.status.idle":"2021-11-27T18:55:21.157127Z","shell.execute_reply.started":"2021-11-27T18:55:20.758544Z","shell.execute_reply":"2021-11-27T18:55:21.156147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='Q41',data=responses_other, order=responses_other.Q41.value_counts().iloc[:20].index, zorder=3, palette='flare')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Tool')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - Primary tool to analyze data')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:21.158638Z","iopub.execute_input":"2021-11-27T18:55:21.158981Z","iopub.status.idle":"2021-11-27T18:55:21.46339Z","shell.execute_reply.started":"2021-11-27T18:55:21.15894Z","shell.execute_reply":"2021-11-27T18:55:21.462486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='Q26',data=responses_other, order=responses_other.Q26.value_counts().iloc[:20].index, zorder=3, palette='flare')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('Earnings')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - Earnings')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:21.46462Z","iopub.execute_input":"2021-11-27T18:55:21.464833Z","iopub.status.idle":"2021-11-27T18:55:21.99371Z","shell.execute_reply.started":"2021-11-27T18:55:21.464807Z","shell.execute_reply":"2021-11-27T18:55:21.992883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='Q23',data=responses_other, order=responses_other.Q23.value_counts().iloc[:20].index, zorder=3, palette='flare')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('ML methods')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - using ML methods')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:21.995148Z","iopub.execute_input":"2021-11-27T18:55:21.996278Z","iopub.status.idle":"2021-11-27T18:55:22.344382Z","shell.execute_reply.started":"2021-11-27T18:55:21.996228Z","shell.execute_reply":"2021-11-27T18:55:22.343555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='Q35',data=responses_other, order=responses_other.Q35.value_counts().iloc[:10].index, zorder=3, palette='flare')\n\nplt.xticks(\n    rotation=45, \n    horizontalalignment='right')\nplt.xlabel('BI tools')\nplt.grid(axis='y', zorder=0)\nplt.title('Job: Other - business intelligence tools')\nNone","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-11-27T18:55:22.34838Z","iopub.execute_input":"2021-11-27T18:55:22.349916Z","iopub.status.idle":"2021-11-27T18:55:22.650662Z","shell.execute_reply.started":"2021-11-27T18:55:22.349868Z","shell.execute_reply":"2021-11-27T18:55:22.649683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusions\n\nConsidering demographic attributes, kagglers from east Europe are similar to other users. \n\nHowever, distributions of experience and salaries are showing that they are more close to India than to USA and the rest of Europe!\n\nOne way how they can get closer to western live standards is high education, followed with years of experience...","metadata":{}}]}