{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np  \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\npd.options.display.max_rows = 100\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"id":"iCtoUGn_br_l","execution":{"iopub.status.busy":"2022-08-05T18:08:21.316508Z","iopub.execute_input":"2022-08-05T18:08:21.317001Z","iopub.status.idle":"2022-08-05T18:08:21.324564Z","shell.execute_reply.started":"2022-08-05T18:08:21.316961Z","shell.execute_reply":"2022-08-05T18:08:21.323186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from math import pi","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:09:41.366769Z","iopub.execute_input":"2022-08-05T18:09:41.367249Z","iopub.status.idle":"2022-08-05T18:09:41.372905Z","shell.execute_reply.started":"2022-08-05T18:09:41.367194Z","shell.execute_reply":"2022-08-05T18:09:41.371867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"../input/kaggle-survey-2021/kaggle_survey_2021_responses.csv\")\ndata\n\n","metadata":{"id":"vgpCE5lfFfGR","execution":{"iopub.status.busy":"2022-08-05T18:08:46.364332Z","iopub.execute_input":"2022-08-05T18:08:46.364756Z","iopub.status.idle":"2022-08-05T18:08:48.420598Z","shell.execute_reply.started":"2022-08-05T18:08:46.364723Z","shell.execute_reply":"2022-08-05T18:08:48.419639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"id":"5ia5MR5gF_mb","outputId":"930938e8-7da4-4b0d-93e6-e7618a856261","execution":{"iopub.status.busy":"2022-08-05T18:08:50.226509Z","iopub.execute_input":"2022-08-05T18:08:50.227554Z","iopub.status.idle":"2022-08-05T18:08:50.491097Z","shell.execute_reply.started":"2022-08-05T18:08:50.227499Z","shell.execute_reply":"2022-08-05T18:08:50.489809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"other_color = \"salmon\"\nkorea_color = \"skyblue\"\ndata['korea'] = data['Q3'] == 'South Korea'","metadata":{"id":"S-haPQHWGBwA","execution":{"iopub.status.busy":"2022-08-05T18:09:04.708830Z","iopub.execute_input":"2022-08-05T18:09:04.709639Z","iopub.status.idle":"2022-08-05T18:09:04.721109Z","shell.execute_reply.started":"2022-08-05T18:09:04.709599Z","shell.execute_reply":"2022-08-05T18:09:04.720131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = data['Q3'].value_counts(normalize = True)\ndf1 = df1.mul(100)           #Each is multiplied by 100\ndf1 = df1.rename('percent').reset_index()\ndf1 = df1.iloc[:20]\ndf1 = df1.drop([2])\ndf1['index'] = df1['index'].replace({'United States of America' : 'USA', \n                                    'United Kingdom of Great Britain and Northern Ireland' : 'UK'})\n\n\ng = sns.catplot(x = 'index', y = 'percent', kind = 'bar', data = df1,\n               height = 6, aspect = 4/2, palette= [other_color, other_color, other_color, other_color, \n                                                   other_color, other_color, other_color, other_color, \n                                                   other_color, other_color, other_color, other_color, \n                                           other_color, other_color, other_color, korea_color], alpha = 0.3)\ng.ax.set_ylim(0, 30)\nfor p in g.ax.patches:\n    txt = str(p.get_height().round(1)) + '%'\n    txt_x = p.get_x() + 0.2\n    txt_y = p.get_height() + 0.5\n    g.ax.text(txt_x, txt_y , txt)\n\ng.ax.set_xlabel('Country')\ng.ax.set_ylabel('Percentage [%]')\ng.ax.set_title('Where do Kagglers Currently Reside?')\nplt.show()","metadata":{"id":"TTIUXr83GUt4","outputId":"04444fc0-1174-49a9-bc9d-3c047d1857a0","execution":{"iopub.status.busy":"2022-08-05T18:09:04.892108Z","iopub.execute_input":"2022-08-05T18:09:04.892546Z","iopub.status.idle":"2022-08-05T18:09:05.438045Z","shell.execute_reply.started":"2022-08-05T18:09:04.892512Z","shell.execute_reply":"2022-08-05T18:09:05.436800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q2'].value_counts().to_frame()\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q2')['amount']\n\ntemp['all'] = temp.sum(axis = 1)\nplt.rcParams.update({'font.size' : 14})\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all'] * 100\n\ntemp = temp.drop('all', axis = 1)\nwidth = 0.5\n\nf, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (6, 6))\nax.bar(['Rest of World', 'South Korea'], temp['Man'].values, \n       width, label = 'Men', color = ['skyblue', 'skyblue'], alpha = 0.3)\nax.bar(['Rest of World', 'South Korea'], temp['Woman'].values, width, \n       bottom = temp['Man'].values, label = 'Women', color = ['salmon', 'salmon'], alpha = 0.3)\n\nax.annotate('18.8 %', [-0.06, 87], fontsize=15)\nax.annotate('20.0 %', [1-0.06, 87], fontsize=15)\n\nax.set_ylabel('Percentage [%]')\nax.set_title('Comparison of Gender Distribution')\nax.legend(loc='lower center')\n\nplt.show()\n","metadata":{"id":"m-t2GeAYGsvz","outputId":"c7421293-75f3-47e1-8689-c4fa8cab2a3a","execution":{"iopub.status.busy":"2022-08-05T18:15:58.158050Z","iopub.execute_input":"2022-08-05T18:15:58.158988Z","iopub.status.idle":"2022-08-05T18:15:58.369540Z","shell.execute_reply.started":"2022-08-05T18:15:58.158945Z","shell.execute_reply":"2022-08-05T18:15:58.368294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q4'].value_counts().to_frame()\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q4')['amount']\n\ntemp['all'] = temp.sum(axis = 1)\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all'] * 100\n\ntemp = temp.drop('all', axis = 1)\ntemp = temp[data['Q4'].value_counts().index]\ntemp.rename(columns={'Some college/university study without earning a bachelor’s degree' : 'another bachelor''s',\n                    'I prefer not to answer' : 'NAN', 'No formal education past high school' : 'high school'}, inplace = True)\n\ncategories = list([c.replace(' ', '\\n') for c in temp.columns])\nangles = [n / float(len(categories)) * 2 * pi for n in range(len(categories))]\nangles += angles[:1]\n\nfig, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (8, 8), subplot_kw=dict(polar = True))\nplt.xticks(angles[:-1], categories)\nplt.ylim(0, 50)\n\nvalues = temp.iloc[0].values.flatten().tolist()\nvalues += values[:1]\nax.plot(angles, values, other_color, linewidth = 3, linestyle = 'solid', label = 'Row')\nax.fill(angles, values, other_color, alpha = 0.3)\n\nvalues = temp.iloc[1].values.flatten().tolist()\nvalues += values[:1]\nax.plot(angles, values, korea_color, linewidth = 3, linestyle = 'solid', label = 'Korea')\nax.fill(angles, values, korea_color, alpha = 0.3)\nplt.title('Comparsion of Degree\\n\\n')\nplt.legend(loc = 'lower left')\nplt.show()","metadata":{"id":"ErjEyuWpG-8Y","outputId":"bca480ab-1fe9-47c8-fbdf-97e54b08e834","execution":{"iopub.status.busy":"2022-08-05T18:16:00.296728Z","iopub.execute_input":"2022-08-05T18:16:00.297326Z","iopub.status.idle":"2022-08-05T18:16:00.965059Z","shell.execute_reply.started":"2022-08-05T18:16:00.297282Z","shell.execute_reply":"2022-08-05T18:16:00.963906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q5'].value_counts().to_frame()\n\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q5')['amount']\n\ntemp['all'] = temp.sum(axis = 1)\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all'] * 100\n\ntemp = temp.drop(['all'], axis = 1)\ntemp = temp[data['Q5'].value_counts().index]\n\ntemp = temp.drop(['Currently not employed', 'Other', 'Student'], axis = 1)\ncategories = list([c.replace(' ', '\\n') for c in temp.columns])\n\nangles = [n / float(len(categories)) * 2 * pi for n in range(len(categories))]\nangles += angles[:1]\n\nfig, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (8, 8), subplot_kw=dict(polar = True))\nplt.xticks(angles[:-1], categories)\nplt.ylim(0, 18)\n\nvalues = temp.iloc[0].values.flatten().tolist()\nvalues += values[:1]\nax.plot(angles, values, other_color, linewidth = 3, linestyle = 'solid', label = 'Row')\nax.fill(angles, values, other_color, alpha = 0.3)\n\nvalues = temp.iloc[1].values.flatten().tolist()\nvalues += values[:1]\nax.plot(angles, values, korea_color, linewidth = 3, linestyle = 'solid', label = 'Korea')\nax.fill(angles, values, korea_color, alpha = 0.3)\nplt.title('Comparsion of Role Landscape \\n')\nplt.legend(loc = 'lower left')\nplt.show()\n","metadata":{"id":"nGginPfSHLiY","outputId":"323fc0a2-dae1-4698-9b6c-079d98550452","execution":{"iopub.status.busy":"2022-08-05T18:16:02.985841Z","iopub.execute_input":"2022-08-05T18:16:02.986860Z","iopub.status.idle":"2022-08-05T18:16:03.775708Z","shell.execute_reply.started":"2022-08-05T18:16:02.986818Z","shell.execute_reply":"2022-08-05T18:16:03.774425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q5'].value_counts().to_frame()\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q5')['amount']\n\ntemp['all'] = temp.sum(axis = 1)\n\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all'] * 100\n\ntemp = temp.drop(['all'], axis = 1)\ntemp = temp[data['Q5'].value_counts().index]\n\ntemp = temp.reset_index(drop = False)\ng = sns.catplot(x = 'korea', y = 'Student', kind = 'bar', data = temp,\n               height = 6, aspect = 1/1,\n               palette = [other_color, korea_color], alpha = 0.3)\n\ng.ax.set_ylim(0, 30)\nfor p in g.ax.patches:\n    txt = str(p.get_height().round(1)) + '%'\n    txt_x = p.get_x() + 0.3\n    txt_y = p.get_height() + 0.5\n    g.ax.text(txt_x, txt_y, txt)\n\ng.ax.set_xlabel('')\ng.ax.set_xticklabels(['Rest of World', 'Korea'])\ng.ax.set_ylabel('Percentage [%]')\ng.ax.set_title('Percentage of Student among kagglers')\nplt.show()\n","metadata":{"id":"_q6UxzPmHmTw","outputId":"1aa9a707-7425-4d2e-bb64-b5712a8ebd5a","execution":{"iopub.status.busy":"2022-08-05T18:16:03.778011Z","iopub.execute_input":"2022-08-05T18:16:03.778755Z","iopub.status.idle":"2022-08-05T18:16:04.048267Z","shell.execute_reply.started":"2022-08-05T18:16:03.778706Z","shell.execute_reply":"2022-08-05T18:16:04.047011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea').Q1.value_counts(normalize = True).to_frame()\ntemp = temp * 100\ntemp.columns = ['amount']\n\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q1')['amount']\ntemp = temp.T\ntemp.columns = ['Row', 'Korea']\nlabels = temp.index.values\ny1 = temp['Row'].values\ny2 = temp['Korea'].values\ny2[-1] = 0\n\nwidth = 0.35\nf, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (12, 6))\nx = np.arange(len(labels))\nrects1 = ax.bar(x - width/2, y1, width, label = 'Rest of World', color = other_color, alpha = 0.3)\nrects2 = ax.bar(x + width/2, y2, width, label = 'Korea', color = korea_color, alpha = 0.3)\nplt.xlabel('Age', )\nplt.ylabel('Percentage [%]')\nax.set_xticks(x)\nax.set_xticklabels(labels, rotation = 360)\nplt.legend()\nplt.title('Age Distribution among kaggelrs')\nplt.show()\n\n","metadata":{"id":"miHKh__EH1cB","outputId":"2e7f86ee-2103-4123-bef3-c55cc60d915f","execution":{"iopub.status.busy":"2022-08-05T18:16:04.736642Z","iopub.execute_input":"2022-08-05T18:16:04.737066Z","iopub.status.idle":"2022-08-05T18:16:05.045731Z","shell.execute_reply.started":"2022-08-05T18:16:04.737022Z","shell.execute_reply":"2022-08-05T18:16:05.044475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q15'].value_counts().to_frame()\n\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q15')['amount']\ntemp['all'] = temp.sum(axis = 1)\n\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all']*100\ntemp = temp.drop(['all'], axis = 1)\ntemp = temp[data['Q15'].value_counts().index]\ntemp = temp.reset_index(drop = False)\ntemp.rename(columns={'I do not use machine learning methods' : 'not Learn'}, inplace = True)\ntemp = temp.T\ntemp.columns = ['Row', 'Korea']\ntemp = temp[1:]\nlabels = temp.index.values\nlabels[-1] = '20+ years'\ny1 = temp['Row'].values\ny2 = temp['Korea'].values\n\n\nwidth = 0.35\nf, ax = plt.subplots(nrows = 1, ncols= 1, figsize = (15, 6))\nx = np.arange(len(labels))\nrects1 = ax.bar(x - width/2, y1, width, label = 'Rest of World', color = other_color, alpha = 0.3)\nrects2 = ax.bar(x + width/2, y2, width, label = 'Korea', color = korea_color, alpha = 0.3)\nplt.xlabel('Career Year')\nplt.ylabel('Percentage [%]')\nplt.xticks(np.arange(10), labels=labels, fontsize = 13)\nplt.xticks\nplt.legend()\nplt.title('Career Distribution among Korean kaggelrs')\nplt.show()","metadata":{"id":"6rNxy1U7IEDf","outputId":"6c8dee8a-9167-4d11-f8a6-b25179ece3c1","execution":{"iopub.status.busy":"2022-08-05T18:16:15.809539Z","iopub.execute_input":"2022-08-05T18:16:15.810001Z","iopub.status.idle":"2022-08-05T18:16:16.144443Z","shell.execute_reply.started":"2022-08-05T18:16:15.809961Z","shell.execute_reply":"2022-08-05T18:16:16.143263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = data.groupby('korea')['Q20'].value_counts().to_frame()\ntemp.columns = ['amount']\ntemp = temp.reset_index(drop = False)\ntemp = temp.pivot(index = 'korea', columns='Q20')['amount']\ntemp['all'] = temp.sum(axis = 1)\n\nfor c in temp.columns:\n    temp[c] = temp[c] / temp['all']*100\ntemp = temp.drop(['all'], axis = 1)\ntemp = temp.T\ntemp.columns = ['Row', 'Korea']\nlabels = temp.index.values\ny1 = temp['Row'].values\ny2 = temp['Korea'].values\ny2[6] = 0\nexplode = [0.05 for i in range(18)]\n\nplt.figure(figsize = (20, 10))\nplt.subplot(1, 2, 1)\ncolor = sns.color_palette('pastel')[:18]\nplt.pie(y1, labels = labels, colors=color, explode = explode, autopct='%1.1f%%', textprops= {'fontsize' : 8})\nplt.title('Rest of World : kinds of Kaggler''s Occupation')\nplt.subplot(1, 2, 2)\nplt.pie(y2, labels = labels, colors=color, explode = explode, autopct='%1.1f%%',  textprops= {'fontsize' : 8})\nplt.title('Korea : kinds of Kaggler''s Occupation')\nplt.suptitle('kinds of Kaggler''s Occupation', fontsize = 25)\nplt.show()\n\n","metadata":{"id":"N_F7nm1xJR7o","outputId":"49bb5c7a-f902-46f0-d51d-7de98d7c35e8","execution":{"iopub.status.busy":"2022-08-05T18:16:20.715340Z","iopub.execute_input":"2022-08-05T18:16:20.715775Z","iopub.status.idle":"2022-08-05T18:16:21.157463Z","shell.execute_reply.started":"2022-08-05T18:16:20.715737Z","shell.execute_reply":"2022-08-05T18:16:21.155454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"M489efN6JWsR"},"execution_count":null,"outputs":[]}]}