{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Shakeup interactive scatterplot maker\n\nThis notebook is a *reductio* of the more detailed notebook [Meta Kaggle: Scatter Plot Competition Shake-up](https://www.kaggle.com/jtrotman/meta-kaggle-scatter-plot-competition-shake-up), written by [jtrotman](https://www.kaggle.com/jtrotman). Here we produce an individual shakeup scatter plot of the [Google Smartphone Decimeter Challenge](https://www.kaggle.com/c/google-smartphone-decimeter-challenge) competition. This notebook also outputs the entire combined public and private leaderboards as a unified `csv` file.\n\nThis notebook is easily adaptable to any finished competition simply by using the pertinent `search_string`. (Note that it usually takes a few days between a competition finishing, and the data becoming publicly available in the [Meta Kaggle dataset](https://www.kaggle.com/kaggle/meta-kaggle)).","metadata":{}},{"cell_type":"code","source":"search_string = \"Smartphone Decimeter\"","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-08-11T03:49:30.448298Z","iopub.execute_input":"2021-08-11T03:49:30.448959Z","iopub.status.idle":"2021-08-11T03:49:30.454069Z","shell.execute_reply.started":"2021-08-11T03:49:30.448920Z","shell.execute_reply":"2021-08-11T03:49:30.453237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check that we have the correct competition:","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport plotly.express as px\nimport warnings\nwarnings.filterwarnings('ignore')\n\ncomps = pd.read_csv('../input/meta-kaggle/Competitions.csv')\nour_competition  = comps[comps['Title'].str.contains(search_string,na=False)]\npd.set_option('display.max_columns', None)\nour_competition","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-11T03:49:30.470804Z","iopub.execute_input":"2021-08-11T03:49:30.471416Z","iopub.status.idle":"2021-08-11T03:49:31.797437Z","shell.execute_reply.started":"2021-08-11T03:49:30.471375Z","shell.execute_reply":"2021-08-11T03:49:31.796342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### And now the scatterplot, along with the 'Top 20' leaderboard","metadata":{}},{"cell_type":"code","source":"CompetitionId       = our_competition[\"Id\"].squeeze()\nCompetitionIndex    = our_competition.index.values.astype(int)[0]\nall_teams           = pd.read_csv('../input/meta-kaggle/Teams.csv')\nteams               = all_teams[all_teams['CompetitionId']==CompetitionId]\nteams               = teams.assign(Medal=teams.Medal.fillna(0).astype(int))\nCOLOR_DICT          = {0: 'deepskyblue', 1: 'gold', 2: 'silver', 3: 'chocolate'}\nMEDAL_NAMES         = np.asarray([\"None\", \"Gold\", \"Silver\", \"Bronze\"])\nMEDAL_COLORS        = dict(zip(MEDAL_NAMES, COLOR_DICT.values()))\nrow                 = comps.loc[CompetitionIndex]\nteams               = teams.assign(Medal=MEDAL_NAMES[teams.Medal])\n\n# remove any teams with NaN scores\nLB_ranks            = teams[['Id','TeamName','PublicLeaderboardRank','PrivateLeaderboardRank', 'PublicLeaderboardSubmissionId', 'PrivateLeaderboardSubmissionId']].dropna(axis=0, how='any')\n\n# read in the file with the score data\nSubmissions         = pd.read_csv('../input/meta-kaggle/Submissions.csv')\n\ndef get_pub_score(PublicLeaderboardSubmissionId):\n    pub  = Submissions.query('Id == @PublicLeaderboardSubmissionId').PublicScoreLeaderboardDisplay.values[0]\n    return(pub)\n\ndef get_priv_score(PrivateLeaderboardSubmissionId):\n    priv = Submissions.query('Id == @PrivateLeaderboardSubmissionId').PrivateScoreLeaderboardDisplay.values[0]\n    return(priv)\n\nLB_ranks['PublicLeaderboardScore']  = LB_ranks.apply(lambda x: get_pub_score(x['PublicLeaderboardSubmissionId']),axis=1)\nLB_ranks['PrivateLeaderboardScore'] = LB_ranks.apply(lambda x: get_priv_score(x['PrivateLeaderboardSubmissionId']),axis=1)\n\n# make a new dataframe for writing out\nLB_ranks_and_scores = (LB_ranks[['Id','TeamName', 'PublicLeaderboardRank', 'PublicLeaderboardScore', 'PrivateLeaderboardRank','PrivateLeaderboardScore']]).set_index('Id')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-11T03:49:31.799957Z","iopub.execute_input":"2021-08-11T03:49:31.800426Z","iopub.status.idle":"2021-08-11T03:52:32.509861Z","shell.execute_reply.started":"2021-08-11T03:49:31.800377Z","shell.execute_reply":"2021-08-11T03:52:32.508018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CompetitionId       = our_competition[\"Id\"].squeeze()\nCompetitionIndex    = our_competition.index.values.astype(int)[0]\nall_teams           = pd.read_csv('../input/meta-kaggle/Teams.csv')\nteams               = all_teams[all_teams['CompetitionId']==CompetitionId]\nteams               = teams.assign(Medal=teams.Medal.fillna(0).astype(int))\nCOLOR_DICT          = {0: 'deepskyblue', 1: 'gold', 2: 'silver', 3: 'chocolate'}\nMEDAL_NAMES         = np.asarray([\"None\", \"Gold\", \"Silver\", \"Bronze\"])\nMEDAL_COLORS        = dict(zip(MEDAL_NAMES, COLOR_DICT.values()))\nrow                 = comps.loc[CompetitionIndex]\nteams               = teams.assign(Medal=MEDAL_NAMES[teams.Medal])\n\n# remove any teams with NaN scores\nLB_ranks            = teams[['Id','TeamName','PublicLeaderboardRank','PrivateLeaderboardRank', 'PublicLeaderboardSubmissionId', 'PrivateLeaderboardSubmissionId']].dropna(axis=0, how='any')\n\n# read in the file with the score data\nSubmissions         = pd.read_csv('../input/meta-kaggle/Submissions.csv')\n\ndef get_pub_score(PublicLeaderboardSubmissionId):\n    pub  = Submissions.query('Id == @PublicLeaderboardSubmissionId').PublicScoreLeaderboardDisplay.values[0]\n    return(pub)\n\ndef get_priv_score(PrivateLeaderboardSubmissionId):\n    priv = Submissions.query('Id == @PrivateLeaderboardSubmissionId').PrivateScoreLeaderboardDisplay.values[0]\n    return(priv)\n\nLB_ranks['PublicLeaderboardScore']  = LB_ranks.apply(lambda x: get_pub_score(x['PublicLeaderboardSubmissionId']),axis=1)\nLB_ranks['PrivateLeaderboardScore'] = LB_ranks.apply(lambda x: get_priv_score(x['PrivateLeaderboardSubmissionId']),axis=1)\n\n# make a new dataframe for writing out\nLB_ranks_and_scores = (LB_ranks[['Id','TeamName', 'PublicLeaderboardRank', 'PublicLeaderboardScore', 'PrivateLeaderboardRank','PrivateLeaderboardScore']]).set_index('Id')\n\nfig = px.scatter(teams,\n                 title='Shakeup plot for: ' + row.Title,\n                 x='PublicLeaderboardRank',\n                 y='PrivateLeaderboardRank',\n                 hover_name='TeamName',\n                 hover_data=[\n                     'PublicLeaderboardRank',\n                     'PrivateLeaderboardRank',\n                     'Medal',\n                 ],\n                 color='Medal',\n                 color_discrete_map=MEDAL_COLORS)\nfig.update_traces(marker=dict(size=5))\nfig.update_layout(showlegend=False)\nfig.show()\n\n# save to a csv file\nLB_ranks_and_scores.to_csv(\"LB_ranks_and_scores.csv\", index=False)\n\n# Take a look at the Top 20 \nLB_ranks_and_scores.sort_values(by='PrivateLeaderboardRank', ascending=True).head(20)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-11T03:52:32.512887Z","iopub.execute_input":"2021-08-11T03:52:32.513274Z","iopub.status.idle":"2021-08-11T03:55:27.688326Z","shell.execute_reply.started":"2021-08-11T03:52:32.513236Z","shell.execute_reply":"2021-08-11T03:55:27.687418Z"},"trusted":true},"execution_count":null,"outputs":[]}]}