{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Taxonomy\n\nThe purpose of this notebook is to join the table `train.csv` with a taxonomy table.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ntaxonomy = pd.read_csv(\"/kaggle/input/happywhalespeciesclassification/species.csv\")\ntaxonomy.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-09T10:42:18.126094Z","iopub.execute_input":"2022-03-09T10:42:18.126857Z","iopub.status.idle":"2022-03-09T10:42:18.138559Z","shell.execute_reply.started":"2022-03-09T10:42:18.126815Z","shell.execute_reply":"2022-03-09T10:42:18.137798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#taxonomy.sort_values(by = ['infraorder','family','genus','specy']).drop(columns = ['wikipedia','image','size']).set_index('specy_id')\ntaxonomy = taxonomy.sort_values(by = ['specy']).drop(columns = ['wikipedia','image','size']).set_index('specy_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-09T10:42:18.140284Z","iopub.execute_input":"2022-03-09T10:42:18.140907Z","iopub.status.idle":"2022-03-09T10:42:18.150108Z","shell.execute_reply.started":"2022-03-09T10:42:18.140874Z","shell.execute_reply":"2022-03-09T10:42:18.149132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy","metadata":{"execution":{"iopub.status.busy":"2022-03-09T10:42:18.15145Z","iopub.execute_input":"2022-03-09T10:42:18.151877Z","iopub.status.idle":"2022-03-09T10:42:18.17155Z","shell.execute_reply.started":"2022-03-09T10:42:18.151846Z","shell.execute_reply":"2022-03-09T10:42:18.1703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/happy-whale-and-dolphin/train.csv\")\ndf.loc[df.species == 'bottlenose_dolpin', 'species'] = 'bottlenose_dolphin'\ndf.loc[df.species == 'kiler_whale', 'species'] = 'killer_whale'\ndf.loc[df.species == 'long_finned_pilot_whale', 'species'] = 'pilot_whale'\ndf.loc[df.species == 'globis', 'species'] = 'pilot_whale'\ndf = df.rename(columns = { 'species': 'specy_id' })\nfinal = pd.merge(df, taxonomy, on = ['specy_id'], how = 'left')\nfinal","metadata":{"execution":{"iopub.status.busy":"2022-03-09T10:42:18.173452Z","iopub.execute_input":"2022-03-09T10:42:18.174266Z","iopub.status.idle":"2022-03-09T10:42:18.30683Z","shell.execute_reply.started":"2022-03-09T10:42:18.17423Z","shell.execute_reply":"2022-03-09T10:42:18.305967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.to_csv(\"/kaggle/working/final.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T10:42:18.308245Z","iopub.execute_input":"2022-03-09T10:42:18.308499Z","iopub.status.idle":"2022-03-09T10:42:18.613273Z","shell.execute_reply.started":"2022-03-09T10:42:18.308469Z","shell.execute_reply":"2022-03-09T10:42:18.612199Z"},"trusted":true},"execution_count":null,"outputs":[]}]}