{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.452253Z","iopub.execute_input":"2022-02-24T06:19:26.452665Z","iopub.status.idle":"2022-02-24T06:19:26.480692Z","shell.execute_reply.started":"2022-02-24T06:19:26.452558Z","shell.execute_reply":"2022-02-24T06:19:26.479794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kflod_10 = pd.read_csv('../input/happywhale-splits/skf_species_10folds.csv')\ntrain_df = pd.read_csv('../input/happy-whale-and-dolphin/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.482702Z","iopub.execute_input":"2022-02-24T06:19:26.483362Z","iopub.status.idle":"2022-02-24T06:19:26.657077Z","shell.execute_reply.started":"2022-02-24T06:19:26.483319Z","shell.execute_reply":"2022-02-24T06:19:26.656211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"individual_id_num\"] = kflod_10[\"individual_id\"]\ntrain_df[\"species_num\"] = kflod_10[\"species\"]","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.658207Z","iopub.execute_input":"2022-02-24T06:19:26.658453Z","iopub.status.idle":"2022-02-24T06:19:26.672224Z","shell.execute_reply.started":"2022-02-24T06:19:26.658423Z","shell.execute_reply":"2022-02-24T06:19:26.671099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"map_dict_spe = train_df[[\"species\", \"species_num\"]].set_index(\"species\").to_dict(orient='dict')['species_num']\nmap_dict_ind = train_df[[\"individual_id\", \"individual_id_num\"]].set_index(\"individual_id\").to_dict(orient='dict')['individual_id_num']","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.673712Z","iopub.execute_input":"2022-02-24T06:19:26.673968Z","iopub.status.idle":"2022-02-24T06:19:26.795705Z","shell.execute_reply.started":"2022-02-24T06:19:26.673917Z","shell.execute_reply":"2022-02-24T06:19:26.794509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['nums']=train_df['individual_id'].map(dict(train_df['individual_id'].value_counts()))","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.798530Z","iopub.execute_input":"2022-02-24T06:19:26.798985Z","iopub.status.idle":"2022-02-24T06:19:26.922686Z","shell.execute_reply.started":"2022-02-24T06:19:26.798926Z","shell.execute_reply":"2022-02-24T06:19:26.921763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[train_df['nums']>1]","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.923733Z","iopub.execute_input":"2022-02-24T06:19:26.923935Z","iopub.status.idle":"2022-02-24T06:19:26.933967Z","shell.execute_reply.started":"2022-02-24T06:19:26.923910Z","shell.execute_reply":"2022-02-24T06:19:26.932950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.935383Z","iopub.execute_input":"2022-02-24T06:19:26.935645Z","iopub.status.idle":"2022-02-24T06:19:26.942383Z","shell.execute_reply.started":"2022-02-24T06:19:26.935615Z","shell.execute_reply":"2022-02-24T06:19:26.941793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:26.943519Z","iopub.execute_input":"2022-02-24T06:19:26.943910Z","iopub.status.idle":"2022-02-24T06:19:28.048239Z","shell.execute_reply.started":"2022-02-24T06:19:26.943875Z","shell.execute_reply":"2022-02-24T06:19:28.047418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder() # 6329\ntrain_df['individual_id_num'] = encoder.fit_transform(train_df['individual_id'])","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:28.049716Z","iopub.execute_input":"2022-02-24T06:19:28.050067Z","iopub.status.idle":"2022-02-24T06:19:28.073695Z","shell.execute_reply.started":"2022-02-24T06:19:28.050019Z","shell.execute_reply":"2022-02-24T06:19:28.072808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"./le_6329.pkl\", \"wb\") as fp:\n    joblib.dump(encoder, fp)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:28.074863Z","iopub.execute_input":"2022-02-24T06:19:28.075178Z","iopub.status.idle":"2022-02-24T06:19:28.081744Z","shell.execute_reply.started":"2022-02-24T06:19:28.075144Z","shell.execute_reply":"2022-02-24T06:19:28.081047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=10)\nfor fold, ( _, val_) in enumerate(skf.split(X=train_df, y=train_df.individual_id_num)):\n      train_df.loc[val_ , \"kfold\"] = fold","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:28.082869Z","iopub.execute_input":"2022-02-24T06:19:28.083445Z","iopub.status.idle":"2022-02-24T06:19:28.304614Z","shell.execute_reply.started":"2022-02-24T06:19:28.083387Z","shell.execute_reply":"2022-02-24T06:19:28.303524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('./train_fold10_onlymore2_6329.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T06:19:28.305710Z","iopub.execute_input":"2022-02-24T06:19:28.305933Z","iopub.status.idle":"2022-02-24T06:19:28.520327Z","shell.execute_reply.started":"2022-02-24T06:19:28.305904Z","shell.execute_reply":"2022-02-24T06:19:28.519455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}