{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom sklearn import datasets\nfrom sklearn import model_selection\n\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-02-19T08:52:24.315712Z","iopub.execute_input":"2022-02-19T08:52:24.316530Z","iopub.status.idle":"2022-02-19T08:52:25.402424Z","shell.execute_reply.started":"2022-02-19T08:52:24.316419Z","shell.execute_reply":"2022-02-19T08:52:25.401584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file = '../input/happy-whale-and-dolphin/train.csv'\ndf = pd.read_csv(train_file)\ndf.species.replace({\"globis\": \"short_finned_pilot_whale\",\n                      \"pilot_whale\": \"short_finned_pilot_whale\",\n                      \"kiler_whale\": \"killer_whale\",\n                      \"bottlenose_dolpin\": \"bottlenose_dolphin\"}, inplace=True)\ndf.to_csv('train_fixed.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T08:52:25.404680Z","iopub.execute_input":"2022-02-19T08:52:25.405192Z","iopub.status.idle":"2022-02-19T08:52:25.619418Z","shell.execute_reply.started":"2022-02-19T08:52:25.405156Z","shell.execute_reply":"2022-02-19T08:52:25.618635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T09:20:11.044367Z","iopub.execute_input":"2022-02-19T09:20:11.044661Z","iopub.status.idle":"2022-02-19T09:20:11.056696Z","shell.execute_reply.started":"2022-02-19T09:20:11.044626Z","shell.execute_reply":"2022-02-19T09:20:11.055930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"species_set = set(df['species'].values)\nprint(\"Now, we have \" + str(len(species_set)) + \" classes of whale and dolphin: \")","metadata":{"execution":{"iopub.status.busy":"2022-02-19T09:01:24.788972Z","iopub.execute_input":"2022-02-19T09:01:24.789681Z","iopub.status.idle":"2022-02-19T09:01:24.795208Z","shell.execute_reply.started":"2022-02-19T09:01:24.789645Z","shell.execute_reply":"2022-02-19T09:01:24.794285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_set = set(df['individual_id'].values)\nprint(\"Now, we have \" + str(len(id_set)) + \" whales and dolphins: \")","metadata":{"execution":{"iopub.status.busy":"2022-02-19T09:19:48.609487Z","iopub.execute_input":"2022-02-19T09:19:48.610309Z","iopub.status.idle":"2022-02-19T09:19:48.619427Z","shell.execute_reply.started":"2022-02-19T09:19:48.610238Z","shell.execute_reply":"2022-02-19T09:19:48.618579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for species in species_set:\n    print(species)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T09:17:18.468653Z","iopub.execute_input":"2022-02-19T09:17:18.469177Z","iopub.status.idle":"2022-02-19T09:17:18.476092Z","shell.execute_reply.started":"2022-02-19T09:17:18.469143Z","shell.execute_reply":"2022-02-19T09:17:18.475332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef create_folds(data, num_splits):\n    data[\"kfold\"] = -1\n    data[\"code\"] = -1\n    cat = 0\n    for species in species_set:\n        for i in range(len(data)):\n            if data[\"species\"][i] == species:                \n                data[\"code\"][i] = cat\n        cat = cat + 1\n    # print(cat)\n    # print(data)\n    num_bins = int(np.floor(1 + math.log(len(data),len(species_set))))\n    \n    data.loc[:, \"bins\"] = pd.cut(data[\"code\"], bins=num_bins, labels=False)\n\n    kf = model_selection.StratifiedKFold(n_splits=num_splits, shuffle=True, random_state=42)\n    \n    for f, (t_, v_) in enumerate(kf.split(X=data, y=data.bins.values)):\n        data.loc[v_, 'kfold'] = f\n    \n    data = data.drop(\"bins\", axis=1)\n    data = data.drop(\"code\", axis=1)\n    return data\n\n\ndf = pd.read_csv(\"./train_fixed.csv\")\n\ndf_5 = create_folds(df, num_splits=5)\ndf_5.to_csv(\"train_5folds.csv\", index=False)\ndf_10 = create_folds(df, num_splits=10)\ndf_10.to_csv(\"train_10folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T09:24:39.882151Z","iopub.execute_input":"2022-02-19T09:24:39.882910Z","iopub.status.idle":"2022-02-19T09:25:17.974810Z","shell.execute_reply.started":"2022-02-19T09:24:39.882869Z","shell.execute_reply":"2022-02-19T09:25:17.973959Z"},"trusted":true},"execution_count":null,"outputs":[]}]}