{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom sklearn import model_selection","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-15T10:42:53.098119Z","iopub.execute_input":"2022-02-15T10:42:53.099122Z","iopub.status.idle":"2022-02-15T10:42:53.103435Z","shell.execute_reply.started":"2022-02-15T10:42:53.099074Z","shell.execute_reply":"2022-02-15T10:42:53.102430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_folds(data,target,num_splits):\n    # we create a new column called kfold and fill it with -1\n    data[\"kfold\"] = -1\n    \n    # the next step is to randomize the rows of the data\n    data = data.sample(frac=1).reset_index(drop=True)\n\n    # calculate number of bins by Sturge's rule\n    # I take the floor of the value, you can also\n    # just round it\n    num_bins = int(np.floor(1 + np.log2(len(data))))\n    \n    # bin targets\n    data.loc[:, \"bins\"] = pd.cut(\n        data[target], bins=num_bins, labels=False\n    )\n    \n    # initiate the kfold class from model_selection module\n    kf = model_selection.StratifiedKFold(n_splits=num_splits)\n    \n    # fill the new kfold column\n    # note that, instead of targets, we use bins!\n    for f, (t_, v_) in enumerate(kf.split(X=data, y=data.bins.values)):\n        data.loc[v_, 'kfold'] = f\n    \n    # drop the bins column\n    data = data.drop(\"bins\", axis=1)\n\n    # return dataframe with folds\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:46:55.779682Z","iopub.execute_input":"2022-02-15T10:46:55.780500Z","iopub.status.idle":"2022-02-15T10:46:55.786637Z","shell.execute_reply.started":"2022-02-15T10:46:55.780460Z","shell.execute_reply":"2022-02-15T10:46:55.786070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df= pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.145229Z","iopub.execute_input":"2022-02-15T10:42:53.146179Z","iopub.status.idle":"2022-02-15T10:42:53.223353Z","shell.execute_reply.started":"2022-02-15T10:42:53.146129Z","shell.execute_reply":"2022-02-15T10:42:53.222582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.224453Z","iopub.execute_input":"2022-02-15T10:42:53.224709Z","iopub.status.idle":"2022-02-15T10:42:53.237325Z","shell.execute_reply.started":"2022-02-15T10:42:53.224678Z","shell.execute_reply":"2022-02-15T10:42:53.236243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## find duplicates\ntrain_df.species.unique()\n## Find the duplicactes and merge\n\nprint(\"Total species before finding duplicates :\",len(train_df.species.unique()))\ntrain_df.species = train_df.species.str.replace('kiler_whale','killer_whale')\ntrain_df.species = train_df.species.str.replace('bottlenose_dolpin','bottlenose_dolphin')\ntrain_df['species'][(train_df['species'] ==\"pilot_whale\") | (train_df['species'] ==\"globis\" )]='short_finned_pilot_whale'\nprint(\"Total species after :\",len(train_df.species.unique()))","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.239122Z","iopub.execute_input":"2022-02-15T10:42:53.239599Z","iopub.status.idle":"2022-02-15T10:42:53.324426Z","shell.execute_reply.started":"2022-02-15T10:42:53.239541Z","shell.execute_reply":"2022-02-15T10:42:53.323095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## lets convert labels into numbers\n## create dictionary\n\nspecies = dict((a,b) for b,a in enumerate(train_df.species.unique()))\nspecies_inv = {(a,b) for b,a in species.items()}","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.326715Z","iopub.execute_input":"2022-02-15T10:42:53.326990Z","iopub.status.idle":"2022-02-15T10:42:53.336539Z","shell.execute_reply.started":"2022-02-15T10:42:53.326958Z","shell.execute_reply":"2022-02-15T10:42:53.335355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"individual = dict((a,b) for b,a in enumerate(train_df.individual_id.unique()))\nindividual_inv = {(a,b) for b,a in species.items()}","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.338063Z","iopub.execute_input":"2022-02-15T10:42:53.338295Z","iopub.status.idle":"2022-02-15T10:42:53.363969Z","shell.execute_reply.started":"2022-02-15T10:42:53.338269Z","shell.execute_reply":"2022-02-15T10:42:53.362624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"species\"]= [species[i] for i in train_df[\"species\"]]","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.365974Z","iopub.execute_input":"2022-02-15T10:42:53.367735Z","iopub.status.idle":"2022-02-15T10:42:53.397399Z","shell.execute_reply.started":"2022-02-15T10:42:53.367661Z","shell.execute_reply":"2022-02-15T10:42:53.395899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['individual_id']=[individual[i] for i in train_df['individual_id']]","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:42:53.398993Z","iopub.execute_input":"2022-02-15T10:42:53.399606Z","iopub.status.idle":"2022-02-15T10:42:53.437193Z","shell.execute_reply.started":"2022-02-15T10:42:53.399560Z","shell.execute_reply":"2022-02-15T10:42:53.435879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:44:20.326224Z","iopub.execute_input":"2022-02-15T10:44:20.326482Z","iopub.status.idle":"2022-02-15T10:44:20.340828Z","shell.execute_reply.started":"2022-02-15T10:44:20.326454Z","shell.execute_reply":"2022-02-15T10:44:20.339929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=create_folds(train_df,\"species\",10)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:47:03.341565Z","iopub.execute_input":"2022-02-15T10:47:03.342327Z","iopub.status.idle":"2022-02-15T10:47:03.383641Z","shell.execute_reply.started":"2022-02-15T10:47:03.342285Z","shell.execute_reply":"2022-02-15T10:47:03.382767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.agg(['min','max','count','nunique'])","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:49:09.238476Z","iopub.execute_input":"2022-02-15T10:49:09.238909Z","iopub.status.idle":"2022-02-15T10:49:09.294526Z","shell.execute_reply.started":"2022-02-15T10:49:09.238878Z","shell.execute_reply":"2022-02-15T10:49:09.293888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(\"train.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T10:51:29.344214Z","iopub.execute_input":"2022-02-15T10:51:29.344681Z","iopub.status.idle":"2022-02-15T10:51:29.444273Z","shell.execute_reply.started":"2022-02-15T10:51:29.344647Z","shell.execute_reply":"2022-02-15T10:51:29.443266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## dont forget the steps we done in the preprocessing we actually neeed it when deploying","metadata":{"execution":{"iopub.status.busy":"2022-02-15T11:03:03.138025Z","iopub.execute_input":"2022-02-15T11:03:03.138305Z","iopub.status.idle":"2022-02-15T11:03:03.143831Z","shell.execute_reply.started":"2022-02-15T11:03:03.138277Z","shell.execute_reply":"2022-02-15T11:03:03.142412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}