{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn import datasets\nfrom sklearn import model_selection","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-06T13:34:20.063386Z","iopub.execute_input":"2022-03-06T13:34:20.064247Z","iopub.status.idle":"2022-03-06T13:34:21.302669Z","shell.execute_reply.started":"2022-03-06T13:34:20.064127Z","shell.execute_reply":"2022-03-06T13:34:21.301769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Article on Cross Validation : https://dev.to/rohitgupta24/cross-validation-5gm3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ndf.species.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:34:26.765570Z","iopub.execute_input":"2022-03-06T13:34:26.765879Z","iopub.status.idle":"2022-03-06T13:34:26.887944Z","shell.execute_reply.started":"2022-03-06T13:34:26.765831Z","shell.execute_reply":"2022-03-06T13:34:26.887038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Data Cleaning\n#Found out some duplicate labels\n\nprint(\"Number of unique species before fixing : \", df['species'].nunique())\n\ndf['species'].replace({\n    'bottlenose_dolpin' : 'bottlenose_dolphin',\n    'kiler_whale' : 'killer_whale',\n},inplace =True)\n\nprint(\"Number of unique species after fixing : \", df['species'].nunique())\n\n\n# df['class'] = df['species'].apply(lambda x: x.split('_')[-1])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:34:32.054755Z","iopub.execute_input":"2022-03-06T13:34:32.055047Z","iopub.status.idle":"2022-03-06T13:34:32.096726Z","shell.execute_reply.started":"2022-03-06T13:34:32.055017Z","shell.execute_reply":"2022-03-06T13:34:32.095758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''visualizing  the data count'''\n\n\n#visualization imports\n#to have better idea, please visit : https://www.kaggle.com/awsaf49/happywhale-data-distribution#Dolphin-Species-6%EF%B8%8F%E2%83%A3\n\nimport plotly.express as px\n\ndata = df.species.value_counts().reset_index()\nfig = px.bar(data, x='index', y='species', color='species',title='Species Count', text_auto=True)\nfig.update_traces(textfont_size=12, textangle=0, textposition=\"outside\", cliponaxis=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:34:37.604556Z","iopub.execute_input":"2022-03-06T13:34:37.605394Z","iopub.status.idle":"2022-03-06T13:34:40.256771Z","shell.execute_reply.started":"2022-03-06T13:34:37.605320Z","shell.execute_reply":"2022-03-06T13:34:40.255940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset is skewed, hence it is an ideal case for stratified Cross Validation\n\n\n\n# we create a new column called kfold and fill it with -1\ndf[\"kfold\"] = -1\n# the next step is to randomize the rows of the data\ndf = df.sample(frac=1).reset_index(drop=True)\n# fetch targets\ny = df.species.values\n# initiate the kfold class from model_selection module\nkf = model_selection.StratifiedKFold(n_splits=10)\n# fill the new kfold column\nfor f, (t_, v_) in enumerate(kf.split(X=df, y=y)):\n    df.loc[v_, 'kfold'] = f\n","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:38:04.598147Z","iopub.execute_input":"2022-03-06T13:38:04.598408Z","iopub.status.idle":"2022-03-06T13:38:04.750895Z","shell.execute_reply.started":"2022-03-06T13:38:04.598381Z","shell.execute_reply":"2022-03-06T13:38:04.749865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save the new csv with kfold column\ndf.to_csv(\"train_folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:38:09.375549Z","iopub.execute_input":"2022-03-06T13:38:09.376067Z","iopub.status.idle":"2022-03-06T13:38:09.583064Z","shell.execute_reply.started":"2022-03-06T13:38:09.376033Z","shell.execute_reply":"2022-03-06T13:38:09.582127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfx = pd.read_csv(\"train_folds.csv\")\ndfx.head","metadata":{"execution":{"iopub.status.busy":"2022-03-06T13:38:12.703625Z","iopub.execute_input":"2022-03-06T13:38:12.703944Z","iopub.status.idle":"2022-03-06T13:38:12.784352Z","shell.execute_reply.started":"2022-03-06T13:38:12.703901Z","shell.execute_reply":"2022-03-06T13:38:12.783516Z"},"trusted":true},"execution_count":null,"outputs":[]}]}