{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Contents\n#### Method 1: Label encoding\n#### Method 2: One hot encoding\n#### Method 3: replace()\n#### Method 4: Dummy Variable Encoding (get_dummies)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import KFold\nfrom sklearn import base","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:35.864249Z","iopub.execute_input":"2022-07-13T09:15:35.865192Z","iopub.status.idle":"2022-07-13T09:15:37.445406Z","shell.execute_reply.started":"2022-07-13T09:15:35.865064Z","shell.execute_reply":"2022-07-13T09:15:37.443810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read data\ndf = pd.read_csv(\"../input/cat-in-the-dat/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:37.448064Z","iopub.execute_input":"2022-07-13T09:15:37.448604Z","iopub.status.idle":"2022-07-13T09:15:39.421304Z","shell.execute_reply.started":"2022-07-13T09:15:37.448542Z","shell.execute_reply":"2022-07-13T09:15:39.420130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:39.422634Z","iopub.execute_input":"2022-07-13T09:15:39.422930Z","iopub.status.idle":"2022-07-13T09:15:39.458662Z","shell.execute_reply.started":"2022-07-13T09:15:39.422903Z","shell.execute_reply":"2022-07-13T09:15:39.457845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:39.460903Z","iopub.execute_input":"2022-07-13T09:15:39.461513Z","iopub.status.idle":"2022-07-13T09:15:40.041885Z","shell.execute_reply.started":"2022-07-13T09:15:39.461479Z","shell.execute_reply":"2022-07-13T09:15:40.040901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(columns=\"target\" , axis=1)\nY = df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:40.043133Z","iopub.execute_input":"2022-07-13T09:15:40.044371Z","iopub.status.idle":"2022-07-13T09:15:40.106041Z","shell.execute_reply.started":"2022-07-13T09:15:40.044309Z","shell.execute_reply":"2022-07-13T09:15:40.104866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:40.107610Z","iopub.execute_input":"2022-07-13T09:15:40.108793Z","iopub.status.idle":"2022-07-13T09:15:40.354704Z","shell.execute_reply.started":"2022-07-13T09:15:40.108753Z","shell.execute_reply":"2022-07-13T09:15:40.353477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Method 1: Label encoding","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/max/772/1*Yp6r7m82IoSnnZDPpDpYNw.png)","metadata":{}},{"cell_type":"markdown","source":"- **Method 1.1 =Encoding to the desired column**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel = LabelEncoder()\nX[\"nom_0\"] =label.fit_transform(X[\"nom_0\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:40.356439Z","iopub.execute_input":"2022-07-13T09:15:40.356854Z","iopub.status.idle":"2022-07-13T09:15:40.554525Z","shell.execute_reply.started":"2022-07-13T09:15:40.356820Z","shell.execute_reply":"2022-07-13T09:15:40.553175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:40.556128Z","iopub.execute_input":"2022-07-13T09:15:40.556562Z","iopub.status.idle":"2022-07-13T09:15:40.581857Z","shell.execute_reply.started":"2022-07-13T09:15:40.556527Z","shell.execute_reply":"2022-07-13T09:15:40.580627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **Method 1.2 = Encoding to all columns**","metadata":{}},{"cell_type":"code","source":"%%time\n\ntrain=pd.DataFrame()\nlabel=LabelEncoder()\nfor c in  X.columns:\n    if(X[c].dtype=='object'):\n        X[c]=label.fit_transform(X[c])\n    else:\n        X[c]=X[c]\n        \n       \n\n    \nX.head(3)\n#çok tercih etmeyebiliriz. 300 farklı unique değere encoding maliyetli olabilir.","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:40.583210Z","iopub.execute_input":"2022-07-13T09:15:40.583583Z","iopub.status.idle":"2022-07-13T09:15:42.560368Z","shell.execute_reply.started":"2022-07-13T09:15:40.583552Z","shell.execute_reply":"2022-07-13T09:15:42.559122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"### Method 2: One hot encoding ","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/max/1400/1*ggtP4a5YaRx6l09KQaYOnw.png)","metadata":{}},{"cell_type":"code","source":"X = df.drop(columns=\"target\" , axis=1)\nY = df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:42.565168Z","iopub.execute_input":"2022-07-13T09:15:42.565686Z","iopub.status.idle":"2022-07-13T09:15:42.621794Z","shell.execute_reply.started":"2022-07-13T09:15:42.565638Z","shell.execute_reply":"2022-07-13T09:15:42.620510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nohe = OneHotEncoder(handle_unknown='ignore')\n\nohe_df = pd.DataFrame(ohe.fit_transform(df[['nom_0']]).toarray())\nX = df.join(ohe_df)  #X'e ekledik\nX.drop('nom_0', axis=1, inplace=True) # nom0 kaldırdık\n\nX.head(3)\n#kolon ismi ekleyebilirsin.","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:42.623479Z","iopub.execute_input":"2022-07-13T09:15:42.624172Z","iopub.status.idle":"2022-07-13T09:15:43.095637Z","shell.execute_reply.started":"2022-07-13T09:15:42.624119Z","shell.execute_reply":"2022-07-13T09:15:43.094348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"--------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"### Method 3: replace()","metadata":{}},{"cell_type":"code","source":"X = df.drop(columns=\"target\" , axis=1)\nY = df[\"target\"]\n\nX[\"nom_0\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.097786Z","iopub.execute_input":"2022-07-13T09:15:43.098626Z","iopub.status.idle":"2022-07-13T09:15:43.202741Z","shell.execute_reply.started":"2022-07-13T09:15:43.098577Z","shell.execute_reply":"2022-07-13T09:15:43.201599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### dict-like","metadata":{}},{"cell_type":"code","source":"X[\"nom_0\"].replace({\"Green\":0 , \"Blue\":1 ,\"Red\":2} , inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.204238Z","iopub.execute_input":"2022-07-13T09:15:43.205123Z","iopub.status.idle":"2022-07-13T09:15:43.478017Z","shell.execute_reply.started":"2022-07-13T09:15:43.205074Z","shell.execute_reply":"2022-07-13T09:15:43.476791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.479599Z","iopub.execute_input":"2022-07-13T09:15:43.480455Z","iopub.status.idle":"2022-07-13T09:15:43.505888Z","shell.execute_reply.started":"2022-07-13T09:15:43.480417Z","shell.execute_reply":"2022-07-13T09:15:43.504990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[\"bin_4\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.507270Z","iopub.execute_input":"2022-07-13T09:15:43.507774Z","iopub.status.idle":"2022-07-13T09:15:43.550383Z","shell.execute_reply.started":"2022-07-13T09:15:43.507742Z","shell.execute_reply":"2022-07-13T09:15:43.549561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### arr_like","metadata":{}},{"cell_type":"code","source":"X['bin_4'].replace(['Y', 'N'], [0, 1], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.551549Z","iopub.execute_input":"2022-07-13T09:15:43.552036Z","iopub.status.idle":"2022-07-13T09:15:43.787570Z","shell.execute_reply.started":"2022-07-13T09:15:43.552004Z","shell.execute_reply":"2022-07-13T09:15:43.786680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.788749Z","iopub.execute_input":"2022-07-13T09:15:43.789238Z","iopub.status.idle":"2022-07-13T09:15:43.811828Z","shell.execute_reply.started":"2022-07-13T09:15:43.789208Z","shell.execute_reply":"2022-07-13T09:15:43.810413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"### Method 4: Dummy Variable Encoding (get_dummies)","metadata":{}},{"cell_type":"code","source":"X = df.drop(columns=\"target\" , axis=1)\nY = df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.813489Z","iopub.execute_input":"2022-07-13T09:15:43.813952Z","iopub.status.idle":"2022-07-13T09:15:43.875362Z","shell.execute_reply.started":"2022-07-13T09:15:43.813904Z","shell.execute_reply":"2022-07-13T09:15:43.874188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummies = pd.get_dummies(df.nom_0)\nmerged = pd.concat([X, dummies], axis='columns')\nX = merged.drop([\"nom_0\",'Green'], axis='columns')\nX.head(3)\n#Blue ve Red kolonları oluştu.","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:15:43.876806Z","iopub.execute_input":"2022-07-13T09:15:43.877997Z","iopub.status.idle":"2022-07-13T09:15:44.250264Z","shell.execute_reply.started":"2022-07-13T09:15:43.877934Z","shell.execute_reply":"2022-07-13T09:15:44.249059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If you like it, please upvote :))**","metadata":{}}]}