{"cells":[{"metadata":{"_uuid":"811e2fce369f330139c9465c8a733dedacc987d5"},"cell_type":"markdown","source":"**Feature Engineering and Data Pre-Processing**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input\"))\n# manipulating dataframes\nimport pandas as pd\nimport numpy as np\n# visualizing libraries\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"72b1b4cbd5ff2810e434c29c4fa31742f99cdd78"},"cell_type":"markdown","source":"**Load Dataset**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"titanic = pd.read_csv('../input/train.csv')\ntitanic.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30ef01c0672e6147785b06be9a9d95d7d65dfef8"},"cell_type":"code","source":"titanic.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"60dfa6a5c8c0ced34f7dc72dc333d6fe114ca33a"},"cell_type":"markdown","source":"**Explore data**"},{"metadata":{"trusted":true,"_uuid":"dff6d2b47c866aba4d831fa02d5df3621cb12689"},"cell_type":"code","source":"titanic.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"579d1047691c4a416b294bcfffceb588df6d6cca"},"cell_type":"code","source":"#summary statistics\ntitanic.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df1be5dd3d3fa4815b9628b32250be6a1a0bbb3e"},"cell_type":"code","source":"#plots to identify any patterns or insights\ntitanic.hist(figsize=(10,10))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"82cd3c8871d9a3e4fdbaf9b2a060cabc707eac19"},"cell_type":"markdown","source":"**Basic Feature Engineering**"},{"metadata":{"_uuid":"9c804c984f8333a18f99c377e0dfed33a18ae96b"},"cell_type":"markdown","source":"Create new features from dataset:\n* title: reflecting a persons title (Mr., Mrs. etc)\n* mother: reflecting if a person is a mother or not"},{"metadata":{"trusted":true,"_uuid":"3f5ea5404ffb02c1c332bf8a95a6342089b366a5"},"cell_type":"code","source":"#New Feature:Title\ndef title(x):\n    if 'Mr.' in x:\n        return 'Mr'\n    elif 'Mrs.' in x:\n        return 'Mrs'\n    elif 'Master' in x:\n        return 'Master'\n    elif 'Miss.' in x:\n        return 'Miss'\n    else:\n        return 'Other'\n# feature for the title of each person\ntitanic['Title'] = titanic['Name'].apply(title)\n\ntitanic['Title'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9fc60708d0a599ce1c9ba9e5246048f0aa81012"},"cell_type":"code","source":"#New Feature: Mother\ndef mother(df):\n    if df['Sex'] == 'female' and df['Parch'] > 0 and df['Age'] > 18 and df['Title'] != 'Miss':\n        return 'Mother'\n    else:\n        return 'Not Mother'\n# feature mother for each person\ntitanic['Mother'] = titanic.apply(mother, axis=1)\n\ntitanic.Mother.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"125127a623a45efa64129ce409646345c482572a"},"cell_type":"code","source":"# Added in 2 new features\ntitanic.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"abe53c92c5315cdfa535e56604ccc4e808203ec6"},"cell_type":"markdown","source":"**Impute Missing Values**"},{"metadata":{"trusted":true,"_uuid":"7a104480fca4a06f6d1b1b2b67c02264e0221f48"},"cell_type":"code","source":"titanic.isnull().sum() / len(titanic)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c8b6f24cec8551ec7bdebaffbeb09bead5f3c72"},"cell_type":"code","source":"#Data Imputation: Age\n# Plot Distribution of Age (Missing)\nplt.subplot(1, 2, 1)\ntitanic['Age'].hist(bins=20, figsize=(15,6), edgecolor='white')\nplt.xlabel('Age', fontsize=12)\nplt.title('Distribution of Age (Missing)', fontsize=18)\n\n# Plot Distribution of Age (Imputed Mean)\nplt.subplot(1, 2, 2)\nmean_age = pd.DataFrame(titanic['Age'].fillna(titanic.Age.mean()))\nmean_age['Age'].hist(bins=20, figsize=(15,6), edgecolor='white', color='r')\nplt.xlabel('Age)', fontsize=12)\nplt.title('Distribution of Age (Imputed Mean)', fontsize=18)\n\nplt.show()    \n \n    \n    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e62839af02c3f688ce73c69aef2a73ea441c073"},"cell_type":"code","source":"#Data Imputation: Age by Passenger Title\ntitanic.groupby(['Title'])['Age'].median()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"50cd9c564c393f8350db1c7becb4797ef8b2c824"},"cell_type":"code","source":"titanic.boxplot(column='Age',by='Title') #Mean Age is different per title\nplt.ylabel('Age')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6164c8152476d879707fb2d75d9987a723b8057"},"cell_type":"code","source":"# Fill in the missing age with the median of their Titles\ntitanic['Age'].fillna(titanic.groupby([\"Title\"])[\"Age\"].transform(np.median),inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20caa2b6e7fe150cc16d19f7a95a1909fb2a0bf1"},"cell_type":"code","source":"#Plot the Age Distribution\ntitanic['Age'].hist(bins=20, figsize=(15,6), edgecolor='white')\nplt.xlabel('Age', fontsize=12)\nplt.title('Distribution of Age (Imputed by Title)', fontsize=18)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd12c21ecb52abb0125edd6318dc017c12ab0885"},"cell_type":"code","source":"#Data Imputation: Embarked\ntitanic.Embarked.value_counts() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10d6876012f235a7d4a779538c248efd7f762a66"},"cell_type":"code","source":"#Impute missing 'Embarked' variable with the most frequent value: (S)\ntitanic['Embarked'] = titanic['Embarked'].fillna(titanic['Embarked'].value_counts().index[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b11c768f49a7a683ffa12749cb73ffffbd1b5db"},"cell_type":"code","source":"#Data Imputation: Cabin\ntitanic.Cabin.isnull().sum() / len(titanic)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"649b93875a57f74a8fe6ecc01d494596c603d7bb"},"cell_type":"code","source":"#Drop Cabin Feature\ntitanic.drop(columns=['Cabin'], inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afcb043877914306a876801068d13d712ed2a2fa"},"cell_type":"markdown","source":"**Numeric Feature Engineering Techniques**"},{"metadata":{"trusted":true,"_uuid":"7236eafc45adb583841038aee594700ebb472453"},"cell_type":"code","source":"titanic['FamilySize'] = titanic['SibSp'] + titanic['Parch'] + 1\ntitanic.FamilySize.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87b78c13b79c9a79fddd166f5791954eba790062"},"cell_type":"code","source":"titanic.FamilySize.hist()\nplt.title('FamilySize Distribution for Titanic Passengers', size=15)\nplt.xlabel(\"Family Size\")\nplt.ylabel('Frequency')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"da9b00adf950cb01ac44457e538a06de8baf44fb"},"cell_type":"markdown","source":"There's A LOT of single riders. \n**Create new feature**: \"IsAlone\""},{"metadata":{"trusted":true,"_uuid":"2632e721db2a75b197abc13b9531b94d9ed4dfe0"},"cell_type":"code","source":"titanic['IsAlone'] = titanic['FamilySize'].apply(lambda x: 1 if x == 1 else 0)\ntitanic.IsAlone.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd2bc26788631249cb916f10764ef620dfab939f"},"cell_type":"code","source":"titanic.IsAlone.hist()\nplt.title('IsAlone Distribution for Titanic Passengers', size=15)\nplt.xlabel(\"IsAlone\")\nplt.ylabel('Frequency')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b3164f9330cb4e353fa6e56a86c7f97b44fdf9fb"},"cell_type":"code","source":"# Examine the Age Distribution\n#Binning helps solve the skewness problem.\ntitanic.Age.hist(bins=25)\nplt.title('Age Distribution for Titanic Passengers', size=15)\nplt.xlabel(\"Age\")\nplt.ylabel('Frequency')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ededacf1e3657ced29335fea726a43b87e605be"},"cell_type":"code","source":"# Fixed Width Binning (Kid, Teen, Adult, Elderly)\n#With domain knowledge, we can safely bin our passengers into different age groups.\nbins = [0,12,17,60,150]\nlabels = [\"kid\",\"teen\",\"adult\",\"elderly\"]\ntitanic['AgeGroup'] = pd.cut(titanic.Age,bins=bins,labels=labels)\ntitanic[['Age','AgeGroup']].head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"844e9d855c9c06e181e33520e95c133812fd4eda"},"cell_type":"code","source":"#Quantile Binning uses the quantiles of our feature\ntitanic.Age.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d68dc7e18a87465676ba61bb008cedc08e837793"},"cell_type":"code","source":"#Create a quantile list\nquantile_list = [0, .25, .5, .75, 1.]\nquantiles = titanic['Age'].quantile(quantile_list)\nquantiles","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ce75c3588829d9392e08354d4d52b18b728a4c1"},"cell_type":"code","source":"titanic['age_quantile_range'] = pd.qcut(titanic.Age, 4)\ntitanic['age_quantile_label'] = pd.qcut(titanic.Age, 4, labels=[0.25, 0.5, 0.75, 1])\ntitanic[['Age','age_quantile_range','age_quantile_label']].head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8cd6a05cee9560655dda4a73437a9b35feb8f8ea"},"cell_type":"markdown","source":"**Power Transformations**\n\n* Power Transformations DOES CHANGE the distribution of your data and tries to make it more \"normal\".\n* Example: Log Transformation, It compresses the long tail of the distribution into a shorter tail, and expands the lower tail of the distribution into the longer head."},{"metadata":{"trusted":true,"_uuid":"c2511b1d8a769243c89fd3875c278d5951266ddd"},"cell_type":"code","source":"# Plot Fare Price Distribution\nplt.subplot(1, 2, 1)\n(titanic['Fare']).plot.hist(bins=15, figsize=(15, 6), edgecolor = 'white')\nplt.xlabel('Fare Price', fontsize=12)\nplt.title('BEFORE', fontsize=24)\n\n#Plot Log Fare Price Distribution\nplt.subplot(1, 2, 2)\nnp.log(titanic['Fare']+1).plot.hist(bins=15,figsize=(15,6), edgecolor='white', color='r')\nplt.xlabel('log(Fare Price+1)', fontsize=12)\nplt.title('AFTER', fontsize=24)\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5251592656b7ba93a7fa526bfc4ae941e009c471"},"cell_type":"markdown","source":"**Normalization**\n\nNormalization( Feature scaling) always divides the feature by a constant.\nAnd it does not change the distribution of your data"},{"metadata":{"trusted":true,"_uuid":"d44d22996fe8a9b9c2488185946243a3a9d1897f"},"cell_type":"code","source":"import sklearn.preprocessing as preproc\ndf_scale = titanic[['Fare']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba4fded0f9816bf30fd4635d46c1105b691adab6"},"cell_type":"code","source":"#Min-Max Scaling\ndf_scale['Min-Max'] = preproc.minmax_scale(titanic[['Fare']])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bb0e50da1c7fb15a92cf9cad9b0940259f0705a"},"cell_type":"code","source":"#Standardization\ndf_scale['Standardization'] = preproc.StandardScaler().fit_transform(titanic[['Fare']])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b6c073c50ff6883d41c16be9e0454e3ac0b9e684"},"cell_type":"markdown","source":"**Plot Scaled Features**"},{"metadata":{"trusted":true,"_uuid":"b3ef3fe6fa8e5d62358816097f2629c4ecafe0a5"},"cell_type":"code","source":"fig, (ax1, ax2, ax3) = plt.subplots(3,1)\nfig.tight_layout()\n\n# Plot Original Price\ndf_scale['Fare'].hist(ax=ax1, bins=50)\nax1.tick_params(labelsize=14)\nax1.set_xlabel(\"Original Price\", fontsize=10)\nax1.set_ylabel(\"Frequency\", fontsize=14)\n\n# Plot Min-Max Scaling on Price\ndf_scale['Min-Max'].hist(ax=ax2, bins=50, color='r')\nax2.tick_params(labelsize=14)\nax2.set_xlabel(\"Min-Max Price\", fontsize=10)\n\n# Plot Standardized Scaling on Price\ndf_scale['Standardization'].hist(ax=ax3, bins=50, color='g')\nax3.tick_params(labelsize=14)\nax3.set_xlabel(\"Standarized Price\", fontsize=10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b878548cc91844ca07f407af364ab58f06b1e096"},"cell_type":"markdown","source":"**Categorical Feature Engineering**"},{"metadata":{"trusted":true,"_uuid":"cb09940aa6bc47856405eee641102055060d58b8"},"cell_type":"code","source":"titanic_cat = titanic.select_dtypes(include=['object','category'])\ntitanic_cat.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c424a3bbb1c398c3edb0adb7c7855ecbb2090d3a"},"cell_type":"code","source":"#Encode our 'AgeGroup' Category\nage_group = np.unique(titanic_cat['AgeGroup'])\nage_group","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e4160b59f022cce5ca8291c3e7f1d0d033a0fef"},"cell_type":"code","source":"#Label Encoding\nfrom sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ngenre_labels = le.fit_transform(titanic['AgeGroup'])\ntitanic_cat['AgeGroup_LE'] = genre_labels\ntitanic_cat[['AgeGroup','AgeGroup_LE']].head(20)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81740a6ce36fecdd49f330bd1566a668456e733a"},"cell_type":"markdown","source":"**Encoding Ordinal Variables**\n\nRequires domain knowledge and assumes order of importance"},{"metadata":{"trusted":true,"_uuid":"86ffbb3c2040e36b2a3ea99a5a2d2af44e4f8804"},"cell_type":"code","source":"# Ordinal Encoding\nfrom sklearn.preprocessing import OrdinalEncoder\nenc = OrdinalEncoder(categories=[['kid','teen','adult','elderly']])\nX = [['adult'], ['teen'], ['kid'], ['elderly'], ['adult']]\nAgeGroup_OE = enc.fit_transform(X)\nAgeGroup_OE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"baa3dc6ee44226c9b27d485e8a5f29d231c19de8"},"cell_type":"code","source":"#Get the original data back\nage_ord_map = {'kid': 1, 'teen': 2, 'adult': 3, 'elderly': 4}\ntitanic_cat['AgeGroup_OE'] = titanic_cat['AgeGroup'].map(age_ord_map)\ntitanic_cat[['AgeGroup','AgeGroup_OE']].head(20)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3fa67ee748db92199a9293ac8198781e6cb05c5c"},"cell_type":"markdown","source":"**Dummy Encoding**\n\n* Use one Dummy encoding where you want each value/category of the feature to be unique.\n* One Hot Encoding fixes the problem of having your model think that different categorical values have some numeric association to it."},{"metadata":{"trusted":true,"_uuid":"39c0107fc5a38f0ea44f3333ea3c6aa394fef6d0"},"cell_type":"code","source":"age = pd.DataFrame(['Kid','Teen','Adult','Elderly'], columns=['AgeGroup'])\nage_dummy_features = pd.get_dummies(age['AgeGroup'])\npd.concat([age, age_dummy_features], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d402489a02dda662b15d340d76b5a34f7498a40"},"cell_type":"code","source":"#Apply Dummy Encoding to 'AgeGroup' feature\ntitanic_dummyage = pd.get_dummies(titanic_cat['AgeGroup'])\npd.concat([titanic_cat, titanic_dummyage], axis=1)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}