{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T07:47:56.499973Z","iopub.execute_input":"2022-08-07T07:47:56.500577Z","iopub.status.idle":"2022-08-07T07:47:56.541752Z","shell.execute_reply.started":"2022-08-07T07:47:56.500487Z","shell.execute_reply":"2022-08-07T07:47:56.541116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Modeling \n> `sklearn.pipeline.Pipeline` Sequentially apply a list of transforms and a final estimator.The purpose of the pipeline is to assemble several steps that can be cross-validated together while setting different parameters.\n\n## 2. Data preprocessing\n* Check for null values\n> if there are null values then impute them\n\n* Check for Categorical feature columns and encode them \n>  feature-column `Sex` can be ecoded using a One-hot-encoder <br>\n>  feature-column `Pclass` can be ecoded using a One-hot-encoder <br>\n>  etc....\n\n* Check the mean and standard deviation of the features <br>\n> Its a good practice to normalize the features that have different scales and range.\nThis is important because the features are multiplied by model weights so the scale of the output and the scale of the gradient are affected by the scale of the inputs\n\n\n### Numerical categorical Features\nLet's encode them using OneHotEncoder\n> * 'Pclass'\n> * 'SibSp'\n> * 'Parch'\n\n### String Categorical Features\nLet's first impute the missing values using a SimpleImputer then\nLet's encode them using OneHotEncoder\n> * 'Sex'\n> * 'Cabin'\n> * 'Embarked'\n\n### Numerical features\nLet's first impute the missing values using a SimpleImputer then\nLet's Standardize features by removing the mean and scaling to unit variance\n> * 'Fare'\n> * 'Age'\n\n## 4. Cross-validate the model\n> Evaluate a score by cross-validation\n\n## 5. Compare different model scores\n> * RandomForestClassifier\n> * GradientBoostingClassifier\n> * LogisticRegression\n> * SGDClassifier\n> * SVC\n> * XGBClassifier\n\n## 6. Hyperparameter tuning / model improvement\n> we can use Grid-search-cv to find best model parameters that give us best score\n\n## 7. Kaggle submission\n> we can use the hypertuned / best model to make kaggle submission","metadata":{}},{"cell_type":"code","source":"# imports\n\n# seaborn for visualization\nimport seaborn as sns\n\n# preprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer\n\n# Pipeline\nfrom sklearn.pipeline import Pipeline\n\n# column transformer\nfrom sklearn.compose import ColumnTransformer\n\n# cross validation\nfrom sklearn.model_selection import cross_val_score\n\n# hyperparameter tuning\nfrom sklearn.model_selection import GridSearchCV\n\n# model\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.svm import SVC\nfrom xgboost import XGBClassifier\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:48:31.192569Z","iopub.execute_input":"2022-08-07T07:48:31.193289Z","iopub.status.idle":"2022-08-07T07:48:31.200410Z","shell.execute_reply.started":"2022-08-07T07:48:31.193261Z","shell.execute_reply":"2022-08-07T07:48:31.199663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest_df = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:48:37.113181Z","iopub.execute_input":"2022-08-07T07:48:37.113444Z","iopub.status.idle":"2022-08-07T07:48:37.142374Z","shell.execute_reply.started":"2022-08-07T07:48:37.113418Z","shell.execute_reply":"2022-08-07T07:48:37.141474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore the data","metadata":{}},{"cell_type":"code","source":"train_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:48:40.600428Z","iopub.execute_input":"2022-08-07T07:48:40.600834Z","iopub.status.idle":"2022-08-07T07:48:40.622075Z","shell.execute_reply.started":"2022-08-07T07:48:40.600790Z","shell.execute_reply":"2022-08-07T07:48:40.621450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().iloc[:3,:]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:48:43.337966Z","iopub.execute_input":"2022-08-07T07:48:43.338326Z","iopub.status.idle":"2022-08-07T07:48:43.370501Z","shell.execute_reply.started":"2022-08-07T07:48:43.338276Z","shell.execute_reply":"2022-08-07T07:48:43.370058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:49:03.036347Z","iopub.execute_input":"2022-08-07T07:49:03.037112Z","iopub.status.idle":"2022-08-07T07:49:03.050243Z","shell.execute_reply.started":"2022-08-07T07:49:03.037076Z","shell.execute_reply":"2022-08-07T07:49:03.049698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for null values in training data\ntrain_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:49:06.880453Z","iopub.execute_input":"2022-08-07T07:49:06.881158Z","iopub.status.idle":"2022-08-07T07:49:06.888762Z","shell.execute_reply.started":"2022-08-07T07:49:06.881105Z","shell.execute_reply":"2022-08-07T07:49:06.888101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for null values in testing data\ntest_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:49:11.680392Z","iopub.execute_input":"2022-08-07T07:49:11.680655Z","iopub.status.idle":"2022-08-07T07:49:11.688022Z","shell.execute_reply.started":"2022-08-07T07:49:11.680622Z","shell.execute_reply":"2022-08-07T07:49:11.687292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize the data","metadata":{}},{"cell_type":"code","source":"sns.heatmap(train_df.corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:49:17.807049Z","iopub.execute_input":"2022-08-07T07:49:17.807540Z","iopub.status.idle":"2022-08-07T07:49:18.270275Z","shell.execute_reply.started":"2022-08-07T07:49:17.807513Z","shell.execute_reply":"2022-08-07T07:49:18.269534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Length of training data', len(train_df))\nX = train_df.drop(columns=['Survived'])\ny = train_df['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:50:27.422912Z","iopub.execute_input":"2022-08-07T07:50:27.423123Z","iopub.status.idle":"2022-08-07T07:50:27.429561Z","shell.execute_reply.started":"2022-08-07T07:50:27.423101Z","shell.execute_reply":"2022-08-07T07:50:27.428334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"#  Features and transformers\n\n#  integer category\nint_cat_features = ['Pclass', 'SibSp', 'Parch']\nint_cat_transformers = Pipeline(steps=[('imputer', SimpleImputer(strategy='most_frequent')),\\\n                                      ('scale', StandardScaler())])\n\n# string category\nstr_cat_features = ['Sex', 'Cabin', 'Embarked']\nstr_cat_transformers = Pipeline(steps=[('imputer', SimpleImputer(strategy='most_frequent')),\\\n                                       ('one-hot', OneHotEncoder(handle_unknown='ignore'))])\n\n# continues neumerical - floats\nnum_features = ['Age', 'Fare']\nnum_transformers = Pipeline(steps=[('imputer', SimpleImputer()),\\\n                                   ('scale', StandardScaler())])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:52:22.064287Z","iopub.execute_input":"2022-08-07T07:52:22.064521Z","iopub.status.idle":"2022-08-07T07:52:22.069818Z","shell.execute_reply.started":"2022-08-07T07:52:22.064499Z","shell.execute_reply":"2022-08-07T07:52:22.069386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model building\n\ndef model_building(model):\n    #applying transformations\n    preprocess = ColumnTransformer(transformers=[('int_cat', int_cat_transformers, int_cat_features),\\\n                                                 ('str_cat', str_cat_transformers, str_cat_features),\\\n                                                 ('numeric', num_transformers, num_features)])\\\n    # preprocessing and modeling pipeline\n    pipe = Pipeline(steps=[('preprocessing', preprocess),\\\n                           ('modeling', model)])\n    \n    return pipe\n    \n# cross validating\ndef cross_validate_pipeline(pipeline, X, y):\n    cv_scores = cross_val_score(pipeline, X, y)\n    return cv_scores","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:52:32.261835Z","iopub.execute_input":"2022-08-07T07:52:32.262331Z","iopub.status.idle":"2022-08-07T07:52:32.267087Z","shell.execute_reply.started":"2022-08-07T07:52:32.262304Z","shell.execute_reply":"2022-08-07T07:52:32.266624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model comparison","metadata":{}},{"cell_type":"code","source":"\nmodels = [('RandomForest',RandomForestClassifier()), \\\n          ('LogisticRegression',LogisticRegression()), \\\n          ('GradientBoosting',GradientBoostingClassifier()), \\\n          ('SVC',SVC()), \\\n          ('SGDClassifier',SGDClassifier()), \\\n          ('XGBClassifier',XGBClassifier(use_label_encoder=False, eval_metric='logloss')) \\\n         ]\n\nfor name,model in models:\n    model_pipeline = model_building(model)\n    cv_scores = cross_validate_pipeline(model_pipeline, X, y)\n    print(f'{name :20} {cv_scores.mean()} ')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:53:04.861329Z","iopub.execute_input":"2022-08-07T07:53:04.862109Z","iopub.status.idle":"2022-08-07T07:53:08.130060Z","shell.execute_reply.started":"2022-08-07T07:53:04.862078Z","shell.execute_reply":"2022-08-07T07:53:08.129576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Improvement ","metadata":{}},{"cell_type":"code","source":"models = [('RandomForest', \\\n           RandomForestClassifier(), \\\n           {'modeling__max_depth':[i for i in range(4,12)]}), \\\n          \n          ('LogisticRegression', \\\n           LogisticRegression(), \\\n           {'modeling__C':[i*0.1 for i in range(10,15)]}), \\\n          \n          ('GradientBoosting', \\\n           GradientBoostingClassifier(), \\\n           {'modeling__n_estimators':[i for i in range(100,300,50)]}), \\\n          \n          ('SVC', \\\n           SVC(), \\\n           {'modeling__C':[i for i in range(1,10)]}), \\\n          \n          ('SGDClassifier',SGDClassifier(), \\\n           {'modeling__warm_start':[True,False], \\\n            'modeling__early_stopping':[True,False], \\\n            'modeling__average':[True,False]}), \\\n          \n          ('XGBClassifier', \\\n           XGBClassifier(use_label_encoder=False, eval_metric='logloss'), \\\n           {'modeling__colsample_bytree':[0.7], \\\n            'modeling__colsample_bylevel':[0.5], \\\n            'modeling__colsample_bynode':[0.7], \\\n            'modeling__subsample':[0.6,0.7]}) \\\n         ]\n\nfor name, model, param_grid in models:\n    pipe = model_building(model)\n    gs = GridSearchCV(pipe, param_grid)\n    gs.fit(X,y)\n    print(f'{name :20} {gs.best_score_}')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:53:18.869680Z","iopub.execute_input":"2022-08-07T07:53:18.869921Z","iopub.status.idle":"2022-08-07T07:53:36.935421Z","shell.execute_reply.started":"2022-08-07T07:53:18.869885Z","shell.execute_reply":"2022-08-07T07:53:36.934843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making Final predictions","metadata":{}},{"cell_type":"code","source":"# classifier for making predictions\nclf = SVC()\n# modeling\nmodel = model_building(clf)\n# training\nmodel.fit(X,y)\n# making predictions\npreds = model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:51:26.306941Z","iopub.execute_input":"2022-08-07T09:51:26.307263Z","iopub.status.idle":"2022-08-07T09:51:26.368777Z","shell.execute_reply.started":"2022-08-07T09:51:26.307239Z","shell.execute_reply":"2022-08-07T09:51:26.368275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kaggle submission","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame(data={'PassengerId':test_df['PassengerId'],'Survived':preds})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:54:42.732648Z","iopub.execute_input":"2022-08-07T07:54:42.732951Z","iopub.status.idle":"2022-08-07T07:54:42.739598Z","shell.execute_reply.started":"2022-08-07T07:54:42.732928Z","shell.execute_reply":"2022-08-07T07:54:42.739102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Your submission was successfully saved!')","metadata":{"execution":{"iopub.status.busy":"2021-11-16T14:32:19.430392Z","iopub.execute_input":"2021-11-16T14:32:19.430746Z","iopub.status.idle":"2021-11-16T14:32:19.438617Z","shell.execute_reply.started":"2021-11-16T14:32:19.430702Z","shell.execute_reply":"2021-11-16T14:32:19.437703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just for fun lets solve this problem using Tensorflow\n## Using Tensorflow","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:41.784747Z","iopub.execute_input":"2022-08-07T09:42:41.785416Z","iopub.status.idle":"2022-08-07T09:42:41.788713Z","shell.execute_reply.started":"2022-08-07T09:42:41.785389Z","shell.execute_reply":"2022-08-07T09:42:41.788073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset creation","metadata":{}},{"cell_type":"code","source":"titanic_types = {}\n# create a dict of column names and type\nfor k,v in train_df.items():\n    titanic_types[k] = v.dtype\n    \ntype_lookup = {\n    'int64' : np.int64(),\n    'float64' : np.float64(),\n    'object': np.str()\n}\n    \ncol_names = [k for k,v in titanic_types.items()]\ncol_types = [type_lookup[str(v)] for k,v in titanic_types.items()]\ncol_types","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:45.165899Z","iopub.execute_input":"2022-08-07T09:42:45.166138Z","iopub.status.idle":"2022-08-07T09:42:45.174176Z","shell.execute_reply.started":"2022-08-07T09:42:45.166114Z","shell.execute_reply":"2022-08-07T09:42:45.173554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = tf.data.experimental.make_csv_dataset('/kaggle/input/titanic/train.csv',\n                                                 batch_size=32,\n                                                 column_defaults=col_types,\n                                                 label_name='Survived',\n                                                 num_epochs=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:47.485596Z","iopub.execute_input":"2022-08-07T09:42:47.485954Z","iopub.status.idle":"2022-08-07T09:42:47.518572Z","shell.execute_reply.started":"2022-08-07T09:42:47.485928Z","shell.execute_reply":"2022-08-07T09:42:47.518091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_batch, label_batch = next(iter(train_ds))\n\nfor k,v in feature_batch.items():\n    print(f'{k !r:15s} {v.dtype} {v[:2]}')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:49.189101Z","iopub.execute_input":"2022-08-07T09:42:49.189536Z","iopub.status.idle":"2022-08-07T09:42:49.251251Z","shell.execute_reply.started":"2022-08-07T09:42:49.189510Z","shell.execute_reply":"2022-08-07T09:42:49.250555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It you're passing a heterogenous DataFrame to Keras, each column may need unique preprocessing. You could do this preprocessing directly in the DataFrame, but for a model to work correctly, inputs always need to be preprocessed the same way. So, the best approach is to build the preprocessing into the model.\n\n\n\n**`\"Symbolic\" tensors`**. <br>\nNormal \"eager\" tensors have a value. In contrast these \"symbolic\" tensors do not. Instead they keep track of which operations are run on them, and build representation of the calculation, that you can run later.\n\nThe functional API operates on Symbolic\" tensors","metadata":{}},{"cell_type":"code","source":"# create Symbolic Tensors \n\ntype_lookup = {\n    'int64' : tf.int64,\n    'float64' : tf.float64,\n    'object' : tf.string  \n}\n\ninputs={}\n\nfor k,v in titanic_types.items():   \n    if k not in ['PassengerId', 'Survived', 'Name', 'Ticket']:\n        inputs[k]=tf.keras.Input(shape=(1,),name=k, dtype=type_lookup[str(v)])\n\n# input features \ninputs","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:53.831922Z","iopub.execute_input":"2022-08-07T09:42:53.832312Z","iopub.status.idle":"2022-08-07T09:42:53.846089Z","shell.execute_reply.started":"2022-08-07T09:42:53.832277Z","shell.execute_reply":"2022-08-07T09:42:53.845300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data preprocessing util\n\ndef encode_numeric(numeric_features, feature_names, df):\n    df = df.copy()\n    df = df[feature_names].fillna(method='ffill')\n    normalizer = tf.keras.layers.Normalization()\n    normalizer.adapt(np.array(df))\n    return normalizer(numeric_features)\n\ndef integer_categorical_encoding(cat_feature, feature_name, df):\n    df = df.copy()\n    df = df[feature_name].fillna(-1)\n    lookup = tf.keras.layers.IntegerLookup()\n    lookup.adapt(df)\n    cat_encode = tf.keras.layers.CategoryEncoding(num_tokens=lookup.vocabulary_size()) \n    return cat_encode(lookup(cat_feature))\n\ndef string_categorical_encoding(cat_feature, feature_name, df):\n    df = df.copy()\n    df = df[feature_name].fillna('missing')\n    lookup = tf.keras.layers.StringLookup()\n    lookup.adapt(df)\n    cat_encode = tf.keras.layers.CategoryEncoding(num_tokens=lookup.vocabulary_size()) \n    return cat_encode(lookup(cat_feature))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:55.737242Z","iopub.execute_input":"2022-08-07T09:42:55.737780Z","iopub.status.idle":"2022-08-07T09:42:55.744803Z","shell.execute_reply.started":"2022-08-07T09:42:55.737742Z","shell.execute_reply":"2022-08-07T09:42:55.744042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Continuous numerical encoding**\n\nNote: If you have many features that need identical preprocessing it's more efficient to concatenate them together befofre applying the preprocessing.","metadata":{}},{"cell_type":"code","source":"numerical_features = [k for k,v in titanic_types.items() if v == np.float64]\nnumerical_features","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:42:59.199209Z","iopub.execute_input":"2022-08-07T09:42:59.199434Z","iopub.status.idle":"2022-08-07T09:42:59.204672Z","shell.execute_reply.started":"2022-08-07T09:42:59.199409Z","shell.execute_reply":"2022-08-07T09:42:59.203846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nnumerical_encoding =[]\n\n# features that need identical preprocessing \nnumerical_inputs = [inputs[name] for name in numerical_features]\n\n# concatenate them together before applying the preprocessing.\nnumerical_inputs = tf.keras.layers.Concatenate()(numerical_inputs)\n\n# applying preprocessing\nencoded = encode_numeric(numerical_inputs, numerical_features, train_df)\n\nnumerical_encoding.append(encoded)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:01.670771Z","iopub.execute_input":"2022-08-07T09:43:01.671172Z","iopub.status.idle":"2022-08-07T09:43:01.846192Z","shell.execute_reply.started":"2022-08-07T09:43:01.671146Z","shell.execute_reply":"2022-08-07T09:43:01.845590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_encoding","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:04.139965Z","iopub.execute_input":"2022-08-07T09:43:04.140196Z","iopub.status.idle":"2022-08-07T09:43:04.146229Z","shell.execute_reply.started":"2022-08-07T09:43:04.140173Z","shell.execute_reply":"2022-08-07T09:43:04.145509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Categorical encoding**","metadata":{}},{"cell_type":"code","source":"int_cat_features = [k for k,v in inputs.items() if v.dtype == tf.int64]\nstr_cat_features = [k for k,v in inputs.items() if v.dtype == tf.string]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:05.994203Z","iopub.execute_input":"2022-08-07T09:43:05.995042Z","iopub.status.idle":"2022-08-07T09:43:06.000062Z","shell.execute_reply.started":"2022-08-07T09:43:05.995015Z","shell.execute_reply":"2022-08-07T09:43:05.999581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encoding categorical integers\nint_cat_encoding = []\nfor name in int_cat_features:\n    feature = inputs[name]\n    encoding = integer_categorical_encoding(feature, name, train_df)\n    int_cat_encoding.append(encoding)\n    \n# encoding categorical string\nstr_cat_encoding = []\nfor name in str_cat_features:\n    feature = inputs[name]\n    encoding = string_categorical_encoding(feature, name, train_df)\n    str_cat_encoding.append(encoding)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:08.061044Z","iopub.execute_input":"2022-08-07T09:43:08.061279Z","iopub.status.idle":"2022-08-07T09:43:08.589457Z","shell.execute_reply.started":"2022-08-07T09:43:08.061257Z","shell.execute_reply":"2022-08-07T09:43:08.588872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int_cat_encoding","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:15.947797Z","iopub.execute_input":"2022-08-07T09:43:15.948494Z","iopub.status.idle":"2022-08-07T09:43:15.953166Z","shell.execute_reply.started":"2022-08-07T09:43:15.948468Z","shell.execute_reply":"2022-08-07T09:43:15.952600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"str_cat_encoding","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:17.609265Z","iopub.execute_input":"2022-08-07T09:43:17.610287Z","iopub.status.idle":"2022-08-07T09:43:17.615827Z","shell.execute_reply.started":"2022-08-07T09:43:17.610245Z","shell.execute_reply":"2022-08-07T09:43:17.615336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all encoded inputs \nencoded_inputs = [*numerical_encoding, *int_cat_encoding, *str_cat_encoding ]\nencoded_inputs","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:19.662806Z","iopub.execute_input":"2022-08-07T09:43:19.663160Z","iopub.status.idle":"2022-08-07T09:43:19.668837Z","shell.execute_reply.started":"2022-08-07T09:43:19.663123Z","shell.execute_reply":"2022-08-07T09:43:19.668245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Concatenate all nunique preprocessed features along the depth axis, so each dictionary-example is converted into a single vector. The vector contains categorical features, numeric features","metadata":{}},{"cell_type":"code","source":"\npreprocessed = tf.keras.layers.Concatenate()(encoded_inputs)\n\ntitanic_preprocessing = tf.keras.Model(inputs, preprocessed)\n\n# lets plot the preprocessing\n\ntf.keras.utils.plot_model(titanic_preprocessing,\n                          show_shapes=True,\n                          rankdir='LR')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:23.877922Z","iopub.execute_input":"2022-08-07T09:43:23.878440Z","iopub.status.idle":"2022-08-07T09:43:24.083587Z","shell.execute_reply.started":"2022-08-07T09:43:23.878414Z","shell.execute_reply":"2022-08-07T09:43:24.082689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now lets build the neural network on top this**","metadata":{}},{"cell_type":"code","source":"# BODY OF THE MODEL\nfully_connected = tf.keras.Sequential([ \\\n    tf.keras.layers.Dense(32, activation='relu'), \\\n    tf.keras.layers.Dropout(0.2), \\\n    tf.keras.layers.Dense(16, activation='relu'),\n    tf.keras.layers.Dense(1), \\\n]) ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:27.691751Z","iopub.execute_input":"2022-08-07T09:43:27.692354Z","iopub.status.idle":"2022-08-07T09:43:27.702150Z","shell.execute_reply.started":"2022-08-07T09:43:27.692325Z","shell.execute_reply":"2022-08-07T09:43:27.701669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Titanic model using Tensorflow**","metadata":{}},{"cell_type":"code","source":"def build_tf_model(inputs, preprocessing, fullyconnected):\n    prep = preprocessing(inputs)\n    result = fullyconnected(prep)\n    model = tf.keras.Model(inputs, result)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:30.191756Z","iopub.execute_input":"2022-08-07T09:43:30.192409Z","iopub.status.idle":"2022-08-07T09:43:30.196899Z","shell.execute_reply.started":"2022-08-07T09:43:30.192374Z","shell.execute_reply":"2022-08-07T09:43:30.195979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Training**","metadata":{}},{"cell_type":"code","source":"# early stopping callback\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=1)\n\n# building titanic model\ntf_model = build_tf_model(inputs,\\\n                          titanic_preprocessing, \\\n                          fully_connected)\n\n# compile the model\ntf_model.compile(optimizer='adam', \\\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True), \\\n              metrics=['accuracy'] )\n\n# training \ntf_history = tf_model.fit(train_ds,\n                          callbacks=[early_stopping], \\\n                          epochs=15)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:32.673179Z","iopub.execute_input":"2022-08-07T09:43:32.673654Z","iopub.status.idle":"2022-08-07T09:43:35.514947Z","shell.execute_reply.started":"2022-08-07T09:43:32.673630Z","shell.execute_reply":"2022-08-07T09:43:35.513881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ploting model performance\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12,5))\nplt.subplot(1,2,1)\nplt.plot(tf_history.history['accuracy'],color='green')\nplt.title(\"MODEL ACCURACY\")\nplt.xlabel('Epochs')\nplt.subplot(1,2,2)\nplt.plot(tf_history.history['loss'],color='salmon')\nplt.title(\"MODEL LOSS\")\nplt.xlabel('Epochs')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:43:58.044529Z","iopub.execute_input":"2022-08-07T09:43:58.045357Z","iopub.status.idle":"2022-08-07T09:43:58.291468Z","shell.execute_reply.started":"2022-08-07T09:43:58.045318Z","shell.execute_reply":"2022-08-07T09:43:58.290648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_types = {}\n# create a dict of column names and type\nfor k,v in test_df.items():\n    titanic_types[k] = v.dtype\n    \ntype_lookup = {\n    'int64' : np.int64(),\n    'float64' : np.float64(),\n    'object': np.str()\n}\n    \ncol_names = [k for k,v in titanic_types.items()]\ncol_types = [type_lookup[str(v)] for k,v in titanic_types.items()]\ntitanic_types","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:47:48.061427Z","iopub.execute_input":"2022-08-07T09:47:48.061692Z","iopub.status.idle":"2022-08-07T09:47:48.069265Z","shell.execute_reply.started":"2022-08-07T09:47:48.061664Z","shell.execute_reply":"2022-08-07T09:47:48.068819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = tf.data.experimental.make_csv_dataset('/kaggle/input/titanic/test.csv',\n                                                 batch_size=32,\n                                                 column_defaults=col_types,\n                                                 num_epochs=1)\n\n\ntest_feature_batch = next(iter(test_ds))\n\nfor k,v in test_feature_batch.items():\n    print(f'{k !r:15s} {v.dtype} {v[:2]}')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:49:43.820562Z","iopub.execute_input":"2022-08-07T09:49:43.820820Z","iopub.status.idle":"2022-08-07T09:49:43.906069Z","shell.execute_reply.started":"2022-08-07T09:49:43.820797Z","shell.execute_reply":"2022-08-07T09:49:43.904814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_preds = tf_model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:51:41.473685Z","iopub.execute_input":"2022-08-07T09:51:41.473915Z","iopub.status.idle":"2022-08-07T09:51:41.581106Z","shell.execute_reply.started":"2022-08-07T09:51:41.473892Z","shell.execute_reply":"2022-08-07T09:51:41.580204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outputs = np.array([0 if x<0 else 1 for x in tf_preds.flatten()])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:56:06.751149Z","iopub.execute_input":"2022-08-07T09:56:06.751501Z","iopub.status.idle":"2022-08-07T09:56:06.756664Z","shell.execute_reply.started":"2022-08-07T09:56:06.751464Z","shell.execute_reply":"2022-08-07T09:56:06.756084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## kaggle submission","metadata":{}},{"cell_type":"code","source":"# tf_output = pd.DataFrame(data={'PassengerId':test_df['PassengerId'],'Survived':outputs})\n# tf_output.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:57:33.477595Z","iopub.execute_input":"2022-08-07T09:57:33.477845Z","iopub.status.idle":"2022-08-07T09:57:33.483970Z","shell.execute_reply.started":"2022-08-07T09:57:33.477822Z","shell.execute_reply":"2022-08-07T09:57:33.483158Z"},"trusted":true},"execution_count":null,"outputs":[]}]}