{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:01:53.572115Z","iopub.execute_input":"2022-07-14T15:01:53.572573Z","iopub.status.idle":"2022-07-14T15:01:54.491557Z","shell.execute_reply.started":"2022-07-14T15:01:53.572460Z","shell.execute_reply":"2022-07-14T15:01:54.490351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![ss-titanic](https://i.imgur.com/voGgul0.png)","metadata":{}},{"cell_type":"code","source":"#load data\ntitanic_train_path = \"../input/spaceship-titanic/train.csv\"\ntitanic_train_raw = pd.read_csv(titanic_train_path)\n\nprint(titanic_train_raw.info())\ntitanic_train_raw.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:01:59.170153Z","iopub.execute_input":"2022-07-14T15:01:59.170662Z","iopub.status.idle":"2022-07-14T15:01:59.261968Z","shell.execute_reply.started":"2022-07-14T15:01:59.170615Z","shell.execute_reply":"2022-07-14T15:01:59.260858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train_raw.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:06.891112Z","iopub.execute_input":"2022-07-14T15:02:06.892081Z","iopub.status.idle":"2022-07-14T15:02:06.904388Z","shell.execute_reply.started":"2022-07-14T15:02:06.892027Z","shell.execute_reply":"2022-07-14T15:02:06.903496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### About Each Features\n\n* `PassengerId`: A unique Id for each passenger. Each Id takes the form gggg_pp where gggg indicates a group the passenger is travelling with and pp is their number within the group. People in a group are often family members, but not always.\n* `HomePlanet`: The planet the passenger departed from, typically their planet of permanent residence.\n* `CryoSleep`: Indicates whether the passenger elected to be put into suspended animation for the duration of the voyage. Passengers in cryosleep are confined to their cabins.\n* `Cabin`: The cabin number where the passenger is staying. Takes the form deck/num/side, where side can be either P for Port or S for Starboard.\n* `Destination`: The planet the passenger will be debarking to.\n* `Age`: The age of the passenger.\n* `VIP`: Whether the passenger has paid for special VIP service during the voyage.\n* `RoomService`, `FoodCourt`, `ShoppingMall`, `Spa`, `VRDeck`: Amount the passenger has billed at each of the Spaceship Titanic's many luxury amenities.\n* `Name`: The first and last names of the passenger.\n* `Transported`: Whether the passenger was transported to another dimension. This is the target, the column you are trying to predict.","metadata":{}},{"cell_type":"markdown","source":"### Extract useful informations from features","metadata":{}},{"cell_type":"code","source":"def extract_info(df):\n    #extract groupID from PassengerID\n    df['GroupID'] = df['PassengerId'].str.split('_', expand=True)[0]\n\n    #split cabin into deck, num, and side\n    cabin_split = df['Cabin'].str.split('/', expand=True)\n    df['CabinDeck'] = cabin_split[0]\n    df['CabinNum'] = cabin_split[1]\n    df['CabinSide'] = cabin_split[2]\n    df.drop('Cabin', inplace=True, axis=1)\n\n    #split first and family name\n    name_split = df['Name'].str.split(' ', expand=True)\n    df['FamilyName'] = name_split[1]\n    df.drop('Name', inplace=True, axis=1)\n    \n    #calculate total amenities fee\n    amenities = ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n    df['TotalFee'] = df[amenities].sum(axis=1)\n    \n    #fix numeric data's dtype\n    numeric = ['CryoSleep','Age','VIP','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck','GroupID','CabinNum','TotalFee']\n    for n in numeric:\n        df[n] = pd.to_numeric(df[n])\n    \n    return df\n\ntitanic_train = titanic_train_raw.copy()\ntitanic_train = extract_info(titanic_train)\n#titanic_train = calculate_columns(titanic_train)\nprint(titanic_train.info())\ntitanic_train","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:11.412135Z","iopub.execute_input":"2022-07-14T15:02:11.413240Z","iopub.status.idle":"2022-07-14T15:02:11.522828Z","shell.execute_reply.started":"2022-07-14T15:02:11.413200Z","shell.execute_reply":"2022-07-14T15:02:11.521839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filling missing values with an educated guess","metadata":{}},{"cell_type":"code","source":"#search groups with different home planet and destination\nlist_group = ['FamilyName', 'GroupID']\nlist_target = ['HomePlanet', 'Destination']\n\nfor group in list_group:\n    for target in list_target:\n        num_target_in_group = titanic_train.groupby([group])[target].nunique().reset_index()\n        print(\"# of %s with different %s: %d\" % (group, target, num_target_in_group[num_target_in_group[target] > 1].shape[0]))\nprint(\"\\ntotal # of families: %s  total # of groups: %s\"%(titanic_train.nunique()['FamilyName'], titanic_train.nunique()['GroupID']))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:18.160538Z","iopub.execute_input":"2022-07-14T15:02:18.161609Z","iopub.status.idle":"2022-07-14T15:02:18.209559Z","shell.execute_reply.started":"2022-07-14T15:02:18.161490Z","shell.execute_reply":"2022-07-14T15:02:18.208528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#passengers in same family and passengers in same group are from same home planet, but does not always goes to same destination\ndef homeplanet_vs_group_family_fill(df):\n    list_group = ['GroupID', 'FamilyName']\n    for group in list_group:\n        home_planet_in_group = df.dropna(subset=[group, 'HomePlanet']).drop_duplicates([group])[[group,'HomePlanet']]\n        home_planet_dict = dict(zip(home_planet_in_group[group], home_planet_in_group['HomePlanet']))\n        df['HomePlanet'] = df['HomePlanet'].fillna(df[group].map(home_planet_dict))\n    \n    return df\n\ntitanic_train = homeplanet_vs_group_family_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:22.001130Z","iopub.execute_input":"2022-07-14T15:02:22.001749Z","iopub.status.idle":"2022-07-14T15:02:22.035017Z","shell.execute_reply.started":"2022-07-14T15:02:22.001707Z","shell.execute_reply":"2022-07-14T15:02:22.034066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#passengers in cryo sleep has no amenities expenses\ndef cryo_vs_amenities_fill(df):\n    df['RoomService'] = np.where(df['CryoSleep'] == True, 0, df['RoomService'])\n    df['FoodCourt'] = np.where(df['CryoSleep'] == True, 0, df['FoodCourt'])\n    df['ShoppingMall'] = np.where(df['CryoSleep'] == True, 0, df['ShoppingMall'])\n    df['Spa'] = np.where(df['CryoSleep'] == True, 0, df['Spa'])\n    df['VRDeck'] = np.where(df['CryoSleep'] == True, 0, df['VRDeck'])\n\n    df['CryoSleep'] = np.where(df['TotalFee'] > 0, False, df['CryoSleep'])\n    \n    return df\n    \ntitanic_train = cryo_vs_amenities_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:25.190849Z","iopub.execute_input":"2022-07-14T15:02:25.191197Z","iopub.status.idle":"2022-07-14T15:02:25.202797Z","shell.execute_reply.started":"2022-07-14T15:02:25.191167Z","shell.execute_reply":"2022-07-14T15:02:25.201861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find age limit for vip and amenities\nunder_21 = titanic_train[titanic_train['Age']<21]\n\nfig, ax = plt.subplots(1, 2, figsize=(18,4))\n\nax[0].set_title(\"Age vs VIP\")\nsns.barplot(x=under_21.Age, y=under_21.VIP, ax=ax[0])\n\nax[1].set_title(\"Age vs Total Amenities Fee\")\nsns.barplot(x=under_21.Age, y=under_21.TotalFee, ax=ax[1])\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:28.891379Z","iopub.execute_input":"2022-07-14T15:02:28.892164Z","iopub.status.idle":"2022-07-14T15:02:30.296599Z","shell.execute_reply.started":"2022-07-14T15:02:28.892127Z","shell.execute_reply":"2022-07-14T15:02:30.295442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def age_limit_fill(df, vip_limit=18, amenities_limit=13):\n    df['VIP'] = np.where(df['Age'] < vip_limit, False, df['VIP'])\n    \n    df['RoomService'] = np.where(df['Age'] < amenities_limit, 0, df['RoomService'])\n    df['FoodCourt'] = np.where(df['Age'] < amenities_limit, 0, df['FoodCourt'])\n    df['ShoppingMall'] = np.where(df['Age'] < amenities_limit, 0, df['ShoppingMall'])\n    df['Spa'] = np.where(df['Age'] < amenities_limit, 0, df['Spa'])\n    df['VRDeck'] = np.where(df['Age'] < amenities_limit, 0, df['VRDeck'])\n\n    df['TotalFee'] = np.where(df['Age'] < amenities_limit, 0, df['TotalFee'])\n    \n    return df\n    \ntitanic_train = age_limit_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:38.038630Z","iopub.execute_input":"2022-07-14T15:02:38.039594Z","iopub.status.idle":"2022-07-14T15:02:38.051062Z","shell.execute_reply.started":"2022-07-14T15:02:38.039553Z","shell.execute_reply":"2022-07-14T15:02:38.050173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mode_by_group_n_family_fill(df, columns=['CabinDeck','CabinNum','CabinSide']):\n    for col in columns:\n        print(col)\n        mode_by_group = df.dropna(subset=[col]).groupby(['GroupID', 'FamilyName'])[col].agg(lambda x: pd.Series.mode(x)[0]).reset_index()\n        mode_dict = dict(zip(mode_by_group.set_index(['GroupID', 'FamilyName']).index, mode_by_group[col]))\n        df[col] = df[col].fillna(df.set_index(['GroupID', 'FamilyName']).index.map(mode_dict).to_series().reset_index(drop=True))\n\n    return df\n\ntitanic_train = mode_by_group_n_family_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:44.193008Z","iopub.execute_input":"2022-07-14T15:02:44.193381Z","iopub.status.idle":"2022-07-14T15:02:46.614961Z","shell.execute_reply.started":"2022-07-14T15:02:44.193349Z","shell.execute_reply":"2022-07-14T15:02:46.613933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mode_by_group_fill(df, columns=['FamilyName','CabinDeck','CabinNum','CabinSide']):\n    for col in columns:\n        print(col)\n        mode_by_group = df.dropna(subset=[col]).groupby('GroupID')[col].agg(lambda x: pd.Series.mode(x)[0]).reset_index()\n        mode_dict = dict(zip(mode_by_group['GroupID'], mode_by_group[col]))\n        df[col] = df[col].fillna(df['GroupID'].map(mode_dict))\n    \n    return df\ntitanic_train = mode_by_group_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:49.948670Z","iopub.execute_input":"2022-07-14T15:02:49.949834Z","iopub.status.idle":"2022-07-14T15:02:53.043661Z","shell.execute_reply.started":"2022-07-14T15:02:49.949785Z","shell.execute_reply":"2022-07-14T15:02:53.042799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mode_fill(df, columns=['HomePlanet','CryoSleep','Destination','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck','VIP','FamilyName','CabinDeck','CabinNum','CabinSide']):\n    for col in columns:\n        print(col)\n        df[col] = df[col].fillna(value = df[col].mode()[0])\n    \n    return df\n\ntitanic_train = mode_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:55.354669Z","iopub.execute_input":"2022-07-14T15:02:55.355051Z","iopub.status.idle":"2022-07-14T15:02:55.376035Z","shell.execute_reply.started":"2022-07-14T15:02:55.355019Z","shell.execute_reply":"2022-07-14T15:02:55.374903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_fill(df, columns=['Age']):\n    for col in columns:\n        print(col)\n        df[col] = df[col].fillna(value = df[col].mean())\n    \n    return df\n\ntitanic_train = mean_fill(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:02:59.700911Z","iopub.execute_input":"2022-07-14T15:02:59.701288Z","iopub.status.idle":"2022-07-14T15:02:59.709417Z","shell.execute_reply.started":"2022-07-14T15:02:59.701255Z","shell.execute_reply":"2022-07-14T15:02:59.708539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:03:03.403431Z","iopub.execute_input":"2022-07-14T15:03:03.404472Z","iopub.status.idle":"2022-07-14T15:03:03.419800Z","shell.execute_reply.started":"2022-07-14T15:03:03.404435Z","shell.execute_reply":"2022-07-14T15:03:03.418640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_group_family(df):\n\n    #count # of family members (passengers with same family name) in same traveling group\n    num_family_members = df.groupby('FamilyName')['PassengerId'].count().reset_index()\n    num_family_members_dict = dict(zip(num_family_members['FamilyName'], num_family_members['PassengerId']))\n    df['NumFamilyMembers'] = df['FamilyName'].map(num_family_members_dict)\n\n    #count # of group memebers\n    num_group_members = df.groupby('GroupID')['PassengerId'].count().reset_index()\n    num_group_members_dict = dict(zip(num_group_members['GroupID'], num_group_members['PassengerId']))\n    df['NumGroupMembers'] = df['GroupID'].map(num_group_members_dict)\n\n    return df\n\ntitanic_train = count_group_family(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:30:26.211383Z","iopub.execute_input":"2022-07-14T15:30:26.211751Z","iopub.status.idle":"2022-07-14T15:30:26.240857Z","shell.execute_reply.started":"2022-07-14T15:30:26.211721Z","shell.execute_reply":"2022-07-14T15:30:26.239968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def age_group(df):\n    df['Age_group'] = df['Age']//20\n    \n    return df\ntitanic_train = age_group(titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:04:30.891747Z","iopub.execute_input":"2022-07-14T15:04:30.892468Z","iopub.status.idle":"2022-07-14T15:04:30.899207Z","shell.execute_reply.started":"2022-07-14T15:04:30.892426Z","shell.execute_reply":"2022-07-14T15:04:30.898100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:04:33.822370Z","iopub.execute_input":"2022-07-14T15:04:33.822759Z","iopub.status.idle":"2022-07-14T15:04:33.835439Z","shell.execute_reply.started":"2022-07-14T15:04:33.822727Z","shell.execute_reply":"2022-07-14T15:04:33.834658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing","metadata":{}},{"cell_type":"code","source":"#onehot_encoding\ndef one_hot(df, columns):\n    for col in columns:\n        dummies = pd.get_dummies(df[col], prefix = col)\n        df = pd.concat([df, dummies], axis = 1)\n    df = df.drop(columns = columns)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:06:34.562348Z","iopub.execute_input":"2022-07-14T15:06:34.563107Z","iopub.status.idle":"2022-07-14T15:06:34.568387Z","shell.execute_reply.started":"2022-07-14T15:06:34.563067Z","shell.execute_reply":"2022-07-14T15:06:34.567354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# compile all into a function\ndef preprocessing(train, test):\n    #train data\n    train = extract_info(train)\n    train = homeplanet_vs_group_family_fill(train)\n    train = cryo_vs_amenities_fill(train)\n    train = age_limit_fill(train)\n    train = mode_by_group_n_family_fill(train)\n    train = mode_by_group_fill(train)\n    train = mode_fill(train)\n    train = mean_fill(train)\n    trian = count_group_family(train)\n    train = age_group(train)\n    \n    #test data\n    test = extract_info(test)\n    test = homeplanet_vs_group_family_fill(test)\n    test = cryo_vs_amenities_fill(test)\n    test = age_limit_fill(test)\n    test = mode_by_group_n_family_fill(test)\n    test = mode_by_group_fill(test)\n    test = mode_fill(test)\n    test = mean_fill(test)\n    test = count_group_family(test)\n    test = age_group(test)\n    \n    remove_col = [\n        'PassengerId',\n        \n    ]\n    \n    \n    one_hot_col = [\n        'HomePlanet',\n        'Destination',\n        'CabinDeck',\n        'CabinSide'\n    ]\n    \n    label_encode_col = [\n        'FamilyName'\n    ]\n    \n    train = train.drop(columns=remove_col)\n    test = test.drop(columns=remove_col)\n    \n    train = one_hot(train, one_hot_col)\n    test = one_hot(test, one_hot_col)\n    \n    concat_data = pd.concat([train[label_encode_col], test[label_encode_col]])\n    label_encoder = LabelEncoder()\n    label_encoder.fit(concat_data[label_encode_col])\n    \n    \n    for col in label_encode_col:\n        train[col] = label_encoder.transform(train[col])\n        test[col] = label_encoder.transform(test[col])\n    \n    return train, test\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:31:33.124392Z","iopub.execute_input":"2022-07-14T15:31:33.125317Z","iopub.status.idle":"2022-07-14T15:31:33.134800Z","shell.execute_reply.started":"2022-07-14T15:31:33.125279Z","shell.execute_reply":"2022-07-14T15:31:33.133864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train_path = \"../input/spaceship-titanic/train.csv\"\ntitanic_test_path = \"../input/spaceship-titanic/test.csv\"\ntitanic_train_raw = pd.read_csv(titanic_train_path)\ntitanic_test_raw = pd.read_csv(titanic_test_path)\n\ntitanic_train_y = titanic_train_raw.Transported\ntitanic_train_X = titanic_train_raw.drop(columns=\"Transported\")\ntitanic_train_X, titanic_test_X = preprocessing(titanic_train_X, titanic_test_raw)\ntitanic_train_X.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:31:37.929316Z","iopub.execute_input":"2022-07-14T15:31:37.930269Z","iopub.status.idle":"2022-07-14T15:31:46.146224Z","shell.execute_reply.started":"2022-07-14T15:31:37.930228Z","shell.execute_reply":"2022-07-14T15:31:46.145470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build Model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom tensorflow.keras import layers ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:33:01.464656Z","iopub.execute_input":"2022-07-14T15:33:01.465807Z","iopub.status.idle":"2022-07-14T15:33:01.470630Z","shell.execute_reply.started":"2022-07-14T15:33:01.465758Z","shell.execute_reply":"2022-07-14T15:33:01.469771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(titanic_train_X.copy(), titanic_train_y.copy())\nX_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:33:04.457707Z","iopub.execute_input":"2022-07-14T15:33:04.458037Z","iopub.status.idle":"2022-07-14T15:33:04.478496Z","shell.execute_reply.started":"2022-07-14T15:33:04.458010Z","shell.execute_reply":"2022-07-14T15:33:04.477772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Sequential([\n    layers.BatchNormalization(input_shape=[31]),\n    layers.Dropout(rate=0.2),\n    layers.Dense(units=1024, activation='swish'),\n    layers.BatchNormalization(),\n    layers.Dropout(rate=0.2),\n    layers.Dense(units=256, activation='swish'),\n    layers.BatchNormalization(),\n    layers.Dropout(rate=0.2),\n    layers.Dense(units=128, activation='swish'),\n    layers.BatchNormalization(),\n    layers.Dense(units=1,activation='sigmoid'),\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:42:43.560941Z","iopub.execute_input":"2022-07-14T15:42:43.561640Z","iopub.status.idle":"2022-07-14T15:42:43.671345Z","shell.execute_reply.started":"2022-07-14T15:42:43.561604Z","shell.execute_reply":"2022-07-14T15:42:43.670422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = keras.callbacks.EarlyStopping(\n    patience=10,\n    min_delta=0.01,\n    restore_best_weights=True,\n)\n\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_valid, y_valid),\n    batch_size=256,\n    epochs=200,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot()\nhistory_df.loc[:, ['binary_accuracy','val_binary_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:42:47.736079Z","iopub.execute_input":"2022-07-14T15:42:47.736760Z","iopub.status.idle":"2022-07-14T15:43:04.931206Z","shell.execute_reply.started":"2022-07-14T15:42:47.736724Z","shell.execute_reply":"2022-07-14T15:43:04.930499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#early stopping 35 epoch\n#validation accuracy 0.8040","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:49:15.492736Z","iopub.execute_input":"2022-07-14T14:49:15.493412Z","iopub.status.idle":"2022-07-14T14:49:15.498046Z","shell.execute_reply.started":"2022-07-14T14:49:15.493375Z","shell.execute_reply":"2022-07-14T14:49:15.497041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets fit train data too to improve accuracy\nmodel.fit(\n    titanic_train_X, titanic_train_y,\n    batch_size=512,\n    epochs=35,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:43:36.382358Z","iopub.execute_input":"2022-07-14T15:43:36.382757Z","iopub.status.idle":"2022-07-14T15:43:38.931898Z","shell.execute_reply.started":"2022-07-14T15:43:36.382722Z","shell.execute_reply":"2022-07-14T15:43:38.930291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y = model.predict(titanic_test_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:38:54.603015Z","iopub.execute_input":"2022-07-14T15:38:54.604001Z","iopub.status.idle":"2022-07-14T15:38:55.000351Z","shell.execute_reply.started":"2022-07-14T15:38:54.603962Z","shell.execute_reply":"2022-07-14T15:38:54.999573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(titanic_test_raw.PassengerId)\nsubmission['Transported'] = np.where(test_y > 0.5, True, False)\nsubmission.to_csv('submission_04.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:39:02.576826Z","iopub.execute_input":"2022-07-14T15:39:02.577199Z","iopub.status.idle":"2022-07-14T15:39:02.592212Z","shell.execute_reply.started":"2022-07-14T15:39:02.577169Z","shell.execute_reply":"2022-07-14T15:39:02.591380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:39:05.501331Z","iopub.execute_input":"2022-07-14T15:39:05.501720Z","iopub.status.idle":"2022-07-14T15:39:05.515189Z","shell.execute_reply.started":"2022-07-14T15:39:05.501685Z","shell.execute_reply":"2022-07-14T15:39:05.514399Z"},"trusted":true},"execution_count":null,"outputs":[]}]}