{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Fait par: \n- Florent Mullet\n- Leo Heummel\n- Romain Martinez\n- William Jacquot\n- Aime Marcant","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport time\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport plotly.express as px\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.model_selection import train_test_split, GridSearchCV, RandomizedSearchCV, StratifiedKFold\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OrdinalEncoder, MinMaxScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.feature_selection import mutual_info_regression\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, callbacks\n\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T17:51:15.430535Z","iopub.execute_input":"2022-07-18T17:51:15.431016Z","iopub.status.idle":"2022-07-18T17:51:28.486328Z","shell.execute_reply.started":"2022-07-18T17:51:15.430915Z","shell.execute_reply":"2022-07-18T17:51:28.484995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Importez la donnee avec pandas","metadata":{}},{"cell_type":"markdown","source":"**Load data**","metadata":{}},{"cell_type":"code","source":"# Save to df\ntrain = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest = pd.read_csv('../input/spaceship-titanic/test.csv')\n\n# Shape and preview\nprint('Train set shape:', train.shape)\nprint('Test set shape:', test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:28.489046Z","iopub.execute_input":"2022-07-18T17:51:28.490218Z","iopub.status.idle":"2022-07-18T17:51:28.578920Z","shell.execute_reply.started":"2022-07-18T17:51:28.490165Z","shell.execute_reply":"2022-07-18T17:51:28.578005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Verifiez les valeurs manquantes","metadata":{}},{"cell_type":"code","source":"print('TRAIN SET MISSING VALUES:')\nprint(train.isna().sum())\nprint('')\nprint('TEST SET MISSING VALUES:')\nprint(test.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:28.580410Z","iopub.execute_input":"2022-07-18T17:51:28.581002Z","iopub.status.idle":"2022-07-18T17:51:28.606795Z","shell.execute_reply.started":"2022-07-18T17:51:28.580962Z","shell.execute_reply":"2022-07-18T17:51:28.604351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.heatmap(train.isna().T, cmap='summer')\nplt.title('Heatmap of missing values')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:28.610164Z","iopub.execute_input":"2022-07-18T17:51:28.611168Z","iopub.status.idle":"2022-07-18T17:51:29.882255Z","shell.execute_reply.started":"2022-07-18T17:51:28.611124Z","shell.execute_reply":"2022-07-18T17:51:29.880654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['na_count']=train.isna().sum(axis=1)\nplt.figure(figsize=(10,4))\nsns.countplot(data=train, x='na_count', hue='Transported')\nplt.title('Number of missing entries by passenger')\ntrain.drop('na_count', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:29.883973Z","iopub.execute_input":"2022-07-18T17:51:29.884358Z","iopub.status.idle":"2022-07-18T17:51:30.137235Z","shell.execute_reply.started":"2022-07-18T17:51:29.884322Z","shell.execute_reply":"2022-07-18T17:51:30.136000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This shows that missing values don't favour either outcome in the target and for the most part are isolated. This means it is reasonable to smartly 'guess' alternatives to the missing values as opposed to dropping these passengers entirely and losing a lot of training data.","metadata":{}},{"cell_type":"markdown","source":"## 3. Remplir les valeurs manquantes","metadata":{}},{"cell_type":"markdown","source":"### a. Fusionner les datas","metadata":{}},{"cell_type":"code","source":"# Labels and features\ny=train['Transported'].copy().astype(int)\nX=train.drop('Transported', axis=1).copy()\n\n#combine train and test data for data preprocessing\ndf_merge=pd.concat([test.assign(ind=\"test\"), train.assign(ind=\"train\")])\ndf_merge.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.138760Z","iopub.execute_input":"2022-07-18T17:51:30.139223Z","iopub.status.idle":"2022-07-18T17:51:30.185700Z","shell.execute_reply.started":"2022-07-18T17:51:30.139189Z","shell.execute_reply":"2022-07-18T17:51:30.184427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### a. Valeurs continues","metadata":{}},{"cell_type":"code","source":"# Impute median (for continuous data)\ndf_merge['Age'].fillna(train['Age'].median(), inplace=True) # be careful of data leakage","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.189455Z","iopub.execute_input":"2022-07-18T17:51:30.190116Z","iopub.status.idle":"2022-07-18T17:51:30.201530Z","shell.execute_reply.started":"2022-07-18T17:51:30.190076Z","shell.execute_reply":"2022-07-18T17:51:30.200583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Valeurs categorielles","metadata":{}},{"cell_type":"code","source":"# Find mode of each categorical feature\ndf_merge[['HomePlanet','CryoSleep','Destination','VIP']].mode()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.203136Z","iopub.execute_input":"2022-07-18T17:51:30.203567Z","iopub.status.idle":"2022-07-18T17:51:30.233031Z","shell.execute_reply.started":"2022-07-18T17:51:30.203532Z","shell.execute_reply":"2022-07-18T17:51:30.231816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Impute most frequent category (for categorical data)\ndf_merge['HomePlanet'].fillna('Earth', inplace=True)\n\ndf_merge['CryoSleep'].fillna(False, inplace=True)\n\ndf_merge['Destination'].fillna('TRAPPIST-1e', inplace=True)\n\ndf_merge['VIP'].fillna(False, inplace=True)\n\n\nexp_feats=['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n\n# Impute 0's (mode)\nfor col in exp_feats:\n    df_merge.loc[df_merge[col].isna(),col]=0","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.235895Z","iopub.execute_input":"2022-07-18T17:51:30.236643Z","iopub.status.idle":"2022-07-18T17:51:30.261477Z","shell.execute_reply.started":"2022-07-18T17:51:30.236573Z","shell.execute_reply":"2022-07-18T17:51:30.260573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### c. Valeurs qualitatives","metadata":{}},{"cell_type":"code","source":"# Impute outliers (for qualitative data)\ndf_merge['Cabin'].fillna('Z/9999/Z', inplace=True)\n\ndf_merge['Name'].fillna('No Name', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.264980Z","iopub.execute_input":"2022-07-18T17:51:30.265816Z","iopub.status.idle":"2022-07-18T17:51:30.275043Z","shell.execute_reply.started":"2022-07-18T17:51:30.265779Z","shell.execute_reply":"2022-07-18T17:51:30.273934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Affichez les statistiques avec .describe()","metadata":{}},{"cell_type":"code","source":"df_merge.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.276755Z","iopub.execute_input":"2022-07-18T17:51:30.277143Z","iopub.status.idle":"2022-07-18T17:51:30.315784Z","shell.execute_reply.started":"2022-07-18T17:51:30.277112Z","shell.execute_reply":"2022-07-18T17:51:30.314652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Utilisez un StandardScaler sur variables numeriques","metadata":{}},{"cell_type":"markdown","source":"Mettre image avec les scalers","metadata":{}},{"cell_type":"code","source":"# Train and test\ntrain=df_merge[df_merge['PassengerId'].isin(train['PassengerId'].values)].copy()\nX = train.drop(labels=['Transported', 'ind'], axis=1)\ny = pd.DataFrame(train['Transported'])\nX_test=df_merge[df_merge['PassengerId'].isin(test['PassengerId'].values)].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.317276Z","iopub.execute_input":"2022-07-18T17:51:30.317650Z","iopub.status.idle":"2022-07-18T17:51:30.334481Z","shell.execute_reply.started":"2022-07-18T17:51:30.317588Z","shell.execute_reply":"2022-07-18T17:51:30.333246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Utilisez un LabelEncoder ou OneHotEncoder sur des variables categorielles si necessaire","metadata":{}},{"cell_type":"markdown","source":"label encoder et one hot image","metadata":{}},{"cell_type":"code","source":"num_pipeline = Pipeline([\n        (\"imputer\", SimpleImputer(strategy=\"median\")),\n        (\"scaler\", StandardScaler())\n    ])\n\ncat_pipeline = Pipeline([\n        (\"imputer\", SimpleImputer(strategy=\"most_frequent\")),\n        (\"cat_encoder\", OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)), \n        (\"scaler\", StandardScaler()), \n    ])\n\nfrom sklearn.compose import ColumnTransformer\n\nnum_attribs = list(X.select_dtypes(exclude='object').columns)\ncat_attribs = list(X.select_dtypes(include='object').columns)\n\npreprocess_pipeline = ColumnTransformer([\n        (\"num\", num_pipeline, num_attribs),\n        (\"cat\", cat_pipeline, cat_attribs),\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.336446Z","iopub.execute_input":"2022-07-18T17:51:30.337137Z","iopub.status.idle":"2022-07-18T17:51:30.350711Z","shell.execute_reply.started":"2022-07-18T17:51:30.337100Z","shell.execute_reply":"2022-07-18T17:51:30.349663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_prepared = pd.DataFrame(preprocess_pipeline.fit_transform(X[num_attribs + cat_attribs]), columns=X.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.352540Z","iopub.execute_input":"2022-07-18T17:51:30.353753Z","iopub.status.idle":"2022-07-18T17:51:30.560644Z","shell.execute_reply.started":"2022-07-18T17:51:30.353710Z","shell.execute_reply":"2022-07-18T17:51:30.559353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder = LabelEncoder()\ny_LE = label_encoder.fit_transform(y)\ny_LE = pd.DataFrame(y_LE, columns=y.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.562072Z","iopub.execute_input":"2022-07-18T17:51:30.562404Z","iopub.status.idle":"2022-07-18T17:51:30.575501Z","shell.execute_reply.started":"2022-07-18T17:51:30.562373Z","shell.execute_reply":"2022-07-18T17:51:30.574491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**On regarde les relations entre les features et les valeurs visee**","metadata":{}},{"cell_type":"code","source":"def make_mi_scores(X, y):\n    mi_scores = mutual_info_regression(X, y)\n    mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=X.columns)\n    mi_scores = mi_scores.sort_values(ascending=False)\n    return mi_scores\n\nmi = pd.DataFrame(make_mi_scores(X_prepared, y_LE))\ncorr = pd.DataFrame(X_prepared[X_prepared.columns].corrwith(y_LE['Transported']),\n                    columns=['Correlation'])\ns_corr = pd.DataFrame(X_prepared[X_prepared.columns].corrwith(y_LE['Transported'],\n                      method='spearman'), columns=['Spearman_Correlation'])\n\nrelation = mi.join(corr)\nrelation = relation.join(s_corr)\nrelation","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:30.576901Z","iopub.execute_input":"2022-07-18T17:51:30.577833Z","iopub.status.idle":"2022-07-18T17:51:31.339078Z","shell.execute_reply.started":"2022-07-18T17:51:30.577794Z","shell.execute_reply":"2022-07-18T17:51:31.338175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X_prepared, y_LE, train_size=0.8,\n                                                      test_size=0.2, random_state=42, stratify = y)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.340385Z","iopub.execute_input":"2022-07-18T17:51:31.340877Z","iopub.status.idle":"2022-07-18T17:51:31.397270Z","shell.execute_reply.started":"2022-07-18T17:51:31.340844Z","shell.execute_reply":"2022-07-18T17:51:31.396087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"**1 neurones**","metadata":{}},{"cell_type":"code","source":"model1 = keras.Sequential([\n     layers.Dense(units=1, input_shape=[13])\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.398751Z","iopub.execute_input":"2022-07-18T17:51:31.399103Z","iopub.status.idle":"2022-07-18T17:51:31.517041Z","shell.execute_reply.started":"2022-07-18T17:51:31.399069Z","shell.execute_reply":"2022-07-18T17:51:31.516080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1 couche de 512 neurones**","metadata":{}},{"cell_type":"code","source":"model2 = keras.Sequential([\n     layers.Dense(units=512, input_shape=[13])\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.518397Z","iopub.execute_input":"2022-07-18T17:51:31.519059Z","iopub.status.idle":"2022-07-18T17:51:31.537130Z","shell.execute_reply.started":"2022-07-18T17:51:31.519018Z","shell.execute_reply":"2022-07-18T17:51:31.536027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**3 couches, 512, 256 et 128**","metadata":{}},{"cell_type":"code","source":"model3 = keras.Sequential([\n    layers.Dense(units=512, activation='relu', input_shape=[13]),\n    layers.Dense(units=256, activation='relu'),\n    layers.Dense(units=128)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.538939Z","iopub.execute_input":"2022-07-18T17:51:31.539552Z","iopub.status.idle":"2022-07-18T17:51:31.578303Z","shell.execute_reply.started":"2022-07-18T17:51:31.539516Z","shell.execute_reply":"2022-07-18T17:51:31.577215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Avec Dropout**","metadata":{}},{"cell_type":"code","source":"model4 = keras.Sequential([\n    layers.Dense(units=512, activation='relu', input_shape=[13]),\n    layers.Dropout(0.3),\n    layers.Dense(units=256, activation='relu'),\n    layers.Dropout(0.3),\n    layers.Dense(units=128),\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.579717Z","iopub.execute_input":"2022-07-18T17:51:31.580073Z","iopub.status.idle":"2022-07-18T17:51:31.628584Z","shell.execute_reply.started":"2022-07-18T17:51:31.580040Z","shell.execute_reply":"2022-07-18T17:51:31.627695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Avec un BatchNormalization**","metadata":{}},{"cell_type":"code","source":"model4 = keras.Sequential([\n    layers.BatchNormalization(input_shape=[13]),\n    layers.Dense(units=512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(units=256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(units=128),\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.629999Z","iopub.execute_input":"2022-07-18T17:51:31.630598Z","iopub.status.idle":"2022-07-18T17:51:31.936798Z","shell.execute_reply.started":"2022-07-18T17:51:31.630563Z","shell.execute_reply":"2022-07-18T17:51:31.935671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**On a choisi cette composition de layers**","metadata":{}},{"cell_type":"code","source":"model = keras.Sequential([\n    layers.BatchNormalization(input_shape=[13]),\n    layers.Dense(512, activation='relu', input_shape=[13]),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(1, activation='sigmoid'),\n])\n\n\nmodel.compile(optimizer='adam',\n              loss='binary_crossentropy',\n              metrics=['binary_accuracy'])\n\n\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=10,\n    min_delta=0.01,\n    restore_best_weights=True,\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:31.938254Z","iopub.execute_input":"2022-07-18T17:51:31.938561Z","iopub.status.idle":"2022-07-18T17:51:32.117408Z","shell.execute_reply.started":"2022-07-18T17:51:31.938533Z","shell.execute_reply":"2022-07-18T17:51:32.116139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    validation_data=(X_valid, y_valid),\n    #batch_size=512,\n    epochs=1000,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy', 'val_binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:51:32.118990Z","iopub.execute_input":"2022-07-18T17:51:32.119744Z","iopub.status.idle":"2022-07-18T17:52:45.330594Z","shell.execute_reply.started":"2022-07-18T17:51:32.119692Z","shell.execute_reply":"2022-07-18T17:52:45.329697Z"},"trusted":true},"execution_count":null,"outputs":[]}]}