{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T16:43:00.192154Z","iopub.execute_input":"2022-07-18T16:43:00.192589Z","iopub.status.idle":"2022-07-18T16:43:00.735452Z","shell.execute_reply.started":"2022-07-18T16:43:00.192553Z","shell.execute_reply":"2022-07-18T16:43:00.734237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_data_train = '../input/spaceship-titanic/train.csv'\n#file_data_test = '../input/spaceship-titanic/test.csv'\ndata_train = pd.read_csv(file_data_train)\n#data_test = pd.read_csv(file_data_test)\nprint (data_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:02.252518Z","iopub.execute_input":"2022-07-18T16:43:02.253476Z","iopub.status.idle":"2022-07-18T16:43:02.345577Z","shell.execute_reply.started":"2022-07-18T16:43:02.253438Z","shell.execute_reply":"2022-07-18T16:43:02.344217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"execution":{"iopub.status.busy":"2022-04-27T12:22:50.567887Z","iopub.execute_input":"2022-04-27T12:22:50.568223Z","iopub.status.idle":"2022-04-27T12:22:50.616673Z","shell.execute_reply.started":"2022-04-27T12:22:50.568184Z","shell.execute_reply":"2022-04-27T12:22:50.615838Z"}}},{"cell_type":"code","source":"missing_values_count_train = data_train.isnull().sum()\nmissing_values_count_train[0:19]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:05.084076Z","iopub.execute_input":"2022-07-18T16:43:05.084623Z","iopub.status.idle":"2022-07-18T16:43:05.109791Z","shell.execute_reply.started":"2022-07-18T16:43:05.084574Z","shell.execute_reply":"2022-07-18T16:43:05.108453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Nettoyage des données catégorielles\nfrom sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(missing_values=np.nan, strategy='most_frequent')\nimputer.fit_transform(data_train[['HomePlanet','CryoSleep','Cabin','Destination','VIP','Name']])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:06.831279Z","iopub.execute_input":"2022-07-18T16:43:06.831687Z","iopub.status.idle":"2022-07-18T16:43:07.046163Z","shell.execute_reply.started":"2022-07-18T16:43:06.831656Z","shell.execute_reply":"2022-07-18T16:43:07.044742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Nettoyage des données numériques\ndata_train['RoomService'] = data_train['RoomService'].fillna(data_train['RoomService'].median())\ndata_train['Age'] = data_train['Age'].fillna(data_train['Age'].median())\ndata_train['FoodCourt'] = data_train['FoodCourt'].fillna(data_train['FoodCourt'].median())\ndata_train['ShoppingMall'] = data_train['ShoppingMall'].fillna(data_train['ShoppingMall'].median())\ndata_train['Spa'] = data_train['Spa'].fillna(data_train['Spa'].median())\ndata_train['VRDeck'] = data_train['VRDeck'].fillna(data_train['VRDeck'].median())\nprint(data_train.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:08.638855Z","iopub.execute_input":"2022-07-18T16:43:08.639485Z","iopub.status.idle":"2022-07-18T16:43:08.675512Z","shell.execute_reply.started":"2022-07-18T16:43:08.639434Z","shell.execute_reply":"2022-07-18T16:43:08.674058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#utilisation du discribes pour afficher les stats\ndataframe1 = data_train.describe()\nprint(\"Statistics are: \\n\")\nprint(dataframe1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:10.796885Z","iopub.execute_input":"2022-07-18T16:43:10.797266Z","iopub.status.idle":"2022-07-18T16:43:10.842960Z","shell.execute_reply.started":"2022-07-18T16:43:10.797235Z","shell.execute_reply":"2022-07-18T16:43:10.841788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Application du StandardScaler\nfrom sklearn.preprocessing import StandardScaler\n\n# create a scaler object\nstd_scaler = StandardScaler()\ndata_sample = data_train[['Age','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']]\n# fit and transform the data\ndf_std = pd.DataFrame(std_scaler.fit_transform(data_sample), columns=data_sample.columns)\n\nprint(df_std)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:13.124110Z","iopub.execute_input":"2022-07-18T16:43:13.124601Z","iopub.status.idle":"2022-07-18T16:43:13.143254Z","shell.execute_reply.started":"2022-07-18T16:43:13.124542Z","shell.execute_reply":"2022-07-18T16:43:13.141860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Passage en variable binaire des features cléfs Homeplanet et destination\nfeatures=[\"HomePlanet\",\"Destination\"]\n\nX=pd.get_dummies(data_train[features])\n\nX.head()\n\nfinal=data_train.join(X)\n\nfinal","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:16.732986Z","iopub.execute_input":"2022-07-18T16:43:16.733371Z","iopub.status.idle":"2022-07-18T16:43:16.792204Z","shell.execute_reply.started":"2022-07-18T16:43:16.733340Z","shell.execute_reply":"2022-07-18T16:43:16.791343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Nous allons utiliser la metrique RMSLE car la conséquences de l'erreur est la vie humaine en ce sens il faut minimiser le risque d'avoir un faux-négatif.","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:59:50.982286Z","iopub.execute_input":"2022-07-18T15:59:50.983648Z","iopub.status.idle":"2022-07-18T15:59:50.988807Z","shell.execute_reply.started":"2022-07-18T15:59:50.983608Z","shell.execute_reply":"2022-07-18T15:59:50.987632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#définition variable d'entrainement\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(df_std, data_train['Transported'], test_size=0.33, random_state=42)\ninput_shape = [X_train.shape[1]]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:33.174001Z","iopub.execute_input":"2022-07-18T16:43:33.174365Z","iopub.status.idle":"2022-07-18T16:43:33.183014Z","shell.execute_reply.started":"2022-07-18T16:43:33.174334Z","shell.execute_reply":"2022-07-18T16:43:33.182248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Couche 1 neurone\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\ndata_train.shape\n\nmodel = keras.Sequential([\n    layers.Dense(units=1, input_shape=[1])\n])\nmodel.weights\nx = tf.linspace(-1.0, 1.0, 100)\ny = model.predict(x)\n\nplt.figure(dpi=100)\nplt.plot(x, y, 'k')\nplt.xlim(-1, 1)\nplt.ylim(-1, 1)\nplt.xlabel(\"Input: x\")\nplt.ylabel(\"Target y\")\nw, b = model.weights # you could also use model.get_weights() here\nplt.title(\"Weight: {:0.2f}\\nBias: {:0.2f}\".format(w[0][0], b[0]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:36.411924Z","iopub.execute_input":"2022-07-18T16:43:36.412344Z","iopub.status.idle":"2022-07-18T16:43:37.121701Z","shell.execute_reply.started":"2022-07-18T16:43:36.412295Z","shell.execute_reply":"2022-07-18T16:43:37.120524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# couche 512 neurones\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nmodel = keras.Sequential([\n    layers.BatchNormalization(input_shape=input_shape),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(), # Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(), # Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(1, activation='sigmoid'),\n])\n\n#compilation\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\n#start\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=5,\n    min_delta=0.001,\n    restore_best_weights=True,\n)\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=200,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy', 'val_binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:36:55.277653Z","iopub.execute_input":"2022-07-18T16:36:55.278581Z","iopub.status.idle":"2022-07-18T16:37:21.080111Z","shell.execute_reply.started":"2022-07-18T16:36:55.278538Z","shell.execute_reply":"2022-07-18T16:37:21.078635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# couche 256 neurones\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nmodel = keras.Sequential([\n    layers.BatchNormalization(input_shape=input_shape),\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(), # Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(), # Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(1, activation='sigmoid'),\n])\n\n#compilation\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\n#start\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=5,\n    min_delta=0.001,\n    restore_best_weights=True,\n)\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=200,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy', 'val_binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:43:45.314611Z","iopub.execute_input":"2022-07-18T16:43:45.315166Z","iopub.status.idle":"2022-07-18T16:43:48.825567Z","shell.execute_reply.started":"2022-07-18T16:43:45.315116Z","shell.execute_reply":"2022-07-18T16:43:48.824751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# couche 128 neurones\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nmodel = keras.Sequential([\n    layers.BatchNormalization(input_shape=input_shape),\n    layers.Dense(128, activation='relu'),\n    layers.BatchNormalization(),# Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(128, activation='relu'),\n    layers.BatchNormalization(),# Ajout du BatchNormalization\n    layers.Dropout(0.3), # Ajout du Dropout\n    layers.Dense(1, activation='sigmoid'),\n])\n\n#compilation\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\n#start\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=5,\n    min_delta=0.001,\n    restore_best_weights=True,\n)\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=200,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy', 'val_binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T16:45:08.460913Z","iopub.execute_input":"2022-07-18T16:45:08.461332Z","iopub.status.idle":"2022-07-18T16:45:18.329034Z","shell.execute_reply.started":"2022-07-18T16:45:08.461280Z","shell.execute_reply":"2022-07-18T16:45:18.328224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Donnée Catégorielle\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:59:51.258355Z","iopub.execute_input":"2022-07-18T15:59:51.259360Z","iopub.status.idle":"2022-07-18T15:59:51.263779Z","shell.execute_reply.started":"2022-07-18T15:59:51.259324Z","shell.execute_reply":"2022-07-18T15:59:51.262496Z"},"trusted":true},"execution_count":null,"outputs":[]}]}