{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, OneHotEncoder, FunctionTransformer, LabelEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom xgboost import XGBClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T07:52:32.607879Z","iopub.execute_input":"2022-07-06T07:52:32.608298Z","iopub.status.idle":"2022-07-06T07:52:32.614958Z","shell.execute_reply.started":"2022-07-06T07:52:32.608264Z","shell.execute_reply":"2022-07-06T07:52:32.613981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"dataset = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv');\ndisplay(dataset)\ndataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:09:56.243335Z","iopub.execute_input":"2022-07-06T08:09:56.244023Z","iopub.status.idle":"2022-07-06T08:09:56.309436Z","shell.execute_reply.started":"2022-07-06T08:09:56.243985Z","shell.execute_reply":"2022-07-06T08:09:56.308241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data analysis","metadata":{}},{"cell_type":"code","source":"print(dataset.shape)\ndataset.dropna(subset=['Cabin'], inplace=True)\nprint('After NA removal', dataset.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:32.830595Z","iopub.execute_input":"2022-07-06T07:52:32.831475Z","iopub.status.idle":"2022-07-06T07:52:32.840782Z","shell.execute_reply.started":"2022-07-06T07:52:32.831434Z","shell.execute_reply":"2022-07-06T07:52:32.839651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['CryoSleep'].value_counts().plot.bar(ylabel='Effectif', xlabel='Cryo sleep')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:18:54.144951Z","iopub.execute_input":"2022-07-06T08:18:54.145364Z","iopub.status.idle":"2022-07-06T08:18:54.325952Z","shell.execute_reply.started":"2022-07-06T08:18:54.145332Z","shell.execute_reply":"2022-07-06T08:18:54.324134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['HomePlanet'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.076170Z","iopub.execute_input":"2022-07-06T07:52:33.076831Z","iopub.status.idle":"2022-07-06T07:52:33.235568Z","shell.execute_reply.started":"2022-07-06T07:52:33.076799Z","shell.execute_reply":"2022-07-06T07:52:33.234543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split cabin infos","metadata":{}},{"cell_type":"code","source":"dataset[['Deck', 'RoomNumber', 'Side']] = dataset['Cabin'].str.split('/', expand=True)\ndataset","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.237222Z","iopub.execute_input":"2022-07-06T07:52:33.237649Z","iopub.status.idle":"2022-07-06T07:52:33.285340Z","shell.execute_reply.started":"2022-07-06T07:52:33.237616Z","shell.execute_reply":"2022-07-06T07:52:33.284468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Side'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.286511Z","iopub.execute_input":"2022-07-06T07:52:33.286901Z","iopub.status.idle":"2022-07-06T07:52:33.440405Z","shell.execute_reply.started":"2022-07-06T07:52:33.286873Z","shell.execute_reply":"2022-07-06T07:52:33.439316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Deck'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.441981Z","iopub.execute_input":"2022-07-06T07:52:33.442505Z","iopub.status.idle":"2022-07-06T07:52:33.615675Z","shell.execute_reply.started":"2022-07-06T07:52:33.442467Z","shell.execute_reply":"2022-07-06T07:52:33.614960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Destination'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.616779Z","iopub.execute_input":"2022-07-06T07:52:33.617229Z","iopub.status.idle":"2022-07-06T07:52:33.783032Z","shell.execute_reply.started":"2022-07-06T07:52:33.617199Z","shell.execute_reply":"2022-07-06T07:52:33.782141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert room number to uint\ndataset['RoomNumber'] = pd.to_numeric(dataset['RoomNumber'], downcast='unsigned')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.784627Z","iopub.execute_input":"2022-07-06T07:52:33.784964Z","iopub.status.idle":"2022-07-06T07:52:33.796630Z","shell.execute_reply.started":"2022-07-06T07:52:33.784934Z","shell.execute_reply":"2022-07-06T07:52:33.795348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(dataset['RoomNumber'].describe())\ndataset['RoomNumber'].plot.hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:33.799440Z","iopub.execute_input":"2022-07-06T07:52:33.800202Z","iopub.status.idle":"2022-07-06T07:52:34.010822Z","shell.execute_reply.started":"2022-07-06T07:52:33.800164Z","shell.execute_reply":"2022-07-06T07:52:34.009675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['VIP'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.012187Z","iopub.execute_input":"2022-07-06T07:52:34.012556Z","iopub.status.idle":"2022-07-06T07:52:34.186199Z","shell.execute_reply.started":"2022-07-06T07:52:34.012524Z","shell.execute_reply":"2022-07-06T07:52:34.185232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(dataset['Age'].describe())\nplt.hist(dataset['Age'])\nplt.xlabel('Age')\nplt.ylabel('Number of passengers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.187318Z","iopub.execute_input":"2022-07-06T07:52:34.187732Z","iopub.status.idle":"2022-07-06T07:52:34.403964Z","shell.execute_reply.started":"2022-07-06T07:52:34.187702Z","shell.execute_reply":"2022-07-06T07:52:34.402975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bivariable analysis","metadata":{}},{"cell_type":"code","source":"def cat_plot(column):\n    fig, ax = plt.subplots()\n    value_counts_trans = dataset[dataset['Transported']][column].value_counts()\n    value_counts_nottrans = dataset[~dataset['Transported']][column].value_counts()\n    labels = value_counts_trans.index.values\n    x = np.arange(len(labels))\n\n    width = 0.35\n    ax.bar(x - width/2, value_counts_trans.values, width, label='Transported')\n    ax.bar(x + width/2, value_counts_nottrans.values, width, label='Not Transported')\n    ax.set_xlabel(column)\n    ax.set_ylabel('Number of passenger')\n    ax.set_xticks(x, labels)\n    ax.legend()\n    plt.show()\n    \ndef int_plot(column):\n    plt.hist([dataset[dataset['Transported']][column], dataset[~dataset['Transported']][column]], label=['Transported', 'Not Transported'])\n    plt.xlabel(column)\n    plt.ylabel('Number of passenger')\n    plt.legend()\n    plt.show()\n    sns.boxplot(x='Transported',y=column, data=dataset)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.405970Z","iopub.execute_input":"2022-07-06T07:52:34.406315Z","iopub.status.idle":"2022-07-06T07:52:34.415485Z","shell.execute_reply.started":"2022-07-06T07:52:34.406287Z","shell.execute_reply":"2022-07-06T07:52:34.414267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encodedSide = dataset['Side'].map({'P': 0, 'S': 1})\nencodedTransported = dataset['Transported'].map({False: 0, True: 1})\ndisplay(pd.concat([encodedSide, encodedTransported], axis=1).corr())\npd.crosstab(dataset['Side'], dataset['Transported'])\n\ncat_plot('Side')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.416962Z","iopub.execute_input":"2022-07-06T07:52:34.417531Z","iopub.status.idle":"2022-07-06T07:52:34.618217Z","shell.execute_reply.started":"2022-07-06T07:52:34.417480Z","shell.execute_reply":"2022-07-06T07:52:34.617231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A bit more of passengers on starboard were transported but the pearson correlation is very low.\n","metadata":{}},{"cell_type":"code","source":"encodedDeck = dataset['Deck'].map({'A': 0, 'B': 1, 'C': 2, 'D': 3, 'E': 4, 'F': 5, 'G': 6, 'T': 7})\ndisplay(pd.concat([encodedDeck, encodedTransported], axis=1).corr())\npd.crosstab(dataset['Deck'], dataset['Transported'])\n\ncat_plot('Deck')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.619564Z","iopub.execute_input":"2022-07-06T07:52:34.619882Z","iopub.status.idle":"2022-07-06T07:52:34.861897Z","shell.execute_reply.started":"2022-07-06T07:52:34.619854Z","shell.execute_reply":"2022-07-06T07:52:34.860825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([dataset['RoomNumber'], encodedTransported], axis=1).corr()\nint_plot('RoomNumber')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:34.863193Z","iopub.execute_input":"2022-07-06T07:52:34.863566Z","iopub.status.idle":"2022-07-06T07:52:35.272525Z","shell.execute_reply.started":"2022-07-06T07:52:34.863527Z","shell.execute_reply":"2022-07-06T07:52:35.271648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encodedVIP = dataset['VIP'].map({False: 0, True: 1})\ndisplay(pd.concat([encodedVIP, encodedTransported], axis=1).corr())\npd.crosstab(dataset['VIP'], dataset['Transported'])\n\ncat_plot('VIP')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:35.273541Z","iopub.execute_input":"2022-07-06T07:52:35.273998Z","iopub.status.idle":"2022-07-06T07:52:35.479464Z","shell.execute_reply.started":"2022-07-06T07:52:35.273965Z","shell.execute_reply":"2022-07-06T07:52:35.478368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.concat([dataset['Age'], encodedTransported], axis=1).corr())\nint_plot('Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:35.480919Z","iopub.execute_input":"2022-07-06T07:52:35.481287Z","iopub.status.idle":"2022-07-06T07:52:35.887678Z","shell.execute_reply.started":"2022-07-06T07:52:35.481256Z","shell.execute_reply":"2022-07-06T07:52:35.886653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['is_child'] = dataset['Age'] <= 4\ndisplay(pd.concat([dataset['is_child'].map({False: 0, True: 1}), encodedTransported], axis=1).corr())\ncat_plot('is_child')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:35.889871Z","iopub.execute_input":"2022-07-06T07:52:35.890210Z","iopub.status.idle":"2022-07-06T07:52:36.085170Z","shell.execute_reply.started":"2022-07-06T07:52:35.890181Z","shell.execute_reply":"2022-07-06T07:52:36.084411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encodedCryo = dataset['CryoSleep'].map({False: 0, True: 1})\ncat_plot('CryoSleep')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:36.086055Z","iopub.execute_input":"2022-07-06T07:52:36.086883Z","iopub.status.idle":"2022-07-06T07:52:36.444214Z","shell.execute_reply.started":"2022-07-06T07:52:36.086849Z","shell.execute_reply":"2022-07-06T07:52:36.443310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encodedHome = pd.get_dummies(dataset['HomePlanet'])\ncat_plot('HomePlanet')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:36.445361Z","iopub.execute_input":"2022-07-06T07:52:36.445679Z","iopub.status.idle":"2022-07-06T07:52:36.630153Z","shell.execute_reply.started":"2022-07-06T07:52:36.445649Z","shell.execute_reply":"2022-07-06T07:52:36.629146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encodedDestination = pd.get_dummies(dataset['Destination'])\ncat_plot('Destination')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:36.633107Z","iopub.execute_input":"2022-07-06T07:52:36.633491Z","iopub.status.idle":"2022-07-06T07:52:36.828480Z","shell.execute_reply.started":"2022-07-06T07:52:36.633458Z","shell.execute_reply":"2022-07-06T07:52:36.826445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(dataset['HomePlanet'], dataset['CryoSleep'])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:36.829637Z","iopub.execute_input":"2022-07-06T07:52:36.830050Z","iopub.status.idle":"2022-07-06T07:52:36.852334Z","shell.execute_reply.started":"2022-07-06T07:52:36.830021Z","shell.execute_reply":"2022-07-06T07:52:36.851269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']:\n    print('Analysing', column)\n    display(dataset[column].describe())\n    int_plot(column)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:36.853550Z","iopub.execute_input":"2022-07-06T07:52:36.853858Z","iopub.status.idle":"2022-07-06T07:52:38.866664Z","shell.execute_reply.started":"2022-07-06T07:52:36.853831Z","shell.execute_reply":"2022-07-06T07:52:38.865486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['TotalSpent'] = dataset[['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']].sum(axis=1)\nint_plot('TotalSpent')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:38.868171Z","iopub.execute_input":"2022-07-06T07:52:38.869295Z","iopub.status.idle":"2022-07-06T07:52:39.297140Z","shell.execute_reply.started":"2022-07-06T07:52:38.869244Z","shell.execute_reply":"2022-07-06T07:52:39.296109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get most corrolated values","metadata":{}},{"cell_type":"code","source":"correlationMatrix = pd.concat([dataset, encodedDeck, encodedSide, encodedVIP, encodedCryo, encodedHome, encodedDestination], axis=1).corr()\ndisplay(correlationMatrix)\nfig, ax = plt.subplots(figsize=(20, 20))\nsns.heatmap(correlationMatrix.abs(), annot=True, ax=ax)\nplt.show()\nprint('Most corrolated values to Transported')\nprint(correlationMatrix['Transported'].drop('Transported').abs().sort_values(ascending=False)[0:10])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:39.298590Z","iopub.execute_input":"2022-07-06T07:52:39.299025Z","iopub.status.idle":"2022-07-06T07:52:41.069969Z","shell.execute_reply.started":"2022-07-06T07:52:39.298980Z","shell.execute_reply":"2022-07-06T07:52:41.069024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model selection","metadata":{}},{"cell_type":"markdown","source":"## Load test data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:41.071448Z","iopub.execute_input":"2022-07-06T07:52:41.071960Z","iopub.status.idle":"2022-07-06T07:52:41.094918Z","shell.execute_reply.started":"2022-07-06T07:52:41.071919Z","shell.execute_reply":"2022-07-06T07:52:41.094168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Try models","metadata":{}},{"cell_type":"code","source":"models = []","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:06:19.509936Z","iopub.execute_input":"2022-07-06T08:06:19.510887Z","iopub.status.idle":"2022-07-06T08:06:19.514688Z","shell.execute_reply.started":"2022-07-06T08:06:19.510851Z","shell.execute_reply":"2022-07-06T08:06:19.513631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#selected_columns = ['PassengerId', 'CryoSleep', 'TotalSpent', 'HomePlanet', 'Cabin', 'Age', 'Destination']\nselected_columns = ['PassengerId', 'CryoSleep', 'RoomService', 'Spa', 'VRDeck', 'HomePlanet', 'Cabin', 'Age', 'Destination', 'FoodCourt', 'ShoppingMall']\nremoved_columns = ['Cabin', 'PassengerId', 'RoomNumber']\ncat_columns = ['CryoSleep', 'HomePlanet', 'Deck', 'Side', 'Destination', 'group_number']\nnum_columns = ['RoomService', 'Spa', 'VRDeck', 'Age', 'FoodCourt', 'ShoppingMall', 'TotalSpent']\n\ndef feature_engineering(dataset):\n    df = dataset.copy()\n    df[['Deck', 'RoomNumber', 'Side']] = df['Cabin'].str.split('/', expand=True)\n    df['group_number'] = df['PassengerId'].str.split('_', expand=True)[1]\n    df['CryoSleep'].fillna(~df[['RoomService', 'FoodCourt','ShoppingMall','Spa','VRDeck']].any(axis=1), inplace=True)\n    df['TotalSpent'] = df[['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']].sum(axis=1)\n    return df.drop(columns=removed_columns)\n            \n\ndef createPipeline(clf):\n    categorical_transformer = ColumnTransformer([\n        ('missing val', Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='constant', fill_value=\"unknown\")),\n            ('onehot with drop', OneHotEncoder(drop=['unknown'] * 2, handle_unknown='ignore'))\n        ]), ['HomePlanet', 'Destination'])\n    ], remainder=OneHotEncoder(handle_unknown='ignore'))\n    \n    numeric_transformer = Pipeline(steps=[\n        ('imputer', SimpleImputer(strategy='median')),\n        ('scaler', StandardScaler())\n    ])\n\n    preprocessor = ColumnTransformer([\n        ('num', numeric_transformer, num_columns),\n        ('cat', categorical_transformer, cat_columns)\n    ])\n\n    return Pipeline([\n        ('feature selection', FunctionTransformer(lambda x: x[selected_columns])),\n        ('feature engineering', FunctionTransformer(feature_engineering)),\n        ('preprocessor', preprocessor),\n        #('log', FunctionTransformer(lambda x: display(x))),\n        ('estimator', clf)\n    ])\n\nlabelEnc = LabelEncoder()\nlabelEnc.fit([True, False])\nX = dataset.drop(columns=['Transported'])#.dropna()\ny = labelEnc.transform(dataset.loc[X.index]['Transported'])\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:06:14.381033Z","iopub.execute_input":"2022-07-06T08:06:14.381465Z","iopub.status.idle":"2022-07-06T08:06:14.407941Z","shell.execute_reply.started":"2022-07-06T08:06:14.381431Z","shell.execute_reply":"2022-07-06T08:06:14.407035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"logreg = createPipeline(LogisticRegression(random_state=1337, max_iter=1e4))\nlogreg.fit(X_train, y_train)\nlogreg.score(X_test, y_test)\n\nparams = [dict(\n    estimator__penalty = ['l1'],\n    estimator__C = np.logspace(0, 2, 3),\n    estimator__solver = ['liblinear', 'saga'],\n), dict(\n    estimator__penalty = ['l2'],\n    estimator__C = np.logspace(0, 2, 3),\n    estimator__solver = ['newton-cg', 'lbfgs'],\n    \n), dict(\n    estimator__penalty = ['elasticnet'],\n    estimator__C = np.logspace(0, 2, 3),\n    estimator__solver = ['saga'],\n    estimator__l1_ratio = np.linspace(0,1, 5),\n)]\ncv = GridSearchCV(logreg, params)\ncv.fit(X_train, y_train)\nprint(cv.best_params_, cv.best_score_)\nmodels.append((cv, cv.best_score_))\n# {'estimator__C': 1.0, 'estimator__l1_ratio': 0.5, 'estimator__penalty': 'elasticnet', 'estimator__solver': 'saga'} 0.7878508356985179","metadata":{"execution":{"iopub.status.busy":"2022-07-06T07:52:41.149164Z","iopub.execute_input":"2022-07-06T07:52:41.149545Z","iopub.status.idle":"2022-07-06T08:00:40.925732Z","shell.execute_reply.started":"2022-07-06T07:52:41.149513Z","shell.execute_reply":"2022-07-06T08:00:40.924477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest","metadata":{}},{"cell_type":"code","source":"rf = createPipeline(RandomForestClassifier(random_state=42))\nrf.fit(X_train, y_train)\nparams = dict(\n    estimator__max_features = ['sqrt'], #['sqrt', 'log2'],\n    estimator__n_estimators = [120], #np.arange(100, 150, 10),\n    estimator__min_samples_split = [4], #np.arange(2, 5, 1),\n    estimator__criterion = ['gini'], #['gini', 'entropy'],\n)\nrfcv = GridSearchCV(rf, params)\nrfcv.fit(X_train, y_train)\nprint(rfcv.best_params_, rfcv.best_score_)\nmodels.append((rfcv, rfcv.best_score_))\n# {'estimator__criterion': 'gini', 'estimator__max_features': 'sqrt', 'estimator__min_samples_split': 4, 'estimator__n_estimators': 120} 0.7995855551733035","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:00:40.927258Z","iopub.execute_input":"2022-07-06T08:00:40.927898Z","iopub.status.idle":"2022-07-06T08:00:47.024243Z","shell.execute_reply.started":"2022-07-06T08:00:40.927851Z","shell.execute_reply":"2022-07-06T08:00:47.023318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gradient boosting","metadata":{}},{"cell_type":"code","source":"gb = createPipeline(GradientBoostingClassifier(random_state=42))\ngb.fit(X_train, y_train)\ngb.score(X_test, y_test)\n\nparams = dict(\n    estimator__learning_rate = [0.1], #np.logspace(-3, -1, 3),\n    estimator__n_estimators = [100], #np.arange(100, 150, 10),\n    estimator__max_depth = [5], #np.arange(1, 7, 2),\n    estimator__loss = ['exponential'], #['deviance', 'exponential'],\n)\ngbcv = GridSearchCV(gb, params)\ngbcv.fit(X_train, y_train)\nprint(gbcv.best_params_, gbcv.best_score_)\nmodels.append((gbcv, gbcv.best_score_))\n\n# {'estimator__learning_rate': 0.1, 'estimator__loss': 'exponential', 'estimator__max_depth': 5, 'estimator__n_estimators': 100} 0.8065879346922393","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:00:47.025400Z","iopub.execute_input":"2022-07-06T08:00:47.025730Z","iopub.status.idle":"2022-07-06T08:00:57.554423Z","shell.execute_reply.started":"2022-07-06T08:00:47.025698Z","shell.execute_reply":"2022-07-06T08:00:57.553377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg = createPipeline(XGBClassifier(tree_method='hist', objective='binary:logistic', use_label_encoder=False, enable_categorical=True, max_leaves=0))\nxg.fit(X_train, y_train)\n\nparams = dict(\n    estimator__learning_rate = [0.023999999999999997], #np.arange(0.0235, 0.0242, 0.0001)\n    estimator__n_estimators = [680], #np.arange(679, 681, 1), \n    estimator__max_depth = [4], #np.arange(4, 10, 2),\n    estimator__grow_policy = ['depthwise'],\n    estimator__colsample_bytree = [0.66], #np.arange(0.66, 0.67, 0.001),\n    estimator__gamma = [0.45], #np.arange(0.40, 0.5, 0.01),\n    estimator__min_child_weight = [0.39999999999999997], #np.arange(0.3, 0.5, 0.05),#[0.2781624502349287],\n    estimator__subsample = [0.5], #np.arange(0.4, 0.6, 0.05),#[0.9717120953891037], \n)\n#params = {'colsample_bytree': 0.6659223566174967,\n#          'gamma': 0.29564889385386356,\n#          'learning_rate': 0.027472179299006416,\n#          'max_depth': 4,\n#          'min_child_weight': 0.2781624502349287,\n#          'n_estimators': 712,\n#          'subsample': 0.9717120953891037}\nxgcv = GridSearchCV(xg, params)\nxgcv.fit(X_train, y_train)\nprint(xgcv.best_params_, xgcv.best_score_)\nmodels.append((xgcv, xgcv.best_score_))\n# {'estimator__colsample_bytree': 0.6659223566174967, 'estimator__gamma': 0.29564889385386356, 'estimator__grow_policy': 'depthwise', 'estimator__learning_rate': 0.027472179299006416, 'estimator__max_depth': 4, 'estimator__min_child_weight': 0.2781624502349287, 'estimator__n_estimators': 712, 'estimator__subsample': 0.9717120953891037, 'preprocessor__num__scaler': StandardScaler()} 0.8007217396290244","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:06:25.622091Z","iopub.execute_input":"2022-07-06T08:06:25.623113Z","iopub.status.idle":"2022-07-06T08:06:46.084005Z","shell.execute_reply.started":"2022-07-06T08:06:25.623048Z","shell.execute_reply":"2022-07-06T08:06:46.082954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SVM","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nsvm = createPipeline(SVC(random_state=42))\nsvm.fit(X_train, y_train)\n\nparams = dict(\n    estimator__C = np.logspace(0, 2, 3),\n    estimator__kernel = ['rbf'], #['linear', 'poly', 'rbf', 'sigmoid'],\n)\nsvmcv = GridSearchCV(svm, params)\nsvmcv.fit(X_train, y_train)\nprint(svmcv.best_params_, svmcv.best_score_)\nmodels.append((svmcv, svmcv.best_score_))\n\n# {'estimator__C': 1.0, 'estimator__kernel': 'rbf'} 0.8016670967002092","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:01:18.242618Z","iopub.execute_input":"2022-07-06T08:01:18.243589Z","iopub.status.idle":"2022-07-06T08:01:56.433492Z","shell.execute_reply.started":"2022-07-06T08:01:18.243548Z","shell.execute_reply":"2022-07-06T08:01:56.432691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SGD","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier\nsgd = createPipeline(SGDClassifier(random_state=42))\nsgd.fit(X_train, y_train)\n\nparams = [dict(\n    estimator__penalty = ['l1'],\n    estimator__alpha = np.logspace(-5, -3, 3),\n), dict(\n    estimator__penalty = ['l2'],\n    \n), dict(\n    estimator__penalty = ['elasticnet'],\n    estimator__alpha = np.logspace(-5, -3, 3),\n    estimator__l1_ratio = np.arange(0.1, 0.9, 0.2),\n)]\nsgdcv = GridSearchCV(sgd, params)\nsgdcv.fit(X_train, y_train)\nprint(sgdcv.best_params_, sgdcv.best_score_)\nmodels.append((sgdcv, sgdcv.best_score_))\n\n# {'estimator__alpha': 0.0001, 'estimator__l1_ratio': 0.7000000000000001, 'estimator__penalty': 'elasticnet'} 0.7961809437802816","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:01:56.434604Z","iopub.execute_input":"2022-07-06T08:01:56.435082Z","iopub.status.idle":"2022-07-06T08:02:15.459461Z","shell.execute_reply.started":"2022-07-06T08:01:56.435037Z","shell.execute_reply":"2022-07-06T08:02:15.458402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nknn = createPipeline(KNeighborsClassifier())\nknn.fit(X_train, y_train)\nprint(knn.score(X_test, y_test))\n\nparams = dict(\n    estimator__n_neighbors = np.linspace(3, 9, 4, dtype='int32'),\n)\nknncv = GridSearchCV(knn, params)\nknncv.fit(X_train, y_train)\nprint(knncv.best_params_, knncv.best_score_)\nmodels.append((knncv, knncv.best_score_))\n# {'estimator__n_neighbors': 7} 0.7810410753705456","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:15.460673Z","iopub.execute_input":"2022-07-06T08:02:15.461000Z","iopub.status.idle":"2022-07-06T08:02:22.764350Z","shell.execute_reply.started":"2022-07-06T08:02:15.460972Z","shell.execute_reply":"2022-07-06T08:02:22.763222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Naives Bayes","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\ngnb = createPipeline(GaussianNB())\ngnb.fit(X_train, y_train)\nprint(gnb.score(X_test, y_test))\n\nparams = dict(\n    estimator__var_smoothing = np.logspace(-8, -10, 3),\n)\ngnb = GridSearchCV(gnb, params)\ngnb.fit(X_train, y_train)\nprint(gnb.best_params_, gnb.best_score_)\nmodels.append((gnb, gnb.best_score_))\n# {'estimator__var_smoothing': 1e-08} 0.7132885740087727","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:22.765813Z","iopub.execute_input":"2022-07-06T08:02:22.766135Z","iopub.status.idle":"2022-07-06T08:02:24.600458Z","shell.execute_reply.started":"2022-07-06T08:02:22.766107Z","shell.execute_reply":"2022-07-06T08:02:24.599455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyse false prediction","metadata":{}},{"cell_type":"code","source":"pred = svmcv.predict(X)\ndf = dataset.loc[X.index]\nwrong_pred = df[df['Transported'] != pred]\nwrong_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:24.601892Z","iopub.execute_input":"2022-07-06T08:02:24.602259Z","iopub.status.idle":"2022-07-06T08:02:26.637891Z","shell.execute_reply.started":"2022-07-06T08:02:24.602228Z","shell.execute_reply":"2022-07-06T08:02:26.636901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1,2, sharey=True)\nfig.suptitle('Number in group')\nfor id, d in enumerate([wrong_pred, dataset]):\n    id_num = d['PassengerId'].str.split('_', expand=True)[1]\n    id_num.astype('int32').plot.hist(ax=ax1 if id == 0 else ax2, title='wrong' if id == 0 else 'total')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:26.639490Z","iopub.execute_input":"2022-07-06T08:02:26.639848Z","iopub.status.idle":"2022-07-06T08:02:26.908161Z","shell.execute_reply.started":"2022-07-06T08:02:26.639819Z","shell.execute_reply":"2022-07-06T08:02:26.907152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO compare with true predictions\nfor col in ['CryoSleep', 'HomePlanet', 'is_child', 'Deck', 'Side', 'VIP']:\n    wvc = wrong_pred[col].value_counts()\n    dvc = dataset[col].value_counts()\n    fig, (ax1, ax2) = plt.subplots(1,2, sharey=True, figsize=(10,6))\n    fig.suptitle(col)\n    wvc.plot.bar(ax=ax1, title='wrong')\n    dvc.plot.bar(ax=ax2, title='total')\n    plt.show()\n    for idx in wvc.index:\n        print(f'{idx} in wrong: {wvc[idx] / wrong_pred.shape[0]}')\n        print(f'{idx} in dataset: {dvc[idx] / dataset.shape[0]}')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:26.909554Z","iopub.execute_input":"2022-07-06T08:02:26.909936Z","iopub.status.idle":"2022-07-06T08:02:28.356463Z","shell.execute_reply.started":"2022-07-06T08:02:26.909906Z","shell.execute_reply":"2022-07-06T08:02:28.355742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['RoomService', 'Spa', 'VRDeck']:\n    fig, (ax1, ax2) = plt.subplots(1,2, sharey=True, figsize=(10,6))\n    fig.suptitle(col)\n    wrong_pred[col].plot.hist(ax=ax1, title='wrong')\n    dataset[col].plot.hist(ax=ax2, title='total')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:02:28.357657Z","iopub.execute_input":"2022-07-06T08:02:28.358178Z","iopub.status.idle":"2022-07-06T08:02:29.524518Z","shell.execute_reply.started":"2022-07-06T08:02:28.358142Z","shell.execute_reply":"2022-07-06T08:02:29.523536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create submission","metadata":{}},{"cell_type":"code","source":"best_model = sorted(models, key=lambda x: x[1], reverse=True)[0]\nbest_model","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:06:46.085716Z","iopub.execute_input":"2022-07-06T08:06:46.086052Z","iopub.status.idle":"2022-07-06T08:06:46.233055Z","shell.execute_reply.started":"2022-07-06T08:06:46.086023Z","shell.execute_reply":"2022-07-06T08:06:46.232034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:04:56.204664Z","iopub.execute_input":"2022-07-06T08:04:56.205032Z","iopub.status.idle":"2022-07-06T08:04:56.221529Z","shell.execute_reply.started":"2022-07-06T08:04:56.205003Z","shell.execute_reply":"2022-07-06T08:04:56.220608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = labelEnc.inverse_transform(best_model[0].predict(test))\nsubmission = pd.DataFrame({'PassengerId': test['PassengerId'], 'Transported': pred})\nsubmission.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T08:06:46.234342Z","iopub.execute_input":"2022-07-06T08:06:46.234629Z","iopub.status.idle":"2022-07-06T08:06:46.322016Z","shell.execute_reply.started":"2022-07-06T08:06:46.234603Z","shell.execute_reply":"2022-07-06T08:06:46.321247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TODO\n\nPour chaque algo, regarder les lignes mal prédites et rechercher des particularités communes à tous les algos.\n\nEssayer d'autres méthodes d'imputation que la médiane pour le meilleur algo. Vérifier par colonne, peut-être que certaines seront mieux jetées.\n\nPour les variables catégorielles, regarder s'il y a assez d'individus par classe.\n\n## Pistes\nOn peut essayer une stratégie d'oversampling qui consiste à dupliquer les lignes comportants une variable mal prédite pour mieux entrainer le modèle dessus.","metadata":{}}]}