{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Titanic Machine Learning from Disaster","metadata":{}},{"cell_type":"code","source":"# Imports \nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.346404Z","iopub.execute_input":"2022-08-02T15:51:47.347048Z","iopub.status.idle":"2022-08-02T15:51:47.375080Z","shell.execute_reply.started":"2022-08-02T15:51:47.346920Z","shell.execute_reply":"2022-08-02T15:51:47.374137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data\ntrain_filename = '../input/titanic/train.csv'\ntrain_df = pd.read_csv(train_filename)\nprint(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.376787Z","iopub.execute_input":"2022-08-02T15:51:47.377306Z","iopub.status.idle":"2022-08-02T15:51:47.408585Z","shell.execute_reply.started":"2022-08-02T15:51:47.377273Z","shell.execute_reply":"2022-08-02T15:51:47.407690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of data input\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.409715Z","iopub.execute_input":"2022-08-02T15:51:47.410669Z","iopub.status.idle":"2022-08-02T15:51:47.415615Z","shell.execute_reply.started":"2022-08-02T15:51:47.410625Z","shell.execute_reply":"2022-08-02T15:51:47.414676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Info of data input\nprint(train_df.info())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.417788Z","iopub.execute_input":"2022-08-02T15:51:47.418409Z","iopub.status.idle":"2022-08-02T15:51:47.446646Z","shell.execute_reply.started":"2022-08-02T15:51:47.418364Z","shell.execute_reply":"2022-08-02T15:51:47.445459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let see where there are Null values\ntrain_df.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.448115Z","iopub.execute_input":"2022-08-02T15:51:47.449124Z","iopub.status.idle":"2022-08-02T15:51:47.461820Z","shell.execute_reply.started":"2022-08-02T15:51:47.449079Z","shell.execute_reply":"2022-08-02T15:51:47.460446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tenemos que el sexo esta como una variable tipo string y no podemos operar con ella en los algoritmos\nsexo = {'male': 0, 'female':1}\ntrain_df['Sex'] = train_df['Sex'].map(sexo)\nprint(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.463129Z","iopub.execute_input":"2022-08-02T15:51:47.463454Z","iopub.status.idle":"2022-08-02T15:51:47.476295Z","shell.execute_reply.started":"2022-08-02T15:51:47.463424Z","shell.execute_reply":"2022-08-02T15:51:47.475160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hallemos la correlacion entre las variables\ncorr_matrix = train_df.corr(method='pearson')\nprint(corr_matrix)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.477655Z","iopub.execute_input":"2022-08-02T15:51:47.477997Z","iopub.status.idle":"2022-08-02T15:51:47.495473Z","shell.execute_reply.started":"2022-08-02T15:51:47.477967Z","shell.execute_reply":"2022-08-02T15:51:47.494663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Antes de cambiar los datos de entrada vamos a visulizar los datos que tenemos\n# Histograma\ntrain_df.hist(sharex=False, sharey=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:47.496687Z","iopub.execute_input":"2022-08-02T15:51:47.497488Z","iopub.status.idle":"2022-08-02T15:51:48.450897Z","shell.execute_reply.started":"2022-08-02T15:51:47.497455Z","shell.execute_reply":"2022-08-02T15:51:48.449605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Densidad\ntrain_df.plot(kind='density', subplots=True, layout=(3,3), sharex=False, sharey=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:48.452754Z","iopub.execute_input":"2022-08-02T15:51:48.453083Z","iopub.status.idle":"2022-08-02T15:51:49.665850Z","shell.execute_reply.started":"2022-08-02T15:51:48.453054Z","shell.execute_reply":"2022-08-02T15:51:49.664687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Matriz de correlacion\nfig = plt.figure()\nfig.suptitle(\"Matriz de correlacion\")\nax = fig.add_subplot(111)\ncax = ax.matshow(corr_matrix, vmin=-1, vmax=1, interpolation='none')\nfig.colorbar(cax)\nnames = ['PassengerId', 'Survived', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare']\nticks = np.arange(0,8,1)\nax.set_xticks(ticks)\nax.set_yticks(ticks)\nax.set_xticklabels(names)\nax.set_yticklabels(names)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:49.670952Z","iopub.execute_input":"2022-08-02T15:51:49.671426Z","iopub.status.idle":"2022-08-02T15:51:50.003197Z","shell.execute_reply.started":"2022-08-02T15:51:49.671393Z","shell.execute_reply":"2022-08-02T15:51:50.001975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill missing data","metadata":{}},{"cell_type":"code","source":"# Vamos a rellenar los datos en la edad que falta\nage_arr = train_df['Age'].values\nprint(\"Total data: %d\" % len(age_arr))\nprint(\"Null data: %d (%f%%)\" % (np.sum(np.isnan(age_arr)), np.sum(np.isnan(age_arr))/len(age_arr) * 100))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.005046Z","iopub.execute_input":"2022-08-02T15:51:50.005814Z","iopub.status.idle":"2022-08-02T15:51:50.014302Z","shell.execute_reply.started":"2022-08-02T15:51:50.005762Z","shell.execute_reply":"2022-08-02T15:51:50.013032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No podemos eliminar la variable pues hay un deficit del 20% de los datos\nmean_age = np.nanmean(age_arr)\nprint(\"La edad media de los pasajeros es de %f\" % mean_age)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.015910Z","iopub.execute_input":"2022-08-02T15:51:50.016851Z","iopub.status.idle":"2022-08-02T15:51:50.026089Z","shell.execute_reply.started":"2022-08-02T15:51:50.016817Z","shell.execute_reply":"2022-08-02T15:51:50.024984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a rellenar los valores faltantes\ntrain_df['Age'] = train_df['Age'].fillna(mean_age)\nprint(train_df.isnull().any())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.027871Z","iopub.execute_input":"2022-08-02T15:51:50.028928Z","iopub.status.idle":"2022-08-02T15:51:50.040031Z","shell.execute_reply.started":"2022-08-02T15:51:50.028881Z","shell.execute_reply":"2022-08-02T15:51:50.038945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a rellenar los datos en 'Cabin' que faltan\ncabin_arr = train_df['Cabin'].values\nprint(\"Total data: %d\" % len(cabin_arr))\nprint(\"Null data: %d (%f%%)\" % (np.sum(pd.isna(cabin_arr)), (np.sum(pd.isna(cabin_arr))/len(cabin_arr))*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.041550Z","iopub.execute_input":"2022-08-02T15:51:50.044875Z","iopub.status.idle":"2022-08-02T15:51:50.053646Z","shell.execute_reply.started":"2022-08-02T15:51:50.044839Z","shell.execute_reply":"2022-08-02T15:51:50.052307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# La columna 'Cabin' podemos desecharla pues faltan demasiados datos \ntrain_df.drop('Cabin', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.056194Z","iopub.execute_input":"2022-08-02T15:51:50.056777Z","iopub.status.idle":"2022-08-02T15:51:50.067137Z","shell.execute_reply.started":"2022-08-02T15:51:50.056730Z","shell.execute_reply":"2022-08-02T15:51:50.066072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.head(5))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.068489Z","iopub.execute_input":"2022-08-02T15:51:50.068844Z","iopub.status.idle":"2022-08-02T15:51:50.081972Z","shell.execute_reply.started":"2022-08-02T15:51:50.068814Z","shell.execute_reply":"2022-08-02T15:51:50.080877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.isnull().any())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.083319Z","iopub.execute_input":"2022-08-02T15:51:50.083747Z","iopub.status.idle":"2022-08-02T15:51:50.092399Z","shell.execute_reply.started":"2022-08-02T15:51:50.083717Z","shell.execute_reply":"2022-08-02T15:51:50.091242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Veamos que pasa con la variable 'Embarked'\nemb_arr = train_df['Embarked'].values\nprint(\"Total data: %d\" % len(emb_arr))\nprint(\"Null data: %d (%f%%)\" % (np.sum(pd.isna(emb_arr)), (np.sum(pd.isna(emb_arr)/len(emb_arr))*100)))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.093776Z","iopub.execute_input":"2022-08-02T15:51:50.094313Z","iopub.status.idle":"2022-08-02T15:51:50.105616Z","shell.execute_reply.started":"2022-08-02T15:51:50.094281Z","shell.execute_reply":"2022-08-02T15:51:50.104575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = set(emb_arr)\nprint(values)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.106802Z","iopub.execute_input":"2022-08-02T15:51:50.107640Z","iopub.status.idle":"2022-08-02T15:51:50.115196Z","shell.execute_reply.started":"2022-08-02T15:51:50.107605Z","shell.execute_reply":"2022-08-02T15:51:50.114131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a calcular las frecuencias de cada uno\nfrec_C = np.sum(emb_arr == 'C')\nfrec_Q = np.sum(emb_arr == 'Q')\nfrec_S = np.sum(emb_arr == 'S')\nprint(\"Frecuencias:\\nC: %d Q: %d S: %d\" % (frec_C, frec_Q, frec_S))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.116659Z","iopub.execute_input":"2022-08-02T15:51:50.117070Z","iopub.status.idle":"2022-08-02T15:51:50.125447Z","shell.execute_reply.started":"2022-08-02T15:51:50.117039Z","shell.execute_reply":"2022-08-02T15:51:50.124577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a rellenar los datos que faltan \ntrain_df['Embarked'] = train_df['Embarked'].fillna('S')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.126496Z","iopub.execute_input":"2022-08-02T15:51:50.127226Z","iopub.status.idle":"2022-08-02T15:51:50.137845Z","shell.execute_reply.started":"2022-08-02T15:51:50.127194Z","shell.execute_reply":"2022-08-02T15:51:50.136909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.isnull().any())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.138945Z","iopub.execute_input":"2022-08-02T15:51:50.139672Z","iopub.status.idle":"2022-08-02T15:51:50.149483Z","shell.execute_reply.started":"2022-08-02T15:51:50.139638Z","shell.execute_reply":"2022-08-02T15:51:50.148717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizacion de los datos arreglados","metadata":{}},{"cell_type":"code","source":"# Histograma\ntrain_df.hist(sharex=False, sharey=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:50.150811Z","iopub.execute_input":"2022-08-02T15:51:50.151293Z","iopub.status.idle":"2022-08-02T15:51:51.079221Z","shell.execute_reply.started":"2022-08-02T15:51:50.151264Z","shell.execute_reply":"2022-08-02T15:51:51.078092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hallemos la correlacion entre las variables\ncorr_matrix = train_df.corr(method='pearson')\nprint(corr_matrix)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.080713Z","iopub.execute_input":"2022-08-02T15:51:51.081081Z","iopub.status.idle":"2022-08-02T15:51:51.092817Z","shell.execute_reply.started":"2022-08-02T15:51:51.081050Z","shell.execute_reply":"2022-08-02T15:51:51.091708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate algorithms","metadata":{}},{"cell_type":"markdown","source":"### Dividamos las variables de los datos de la variable objetivo","metadata":{}},{"cell_type":"code","source":"X = train_df[['PassengerId', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare']].values\nY = train_df['Survived'].values","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.094187Z","iopub.execute_input":"2022-08-02T15:51:51.094641Z","iopub.status.idle":"2022-08-02T15:51:51.101890Z","shell.execute_reply.started":"2022-08-02T15:51:51.094586Z","shell.execute_reply":"2022-08-02T15:51:51.100903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test harness","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.103197Z","iopub.execute_input":"2022-08-02T15:51:51.104064Z","iopub.status.idle":"2022-08-02T15:51:51.231061Z","shell.execute_reply.started":"2022-08-02T15:51:51.104030Z","shell.execute_reply":"2022-08-02T15:51:51.229704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build Models","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVC","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.232647Z","iopub.execute_input":"2022-08-02T15:51:51.233156Z","iopub.status.idle":"2022-08-02T15:51:51.423438Z","shell.execute_reply.started":"2022-08-02T15:51:51.233111Z","shell.execute_reply":"2022-08-02T15:51:51.422183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.431180Z","iopub.execute_input":"2022-08-02T15:51:51.431562Z","iopub.status.idle":"2022-08-02T15:51:51.436309Z","shell.execute_reply.started":"2022-08-02T15:51:51.431529Z","shell.execute_reply":"2022-08-02T15:51:51.435419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Spot-check algorithms\nmodels = []\nmodels.append(('LR', LogisticRegression()))\nmodels.append(('CART', DecisionTreeClassifier()))\nmodels.append(('KNN', KNeighborsClassifier()))\nmodels.append(('LDA', LinearDiscriminantAnalysis()))\nmodels.append(('GNB', GaussianNB()))\nmodels.append(('SVC', SVC()))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.437379Z","iopub.execute_input":"2022-08-02T15:51:51.438190Z","iopub.status.idle":"2022-08-02T15:51:51.447982Z","shell.execute_reply.started":"2022-08-02T15:51:51.438142Z","shell.execute_reply":"2022-08-02T15:51:51.446997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = []\nnames = []\nkfold = KFold(n_splits=10, shuffle=True, random_state=7)\nfor name, model in models:\n    cv_results = cross_val_score(model, X, Y, cv=kfold)\n    results.append(cv_results)\n    names.append(name)\n    msg = \"%s: %f (%f)\" % (name, cv_results.mean(), cv_results.std())\n    print(msg)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:51.449400Z","iopub.execute_input":"2022-08-02T15:51:51.449965Z","iopub.status.idle":"2022-08-02T15:51:52.179632Z","shell.execute_reply.started":"2022-08-02T15:51:51.449925Z","shell.execute_reply":"2022-08-02T15:51:52.178299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Select the best model","metadata":{}},{"cell_type":"code","source":"# Compare algoritms\nfig = plt.figure()\nfig.suptitle(\"Algorithm Comparison\")\nax = fig.add_subplot(111)\nplt.boxplot(results)\nax.set_xticklabels(names)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.181037Z","iopub.execute_input":"2022-08-02T15:51:52.181355Z","iopub.status.idle":"2022-08-02T15:51:52.405574Z","shell.execute_reply.started":"2022-08-02T15:51:52.181327Z","shell.execute_reply":"2022-08-02T15:51:52.404374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Algorithm and Make Predictions","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.406873Z","iopub.execute_input":"2022-08-02T15:51:52.407814Z","iopub.status.idle":"2022-08-02T15:51:52.413986Z","shell.execute_reply.started":"2022-08-02T15:51:52.407776Z","shell.execute_reply":"2022-08-02T15:51:52.412481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import test dataset\ntest_filename = '../input/titanic/test.csv'\ntest_df = pd.read_csv(test_filename)\nprint(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.415808Z","iopub.execute_input":"2022-08-02T15:51:52.416448Z","iopub.status.idle":"2022-08-02T15:51:52.439125Z","shell.execute_reply.started":"2022-08-02T15:51:52.416416Z","shell.execute_reply":"2022-08-02T15:51:52.437981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.440546Z","iopub.execute_input":"2022-08-02T15:51:52.440958Z","iopub.status.idle":"2022-08-02T15:51:52.451929Z","shell.execute_reply.started":"2022-08-02T15:51:52.440925Z","shell.execute_reply":"2022-08-02T15:51:52.450795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Para que coincidan los datasets de entrenamiento y de prueba vamos a cambiar los el conjunto de prueba de la siguiente forma","metadata":{}},{"cell_type":"code","source":"# Veamos si hay algun dato que falte\nprint(test_df.isnull().any())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.453447Z","iopub.execute_input":"2022-08-02T15:51:52.454150Z","iopub.status.idle":"2022-08-02T15:51:52.465985Z","shell.execute_reply.started":"2022-08-02T15:51:52.454115Z","shell.execute_reply":"2022-08-02T15:51:52.464672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.467564Z","iopub.execute_input":"2022-08-02T15:51:52.468016Z","iopub.status.idle":"2022-08-02T15:51:52.485252Z","shell.execute_reply.started":"2022-08-02T15:51:52.467974Z","shell.execute_reply":"2022-08-02T15:51:52.484280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hagamos las mismas transformaciones que habiamos hecho antes\nsexo = {'male': 0, 'female':1}\ntest_df['Sex'] = test_df['Sex'].map(sexo)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.486335Z","iopub.execute_input":"2022-08-02T15:51:52.487531Z","iopub.status.idle":"2022-08-02T15:51:52.493978Z","shell.execute_reply.started":"2022-08-02T15:51:52.487493Z","shell.execute_reply":"2022-08-02T15:51:52.492862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a rellenar los valores faltantes\ntest_df['Age'] = test_df['Age'].fillna(mean_age)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.495215Z","iopub.execute_input":"2022-08-02T15:51:52.496163Z","iopub.status.idle":"2022-08-02T15:51:52.505499Z","shell.execute_reply.started":"2022-08-02T15:51:52.496128Z","shell.execute_reply":"2022-08-02T15:51:52.504271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.507001Z","iopub.execute_input":"2022-08-02T15:51:52.508173Z","iopub.status.idle":"2022-08-02T15:51:52.522666Z","shell.execute_reply.started":"2022-08-02T15:51:52.508128Z","shell.execute_reply":"2022-08-02T15:51:52.521266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# La columna 'Cabin' podemos desecharla pues faltan demasiados datos \ntest_df.drop('Cabin', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.524076Z","iopub.execute_input":"2022-08-02T15:51:52.524963Z","iopub.status.idle":"2022-08-02T15:51:52.531084Z","shell.execute_reply.started":"2022-08-02T15:51:52.524927Z","shell.execute_reply":"2022-08-02T15:51:52.530157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.532624Z","iopub.execute_input":"2022-08-02T15:51:52.533218Z","iopub.status.idle":"2022-08-02T15:51:52.546963Z","shell.execute_reply.started":"2022-08-02T15:51:52.533186Z","shell.execute_reply":"2022-08-02T15:51:52.545911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vamos a llenar los valores en la variable Fare que faltan\nfare_mean = np.nanmean(test_df['Fare'].values)\ntest_df['Fare'] = test_df['Fare'].fillna(fare_mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.548203Z","iopub.execute_input":"2022-08-02T15:51:52.548853Z","iopub.status.idle":"2022-08-02T15:51:52.558105Z","shell.execute_reply.started":"2022-08-02T15:51:52.548818Z","shell.execute_reply":"2022-08-02T15:51:52.557037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.559545Z","iopub.execute_input":"2022-08-02T15:51:52.560245Z","iopub.status.idle":"2022-08-02T15:51:52.575300Z","shell.execute_reply.started":"2022-08-02T15:51:52.560209Z","shell.execute_reply":"2022-08-02T15:51:52.574120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definimos el dataset de prueba\nX_test = test_df[['PassengerId', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare']].values","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.576477Z","iopub.execute_input":"2022-08-02T15:51:52.577128Z","iopub.status.idle":"2022-08-02T15:51:52.586442Z","shell.execute_reply.started":"2022-08-02T15:51:52.577093Z","shell.execute_reply":"2022-08-02T15:51:52.585473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definimos nuestro modelo\nlrm = LogisticRegression()\nlrm.fit(X, Y)\npredictions = lrm.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.587929Z","iopub.execute_input":"2022-08-02T15:51:52.588846Z","iopub.status.idle":"2022-08-02T15:51:52.627064Z","shell.execute_reply.started":"2022-08-02T15:51:52.588811Z","shell.execute_reply":"2022-08-02T15:51:52.625823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compute accuracy","metadata":{}},{"cell_type":"code","source":"gender_sub = pd.read_csv('../input/titanic/gender_submission.csv')\nprint(gender_sub.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.628375Z","iopub.execute_input":"2022-08-02T15:51:52.628742Z","iopub.status.idle":"2022-08-02T15:51:52.642132Z","shell.execute_reply.started":"2022-08-02T15:51:52.628709Z","shell.execute_reply":"2022-08-02T15:51:52.640682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy score: \")\nprint(accuracy_score(gender_sub['Survived'], predictions))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.643746Z","iopub.execute_input":"2022-08-02T15:51:52.644175Z","iopub.status.idle":"2022-08-02T15:51:52.651910Z","shell.execute_reply.started":"2022-08-02T15:51:52.644144Z","shell.execute_reply":"2022-08-02T15:51:52.650838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Confusion matrix: \")\nprint(confusion_matrix(gender_sub['Survived'], predictions))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.653478Z","iopub.execute_input":"2022-08-02T15:51:52.654188Z","iopub.status.idle":"2022-08-02T15:51:52.667640Z","shell.execute_reply.started":"2022-08-02T15:51:52.654143Z","shell.execute_reply":"2022-08-02T15:51:52.666677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Classification report: \")\nprint(classification_report(gender_sub['Survived'], predictions))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.669217Z","iopub.execute_input":"2022-08-02T15:51:52.669890Z","iopub.status.idle":"2022-08-02T15:51:52.680053Z","shell.execute_reply.started":"2022-08-02T15:51:52.669854Z","shell.execute_reply":"2022-08-02T15:51:52.678861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save my predictions","metadata":{}},{"cell_type":"code","source":"my_predict = pd.DataFrame()\nmy_predict['PassengerId'] = gender_sub['PassengerId']\nmy_predict['Survived'] = predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.681649Z","iopub.execute_input":"2022-08-02T15:51:52.682362Z","iopub.status.idle":"2022-08-02T15:51:52.689570Z","shell.execute_reply.started":"2022-08-02T15:51:52.682328Z","shell.execute_reply":"2022-08-02T15:51:52.688450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Salvamos mis predicciones\nmy_predict.to_csv('my_predictions.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T15:51:52.690767Z","iopub.execute_input":"2022-08-02T15:51:52.691574Z","iopub.status.idle":"2022-08-02T15:51:52.703624Z","shell.execute_reply.started":"2022-08-02T15:51:52.691541Z","shell.execute_reply":"2022-08-02T15:51:52.702705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}