{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Starting with scikit-learn","metadata":{}},{"cell_type":"markdown","source":"### 1st iteration: our dummy model","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\ntitanic = pd.read_csv(\"../input/titanic/train.csv\")\ntitanic.head()\ntitanic.info()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T12:56:10.564143Z","iopub.execute_input":"2022-07-19T12:56:10.564999Z","iopub.status.idle":"2022-07-19T12:56:10.601607Z","shell.execute_reply.started":"2022-07-19T12:56:10.564951Z","shell.execute_reply":"2022-07-19T12:56:10.600763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.603557Z","iopub.execute_input":"2022-07-19T12:56:10.604087Z","iopub.status.idle":"2022-07-19T12:56:10.609295Z","shell.execute_reply.started":"2022-07-19T12:56:10.604038Z","shell.execute_reply":"2022-07-19T12:56:10.608624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.610237Z","iopub.execute_input":"2022-07-19T12:56:10.610738Z","iopub.status.idle":"2022-07-19T12:56:10.640818Z","shell.execute_reply.started":"2022-07-19T12:56:10.610705Z","shell.execute_reply":"2022-07-19T12:56:10.639817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.643265Z","iopub.execute_input":"2022-07-19T12:56:10.643826Z","iopub.status.idle":"2022-07-19T12:56:10.654939Z","shell.execute_reply.started":"2022-07-19T12:56:10.643781Z","shell.execute_reply":"2022-07-19T12:56:10.654010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic.groupby(['Sex','Survived']).agg(count = ('Survived','count'))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.656154Z","iopub.execute_input":"2022-07-19T12:56:10.657076Z","iopub.status.idle":"2022-07-19T12:56:10.678683Z","shell.execute_reply.started":"2022-07-19T12:56:10.657026Z","shell.execute_reply":"2022-07-19T12:56:10.678039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create our dummy model (\"Train\")¶","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def my_dummy_model(sex): \n    if sex == 'female': \n        return 1\n    elif sex == 'male':\n        return 0\n\ntitanic['preds'] = [my_dummy_model(sex) for sex in titanic['Sex']]\ntitanic.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.679807Z","iopub.execute_input":"2022-07-19T12:56:10.680148Z","iopub.status.idle":"2022-07-19T12:56:10.701703Z","shell.execute_reply.started":"2022-07-19T12:56:10.680120Z","shell.execute_reply":"2022-07-19T12:56:10.700744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate","metadata":{}},{"cell_type":"code","source":"error_check = titanic.filter(['Survived','preds']).assign(check = lambda x: x['Survived'] == x['preds'])\nerror_check.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.702935Z","iopub.execute_input":"2022-07-19T12:56:10.703195Z","iopub.status.idle":"2022-07-19T12:56:10.722134Z","shell.execute_reply.started":"2022-07-19T12:56:10.703164Z","shell.execute_reply":"2022-07-19T12:56:10.721484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# total number of correct answers\nerror_check['check'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.723455Z","iopub.execute_input":"2022-07-19T12:56:10.723872Z","iopub.status.idle":"2022-07-19T12:56:10.729852Z","shell.execute_reply.started":"2022-07-19T12:56:10.723836Z","shell.execute_reply":"2022-07-19T12:56:10.729022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic.groupby(['Survived']).agg(n=('Survived','count')).assign(perc = lambda x: x['n']/x['n'].sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.731524Z","iopub.execute_input":"2022-07-19T12:56:10.731855Z","iopub.status.idle":"2022-07-19T12:56:10.751054Z","shell.execute_reply.started":"2022-07-19T12:56:10.731817Z","shell.execute_reply":"2022-07-19T12:56:10.750412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic.drop(columns=['preds'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.753065Z","iopub.execute_input":"2022-07-19T12:56:10.753400Z","iopub.status.idle":"2022-07-19T12:56:10.757883Z","shell.execute_reply.started":"2022-07-19T12:56:10.753371Z","shell.execute_reply":"2022-07-19T12:56:10.757244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split the Dataset","metadata":{}},{"cell_type":"markdown","source":"### 2nd iteration: creating train and test¶\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = titanic.drop(columns=['Survived'])\ny = titanic['Survived']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15, random_state=8)\n\nX_train = pd.DataFrame(X_train, columns=X.columns)\nX_test = pd.DataFrame(X_test, columns=X.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.758880Z","iopub.execute_input":"2022-07-19T12:56:10.759435Z","iopub.status.idle":"2022-07-19T12:56:10.774437Z","shell.execute_reply.started":"2022-07-19T12:56:10.759404Z","shell.execute_reply":"2022-07-19T12:56:10.773423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Short exploration","metadata":{}},{"cell_type":"code","source":"(\n    X_train\n    .assign(Survived = y_train)\n    .groupby(['Sex','Survived'])\n    .agg(count = ('Survived','count'))\n    )","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.775768Z","iopub.execute_input":"2022-07-19T12:56:10.776595Z","iopub.status.idle":"2022-07-19T12:56:10.798721Z","shell.execute_reply.started":"2022-07-19T12:56:10.776552Z","shell.execute_reply":"2022-07-19T12:56:10.797854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test our model","metadata":{}},{"cell_type":"code","source":"# Let's predict if someone will have survived or not with my dummy mdoel\nX_train['preds'] = [my_dummy_model(sex) for sex in X_train['Sex']]\nX_test['preds'] = [my_dummy_model(sex) for sex in X_test['Sex']]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.799920Z","iopub.execute_input":"2022-07-19T12:56:10.800154Z","iopub.status.idle":"2022-07-19T12:56:10.807481Z","shell.execute_reply.started":"2022-07-19T12:56:10.800125Z","shell.execute_reply":"2022-07-19T12:56:10.806891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results","metadata":{}},{"cell_type":"code","source":"# results train data\n(\n    pd.DataFrame({\n        'Survived':y_train,\n        'preds':[my_dummy_model(sex) for sex in X_train['Sex']]\n        })\n    .assign(check = lambda x: x['Survived'] == x['preds'])['check']\n    .sum()\n) / len(y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.808418Z","iopub.execute_input":"2022-07-19T12:56:10.809307Z","iopub.status.idle":"2022-07-19T12:56:10.827721Z","shell.execute_reply.started":"2022-07-19T12:56:10.809271Z","shell.execute_reply":"2022-07-19T12:56:10.826865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Results 2 ","metadata":{}},{"cell_type":"code","source":"# results train data\nacc_2nd = (\n    pd.DataFrame({\n        'Survived':y_test,\n        'preds':[my_dummy_model(sex) for sex in X_test['Sex']]\n        })\n    .assign(check = lambda x: x['Survived'] == x['preds'])['check']\n    .sum()\n) / len(y_test)\nacc_2nd","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.828836Z","iopub.execute_input":"2022-07-19T12:56:10.829498Z","iopub.status.idle":"2022-07-19T12:56:10.841997Z","shell.execute_reply.started":"2022-07-19T12:56:10.829466Z","shell.execute_reply":"2022-07-19T12:56:10.841273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3rd iteration: our first scikit-learning model¶\n","metadata":{}},{"cell_type":"code","source":"X = titanic.filter(['Sex'])\ny = titanic.filter(['Survived'])\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15, random_state=8)\n\nX_train = pd.DataFrame(X_train, columns=X.columns)\nX_test = pd.DataFrame(X_test, columns=X.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:56:10.843116Z","iopub.execute_input":"2022-07-19T12:56:10.843804Z","iopub.status.idle":"2022-07-19T12:56:10.857479Z","shell.execute_reply.started":"2022-07-19T12:56:10.843765Z","shell.execute_reply":"2022-07-19T12:56:10.856597Z"},"trusted":true},"execution_count":null,"outputs":[]}]}