{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.140845Z","iopub.execute_input":"2022-07-21T17:55:35.141502Z","iopub.status.idle":"2022-07-21T17:55:35.151618Z","shell.execute_reply.started":"2022-07-21T17:55:35.141454Z","shell.execute_reply":"2022-07-21T17:55:35.150965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a practice run in terms of my understanding of how to submit to a Kaggle competitions.\n\nThe analysis here likely could be expanded upon and further organized for a fuller in depth analysis. ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.155231Z","iopub.execute_input":"2022-07-21T17:55:35.155657Z","iopub.status.idle":"2022-07-21T17:55:35.553758Z","shell.execute_reply.started":"2022-07-21T17:55:35.155633Z","shell.execute_reply":"2022-07-21T17:55:35.553112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prefix = \"/kaggle/input/titanic/\"\ntrain = pd.read_csv(prefix + \"train.csv\")\ntest = pd.read_csv(prefix + \"test.csv\")\n\n\ntrain.describe(include = 'all')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.554707Z","iopub.execute_input":"2022-07-21T17:55:35.555040Z","iopub.status.idle":"2022-07-21T17:55:35.600488Z","shell.execute_reply.started":"2022-07-21T17:55:35.555018Z","shell.execute_reply":"2022-07-21T17:55:35.599887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.602938Z","iopub.execute_input":"2022-07-21T17:55:35.603206Z","iopub.status.idle":"2022-07-21T17:55:35.618027Z","shell.execute_reply.started":"2022-07-21T17:55:35.603184Z","shell.execute_reply":"2022-07-21T17:55:35.616643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.619652Z","iopub.execute_input":"2022-07-21T17:55:35.620242Z","iopub.status.idle":"2022-07-21T17:55:35.662830Z","shell.execute_reply.started":"2022-07-21T17:55:35.620211Z","shell.execute_reply":"2022-07-21T17:55:35.662093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.663836Z","iopub.execute_input":"2022-07-21T17:55:35.664885Z","iopub.status.idle":"2022-07-21T17:55:35.676467Z","shell.execute_reply.started":"2022-07-21T17:55:35.664834Z","shell.execute_reply":"2022-07-21T17:55:35.675471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Nothing else seems out of the ordinary probably just replace with median fare price\ntest[test['Fare'].isna()]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.677743Z","iopub.execute_input":"2022-07-21T17:55:35.678454Z","iopub.status.idle":"2022-07-21T17:55:35.692052Z","shell.execute_reply.started":"2022-07-21T17:55:35.678422Z","shell.execute_reply":"2022-07-21T17:55:35.690894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Numerical columns Analysis","metadata":{}},{"cell_type":"markdown","source":"Will have to deal with Age and Cabin in more interesting manners potentially\nEmbarked and Sex have few unique values relative to total number of values may be useful to OHE them.\n\nGeneral Observations: \n\nAll the categorical information may contain relevant info to predict survivability however, will need some more fine tuning.\n\nHypothesis: \n- Cabin/Ticket/Embarked categories could correlate with location on ship during the crash\n  \n- Name/Sex categories could determine survivability based on societal tendencies on who to save. \n    Name could be split and or grouped by desired prefix (Mr., Mrs, Doc., Rev.) or by similarity of name.\n    Going by similarity of name could skew results interestingly... (family names and families sticking together)\n","metadata":{}},{"cell_type":"code","source":"# We will separate by the initial numerical and categorical columns first for comparison\n\ncolumns_missing_values = ['Age', 'Cabin']\nnum_columns = [ x  for x in train.columns if train[x].dtype != object and x != 'Survived']\ncat_columns = [ x  for x in train.columns if train[x].dtype == object]\nnum_columns += ['Survived']\ncat_columns += ['Survived']\n\nprint(\"Numerical Columns:\")\n[print(x) for x in num_columns]\nprint()\nprint(\"Categorical Columns:\")\nx = [print(x) for x in cat_columns]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.693395Z","iopub.execute_input":"2022-07-21T17:55:35.693719Z","iopub.status.idle":"2022-07-21T17:55:35.701651Z","shell.execute_reply.started":"2022-07-21T17:55:35.693697Z","shell.execute_reply":"2022-07-21T17:55:35.699942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Looking at Numerical features","metadata":{}},{"cell_type":"code","source":"# Generating Histograms for numeric columns\nfig, axs = plt.subplots(4,2,figsize=(12,10))\n\nfor ax, column in zip(axs.flatten(), num_columns):\n    ax.hist(train[column],bins=10, )\n    ax.set_title(column)\n\naxs.flatten()[-1].set_visible(False)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:35.702736Z","iopub.execute_input":"2022-07-21T17:55:35.702998Z","iopub.status.idle":"2022-07-21T17:55:36.554689Z","shell.execute_reply.started":"2022-07-21T17:55:35.702972Z","shell.execute_reply":"2022-07-21T17:55:36.553775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(len(num_columns), len(num_columns), figsize = (20,20))\naxs = pd.plotting.scatter_matrix(train[num_columns], ax = axs, diagonal = 'kde')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:36.555723Z","iopub.execute_input":"2022-07-21T17:55:36.556240Z","iopub.status.idle":"2022-07-21T17:55:39.699490Z","shell.execute_reply.started":"2022-07-21T17:55:36.556216Z","shell.execute_reply":"2022-07-21T17:55:39.698810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(train[num_columns].corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:39.700354Z","iopub.execute_input":"2022-07-21T17:55:39.700678Z","iopub.status.idle":"2022-07-21T17:55:39.944444Z","shell.execute_reply.started":"2022-07-21T17:55:39.700654Z","shell.execute_reply":"2022-07-21T17:55:39.943402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[num_columns].pivot_table(index='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:39.946725Z","iopub.execute_input":"2022-07-21T17:55:39.947795Z","iopub.status.idle":"2022-07-21T17:55:39.975627Z","shell.execute_reply.started":"2022-07-21T17:55:39.947755Z","shell.execute_reply":"2022-07-21T17:55:39.974884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['Pclass', 'Sex', 'Embarked', ]\n\nfor col in cols:\n    print(train.pivot_table(index='Survived', columns=col, values= 'PassengerId', aggfunc=\"count\"))\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:39.980556Z","iopub.execute_input":"2022-07-21T17:55:39.981416Z","iopub.status.idle":"2022-07-21T17:55:40.011211Z","shell.execute_reply.started":"2022-07-21T17:55:39.981386Z","shell.execute_reply":"2022-07-21T17:55:40.010341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Will have to deal with Age and Cabin in more interesting manners potentially\nEmbarked and Sex have few unique values relative to total number of values may be useful to OHE them.\n\nGeneral Observations: \n\nAll the categorical information may contain relevant info to predict survivability however, will need some more fine tuning.\n\nHypothesis: \n- Cabin/Ticket/Embarked categories could correlate with location on ship during the crash\n  \n- Name/Sex categories could determine survivability based on societal tendencies on who to save. \n    Name could be split and or grouped by desired prefix (Mr., Mrs, Doc., Rev.) or by similarity of name.\n    Going by similarity of name could skew results interestingly... (family names and families sticking together)\n","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.012552Z","iopub.execute_input":"2022-07-21T17:55:40.013232Z","iopub.status.idle":"2022-07-21T17:55:40.026913Z","shell.execute_reply.started":"2022-07-21T17:55:40.013170Z","shell.execute_reply":"2022-07-21T17:55:40.025209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since age and survivability had only a slight correlation I impute the Nan age data by replacing with the mean\ndef ImputeAge(df):\n    train.loc[train.Age.isna(), 'Age'] = np.round(df.Age.mean(), decimals = 1)\n    return df\n\n\ntrain = ImputeAge(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.029060Z","iopub.execute_input":"2022-07-21T17:55:40.029953Z","iopub.status.idle":"2022-07-21T17:55:40.041715Z","shell.execute_reply.started":"2022-07-21T17:55:40.029912Z","shell.execute_reply":"2022-07-21T17:55:40.040363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cabins = np.unique(train.Cabin[~train.Cabin.isna()]) # 147 unique Cabins\ncabins","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.043079Z","iopub.execute_input":"2022-07-21T17:55:40.043718Z","iopub.status.idle":"2022-07-21T17:55:40.056390Z","shell.execute_reply.started":"2022-07-21T17:55:40.043692Z","shell.execute_reply":"2022-07-21T17:55:40.055793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems like there are only A-T letter cabins \nAssuming this is the same for the test dataset I decide create a new column based off cabin letter\n","metadata":{}},{"cell_type":"code","source":"cabin_letters = ['A','B', 'C', 'D', 'E', 'F', 'G', 'T']\n\n# cabin_numbers = np.unique([np.int32(x[0]) for x in [(\"\".join([elem for elem in cab if elem.isdigit() or elem == \" \"])).split(\" \") for cab in cabins] if x[0].isdigit()])\ncabin_numbers = np.arange(151)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.057672Z","iopub.execute_input":"2022-07-21T17:55:40.058322Z","iopub.status.idle":"2022-07-21T17:55:40.072440Z","shell.execute_reply.started":"2022-07-21T17:55:40.058295Z","shell.execute_reply":"2022-07-21T17:55:40.071625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Used a helper function which checks to make sure ","metadata":{}},{"cell_type":"code","source":"def check_cabin(cabin):\n    l = [letter for letter in cabin_letters if letter in cabin][-1]\n    numbers = [number for number in cabin_numbers if str(number) in cabin]\n    if len(numbers)>0:\n        n = [number for number in cabin_numbers if str(number) in cabin][-1]\n        return l, n\n    else:\n        return l, -1\n\n\ndef create_cabin_columns(df):\n\n    temp_df = df.Cabin[~df.Cabin.isna()].apply(lambda x: check_cabin(x)).to_frame(\"Tuple\")\n\n    df.loc[~df.Cabin.isna(), 'Cabin_Letter'], df.loc[~df.Cabin.isna(), 'Cabin_Number'] = zip(*temp_df['Tuple'])\n\n    # We create a new class of the passengers without a cabin number or cabin number\n    df.loc[df.Cabin.isna(), 'Cabin_Letter'] = 'n'\n    df.loc[df.Cabin.isna(), 'Cabin_Number'] = -1\n\n    return df\n\ntrain = create_cabin_columns(train)\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.073739Z","iopub.execute_input":"2022-07-21T17:55:40.074238Z","iopub.status.idle":"2022-07-21T17:55:40.139920Z","shell.execute_reply.started":"2022-07-21T17:55:40.074203Z","shell.execute_reply":"2022-07-21T17:55:40.138900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Name and ticket are next on the chopping block","metadata":{}},{"cell_type":"markdown","source":"## Name Analysis","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.142142Z","iopub.execute_input":"2022-07-21T17:55:40.142932Z","iopub.status.idle":"2022-07-21T17:55:40.158853Z","shell.execute_reply.started":"2022-07-21T17:55:40.142892Z","shell.execute_reply":"2022-07-21T17:55:40.157672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train.copy() # Change this to test to see Titles for test set.\n\nlast_names = [ name.split(',')[0] for name in df.Name.values]\n\nprefix_names = [ name.split(',')[1] for name in df.Name.values]\n\n\ntitles = [prefix_name.split('.')[0] for prefix_name in prefix_names]\nuni_titles, count_titles = np.unique(titles, return_counts=True)\n\na = [print(uni_titles[i], count_titles[i]) for i in range(len(uni_titles))]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.160845Z","iopub.execute_input":"2022-07-21T17:55:40.161687Z","iopub.status.idle":"2022-07-21T17:55:40.179479Z","shell.execute_reply.started":"2022-07-21T17:55:40.161654Z","shell.execute_reply":"2022-07-21T17:55:40.178919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uni_lnames, count_lnames = np.unique(last_names, return_counts=True)\n\nplt.hist(count_lnames)\nplt.xlabel(\"Names shared by x people\")\nplt.ylabel('Count of Names')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.180887Z","iopub.execute_input":"2022-07-21T17:55:40.181123Z","iopub.status.idle":"2022-07-21T17:55:40.337024Z","shell.execute_reply.started":"2022-07-21T17:55:40.181102Z","shell.execute_reply":"2022-07-21T17:55:40.335835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"# of Different Last Names\", len(count_lnames))\nsorted(zip(count_lnames, uni_lnames), reverse=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.338337Z","iopub.execute_input":"2022-07-21T17:55:40.339481Z","iopub.status.idle":"2022-07-21T17:55:40.400002Z","shell.execute_reply.started":"2022-07-21T17:55:40.339447Z","shell.execute_reply":"2022-07-21T17:55:40.398845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is rather unexpected, I thought some names would be shared by more than 9 people. (Like an Alex, Maybe this could be a modern bias of mine)\nNonetheless First Names are typically random at most giving geographical information. \n\nHypothesis:\nSurnames should correlate certain people with families and/or geographic information and be a better predictor of survivability.","metadata":{}},{"cell_type":"code","source":"first_names = [prefix.split('.')[1].split(' ')[1] for prefix in prefix_names]\n# print(first_names)\n\nuni_fnames, count_fnames = np.unique(first_names, return_counts=True)\n\nplt.hist(count_fnames)\nplt.xlabel(\"First names shared by x people\")\nplt.ylabel('Count of Names')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.402774Z","iopub.execute_input":"2022-07-21T17:55:40.403477Z","iopub.status.idle":"2022-07-21T17:55:40.553356Z","shell.execute_reply.started":"2022-07-21T17:55:40.403428Z","shell.execute_reply":"2022-07-21T17:55:40.552349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"# of Passengers\", len(count_fnames))\nsorted(zip(count_fnames, uni_fnames), reverse=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.554169Z","iopub.execute_input":"2022-07-21T17:55:40.554401Z","iopub.status.idle":"2022-07-21T17:55:40.583778Z","shell.execute_reply.started":"2022-07-21T17:55:40.554379Z","shell.execute_reply":"2022-07-21T17:55:40.582650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I noticed from printing the names there were more traditionally male names. There's probably an underlying reason worth further investigation.\nI'll likely ignore first names as it is also in general pretty messy","metadata":{}},{"cell_type":"code","source":"Full_First_Name = [prefix.split('.')[1].strip() for prefix in prefix_names]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.584992Z","iopub.execute_input":"2022-07-21T17:55:40.585259Z","iopub.status.idle":"2022-07-21T17:55:40.590350Z","shell.execute_reply.started":"2022-07-21T17:55:40.585237Z","shell.execute_reply":"2022-07-21T17:55:40.589366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Full_First_Name","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-21T17:55:40.591435Z","iopub.execute_input":"2022-07-21T17:55:40.591715Z","iopub.status.idle":"2022-07-21T17:55:40.615429Z","shell.execute_reply.started":"2022-07-21T17:55:40.591694Z","shell.execute_reply":"2022-07-21T17:55:40.614386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(Full_First_Name)[[('(' in prefix) for prefix in Full_First_Name]]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.617131Z","iopub.execute_input":"2022-07-21T17:55:40.617423Z","iopub.status.idle":"2022-07-21T17:55:40.639796Z","shell.execute_reply.started":"2022-07-21T17:55:40.617396Z","shell.execute_reply":"2022-07-21T17:55:40.638344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at the Names that have '('  I notice that generally a traditionally male name was given first with a traditionally female name in the parenthesis.\nUsing the first names would likely require further investigation of the data set.\n\nFor now I will ignore first names and keep last names and titles as a new column in the dataset","metadata":{}},{"cell_type":"markdown","source":"Since titles in training set may not show up in test set and vice versa it may be wiser instead of keeping all sur-title data to replace it with categorize people based on common title vs uncommon title. \nThis could be tweaked with another class for professional titles (Dr, Major, Captain). \n\nFor now I'll break it up for common and uncommon titles.\nThis could be a source for bias in the model as I'm mainly using common titles as an american understands it.\nA more in depth analysis could try to determine it in a more precise manner with the context of the titles held during the time period.","metadata":{}},{"cell_type":"code","source":"def check_title(element):\n    common_titles = ['Mr', 'Miss', 'Ms', 'Mrs']\n    \n    return np.any([title in element for title in common_titles])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.641302Z","iopub.execute_input":"2022-07-21T17:55:40.641552Z","iopub.status.idle":"2022-07-21T17:55:40.649251Z","shell.execute_reply.started":"2022-07-21T17:55:40.641530Z","shell.execute_reply":"2022-07-21T17:55:40.648330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef create_title_lastname_columns(df):\n\n    temp_df = df.Name.apply(lambda x: check_title(x)).to_frame(\"temp\")\n    df.loc[:, 'Common_Title'] = temp_df['temp']\n\n    df = df.copy() # Change this to test to see Titles for test set.\n\n    last_names = [ name.split(',')[0] for name in df.Name.values]\n\n    df['Last_Name'] = last_names\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.650172Z","iopub.execute_input":"2022-07-21T17:55:40.650451Z","iopub.status.idle":"2022-07-21T17:55:40.663291Z","shell.execute_reply.started":"2022-07-21T17:55:40.650424Z","shell.execute_reply":"2022-07-21T17:55:40.661410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = create_title_lastname_columns(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.664942Z","iopub.execute_input":"2022-07-21T17:55:40.665268Z","iopub.status.idle":"2022-07-21T17:55:40.685504Z","shell.execute_reply.started":"2022-07-21T17:55:40.665246Z","shell.execute_reply":"2022-07-21T17:55:40.683989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.Common_Title.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.686990Z","iopub.execute_input":"2022-07-21T17:55:40.687245Z","iopub.status.idle":"2022-07-21T17:55:40.697353Z","shell.execute_reply.started":"2022-07-21T17:55:40.687223Z","shell.execute_reply":"2022-07-21T17:55:40.696345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.698304Z","iopub.execute_input":"2022-07-21T17:55:40.698565Z","iopub.status.idle":"2022-07-21T17:55:40.733411Z","shell.execute_reply.started":"2022-07-21T17:55:40.698535Z","shell.execute_reply":"2022-07-21T17:55:40.732647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ticket Analysis","metadata":{}},{"cell_type":"markdown","source":"I split the tickets into its numeric and prefix components as it could contain useful information?","metadata":{}},{"cell_type":"code","source":"def get_ticket_number(ticket):\n    for tick_comp in ticket.split(' '):\n        if tick_comp.isnumeric():\n            return tick_comp\n\n\ndef get_ticket_prefix(ticket):\n    if not ticket.isnumeric():\n        return ' '.join([tick_comp for tick_comp in ticket.split(' ') if not tick_comp.isnumeric()])\n    else:\n        return 'None'","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.734537Z","iopub.execute_input":"2022-07-21T17:55:40.734898Z","iopub.status.idle":"2022-07-21T17:55:40.742452Z","shell.execute_reply.started":"2022-07-21T17:55:40.734871Z","shell.execute_reply":"2022-07-21T17:55:40.741297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_ticket_columns(df):\n    temp_df = df.Ticket.apply(lambda x: get_ticket_number(x)).to_frame(\"temp\")\n    df.loc[:, 'Ticket_Number'] = temp_df['temp']\n\n    temp_df = df.Ticket.apply(lambda x: get_ticket_prefix(x)).to_frame(\"temp\")\n    df.loc[:, 'Ticket_Prefix'] = temp_df['temp']\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.743665Z","iopub.execute_input":"2022-07-21T17:55:40.744037Z","iopub.status.idle":"2022-07-21T17:55:40.760163Z","shell.execute_reply.started":"2022-07-21T17:55:40.743957Z","shell.execute_reply":"2022-07-21T17:55:40.758718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df = create_ticket_columns(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.761097Z","iopub.execute_input":"2022-07-21T17:55:40.761360Z","iopub.status.idle":"2022-07-21T17:55:40.775341Z","shell.execute_reply.started":"2022-07-21T17:55:40.761337Z","shell.execute_reply":"2022-07-21T17:55:40.774685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I notice many of the tickets non-numeric components are similiar to one another and could be grouped into few columns rather than the 45 I currently have. I'll leave this for future improvements","metadata":{}},{"cell_type":"code","source":"np.unique(temp_df.Ticket_Prefix)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.776271Z","iopub.execute_input":"2022-07-21T17:55:40.776631Z","iopub.status.idle":"2022-07-21T17:55:40.792027Z","shell.execute_reply.started":"2022-07-21T17:55:40.776598Z","shell.execute_reply":"2022-07-21T17:55:40.791284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Putting it all together","metadata":{}},{"cell_type":"code","source":"# Functions to Create Useful Columns\ndef ImputeAge(df):\n    df.loc[df.Age.isna(), 'Age'] = np.round(df.Age.mean(), decimals = 1)\n    return df\n\ndef ImputeFare(df):\n    df.loc[df.Fare.isna(), 'Fare'] = np.round(df.Fare.mean(), decimals = 1)\n    return df\n\ndef check_cabin(cabin):\n    cabin_letters = ['A','B', 'C', 'D', 'E', 'F', 'G', 'T'] # This is specific to titanic dataset\n    cabin_numbers = np.arange(151) # This is specific to titanic dataset\n\n    l = [letter for letter in cabin_letters if letter in cabin][-1]\n    numbers = [number for number in cabin_numbers if str(number) in cabin]\n    if len(numbers)>0:\n        n = [number for number in cabin_numbers if str(number) in cabin][-1]\n        return l, n\n    else:\n        return l, np.nan\n\n\ndef create_cabin_columns(df):\n\n    temp_df = df.Cabin[~df.Cabin.isna()].apply(lambda x: check_cabin(x)).to_frame(\"Tuple\")\n\n    df.loc[~df.Cabin.isna(), 'Cabin_Letter'], df.loc[~df.Cabin.isna(), 'Cabin_Number'] = zip(*temp_df['Tuple'])\n\n    # # We create a new class of the passengers without a cabin number or cabin number\n    df.loc[df.Cabin.isna(), 'Cabin_Letter'] = None\n    # df.loc[df.Cabin.isna(), 'Cabin_Number'] = np.nan\n\n    return df\n\n\n\ndef check_title(element):\n    common_titles = ['Mr', 'Miss', 'Ms', 'Mrs']\n    \n    return np.any([title in element for title in common_titles])\n\n\ndef create_title_lastname_columns(df):\n\n    temp_df = df.Name.apply(lambda x: check_title(x)).to_frame(\"temp\")\n    df.loc[:, 'Common_Title'] = temp_df['temp']\n\n    df = df.copy() # Change this to test to see Titles for test set.\n\n    last_names = [ name.split(',')[0] for name in df.Name.values]\n\n    df['Last_Name'] = last_names\n    \n    return df\n\n\n\ndef get_ticket_number(ticket):\n    for tick_comp in ticket.split(' '):\n        if tick_comp.isnumeric():\n            return np.int32(tick_comp)\n\ndef get_ticket_prefix(ticket):\n    if not ticket.isnumeric():\n        return ' '.join([tick_comp for tick_comp in ticket.split(' ') if not tick_comp.isnumeric()])\n\n        \ndef create_ticket_columns(df):\n    temp_df = df.Ticket.apply(lambda x: get_ticket_number(x)).to_frame(\"temp\")\n    df.loc[:, 'Ticket_Number'] = temp_df['temp']\n\n    df.loc[df.Ticket_Number.isna(), 'Ticket_Number'] = np.int32(df.Ticket_Number.mean())\n\n\n    temp_df = df.Ticket.apply(lambda x: get_ticket_prefix(x)).to_frame(\"temp\")\n    df.loc[:, 'Ticket_Prefix'] = temp_df['temp']\n    return df\n\n\n\ndef create_useful_columns(df, return_labels = False):\n    df = ImputeAge(df)\n    df = ImputeFare(df)\n    df = create_cabin_columns(df)\n    df = create_title_lastname_columns(df)\n    df = create_ticket_columns(df)\n    # Only keeping relevant columns\n    df = df.drop(index = df[df.Embarked.isna()].index) # Dropping Embarked Nan values\n    df.Sex = df.Sex.apply(lambda x: 0 if x == 'male' else 1)\n\n\n    \n    if return_labels == True:\n        labels = df.Survived\n        df = df.loc[:, ['Common_Title', 'Last_Name', 'Sex', 'Age', 'Ticket_Number', 'Cabin_Letter', 'Cabin_Number', 'Fare', 'Embarked', 'Pclass', 'SibSp', 'Parch']]\n        return df, labels\n\n    else:\n        df = df.loc[:, ['PassengerId','Common_Title', 'Last_Name' ,'Sex', 'Age', 'Ticket_Number', 'Cabin_Letter', 'Cabin_Number', 'Fare', 'Embarked', 'Pclass', 'SibSp', 'Parch']]\n        return df\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.793620Z","iopub.execute_input":"2022-07-21T17:55:40.793883Z","iopub.status.idle":"2022-07-21T17:55:40.820973Z","shell.execute_reply.started":"2022-07-21T17:55:40.793856Z","shell.execute_reply":"2022-07-21T17:55:40.819925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prefix = \"/kaggle/input/titanic/\"\ntrain = pd.read_csv(prefix + \"train.csv\")\ntest = pd.read_csv(prefix + \"test.csv\")\n\n\ntrain.describe(include = 'all')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.822547Z","iopub.execute_input":"2022-07-21T17:55:40.823190Z","iopub.status.idle":"2022-07-21T17:55:40.891014Z","shell.execute_reply.started":"2022-07-21T17:55:40.823148Z","shell.execute_reply":"2022-07-21T17:55:40.889706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building Models","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\n\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.892749Z","iopub.execute_input":"2022-07-21T17:55:40.893123Z","iopub.status.idle":"2022-07-21T17:55:40.971098Z","shell.execute_reply.started":"2022-07-21T17:55:40.893087Z","shell.execute_reply":"2022-07-21T17:55:40.970459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import LabelEncoder\nfrom category_encoders import TargetEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:40.976159Z","iopub.execute_input":"2022-07-21T17:55:40.976956Z","iopub.status.idle":"2022-07-21T17:55:41.031981Z","shell.execute_reply.started":"2022-07-21T17:55:40.976932Z","shell.execute_reply":"2022-07-21T17:55:41.030548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, train_labels = create_useful_columns(train.copy(), return_labels=True)\ntest_df = create_useful_columns(test.copy())\n\n\n\n# Binning Cabin Number to 10 Categories and Label Encoding as ordering likely matters\ntrain_df.Cabin_Number, bins = pd.cut(train_df.Cabin_Number, 10, retbins=True)\ntest_df.Cabin_Number= pd.cut(test_df.Cabin_Number, bins =bins)\n\nlabel_encoder = LabelEncoder()\ntrain_df.Cabin_Number = label_encoder.fit_transform(train_df.Cabin_Number)\ntest_df.Cabin_Number = label_encoder.transform(test_df.Cabin_Number)\n\n\nlabel_encoder = LabelEncoder()\ntrain_df.Cabin_Letter = label_encoder.fit_transform(train_df.Cabin_Letter)\ntest_df.Cabin_Letter = label_encoder.transform(test_df.Cabin_Letter)\n\n# Target Encoding Last Name with Age\n# Will essentially place the mean age by last name for each last name\n# trials\n\n# Target Encoding Last Name with Age\n# Will essentially place the mean age by last name for each last name\n# \n\ntarget_encoder = TargetEncoder(smoothing=0.0, min_samples_leaf=0)\n\n# This isn't best practice, grouping Train-Test data here could be data leakage\n# The reason why its not should be similiar to Label Encoded example, I'm essentially just renaming the data with average age as a weight. \ntarget_encoder.fit( pd.concat([train_df.Last_Name, test_df.Last_Name]), pd.concat([train_df.Age, test_df.Age]))\ntrain_df['Last_Name'] = target_encoder.transform(train_df.Last_Name)\ntest_df['Last_Name'] = target_encoder.transform(test_df.Last_Name)\n\n# train_df['Last_Name'] = target_encoder.fit_transform(train_df.Last_Name, train_df.Age)\n# test_df['Last_Name'] = target_encoder.transform(test_df.Last_Name)\n\n\n\ntrain_df = pd.get_dummies(train_df)\ntest_df = pd.get_dummies(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.033394Z","iopub.execute_input":"2022-07-21T17:55:41.033698Z","iopub.status.idle":"2022-07-21T17:55:41.166895Z","shell.execute_reply.started":"2022-07-21T17:55:41.033674Z","shell.execute_reply":"2022-07-21T17:55:41.165841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.values.copy()\ny = train_labels.copy()\nX_train, X_validation, Y_train, Y_validation = train_test_split(X,y, test_size = .1, random_state = 2516, shuffle=True )","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.170474Z","iopub.execute_input":"2022-07-21T17:55:41.171114Z","iopub.status.idle":"2022-07-21T17:55:41.180769Z","shell.execute_reply.started":"2022-07-21T17:55:41.171079Z","shell.execute_reply":"2022-07-21T17:55:41.179099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc = StandardScaler()\n\nX_train = sc.fit_transform(X_train)\nX_validation = sc.transform (X_validation)\npassengerId = test_df.iloc[:, :1]\ntest_df = sc.transform(test_df.iloc[:, 1:])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.182136Z","iopub.execute_input":"2022-07-21T17:55:41.182406Z","iopub.status.idle":"2022-07-21T17:55:41.195280Z","shell.execute_reply.started":"2022-07-21T17:55:41.182380Z","shell.execute_reply":"2022-07-21T17:55:41.193778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\n# Y_pred = logreg.predict(X_validation)\nacc_log = round(logreg.score(X_validation, Y_validation) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.196652Z","iopub.execute_input":"2022-07-21T17:55:41.197133Z","iopub.status.idle":"2022-07-21T17:55:41.217228Z","shell.execute_reply.started":"2022-07-21T17:55:41.197104Z","shell.execute_reply":"2022-07-21T17:55:41.216326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coeff_df = pd.DataFrame(train_df.columns)\ncoeff_df.columns = ['Feature']\ncoeff_df[\"Correlation\"] = pd.Series(logreg.coef_[0])\n\ncoeff_df.sort_values(by='Correlation', ascending=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.218714Z","iopub.execute_input":"2022-07-21T17:55:41.219278Z","iopub.status.idle":"2022-07-21T17:55:41.238769Z","shell.execute_reply.started":"2022-07-21T17:55:41.219242Z","shell.execute_reply":"2022-07-21T17:55:41.237787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\n\nsvc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_validation)\nacc_svc = round(svc.score(X_validation, Y_validation) * 100, 2)\nacc_svc\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.243117Z","iopub.execute_input":"2022-07-21T17:55:41.243454Z","iopub.status.idle":"2022-07-21T17:55:41.315967Z","shell.execute_reply.started":"2022-07-21T17:55:41.243420Z","shell.execute_reply":"2022-07-21T17:55:41.315083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# K-Nearest Neighbors\n\nknn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_validation)\nacc_knn = round(knn.score(X_validation, Y_validation) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.317291Z","iopub.execute_input":"2022-07-21T17:55:41.317789Z","iopub.status.idle":"2022-07-21T17:55:41.341187Z","shell.execute_reply.started":"2022-07-21T17:55:41.317756Z","shell.execute_reply":"2022-07-21T17:55:41.340229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gaussian Naive Bayes\n\ngaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_validation)\nacc_gaussian = round(gaussian.score(X_validation, Y_validation) * 100, 2)\nacc_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.342326Z","iopub.execute_input":"2022-07-21T17:55:41.342549Z","iopub.status.idle":"2022-07-21T17:55:41.353934Z","shell.execute_reply.started":"2022-07-21T17:55:41.342529Z","shell.execute_reply":"2022-07-21T17:55:41.352979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\n\nsgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_validation)\nacc_sgd = round(sgd.score(X_validation, Y_validation) * 100, 2)\nacc_sgd","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.355836Z","iopub.execute_input":"2022-07-21T17:55:41.356755Z","iopub.status.idle":"2022-07-21T17:55:41.374296Z","shell.execute_reply.started":"2022-07-21T17:55:41.356722Z","shell.execute_reply":"2022-07-21T17:55:41.373200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decision Tree\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_validation)\nacc_decision_tree = round(decision_tree.score(X_validation, Y_validation) * 100, 2)\nacc_decision_tree\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.375411Z","iopub.execute_input":"2022-07-21T17:55:41.375712Z","iopub.status.idle":"2022-07-21T17:55:41.388189Z","shell.execute_reply.started":"2022-07-21T17:55:41.375688Z","shell.execute_reply":"2022-07-21T17:55:41.386876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Random Forest\n\nrandom_forest = RandomForestClassifier(n_estimators=150)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_validation)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_validation, Y_validation) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.389623Z","iopub.execute_input":"2022-07-21T17:55:41.390027Z","iopub.status.idle":"2022-07-21T17:55:41.835811Z","shell.execute_reply.started":"2022-07-21T17:55:41.389989Z","shell.execute_reply":"2022-07-21T17:55:41.834882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Logistic Regression',  'Support Vector Machines', 'KNN', \n              'Naive Bayes', 'Stochastic Gradient Decent', 'Decision Tree', 'Random Forest'],\n    'Score': [acc_log, acc_svc, acc_knn,  \n            acc_gaussian, acc_sgd, acc_decision_tree, acc_random_forest]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.837485Z","iopub.execute_input":"2022-07-21T17:55:41.837963Z","iopub.status.idle":"2022-07-21T17:55:41.853200Z","shell.execute_reply.started":"2022-07-21T17:55:41.837926Z","shell.execute_reply":"2022-07-21T17:55:41.851850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final Prediction\n\nY_pred = random_forest.predict(test_df)\n\nsubmission = pd.DataFrame({\n        \"PassengerId\": passengerId.values.flatten(),\n        \"Survived\": Y_pred\n    })","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.854785Z","iopub.execute_input":"2022-07-21T17:55:41.855118Z","iopub.status.idle":"2022-07-21T17:55:41.898855Z","shell.execute_reply.started":"2022-07-21T17:55:41.855089Z","shell.execute_reply":"2022-07-21T17:55:41.897835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:55:41.900196Z","iopub.execute_input":"2022-07-21T17:55:41.900472Z","iopub.status.idle":"2022-07-21T17:55:41.913341Z","shell.execute_reply.started":"2022-07-21T17:55:41.900445Z","shell.execute_reply":"2022-07-21T17:55:41.911717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:13:58.185199Z","iopub.execute_input":"2022-07-21T18:13:58.185525Z","iopub.status.idle":"2022-07-21T18:13:58.194308Z","shell.execute_reply.started":"2022-07-21T18:13:58.185500Z","shell.execute_reply":"2022-07-21T18:13:58.193302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:13:08.995788Z","iopub.execute_input":"2022-07-21T18:13:08.996149Z","iopub.status.idle":"2022-07-21T18:13:09.010029Z","shell.execute_reply.started":"2022-07-21T18:13:08.996121Z","shell.execute_reply":"2022-07-21T18:13:09.009033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:11:09.065541Z","iopub.execute_input":"2022-07-21T18:11:09.065957Z","iopub.status.idle":"2022-07-21T18:11:09.073669Z","shell.execute_reply.started":"2022-07-21T18:11:09.065926Z","shell.execute_reply":"2022-07-21T18:11:09.072309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}