{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"293e6a2b21672c5baa3840a7257e9f0dc93a67f8"},"cell_type":"code","source":"df1 = pd.read_csv('../input/train.csv')\ndf2 = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"278522b5c208869a3ff43373208206dec48b37a8"},"cell_type":"code","source":"df1[\"Sex\"] = df1[\"Sex\"].replace('male', 0)\ndf1[\"Sex\"] = df1[\"Sex\"].replace('female', 1)\n\ndf2[\"Sex\"] = df2[\"Sex\"].replace('male', 0)\ndf2[\"Sex\"] = df2[\"Sex\"].replace('female', 1)\n\ndf1[\"Age\"] = df1[\"Age\"].fillna(df1[\"Age\"].dropna().median())\ndf2[\"Age\"] = df2[\"Age\"].fillna(df1[\"Age\"].dropna().median())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cb016ccb31dd58eb949f0c9a8f0e7e9adf68348"},"cell_type":"code","source":"df1.loc[ df1['Age'] <= 16, 'Age'] = 0,\ndf1.loc[(df1['Age'] > 16) & (df1['Age'] <= 26), 'Age'] = 1,\ndf1.loc[(df1['Age'] > 26) & (df1['Age'] <= 36), 'Age'] = 2,\ndf1.loc[(df1['Age'] > 36) & (df1['Age'] <= 62), 'Age'] = 3,\ndf1.loc[ df1['Age'] > 62, 'Age'] = 4\n\ndf2.loc[ df2['Age'] <= 16, 'Age'] = 0,\ndf2.loc[(df2['Age'] > 16) & (df2['Age'] <= 26), 'Age'] = 1,\ndf2.loc[(df2['Age'] > 26) & (df2['Age'] <= 36), 'Age'] = 2,\ndf2.loc[(df2['Age'] > 36) & (df2['Age'] <= 62), 'Age'] = 3,\ndf2.loc[ df2['Age'] > 62, 'Age'] = 4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9d0499cc9550c86c869ff9ebfad36b32330e3e3"},"cell_type":"code","source":"df1[\"Fare\"] = df1[\"Fare\"].fillna( df1[\"Fare\"].dropna().median() )\ndf2[\"Fare\"] = df2[\"Fare\"].fillna( df2[\"Fare\"].dropna().median() )\n\ndf1.loc[ df1['Fare'] <= 17, 'Fare'] = 0,\ndf1.loc[(df1['Fare'] > 17) & (df1['Fare'] <= 30), 'Fare'] = 1,\ndf1.loc[(df1['Fare'] > 30) & (df1['Fare'] <= 100), 'Fare'] = 2,\ndf1.loc[ df1['Fare'] > 100, 'Fare'] = 3\n\ndf2.loc[ df2['Fare'] <= 17, 'Fare'] = 0,\ndf2.loc[(df2['Fare'] > 17) & (df2['Fare'] <= 30), 'Fare'] = 1,\ndf2.loc[(df2['Fare'] > 30) & (df2['Fare'] <= 100), 'Fare'] = 2,\ndf2.loc[ df2['Fare'] > 100, 'Fare'] = 3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb4a0019fe632c4160b3b7e1c0c0f9e1797fd576"},"cell_type":"code","source":"\ndf1[\"Embarked\"] = df1[\"Embarked\"].fillna(\"C\") \ndf1[\"Embarked\"] = df1[\"Embarked\"].replace(\"S\", 0)\ndf1[\"Embarked\"] = df1[\"Embarked\"].replace(\"C\", 1)\ndf1[\"Embarked\"] = df1[\"Embarked\"].replace(\"Q\", 2)\n\ndf2[\"Embarked\"] = df2[\"Embarked\"].fillna(\"C\") \ndf2[\"Embarked\"] = df2[\"Embarked\"].replace(\"S\", 0)\ndf2[\"Embarked\"] = df2[\"Embarked\"].replace(\"C\", 1)\ndf2[\"Embarked\"] = df2[\"Embarked\"].replace(\"Q\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03ee7f636d7d19981bc4f4c471dafd0d2df39209"},"cell_type":"code","source":"df1[\"Family Size\"] = df1[\"SibSp\"]  + df1[\"Parch\"] + 1\ndf2[\"Family Size\"] = df2[\"SibSp\"]  + df2[\"Parch\"] + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"746786c2c0cefa8396828ddb04e5e035ddd0285d"},"cell_type":"code","source":"df1['Title'] = df1['Name'].str.extract('([A-Za-z]+)\\.', expand = False)\ndf2['Title'] = df2['Name'].str.extract('([A-Za-z]+)\\.', expand = False)\n\ndf1['Title'].value_counts()\ndf2['Title'].value_counts()\n\ntitle_mapping = {\"Mr\": 0, \"Miss\": 1, \"Mrs\": 2, \n                 \"Master\": 3, \"Dr\": 3, \"Rev\": 3, \"Col\": 3, \"Major\": 3, \"Mlle\": 3,\"Countess\": 3,\n                 \"Ms\": 3, \"Lady\": 3, \"Jonkheer\": 3, \"Don\": 3, \"Dona\" : 3, \"Mme\": 3,\"Capt\": 3,\"Sir\": 3 }\n\ndf1['Title'] = df1['Title'].map(title_mapping)\ndf2['Title'] = df2['Title'].map(title_mapping)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5195ab1924e662b6a3b2b830fa2dae016f8ce28a"},"cell_type":"code","source":"split = 600\n\ndata = np.array(df1)  \ndata_2 = np.array(df2)\nnp.set_printoptions(suppress=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"896b73b498b0f503327e9dc7486f81f5eeae790a"},"cell_type":"code","source":"#   0           1             2      3       4     5     6        7      8       9       10       11\n#PassengerId\tSurvived\tPclass\tName\tSex\t  Age\tSibSp\tParch\tTicket\tFare\tCabin\tEmbarked\n\nX_train = np.asarray( data[0:split, [2, 4, 5, 9, 11, 12, 13] ], dtype=np.float32 )\ny_train = np.asarray( data[0:split, 1:2], dtype=np.float32)\n\nX_val = np.asarray( data[split:891, [2, 4, 5, 9, 11, 12, 13] ], dtype=np.float32)\ny_val = np.asarray( data[split:891, 1:2], dtype=np.float32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7bfb3c0174d5dd178f21621e7fddf83245eb7dc"},"cell_type":"code","source":"#   0           1         2      3     4      5       6       7        8      9       10\n#PassengerId\tPclass\tName\tSex\t  Age\tSibSp\tParch\tTicket\t Fare\tCabin\tEmbarked\n\nX_test = np.asarray( data_2[:, [1, 3, 4, 8, 10, 11, 12] ], dtype=np.float32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1cdabf3f7bf5001c708573c04185a04d989e6b7"},"cell_type":"code","source":"global m\nglobal n    \nm,n = X_train.shape ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd0aba03f2d7dbfaf528593fc3d23fb258afeac0"},"cell_type":"code","source":"def featurenormal(X):\n    mu = np.mean(X, axis = 0)\n    sigma = np.std(X, axis = 0)\n    X = (X - mu) / sigma\n    return X","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cfb874e65e52d9c0a9817cf7e4f55449013e634"},"cell_type":"code","source":"X_train = featurenormal(X_train)\nX_val = featurenormal(X_val)\nX_test = featurenormal(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d91752c3fb3150e710143cf6bb04510a3c5915c1"},"cell_type":"code","source":"ones = np.ones((X_train[:,0:1].shape))\nX_train = np.hstack([ones, X_train])\n\nones_1 = np.ones((X_val[:,0:1].shape))\nX_val = np.hstack([ones_1,X_val])\n\nones_2 = np.ones((X_test[:,0:1].shape))\nX_test = np.hstack([ones_2, X_test])\ninitial_theta = np.zeros((n+1,1)) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3633c36e3d8916d30472354e9672b94106027581"},"cell_type":"code","source":"#z = 0\ndef sigmoid(z):\n    z = 1 / (1 + np.exp(-z))\n    return z\nglobal lam\nlam = .06\ndef costfunction(initial_theta, X_train, y_train, lam):\n    \n    #lam = 0\n    h = sigmoid(X_train.dot(initial_theta))\n    l = np.log(h)\n    l1 = np.log(1-h)\n    J = -(1/m) * ((y_train.T.dot(l)) + ((1-y_train).T.dot(l1))) + (lam/(2*m) * (initial_theta[1:].T.dot(initial_theta[1:])) )\n    \n    #grad = ((1/m) * (X_train.T.dot(h-y_train)))  +  ((lam/m)*initial_theta)\n    J= J.flatten()\n    #grad=grad.flatten()\n    return J","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05c795901a8dc9e43c2a2dcdf012b3dfaf17e71e"},"cell_type":"code","source":"J = costfunction(initial_theta, X_train, y_train, lam)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bee9065852eda64da44060e39d36be8a8fc6a41"},"cell_type":"code","source":"import scipy.optimize as opt\nfinal_theta  = opt.fmin( costfunction, x0=initial_theta, args=(X_train, y_train, lam), maxiter=10000, full_output=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4d1dc3becb95933dbb43624b228d6e284d56f35"},"cell_type":"code","source":"print('Cost with Theta [0,0,0] is :', J)\n#print('Gradient with Theta [0,0,0] is :', grad)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cea6f8ba165913724fe53e498b8a5e45b53f2d2"},"cell_type":"code","source":"print(\"Final theta found by minimization function is :\", final_theta[0])\ny_pred_test = sigmoid(X_test.dot(final_theta[0].T))\ny_pred_test = y_pred_test.reshape(418,1)\ny_pred_test = np.round(y_pred_test, decimals=0)\n\ny_trainpred = sigmoid(X_train.dot(final_theta[0].T))\ny_trainpred = y_trainpred.reshape(split,1)\ny_trainpred = np.round(y_trainpred)\n\nAccuracy = (np.sum(y_trainpred == y_train) / y_trainpred.size) * 100\nprint('Training set accuracy is ', Accuracy)\n\nt = 891 - split\ny_valpred = sigmoid(X_val.dot(final_theta[0].T))\ny_valpred = y_valpred.reshape(t,1)\ny_valpred = np.round(y_valpred)\n\nAccuracy_val = (np.sum(y_valpred == y_val) / y_valpred.size) * 100\nprint('Validation set accuracy is ', Accuracy_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc2179a290a60c5ac3029a66aaf84f858d8b9062"},"cell_type":"code","source":"prediction = pd.read_csv(\"../input/gender_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56237ca690b0b611ed2bed77ef5f1acaa8bb6cdb"},"cell_type":"code","source":"#np.argmax(y_pred_test, axis = 1)\nprediction['Survived'] = np.int64(y_pred_test)\nprediction.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2aa99b7a1560fc12766c4fc89376ae6897f5eea4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}