{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install lightgbm","metadata":{"id":"IvOnBuGf48ZA","outputId":"1a822f70-35c2-4868-a38d-52605c51750d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport sklearn as sk\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn import linear_model\nfrom sklearn.experimental import enable_iterative_imputer \nfrom sklearn.impute import IterativeImputer\nfrom sklearn.pipeline import Pipeline\nimport lightgbm as lgb\nfrom sklearn.neighbors import KernelDensity\nimport statsmodels.formula.api as sm","metadata":{"id":"5WFz3HlHuJFd","execution":{"iopub.status.busy":"2022-08-10T19:15:23.400419Z","iopub.execute_input":"2022-08-10T19:15:23.400734Z","iopub.status.idle":"2022-08-10T19:15:23.405910Z","shell.execute_reply.started":"2022-08-10T19:15:23.400710Z","shell.execute_reply":"2022-08-10T19:15:23.405279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train = pd.read_csv(\"/content/drive/MyDrive/Kaggle/train.csv\",sep = ',', index_col= 0)\n#test = pd.read_csv('/content/drive/MyDrive/Kaggle/test.csv', sep = ',', index_col= 0)\n\ntrain = pd.read_csv(\"../input/titanic/train.csv\",sep = ',', index_col= 0)\ntest = pd.read_csv('../input/titanic/test.csv', sep = ',', index_col= 0)\ntrain.info()","metadata":{"id":"wukAn-0PwJt8","outputId":"1c4d35c2-3832-4bcd-d9fa-e41193f16572","execution":{"iopub.status.busy":"2022-08-10T19:05:16.939765Z","iopub.execute_input":"2022-08-10T19:05:16.940105Z","iopub.status.idle":"2022-08-10T19:05:16.988845Z","shell.execute_reply.started":"2022-08-10T19:05:16.940078Z","shell.execute_reply":"2022-08-10T19:05:16.987519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, y_train = train.drop(['Survived', 'Name', 'Ticket', 'Cabin', 'Embarked'], axis = 1), train.loc[:,['Survived']]\nX_test = test.drop(['Name', 'Ticket', 'Cabin', 'Embarked'], axis = 1)\nX_test.Fare.fillna(0, inplace=True)\nbefore = X_train.loc[:, ['Age', 'Fare']]\nprint(np.sum(X_test.Fare.isna()))","metadata":{"id":"4USUZNrJlQty","outputId":"a2b9947f-1a82-4afc-cbd1-387b73f45ec6","execution":{"iopub.status.busy":"2022-08-10T19:05:20.025791Z","iopub.execute_input":"2022-08-10T19:05:20.026253Z","iopub.status.idle":"2022-08-10T19:05:20.041502Z","shell.execute_reply.started":"2022-08-10T19:05:20.026213Z","shell.execute_reply":"2022-08-10T19:05:20.040364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the categorical features\n#train.Pclass, test.Pclass = pd.Categorical(pd.factorize(train.Pclass)[0]), pd.Categorical(pd.factorize(test.Pclass)[0])\n#train.Sex, test.Sex = pd.Categorical(pd.factorize(train.Sex)[0]),  pd.Categorical(pd.factorize(test.Sex)[0])\n\ntrain.Pclass, test.Pclass = pd.Categorical(train.Pclass), pd.Categorical(test.Pclass)\ntrain.Sex, test.Sex = pd.Categorical(train.Sex), pd.Categorical(test.Sex)","metadata":{"id":"uGhOkc89kd0-","execution":{"iopub.status.busy":"2022-08-10T19:05:22.555530Z","iopub.execute_input":"2022-08-10T19:05:22.555961Z","iopub.status.idle":"2022-08-10T19:05:22.564434Z","shell.execute_reply.started":"2022-08-10T19:05:22.555899Z","shell.execute_reply":"2022-08-10T19:05:22.563708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Iterative Imputation for Missing Values in Age**\nFit the imputer to your non-categorical data to impute values for Age.\n\n*Note*: there was a negative term in the imputed data. I didn't look to far into it, and just applied a bandaid solution by zeroing out the negative values since there weren't that many (like 1 or 2).","metadata":{}},{"cell_type":"code","source":"imputer_clf = sk.impute.IterativeImputer(missing_values = np.nan, verbose=1)\nimputer_clf.fit(X_train.drop(['Pclass', 'Sex'], axis = 1))\nimputed_values = imputer_clf.transform(X_train.drop(['Pclass', 'Sex'], axis = 1))\nimputed_values = np.where(imputed_values < 0, 0, imputed_values)\nimputed_values","metadata":{"id":"nsevSS0UpUuk","outputId":"9bbd9c78-4d0a-49a7-82ab-d3c97695d3ad","execution":{"iopub.status.busy":"2022-08-10T19:05:27.159647Z","iopub.execute_input":"2022-08-10T19:05:27.159969Z","iopub.status.idle":"2022-08-10T19:05:27.198435Z","shell.execute_reply.started":"2022-08-10T19:05:27.159916Z","shell.execute_reply":"2022-08-10T19:05:27.197498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace the NaN values in Age with the imputed values.","metadata":{}},{"cell_type":"code","source":"X_train.Age = imputed_values[:, 0]\nafter = X_train.loc[:, ['Age', 'Fare']]\nnp.sum(X_train.Age.isna())","metadata":{"id":"w9qNJLCRtmSU","outputId":"09678421-241f-4322-99df-6ad1612584c1","execution":{"iopub.status.busy":"2022-08-10T19:05:30.959032Z","iopub.execute_input":"2022-08-10T19:05:30.959417Z","iopub.status.idle":"2022-08-10T19:05:30.968706Z","shell.execute_reply.started":"2022-08-10T19:05:30.959388Z","shell.execute_reply":"2022-08-10T19:05:30.967859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Kernel Desity Estimation for Missing Values**\nTesting out KDE for missing values in age, as the imputed values narrowed the distrubtion around the mean too much.\nFit the density estimation around the non-NaN Age values (this makes a lot of assumptions about the underlying distrubtion, but in the I'm just exploring), then use the estimated distribution to predict Age values and replacing NaN's.","metadata":{}},{"cell_type":"code","source":"density_clf = KernelDensity(bandwidth = 5, kernel='gaussian')\nage = X_train.Age.dropna()\nestimator = density_clf.fit(age.to_numpy().reshape(-1, 1))\nkde = before.copy()\nsamples = estimator.sample(177, random_state = 42)\nsamples = np.where(samples < 0, 0, samples)\nkde.loc[kde.Age.isnull(), 'Age'] = samples.flatten()","metadata":{"id":"gsHp_e_D61nK","execution":{"iopub.status.busy":"2022-08-10T19:05:36.325200Z","iopub.execute_input":"2022-08-10T19:05:36.326242Z","iopub.status.idle":"2022-08-10T19:05:36.334227Z","shell.execute_reply.started":"2022-08-10T19:05:36.326213Z","shell.execute_reply":"2022-08-10T19:05:36.333050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Comparison of Imputed vs KDE Distributions (Before and After)**","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows = 2)\nsns.histplot(before.loc[:, ['Age']], kde = True,ax=axs[0][0])\nsns.histplot(after.loc[: , ['Age']], kde = True, ax = axs[0][1])\nsns.histplot(samples, ax = axs[1][0], kde = True)\nsns.histplot(kde.Age, kde = True,ax = axs[1][1])\nplt.show()","metadata":{"id":"xsuOVbu33ZJl","outputId":"3de55b8d-d9e9-47ba-ba0e-013cf54fd5b7","execution":{"iopub.status.busy":"2022-08-10T19:05:39.106982Z","iopub.execute_input":"2022-08-10T19:05:39.107311Z","iopub.status.idle":"2022-08-10T19:05:39.751003Z","shell.execute_reply.started":"2022-08-10T19:05:39.107285Z","shell.execute_reply":"2022-08-10T19:05:39.749727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.Age = np.where(kde.Age < 0, 0, kde.Age)\nX_test.loc[X_test.Age.isnull(), 'Age'] = estimator.sample(np.sum(X_test.Age.isna()), random_state = 42).flatten()","metadata":{"id":"8nkrrAjCJbxU","execution":{"iopub.status.busy":"2022-08-10T19:06:41.265739Z","iopub.execute_input":"2022-08-10T19:06:41.266089Z","iopub.status.idle":"2022-08-10T19:06:41.275302Z","shell.execute_reply.started":"2022-08-10T19:06:41.266064Z","shell.execute_reply":"2022-08-10T19:06:41.274400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_temp = X_train.copy()\nX_temp['SexStatus'] = pd.factorize(X_temp.Pclass)[0]*pd.factorize(X_temp.Sex)[0]\nX_temp.insert(0, 'Survived', train.Survived)\n\n#X_temp['Family'] = (4-X_temp.Pclass)*(X_temp.SibSp + X_temp.Parch)**pd.factorize(X_temp.Sex)[0]\n#X_temp['FTree'] = X_temp.Parch*X_temp.SibSp\n#X_temp['logFare'] = np.log(X_train.Fare + 1)\n#X_temp['logAge'] = np.log(X_train.Age + 1)","metadata":{"id":"-WuB4MdrZsXs","execution":{"iopub.status.busy":"2022-08-10T19:06:43.583117Z","iopub.execute_input":"2022-08-10T19:06:43.583449Z","iopub.status.idle":"2022-08-10T19:06:43.592400Z","shell.execute_reply.started":"2022-08-10T19:06:43.583423Z","shell.execute_reply":"2022-08-10T19:06:43.590225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inference for Feature Selection/Engineering**  \nAn attempt to apply some more traditional statistical inference techniques for the purposes of feature selection and engineering.","metadata":{}},{"cell_type":"code","source":"explore_log =  sm.logit(\"Survived ~ Pclass + Age + Sex + SibSp + SexStatus\", data=X_temp).fit()\nprint(explore_log.summary())","metadata":{"id":"f9aZZVnIcXip","outputId":"6c1107c5-bd49-4621-fe9d-ba78423ba8ab","execution":{"iopub.status.busy":"2022-08-10T19:06:47.168441Z","iopub.execute_input":"2022-08-10T19:06:47.168748Z","iopub.status.idle":"2022-08-10T19:06:47.219447Z","shell.execute_reply.started":"2022-08-10T19:06:47.168724Z","shell.execute_reply":"2022-08-10T19:06:47.218706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Playing around with some engineered features.","metadata":{}},{"cell_type":"code","source":"# Testing out some custom features\nX_train['SexStatus'] = pd.factorize(X_train.Pclass)[0]*pd.factorize(X_train.Sex)[0]\nX_test['SexStatus'] = pd.factorize(X_test.Pclass)[0]*pd.factorize(X_test.Sex)[0]\n\n#X_train['Family'] = (4-X_train.Pclass)*(X_train.SibSp + X_train.Parch)**pd.factorize(X_train.Sex)[0]\n#X_test['Family'] = (4-X_test.Pclass)*(X_test.SibSp + X_test.Parch)**pd.factorize(X_test.Sex)[0]\n\n#X_train['FTree'] = X_train.Parch*X_train.SibSp\n#X_test['FTree'] = X_test.Parch*X_test.SibSp\n\n# Log features don't seem to contribute much\n#X_train['logFare'] = np.log(X_train.Fare + 1)\n#X_test['logFare'] = np.log(X_test.Fare + 1)\n#X_train['logAge'] = np.log(X_train.Age + 1)\n#X_test['logAge'] = np.log(X_test.Age + 1)\n\nX_train = pd.get_dummies(X_train, prefix = ['Class', 'Sex'], columns = ['Pclass', 'Sex'], drop_first = True)\nX_test = pd.get_dummies(X_test, prefix = ['Class', 'Sex'], columns = ['Pclass', 'Sex'], drop_first = True)\n\nX_train = X_train.drop(['Parch', 'Fare'], axis = 1)\nX_test = X_test.drop(['Parch', 'Fare'], axis = 1)","metadata":{"id":"loXpm83gZrya","execution":{"iopub.status.busy":"2022-08-10T19:07:08.510042Z","iopub.execute_input":"2022-08-10T19:07:08.510361Z","iopub.status.idle":"2022-08-10T19:07:08.527711Z","shell.execute_reply.started":"2022-08-10T19:07:08.510336Z","shell.execute_reply":"2022-08-10T19:07:08.526368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().any(), X_test.isna().any()","metadata":{"id":"2YScWIgBTDpd","outputId":"bc5d83ba-8bc8-44f8-9be0-2aa0a1558bd6","execution":{"iopub.status.busy":"2022-08-10T19:07:12.022911Z","iopub.execute_input":"2022-08-10T19:07:12.023271Z","iopub.status.idle":"2022-08-10T19:07:12.034389Z","shell.execute_reply.started":"2022-08-10T19:07:12.023244Z","shell.execute_reply":"2022-08-10T19:07:12.033243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Logistic Regression with Multiple Regularizers**\nPlaying around with various implementations of logistic regression with $L_2$, and elastic net.","metadata":{"id":"-zpQ8RjI3Oxk"}},{"cell_type":"code","source":"X_train","metadata":{"id":"tCsk4Ga5vlaF","outputId":"0a968462-232c-47ab-93f3-63b86fd72fe8","execution":{"iopub.status.busy":"2022-08-10T19:07:16.914399Z","iopub.execute_input":"2022-08-10T19:07:16.915220Z","iopub.status.idle":"2022-08-10T19:07:16.933252Z","shell.execute_reply.started":"2022-08-10T19:07:16.915192Z","shell.execute_reply":"2022-08-10T19:07:16.932267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_clf = linear_model.LogisticRegressionCV(Cs = [0.00001, 0.0001, 0.001, 0.01, 0.1, 1, 10, 100, 1000], \n                                                 cv = 10, random_state = 42, n_jobs = -1, \n                                                 max_iter=1000)\nlogistic_fit = logistic_clf.fit(X_train, y_train.to_numpy().ravel())","metadata":{"id":"BnLz0fzcedps","execution":{"iopub.status.busy":"2022-08-10T19:30:30.957364Z","iopub.execute_input":"2022-08-10T19:30:30.957696Z","iopub.status.idle":"2022-08-10T19:30:31.227256Z","shell.execute_reply.started":"2022-08-10T19:30:30.957671Z","shell.execute_reply":"2022-08-10T19:30:31.226278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(logistic_clf.scores_)\n# Should ROC/AUC\nmetrics.RocCurveDisplay.from_estimator(logistic_clf, X_train, y_train)\nplt.show()","metadata":{"id":"a8eF8vuGu3lw","outputId":"7121b58b-392a-4d4a-c723-a334a0a06756","execution":{"iopub.status.busy":"2022-08-10T19:30:34.144452Z","iopub.execute_input":"2022-08-10T19:30:34.144780Z","iopub.status.idle":"2022-08-10T19:30:34.291474Z","shell.execute_reply.started":"2022-08-10T19:30:34.144753Z","shell.execute_reply":"2022-08-10T19:30:34.289965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_predictions = logistic_fit.predict(X_test)","metadata":{"id":"cELiNrwXUyej","execution":{"iopub.status.busy":"2022-08-10T19:30:50.893244Z","iopub.execute_input":"2022-08-10T19:30:50.893564Z","iopub.status.idle":"2022-08-10T19:30:50.899893Z","shell.execute_reply.started":"2022-08-10T19:30:50.893538Z","shell.execute_reply":"2022-08-10T19:30:50.898996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'PassengerId': X_test.index,\n               'Survived': logistic_predictions})\nsubmission.set_index(['PassengerId'])","metadata":{"id":"7LvZOf5mWtd2","outputId":"5649b8f8-a738-4a39-b823-9f601aae196c","execution":{"iopub.status.busy":"2022-08-10T19:30:55.694630Z","iopub.execute_input":"2022-08-10T19:30:55.694989Z","iopub.status.idle":"2022-08-10T19:30:55.709280Z","shell.execute_reply.started":"2022-08-10T19:30:55.694962Z","shell.execute_reply":"2022-08-10T19:30:55.708213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)\n#!cp predictions_enet.csv \"drive/My Drive/Kaggle\"","metadata":{"id":"uW4DRoeyXIz2","execution":{"iopub.status.busy":"2022-08-10T19:30:59.581109Z","iopub.execute_input":"2022-08-10T19:30:59.581482Z","iopub.status.idle":"2022-08-10T19:30:59.589865Z","shell.execute_reply.started":"2022-08-10T19:30:59.581457Z","shell.execute_reply":"2022-08-10T19:30:59.589111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.loc[:, ['Age', 'Fare']]\nprint(temp.Fare.value_counts())\nsns.pairplot(temp)\nplt.show()","metadata":{"id":"yy5MVSsjwtVi","outputId":"48fafc0f-28c1-4bbb-d9d2-aaad132157fa","execution":{"iopub.status.busy":"2022-07-27T19:06:18.549461Z","iopub.execute_input":"2022-07-27T19:06:18.550445Z","iopub.status.idle":"2022-07-27T19:06:19.894003Z","shell.execute_reply.started":"2022-07-27T19:06:18.550405Z","shell.execute_reply":"2022-07-27T19:06:19.892722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.catplot(y='Ticket', kind = 'count', data = train)\nsns.countplot(data=train, x=\"Pclass\", hue='Survived')\n#sns.catplot(x=\"Pclass\", kind=\"count\", data=train)\nplt.show()\nsns.countplot(data=train, x=\"SibSp\", hue='Survived')\n#sns.catplot(x=\"SibSp\", kind=\"count\", data=train)\nplt.show()\n","metadata":{"id":"TyaStqvT1oo3","outputId":"3a95b3a1-816d-4ba1-df27-50b07eab4099","execution":{"iopub.status.busy":"2022-07-27T19:06:31.215492Z","iopub.execute_input":"2022-07-27T19:06:31.215905Z","iopub.status.idle":"2022-07-27T19:06:31.548465Z","shell.execute_reply.started":"2022-07-27T19:06:31.215869Z","shell.execute_reply":"2022-07-27T19:06:31.547556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **LightGBM**\nMessing around with a lightgbm implementation. Might be sensitive to full featured data, going to retry on reduced data inferred in logistic regression process.  \n(Disabled in Kaggle Notebooks)","metadata":{"id":"qdluvzBc5PTE"}},{"cell_type":"raw","source":"lgb_train = lgb.Dataset(X_train, label = y_train)\n# num_round = 10\nlrs = [0.5, 0.1, 0.01, 0.001]\ncv_models = []","metadata":{"id":"K6GjWd7H5TaV"}},{"cell_type":"raw","source":"for eta in lrs:\n    num_round = 100\n    param = {'num_leaves': 31, \n        'learning_rate': eta,\n        'objective': 'binary', \n        'metric': ['auc', 'binary_logloss']}\n    \n    model = lgb.cv(param, lgb_train, num_round, nfold=10, verbose_eval=True)\n    cv_models.append(model)\n    print(25*'-',f'Finished training on LEARNING RATE {eta}',25*'-')","metadata":{"id":"6g9-Zxmr8z4e"}},{"cell_type":"raw","source":"count = 0\nfor m in cv_models:\n    t = np.mean(m['auc-mean'])\n    print(f'Average AUC for Model {count}: {t}')\n    count+=1","metadata":{"id":"TP7BusOF_I0M","outputId":"83e5dddb-3fb3-4a04-8554-72ce2bf47320"}},{"cell_type":"raw","source":"param = {'num_leaves': 31, \n        'learning_rate': 0.1,\n        'objective': 'binary', \n        'metric': ['auc', 'binary_logloss']\n}\n\nbest_model = lgb.train(param, lgb_train, num_round)","metadata":{"id":"1aMX58rZELs_"}},{"cell_type":"raw","source":"predictions = best_model.predict(X_test)","metadata":{"id":"aafg9EpAESU4"}},{"cell_type":"raw","source":"zop = np.where(predictions > 0.6, 1, 0)\nfinal = pd.DataFrame({'PassengerId': np.arange(892, 1309 + 1),\n               'Survived': zop})\nfinal.set_index(['PassengerId'])","metadata":{"id":"2aiN97oxFZYf","outputId":"61bbba62-e77b-4ef8-db6f-f1c9ed9972c3"}},{"cell_type":"raw","source":"final.to_csv('predictions.csv', index = 0)\n!cp predictions.csv \"drive/My Drive/Kaggle\"","metadata":{"id":"r-ZfJqDTMifD"}},{"cell_type":"raw","source":"plt.hist(predictions)\nplt.show()","metadata":{"id":"7woh-Q2XHRTe","outputId":"0a283fcf-ee01-4c97-a006-96181d5c0cae"}},{"cell_type":"raw","source":"zop","metadata":{"id":"gjMuEG1DHrmj","outputId":"6e29dee3-965b-49de-8e80-636f4f24bf8e"}}]}