{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q fastai\nfrom fastai.imports import *\nfrom fastai.tabular.all import *\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeClassifier, export_graphviz\nfrom sklearn.metrics import mean_absolute_error\nimport graphviz","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:06.804149Z","iopub.execute_input":"2022-08-07T23:39:06.804596Z","iopub.status.idle":"2022-08-07T23:39:17.594395Z","shell.execute_reply.started":"2022-08-07T23:39:06.804558Z","shell.execute_reply":"2022-08-07T23:39:17.593000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/titanic')\ndf = pd.read_csv(path/\"train.csv\")\ntst_df = pd.read_csv(path/\"test.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.597071Z","iopub.execute_input":"2022-08-07T23:39:17.597549Z","iopub.status.idle":"2022-08-07T23:39:17.647232Z","shell.execute_reply.started":"2022-08-07T23:39:17.597479Z","shell.execute_reply":"2022-08-07T23:39:17.646377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.648740Z","iopub.execute_input":"2022-08-07T23:39:17.649070Z","iopub.status.idle":"2022-08-07T23:39:17.658188Z","shell.execute_reply.started":"2022-08-07T23:39:17.649041Z","shell.execute_reply":"2022-08-07T23:39:17.657343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def enhance_features(df):\n    df['LogFare'] = np.log1p(df['Fare'])\n    df[\"Deck\"] = df.Cabin.str[0].map(dict(A=\"ABC\", B=\"ABC\", C=\"ABC\", D=\"DE\", E=\"DE\", F=\"FG\", G=\"FG\"))\n    df[\"Family\"] = df.SibSp + df.Parch\n    df[\"Alone\"] = df.Family == 1\n    df['TicketFreq'] = df.groupby('Ticket')['Ticket'].transform('count')\n    df['Title'] = df.Name.str.split(', ', expand=True)[1].str.split('.', expand=True)[0]\n    df['Title'] = df.Title.map(dict(Mr=\"Mr\",Miss=\"Miss\",Mrs=\"Mrs\",Master=\"Master\")).value_counts(dropna=False)\nenhance_features(df)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.660849Z","iopub.execute_input":"2022-08-07T23:39:17.661331Z","iopub.status.idle":"2022-08-07T23:39:17.711678Z","shell.execute_reply.started":"2022-08-07T23:39:17.661302Z","shell.execute_reply":"2022-08-07T23:39:17.710848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits = RandomSplitter()(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.712920Z","iopub.execute_input":"2022-08-07T23:39:17.713395Z","iopub.status.idle":"2022-08-07T23:39:17.717804Z","shell.execute_reply.started":"2022-08-07T23:39:17.713365Z","shell.execute_reply":"2022-08-07T23:39:17.716971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = TabularPandas(\n    df, \n    splits=splits, \n    procs=[Categorify, FillMissing, Normalize], \n    cat_names=[\"Sex\",\"Pclass\",\"Embarked\",\"Deck\", \"Title\"], \n    cont_names=['Age', 'SibSp', 'Parch', 'LogFare', 'Alone', 'TicketFreq', 'Family'], \n    y_names=\"Survived\", \n    y_block=CategoryBlock()\n).dataloaders(path=\".\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.718880Z","iopub.execute_input":"2022-08-07T23:39:17.719564Z","iopub.status.idle":"2022-08-07T23:39:17.778982Z","shell.execute_reply.started":"2022-08-07T23:39:17.719529Z","shell.execute_reply":"2022-08-07T23:39:17.778061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learner = tabular_learner(dls, metrics=accuracy, layers=[10, 10])\nlearner.lr_find(suggest_funcs=(slide, valley))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:17.780215Z","iopub.execute_input":"2022-08-07T23:39:17.780766Z","iopub.status.idle":"2022-08-07T23:39:19.517135Z","shell.execute_reply.started":"2022-08-07T23:39:17.780727Z","shell.execute_reply":"2022-08-07T23:39:19.516033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = 0.03\nepochs = 16\nlearner.fit(epochs, lr=lr)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:19.518743Z","iopub.execute_input":"2022-08-07T23:39:19.519332Z","iopub.status.idle":"2022-08-07T23:39:21.770504Z","shell.execute_reply.started":"2022-08-07T23:39:19.519299Z","shell.execute_reply":"2022-08-07T23:39:21.769329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_df = pd.read_csv(path/\"test.csv\")\ntesting_df[\"Fare\"] = testing_df.Fare.fillna(0)\nenhance_features(testing_df)\ntesting_df","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:21.772093Z","iopub.execute_input":"2022-08-07T23:39:21.772680Z","iopub.status.idle":"2022-08-07T23:39:21.814903Z","shell.execute_reply.started":"2022-08-07T23:39:21.772638Z","shell.execute_reply":"2022-08-07T23:39:21.813801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_dl = learner.dls.test_dl(testing_df)\npreds,_ = learner.get_preds(dl=testing_dl)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:21.819998Z","iopub.execute_input":"2022-08-07T23:39:21.820523Z","iopub.status.idle":"2022-08-07T23:39:21.897531Z","shell.execute_reply.started":"2022-08-07T23:39:21.820459Z","shell.execute_reply":"2022-08-07T23:39:21.896717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = (preds[:, 1] > 0.5).int()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:21.898842Z","iopub.execute_input":"2022-08-07T23:39:21.899351Z","iopub.status.idle":"2022-08-07T23:39:21.903430Z","shell.execute_reply.started":"2022-08-07T23:39:21.899322Z","shell.execute_reply":"2022-08-07T23:39:21.902642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_df[\"Survived\"] = test_predictions\nsubmission_df = testing_df[[\"PassengerId\", \"Survived\"]]\nsubmission_df.to_csv(\"sub1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:21.904669Z","iopub.execute_input":"2022-08-07T23:39:21.905184Z","iopub.status.idle":"2022-08-07T23:39:21.915914Z","shell.execute_reply.started":"2022-08-07T23:39:21.905154Z","shell.execute_reply":"2022-08-07T23:39:21.915057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head sub.csv","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:21.917338Z","iopub.execute_input":"2022-08-07T23:39:21.917989Z","iopub.status.idle":"2022-08-07T23:39:22.986253Z","shell.execute_reply.started":"2022-08-07T23:39:21.917959Z","shell.execute_reply":"2022-08-07T23:39:22.985052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ensemble():\n    learn = tabular_learner(dls, metrics=accuracy, layers=[10, 10])\n    with learn.no_bar(), learn.no_logging():\n        learn.fit(16, lr=0.03)\n    return learn.get_preds(dl=testing_dl)[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:22.988333Z","iopub.execute_input":"2022-08-07T23:39:22.989120Z","iopub.status.idle":"2022-08-07T23:39:22.995867Z","shell.execute_reply.started":"2022-08-07T23:39:22.989082Z","shell.execute_reply":"2022-08-07T23:39:22.994704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learners = [ensemble() for _ in range(20)]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:22.999553Z","iopub.execute_input":"2022-08-07T23:39:23.000705Z","iopub.status.idle":"2022-08-07T23:39:57.448965Z","shell.execute_reply.started":"2022-08-07T23:39:23.000662Z","shell.execute_reply":"2022-08-07T23:39:57.447829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ensamble_preds = torch.stack(learners).mean(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.450644Z","iopub.execute_input":"2022-08-07T23:39:57.451740Z","iopub.status.idle":"2022-08-07T23:39:57.457355Z","shell.execute_reply.started":"2022-08-07T23:39:57.451686Z","shell.execute_reply":"2022-08-07T23:39:57.456294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_df['Survived'] = (ensamble_preds[:,1]>0.5).int()\nsubmission_df = testing_df[['PassengerId','Survived']]\nsubmission_df.to_csv('sub2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.459052Z","iopub.execute_input":"2022-08-07T23:39:57.459574Z","iopub.status.idle":"2022-08-07T23:39:57.473846Z","shell.execute_reply.started":"2022-08-07T23:39:57.459533Z","shell.execute_reply":"2022-08-07T23:39:57.472861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_df, validation_df = train_test_split(df, test_size=0.25)\nmodes = df.mode().iloc[0]\ndef proc_data(df):\n    df['Fare'] = df.Fare.fillna(0)\n    df.fillna(modes, inplace=True)\n    df['LogFare'] = np.log1p(df['Fare'])\n    df['Embarked'] = pd.Categorical(df.Embarked)\n    df['Sex'] = pd.Categorical(df.Sex)\n\nproc_data(training_df)\nproc_data(validation_df)\nproc_data(testing_df)\n\ncats=[\"Sex\",\"Embarked\"]\nconts=['Age', 'SibSp', 'Parch', 'LogFare',\"Pclass\"]\ndep=\"Survived\"\n\ntraining_df[cats] = training_df[cats].apply(lambda x: x.cat.codes)\nvalidation_df[cats] = validation_df[cats].apply(lambda x: x.cat.codes)\ntesting_df[cats] = testing_df[cats].apply(lambda x: x.cat.codes)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.475880Z","iopub.execute_input":"2022-08-07T23:39:57.476280Z","iopub.status.idle":"2022-08-07T23:39:57.527789Z","shell.execute_reply.started":"2022-08-07T23:39:57.476241Z","shell.execute_reply":"2022-08-07T23:39:57.526467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def xs_y(df):\n    xs = df[cats+conts].copy()\n    return xs,df[dep] if dep in df else None\n\ntrn_xs,trn_y = xs_y(training_df)\nval_xs,val_y = xs_y(validation_df)\ntst_xs, _ = xs_y(testing_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.529521Z","iopub.execute_input":"2022-08-07T23:39:57.530082Z","iopub.status.idle":"2022-08-07T23:39:57.541444Z","shell.execute_reply.started":"2022-08-07T23:39:57.530037Z","shell.execute_reply":"2022-08-07T23:39:57.540530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrf = RandomForestClassifier(100, min_samples_leaf=5)\nrf.fit(trn_xs, trn_y);\nmean_absolute_error(val_y, rf.predict(val_xs))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.543001Z","iopub.execute_input":"2022-08-07T23:39:57.544229Z","iopub.status.idle":"2022-08-07T23:39:57.743503Z","shell.execute_reply.started":"2022-08-07T23:39:57.544196Z","shell.execute_reply":"2022-08-07T23:39:57.741961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_df[\"Survived\"] = rf.predict(tst_xs)\nsub_df = tst_df[['PassengerId','Survived']]\nsub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.745403Z","iopub.execute_input":"2022-08-07T23:39:57.745892Z","iopub.status.idle":"2022-08-07T23:39:57.775092Z","shell.execute_reply.started":"2022-08-07T23:39:57.745848Z","shell.execute_reply":"2022-08-07T23:39:57.774294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_tree(t, df, size=10, ratio=0.6, precision=2, **kwargs):\n    s=export_graphviz(t, out_file=None, feature_names=df.columns, filled=True, rounded=True,\n                      special_characters=True, rotate=False, precision=precision, **kwargs)\n    return graphviz.Source(re.sub('Tree {', f'Tree {{ size={size}; ratio={ratio}', s))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.776822Z","iopub.execute_input":"2022-08-07T23:39:57.777234Z","iopub.status.idle":"2022-08-07T23:39:57.783244Z","shell.execute_reply.started":"2022-08-07T23:39:57.777193Z","shell.execute_reply":"2022-08-07T23:39:57.782169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = DecisionTreeClassifier(min_samples_leaf=50)\nm.fit(trn_xs, trn_y)\ndraw_tree(m, trn_xs, size=12)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.784664Z","iopub.execute_input":"2022-08-07T23:39:57.785229Z","iopub.status.idle":"2022-08-07T23:39:57.856051Z","shell.execute_reply.started":"2022-08-07T23:39:57.785198Z","shell.execute_reply":"2022-08-07T23:39:57.854572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(dict(cols=trn_xs.columns, imp=m.feature_importances_)).plot('cols', 'imp', 'barh');","metadata":{"execution":{"iopub.status.busy":"2022-08-07T23:39:57.860199Z","iopub.execute_input":"2022-08-07T23:39:57.860578Z","iopub.status.idle":"2022-08-07T23:39:58.110171Z","shell.execute_reply.started":"2022-08-07T23:39:57.860543Z","shell.execute_reply":"2022-08-07T23:39:58.109353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}