{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# A fastai implementation for the Amex Competition","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"In this  contest, we are trying to predict the likelihood of loan default.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**Version History**\n* *Version 1* - Getting the notebook setup, importing basic modules, reading the dataset, building a simple model\n* *Version 2* - Update to make only one submission per `customer_ID`\n* *Version 3* - Fixed error splitting large dataframe; added `ReduceLROnPlateau` to callbacks\n* *Version 4* - Adding K-folds cross-validation (Training Gini coef=0.7708 with 2 folds and 1 training epoch)\n* *Version 5* - K-folds cross-validation (n=5) and 20 training epochs per CV max\n* *Version 6* - Pinned learning rate","metadata":{}},{"cell_type":"markdown","source":"## Background\n\n@TODO: update\n\n### For beginners forking this notebook\n\nI highly suggest for anyone interested in solving problems like this to look into the course by Sylvain Gugger and Jeremy Howard. They spent a lot of time building `fastai` into a great library for accessing the latest deep learning techniques.\n\n[Practical Deep Learning for Coders](https://course.fast.ai/)","metadata":{}},{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:25.044331Z","iopub.execute_input":"2022-08-01T14:57:25.044691Z","iopub.status.idle":"2022-08-01T14:57:25.086556Z","shell.execute_reply.started":"2022-08-01T14:57:25.044612Z","shell.execute_reply":"2022-08-01T14:57:25.085758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\nfrom fastai.tabular.all import *\nimport gc\nfrom tqdm import tqdm\nfrom fastprogress.fastprogress import progress_bar\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:25.087982Z","iopub.execute_input":"2022-08-01T14:57:25.088359Z","iopub.status.idle":"2022-08-01T14:57:27.680684Z","shell.execute_reply.started":"2022-08-01T14:57:25.088324Z","shell.execute_reply":"2022-08-01T14:57:27.679836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)\npd.set_option('display.width', None)\npd.set_option('display.max_colwidth', None)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:27.682326Z","iopub.execute_input":"2022-08-01T14:57:27.682655Z","iopub.status.idle":"2022-08-01T14:57:27.718504Z","shell.execute_reply.started":"2022-08-01T14:57:27.682619Z","shell.execute_reply":"2022-08-01T14:57:27.717607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = Path(\"../input/amexfeather\")\nINPUT_DIR.ls()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:27.721178Z","iopub.execute_input":"2022-08-01T14:57:27.721666Z","iopub.status.idle":"2022-08-01T14:57:27.764999Z","shell.execute_reply.started":"2022-08-01T14:57:27.721624Z","shell.execute_reply":"2022-08-01T14:57:27.764168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"WORKING_DIR = Path(\"./\")\nWORKING_DIR.ls()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:27.767658Z","iopub.execute_input":"2022-08-01T14:57:27.767952Z","iopub.status.idle":"2022-08-01T14:57:27.805001Z","shell.execute_reply.started":"2022-08-01T14:57:27.767925Z","shell.execute_reply":"2022-08-01T14:57:27.804163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_feather(INPUT_DIR / 'train_data.ftr')\ncols = list(train_df.columns)\ncols.sort()\ntrain_df = train_df[cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:27.806678Z","iopub.execute_input":"2022-08-01T14:57:27.807076Z","iopub.status.idle":"2022-08-01T14:57:38.048763Z","shell.execute_reply.started":"2022-08-01T14:57:27.807026Z","shell.execute_reply":"2022-08-01T14:57:38.047759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"rows, cols = train_df.shape\nprint(f'There are {rows} rows and {cols} columns in the dataset.')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:38.051234Z","iopub.execute_input":"2022-08-01T14:57:38.051575Z","iopub.status.idle":"2022-08-01T14:57:38.088428Z","shell.execute_reply.started":"2022-08-01T14:57:38.051539Z","shell.execute_reply":"2022-08-01T14:57:38.087332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.sample(n=7)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:38.090392Z","iopub.execute_input":"2022-08-01T14:57:38.090869Z","iopub.status.idle":"2022-08-01T14:57:38.563686Z","shell.execute_reply.started":"2022-08-01T14:57:38.090832Z","shell.execute_reply":"2022-08-01T14:57:38.562857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum().to_frame().T.style.background_gradient(cmap=\"Blues\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:38.564852Z","iopub.execute_input":"2022-08-01T14:57:38.565293Z","iopub.status.idle":"2022-08-01T14:57:44.396653Z","shell.execute_reply.started":"2022-08-01T14:57:38.565257Z","shell.execute_reply":"2022-08-01T14:57:44.395792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop('customer_ID', axis=1).nunique().to_frame().T.style.background_gradient(cmap=\"Blues\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:57:44.398058Z","iopub.execute_input":"2022-08-01T14:57:44.398570Z","iopub.status.idle":"2022-08-01T14:58:08.916407Z","shell.execute_reply.started":"2022-08-01T14:57:44.398522Z","shell.execute_reply":"2022-08-01T14:58:08.914160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inspecting low numer of unique values","metadata":{}},{"cell_type":"code","source":"def print_unique_and_dtype(df, cols):\n    for col in cols:\n        print(f\"{col}: \", train_df['B_30'].dtype, \"\\n\", train_df[col].unique(),\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:08.920112Z","iopub.execute_input":"2022-08-01T14:58:08.922064Z","iopub.status.idle":"2022-08-01T14:58:08.965904Z","shell.execute_reply.started":"2022-08-01T14:58:08.922010Z","shell.execute_reply":"2022-08-01T14:58:08.965100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_unique_and_dtype(train_df, ['B_30', 'B_31', 'B_38', 'D_63', 'D_64', 'D_66', 'D_68', 'D_87', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126'])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:08.966998Z","iopub.execute_input":"2022-08-01T14:58:08.967383Z","iopub.status.idle":"2022-08-01T14:58:09.523755Z","shell.execute_reply.started":"2022-08-01T14:58:08.967348Z","shell.execute_reply":"2022-08-01T14:58:09.522958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select last statement per customer and make customer_ID the index\n\ntrain_df = train_df.groupby('customer_ID')\ntrain_df = train_df.tail(1)\ntrain_df = train_df.drop(['S_2'], axis=1)\ntrain_df.set_index('customer_ID', inplace=True) \ntrain_df.tail(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:09.525057Z","iopub.execute_input":"2022-08-01T14:58:09.525455Z","iopub.status.idle":"2022-08-01T14:58:12.341828Z","shell.execute_reply.started":"2022-08-01T14:58:09.525414Z","shell.execute_reply":"2022-08-01T14:58:12.340925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"reference: https://www.kaggle.com/competitions/amex-default-prediction/discussion/327094","metadata":{}},{"cell_type":"code","source":"rows, cols = train_df.shape\nprint(f'There are {rows} rows and {cols} columns in the dataset.')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.342962Z","iopub.execute_input":"2022-08-01T14:58:12.343432Z","iopub.status.idle":"2022-08-01T14:58:12.379936Z","shell.execute_reply.started":"2022-08-01T14:58:12.343272Z","shell.execute_reply":"2022-08-01T14:58:12.379129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.381152Z","iopub.execute_input":"2022-08-01T14:58:12.381666Z","iopub.status.idle":"2022-08-01T14:58:12.525529Z","shell.execute_reply.started":"2022-08-01T14:58:12.381625Z","shell.execute_reply":"2022-08-01T14:58:12.524567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tabular object setup for `TabularPandas`","metadata":{}},{"cell_type":"code","source":"cat_names = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.527003Z","iopub.execute_input":"2022-08-01T14:58:12.527422Z","iopub.status.idle":"2022-08-01T14:58:12.563384Z","shell.execute_reply.started":"2022-08-01T14:58:12.527387Z","shell.execute_reply":"2022-08-01T14:58:12.562248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_set = set(train_df.columns.to_list())\ncat_set = set(cat_names)\n\ncont_names = list(all_set - cat_set - set(['target']))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.564775Z","iopub.execute_input":"2022-08-01T14:58:12.565160Z","iopub.status.idle":"2022-08-01T14:58:12.600862Z","shell.execute_reply.started":"2022-08-01T14:58:12.565125Z","shell.execute_reply":"2022-08-01T14:58:12.599955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"procs = [Categorify, FillMissing, Normalize]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.602786Z","iopub.execute_input":"2022-08-01T14:58:12.603530Z","iopub.status.idle":"2022-08-01T14:58:12.637415Z","shell.execute_reply.started":"2022-08-01T14:58:12.603490Z","shell.execute_reply":"2022-08-01T14:58:12.636571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cbs = [\n#     ShowGraphCallback(),\n#     GradientAccumulation(),\n    MixedPrecision(),\n    SaveModelCallback(monitor='_rmse', comp=np.less, min_delta=0.001),\n    ReduceLROnPlateau(monitor='_rmse', comp=np.less, min_delta=0.001, patience=3),\n    EarlyStoppingCallback(monitor='_rmse', comp=np.less, min_delta=0.001, patience=7),\n#     GradientClip(0.1),\n      ]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.641787Z","iopub.execute_input":"2022-08-01T14:58:12.642076Z","iopub.status.idle":"2022-08-01T14:58:12.677000Z","shell.execute_reply.started":"2022-08-01T14:58:12.642041Z","shell.execute_reply":"2022-08-01T14:58:12.676185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds, targs = learn.get_preds()\n# y_pred = pd.DataFrame(data={'prediction': preds.numpy()[:,0]})\n# y_true = pd.DataFrame(data={'target': targs.numpy()[:,0]})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.679355Z","iopub.execute_input":"2022-08-01T14:58:12.679795Z","iopub.status.idle":"2022-08-01T14:58:12.713918Z","shell.execute_reply.started":"2022-08-01T14:58:12.679760Z","shell.execute_reply":"2022-08-01T14:58:12.712958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.717241Z","iopub.execute_input":"2022-08-01T14:58:12.717578Z","iopub.status.idle":"2022-08-01T14:58:12.847655Z","shell.execute_reply.started":"2022-08-01T14:58:12.717551Z","shell.execute_reply":"2022-08-01T14:58:12.846608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Competition Metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.851091Z","iopub.execute_input":"2022-08-01T14:58:12.851562Z","iopub.status.idle":"2022-08-01T14:58:12.895890Z","shell.execute_reply.started":"2022-08-01T14:58:12.851523Z","shell.execute_reply":"2022-08-01T14:58:12.895089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"SAMPLE_SUBMISSION = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:12.897088Z","iopub.execute_input":"2022-08-01T14:58:12.897740Z","iopub.status.idle":"2022-08-01T14:58:14.296745Z","shell.execute_reply.started":"2022-08-01T14:58:12.897701Z","shell.execute_reply":"2022-08-01T14:58:14.295848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SUBMISSION.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:14.298073Z","iopub.execute_input":"2022-08-01T14:58:14.298453Z","iopub.status.idle":"2022-08-01T14:58:14.340886Z","shell.execute_reply.started":"2022-08-01T14:58:14.298400Z","shell.execute_reply":"2022-08-01T14:58:14.340107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather('../input/amexfeather/test_data.ftr')\ntest_df = test_df.groupby('customer_ID')\ntest_df = test_df.tail(1)\ntest_df = test_df.drop(['S_2'], axis=1)\ntest_df = test_df.set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:14.342226Z","iopub.execute_input":"2022-08-01T14:58:14.342631Z","iopub.status.idle":"2022-08-01T14:58:33.596224Z","shell.execute_reply.started":"2022-08-01T14:58:14.342592Z","shell.execute_reply":"2022-08-01T14:58:33.595304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.597492Z","iopub.execute_input":"2022-08-01T14:58:33.597845Z","iopub.status.idle":"2022-08-01T14:58:33.637202Z","shell.execute_reply.started":"2022-08-01T14:58:33.597809Z","shell.execute_reply":"2022-08-01T14:58:33.636428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill missing data","metadata":{}},{"cell_type":"code","source":"param = 'B_37'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.638639Z","iopub.execute_input":"2022-08-01T14:58:33.638996Z","iopub.status.idle":"2022-08-01T14:58:33.683682Z","shell.execute_reply.started":"2022-08-01T14:58:33.638959Z","shell.execute_reply":"2022-08-01T14:58:33.682842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'B_40'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.685308Z","iopub.execute_input":"2022-08-01T14:58:33.685960Z","iopub.status.idle":"2022-08-01T14:58:33.728814Z","shell.execute_reply.started":"2022-08-01T14:58:33.685921Z","shell.execute_reply":"2022-08-01T14:58:33.727819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'B_41'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.730848Z","iopub.execute_input":"2022-08-01T14:58:33.731350Z","iopub.status.idle":"2022-08-01T14:58:33.773158Z","shell.execute_reply.started":"2022-08-01T14:58:33.731310Z","shell.execute_reply":"2022-08-01T14:58:33.772362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'D_86'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.774427Z","iopub.execute_input":"2022-08-01T14:58:33.774970Z","iopub.status.idle":"2022-08-01T14:58:33.817490Z","shell.execute_reply.started":"2022-08-01T14:58:33.774932Z","shell.execute_reply":"2022-08-01T14:58:33.816579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'D_102'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.818871Z","iopub.execute_input":"2022-08-01T14:58:33.819523Z","iopub.status.idle":"2022-08-01T14:58:33.862721Z","shell.execute_reply.started":"2022-08-01T14:58:33.819481Z","shell.execute_reply":"2022-08-01T14:58:33.861919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'D_133'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.864053Z","iopub.execute_input":"2022-08-01T14:58:33.864386Z","iopub.status.idle":"2022-08-01T14:58:33.907735Z","shell.execute_reply.started":"2022-08-01T14:58:33.864351Z","shell.execute_reply":"2022-08-01T14:58:33.906794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'D_140'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.909838Z","iopub.execute_input":"2022-08-01T14:58:33.910338Z","iopub.status.idle":"2022-08-01T14:58:33.952467Z","shell.execute_reply.started":"2022-08-01T14:58:33.910302Z","shell.execute_reply":"2022-08-01T14:58:33.951661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'D_144'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.953830Z","iopub.execute_input":"2022-08-01T14:58:33.954421Z","iopub.status.idle":"2022-08-01T14:58:33.997069Z","shell.execute_reply.started":"2022-08-01T14:58:33.954380Z","shell.execute_reply":"2022-08-01T14:58:33.996176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'S_12'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:33.998196Z","iopub.execute_input":"2022-08-01T14:58:33.998524Z","iopub.status.idle":"2022-08-01T14:58:34.041072Z","shell.execute_reply.started":"2022-08-01T14:58:33.998486Z","shell.execute_reply":"2022-08-01T14:58:34.040101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'S_17'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.042180Z","iopub.execute_input":"2022-08-01T14:58:34.042502Z","iopub.status.idle":"2022-08-01T14:58:34.084827Z","shell.execute_reply.started":"2022-08-01T14:58:34.042467Z","shell.execute_reply":"2022-08-01T14:58:34.084102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = 'S_26'\nmean = train_df[param].astype(np.float32).mean().astype(np.float16)\ntest_df[param] = test_df[param].fillna(mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.085944Z","iopub.execute_input":"2022-08-01T14:58:34.086293Z","iopub.status.idle":"2022-08-01T14:58:34.129150Z","shell.execute_reply.started":"2022-08-01T14:58:34.086259Z","shell.execute_reply":"2022-08-01T14:58:34.128109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split_dataframes = np.array_split(test_df, 25)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.130489Z","iopub.execute_input":"2022-08-01T14:58:34.130822Z","iopub.status.idle":"2022-08-01T14:58:34.429198Z","shell.execute_reply.started":"2022-08-01T14:58:34.130788Z","shell.execute_reply":"2022-08-01T14:58:34.428346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-folds cross-validation","metadata":{}},{"cell_type":"code","source":"# setup k-folds\nfolds = 10\nskf = StratifiedKFold(n_splits=folds, shuffle=False, random_state=None)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.430389Z","iopub.execute_input":"2022-08-01T14:58:34.430713Z","iopub.status.idle":"2022-08-01T14:58:34.466492Z","shell.execute_reply.started":"2022-08-01T14:58:34.430679Z","shell.execute_reply":"2022-08-01T14:58:34.465677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idx = train_df.index","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.468837Z","iopub.execute_input":"2022-08-01T14:58:34.469265Z","iopub.status.idle":"2022-08-01T14:58:34.503695Z","shell.execute_reply.started":"2022-08-01T14:58:34.469235Z","shell.execute_reply":"2022-08-01T14:58:34.502768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_idx)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.504866Z","iopub.execute_input":"2022-08-01T14:58:34.505237Z","iopub.status.idle":"2022-08-01T14:58:34.539785Z","shell.execute_reply.started":"2022-08-01T14:58:34.505193Z","shell.execute_reply":"2022-08-01T14:58:34.538941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_submission_votes(dfs, to, dls):\n    preds_all = pd.DataFrame(columns=['prediction'])\n    customer_ID = pd.DataFrame(columns=['customer_ID'])\n    for item in dfs:\n        to_test = to.new(item)\n        to_test.process()\n        test_dl = dls.valid.new(to_test)\n        preds, _ = learn.get_preds(dl=test_dl)\n        y_pred = pd.DataFrame(data={'prediction': preds.numpy()[:,0]})\n        preds_all = pd.concat([preds_all, y_pred])\n        customers = pd.DataFrame(data={'customer_ID': item.index.to_list()})\n        customer_ID = pd.concat([customer_ID, customers])\n    return preds_all, customer_ID","metadata":{"execution":{"iopub.status.busy":"2022-08-01T14:58:34.541018Z","iopub.execute_input":"2022-08-01T14:58:34.541581Z","iopub.status.idle":"2022-08-01T14:58:34.575871Z","shell.execute_reply.started":"2022-08-01T14:58:34.541544Z","shell.execute_reply":"2022-08-01T14:58:34.575110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_epochs = 20\nbs = 64\nlearning_rate = 6.5e-3   # set to None finds a suitable learning rate at training time in next code block","metadata":{"execution":{"iopub.status.busy":"2022-08-01T15:18:47.902482Z","iopub.execute_input":"2022-08-01T15:18:47.902889Z","iopub.status.idle":"2022-08-01T15:18:47.938136Z","shell.execute_reply.started":"2022-08-01T15:18:47.902844Z","shell.execute_reply":"2022-08-01T15:18:47.937135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = []\ny_preds_all = []\ntraining_test_set = RandomSplitter(valid_pct=0.2)(range_of(train_idx))\nfor train_idx, val_idx in skf.split(training_test_set[0], train_df.target.iloc[training_test_set[0]]):\n#     splits = (L(train_idx, use_list=True), L(val_idx, use_list=True))  # generates the same as fastai\n    splits = RandomSplitter(valid_pct=0.2)(range_of(train_idx))\n    to = TabularPandas(train_df, procs, cat_names, cont_names, y_names=\"target\", splits=splits)\n    dls = to.dataloaders(bs = bs)\n    \n    learn = tabular_learner(dls, metrics=rmse)\n    if learning_rate is None:\n        lrs = learn.lr_find()\n        learning_rate = lrs.lr_min\n        print(f'Training time learning rate: {learning_rate}')\n    learn.fit_one_cycle(training_epochs, learning_rate/2, cbs=cbs)\n    to_training_test = to.new(train_df.iloc[training_test_set[1]])\n    to_training_test.process()\n    train_test_dl = dls.valid.new(to_training_test)\n    preds, targs = learn.get_preds(dl=train_test_dl)\n    y_pred = pd.DataFrame(data={'prediction': preds.numpy()[:,0]})\n    y_true = pd.DataFrame(data={'target': targs.numpy()[:,0]})\n    print(f'Gini Score: {amex_metric(y_true,y_pred)}')\n    \n    y_preds_all.append(y_pred)\n    \n    preds_all, customer_ID = get_submission_votes(split_dataframes, to, dls)\n    test_preds.append(preds_all)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T15:18:52.533951Z","iopub.execute_input":"2022-08-01T15:18:52.534347Z","iopub.status.idle":"2022-08-01T17:33:37.391144Z","shell.execute_reply.started":"2022-08-01T15:18:52.534312Z","shell.execute_reply":"2022-08-01T17:33:37.390254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_result = y_preds_all[0]\nfor one_vote in y_preds_all[1:]:\n    vote_result += one_vote\n\nvote_result /= len(y_preds_all)\n# vote_result","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:36.590655Z","iopub.execute_input":"2022-08-01T18:07:36.590977Z","iopub.status.idle":"2022-08-01T18:07:36.632658Z","shell.execute_reply.started":"2022-08-01T18:07:36.590947Z","shell.execute_reply":"2022-08-01T18:07:36.631814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Gini Score: {amex_metric(y_true,vote_result)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:38.435752Z","iopub.execute_input":"2022-08-01T18:07:38.436086Z","iopub.status.idle":"2022-08-01T18:07:38.589960Z","shell.execute_reply.started":"2022-08-01T18:07:38.436048Z","shell.execute_reply":"2022-08-01T18:07:38.589086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make `submission.csv file`","metadata":{}},{"cell_type":"code","source":"test_preds","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:41.122690Z","iopub.execute_input":"2022-08-01T18:07:41.123327Z","iopub.status.idle":"2022-08-01T18:07:41.187243Z","shell.execute_reply.started":"2022-08-01T18:07:41.123280Z","shell.execute_reply":"2022-08-01T18:07:41.186417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_result = test_preds[0]\nfor one_vote in test_preds[1:]:\n    vote_result += one_vote\n\nvote_result /= len(test_preds)\nvote_result","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:43.091507Z","iopub.execute_input":"2022-08-01T18:07:43.091835Z","iopub.status.idle":"2022-08-01T18:07:43.152694Z","shell.execute_reply.started":"2022-08-01T18:07:43.091806Z","shell.execute_reply":"2022-08-01T18:07:43.151906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([vote_result, customer_ID], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:44.327831Z","iopub.execute_input":"2022-08-01T18:07:44.328180Z","iopub.status.idle":"2022-08-01T18:07:44.392211Z","shell.execute_reply.started":"2022-08-01T18:07:44.328137Z","shell.execute_reply":"2022-08-01T18:07:44.391377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SUBMISSION = pd.concat([vote_result, customer_ID], axis=1)\nSAMPLE_SUBMISSION.head()\nSAMPLE_SUBMISSION.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:07:45.573205Z","iopub.execute_input":"2022-08-01T18:07:45.573529Z","iopub.status.idle":"2022-08-01T18:07:50.662464Z","shell.execute_reply.started":"2022-08-01T18:07:45.573500Z","shell.execute_reply":"2022-08-01T18:07:50.661563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}