{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is a quick demonstration, who to use the Fastai v2 library for a Kaggle tabular competition. Fastai v2 is based on pytorch and allows you, to build a decent machine learning application. For more information please visit the Fastai documentation: https://docs.fast.ai/. I will link to \"Chapter 9, Tabular Modelling Deep Dive\" and the notebook \"09_tabular.ipynb\".\n\nThis monthly competition is a regression problem: predict the probability of each team scoring within the next 10 second. The big challenge is the large amount of training data. There are so many records that using them fully will exceed available memory.\n\nThis notebook is inspired by Patricks notebook, which uses an older version of Fastai: https://www.kaggle.com/code/paddykb/tps-2022-10-fastai\nI also use the prepared data sets from Ayush Bihan, his notebook is here:  https://www.kaggle.com/code/paddykb/tps-2022-10-fastai/data\n\nIn this notebook i will use a neural network approach and i will train this network with the offered traing data set.\n\nLet's start and import the needed stuff ..","metadata":{}},{"cell_type":"code","source":"from fastai.tabular.all import * \nfrom fastai.test_utils import show_install\nimport gc\n\nfrom IPython.display import display, clear_output","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-31T09:22:31.063523Z","iopub.execute_input":"2022-10-31T09:22:31.064228Z","iopub.status.idle":"2022-10-31T09:22:32.524840Z","shell.execute_reply.started":"2022-10-31T09:22:31.064140Z","shell.execute_reply":"2022-10-31T09:22:32.523613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_install()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:32.530148Z","iopub.execute_input":"2022-10-31T09:22:32.531676Z","iopub.status.idle":"2022-10-31T09:22:32.709826Z","shell.execute_reply.started":"2022-10-31T09:22:32.531632Z","shell.execute_reply":"2022-10-31T09:22:32.708554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:32.711983Z","iopub.execute_input":"2022-10-31T09:22:32.712739Z","iopub.status.idle":"2022-10-31T09:22:32.725106Z","shell.execute_reply.started":"2022-10-31T09:22:32.712694Z","shell.execute_reply":"2022-10-31T09:22:32.723779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed_value(seed=718):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nset_seed_value()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:32.729359Z","iopub.execute_input":"2022-10-31T09:22:32.730230Z","iopub.status.idle":"2022-10-31T09:22:32.736621Z","shell.execute_reply.started":"2022-10-31T09:22:32.730190Z","shell.execute_reply":"2022-10-31T09:22:32.735548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/tabular-playground-series-oct-2022/')\nfeather_path = Path('../input/fast-loading-high-compression-with-feather/feather_data')\nPath.BASE_PATH = path\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:32.738421Z","iopub.execute_input":"2022-10-31T09:22:32.739844Z","iopub.status.idle":"2022-10-31T09:22:32.756014Z","shell.execute_reply.started":"2022-10-31T09:22:32.739806Z","shell.execute_reply":"2022-10-31T09:22:32.754943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dut to the large amount of offered traing data sets and the the fact, that the memory is limited we have to limit the number of used training data sets. More than 3 files can't be processed later on.","metadata":{}},{"cell_type":"code","source":"dtypes_df = pd.read_csv(os.path.join(path, 'train_dtypes.csv'))\ntrain_types = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\n\ntrain_df = pd.DataFrame(columns=train_types.keys())\nfor i in range(2):\n    print(f\"Read train_{i}.csv\")\n    # dt_read = pd.read_csv(os.path.join(path, f'train_{i}.csv'), dtype=dtypes)\n    \n    dt_read = pd.read_feather(os.path.join(feather_path, f\"train_{i}_compressed.ftr\"))\n    \n    train_df = pd.concat([train_df, dt_read])    \n    del dt_read    \n    gc.collect()\n\nsample_submission = pd.read_csv(os.path.join(path, 'sample_submission.csv')).set_index('id')\n\ntypes_df = pd.read_csv(os.path.join(path, 'test_dtypes.csv'))\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\ntest_df= pd.read_csv(os.path.join(path, 'test.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:32.757972Z","iopub.execute_input":"2022-10-31T09:22:32.759575Z","iopub.status.idle":"2022-10-31T09:22:49.719361Z","shell.execute_reply.started":"2022-10-31T09:22:32.759536Z","shell.execute_reply":"2022-10-31T09:22:49.718219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 50\nprint_missing_values = False\nimpute_missing_values = True\nscale_position_values = True\nscale_timer_values = True\nadd_new_features = True\n\ndep_var_A = 'team_A_scoring_within_10sec'\ndep_var_B = 'team_B_scoring_within_10sec'\ndep_vars = [dep_var_A, dep_var_B]","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:49.724314Z","iopub.execute_input":"2022-10-31T09:22:49.724963Z","iopub.status.idle":"2022-10-31T09:22:49.731791Z","shell.execute_reply.started":"2022-10-31T09:22:49.724923Z","shell.execute_reply":"2022-10-31T09:22:49.730582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:49.733351Z","iopub.execute_input":"2022-10-31T09:22:49.734107Z","iopub.status.idle":"2022-10-31T09:22:49.761273Z","shell.execute_reply.started":"2022-10-31T09:22:49.734068Z","shell.execute_reply":"2022-10-31T09:22:49.760345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:49.765521Z","iopub.execute_input":"2022-10-31T09:22:49.767844Z","iopub.status.idle":"2022-10-31T09:22:49.813505Z","shell.execute_reply.started":"2022-10-31T09:22:49.767774Z","shell.execute_reply":"2022-10-31T09:22:49.812610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[dep_var_A] = train_df[dep_var_A].astype('int8')\ntrain_df[dep_var_B] = train_df[dep_var_B].astype('int8')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:49.817940Z","iopub.execute_input":"2022-10-31T09:22:49.820269Z","iopub.status.idle":"2022-10-31T09:22:50.485218Z","shell.execute_reply.started":"2022-10-31T09:22:49.820228Z","shell.execute_reply":"2022-10-31T09:22:50.483938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_missing_values =  train_df.columns[train_df.isnull().any()]\n\nfor col in cols_with_missing_values:\n    print(col, train_df[col].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:50.490457Z","iopub.execute_input":"2022-10-31T09:22:50.493099Z","iopub.status.idle":"2022-10-31T09:22:53.824531Z","shell.execute_reply.started":"2022-10-31T09:22:50.493054Z","shell.execute_reply":"2022-10-31T09:22:53.823148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_cols = list()\nfor i in range(0,6):\n    used_cols +=  [col for col in train_types.keys() if col.startswith(f'p{i}')]\n    \nused_cols += [col for col in train_types.keys() if col.startswith(f'ball') or col.startswith(f'boost')]\nprint(used_cols)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:53.826432Z","iopub.execute_input":"2022-10-31T09:22:53.827031Z","iopub.status.idle":"2022-10-31T09:22:53.836262Z","shell.execute_reply.started":"2022-10-31T09:22:53.826992Z","shell.execute_reply":"2022-10-31T09:22:53.835194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[['game_num'] + used_cols + dep_vars]\ntest_df = test_df[used_cols]\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:53.843249Z","iopub.execute_input":"2022-10-31T09:22:53.843910Z","iopub.status.idle":"2022-10-31T09:22:55.638184Z","shell.execute_reply.started":"2022-10-31T09:22:53.843869Z","shell.execute_reply":"2022-10-31T09:22:55.637066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_add_new_features(df):\n    print(f\"Add new features\")\n    for i in range(6):\n        df[f'p{i}_ball_distance'] = ((df['ball_pos_x']-df[f'p{i}_pos_x'])**2 + (df['ball_pos_y']-df[f'p{i}_pos_y'])**2 + (df['ball_pos_z']-df[f'p{i}_pos_z'])**2)**0.5\n       # df[f'p{i}_boost_timer'] = data[f'p{i}_boost']*df[f'boost{i}_timer']\n        df[f'p{i}_pos'] = np.sqrt(df[f'p{i}_pos_x']**2 + df[f'p{i}_pos_y']**2 + df[f'p{i}_pos_z']**2)\n        df[f'p{i}_vel'] = np.sqrt(df[f'p{i}_vel_x']**2 + df[f'p{i}_vel_y']**2 + df[f'p{i}_vel_z']**2)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.641483Z","iopub.execute_input":"2022-10-31T09:22:55.642168Z","iopub.status.idle":"2022-10-31T09:22:55.650276Z","shell.execute_reply.started":"2022-10-31T09:22:55.642129Z","shell.execute_reply":"2022-10-31T09:22:55.649352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_scale_boost_timer(df):\n    print(f\"Scale boost timer values\")\n    \n    for col in df.columns:\n        if col.endswith(f'timer'):\n            df[col] = df[col]/10.0\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.651808Z","iopub.execute_input":"2022-10-31T09:22:55.652524Z","iopub.status.idle":"2022-10-31T09:22:55.667296Z","shell.execute_reply.started":"2022-10-31T09:22:55.652488Z","shell.execute_reply":"2022-10-31T09:22:55.666277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_scale_position(df):\n    print(f\"Scale the position values\")\n    \n    for col in df.columns:\n        if col.endswith(\"pos_x\"):\n            df[col] = df[col]/120.0\n\n        if col.endswith(\"pos_y\"):\n            df[col] = df[col]/82.0\n        \n        if col.endswith(\"pos_z\"):\n            df[col] = df[col]/40.0\n            \n        if \"vel_\" in col:\n            df[col] = df[col]/45.0\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.668951Z","iopub.execute_input":"2022-10-31T09:22:55.669657Z","iopub.status.idle":"2022-10-31T09:22:55.678255Z","shell.execute_reply.started":"2022-10-31T09:22:55.669601Z","shell.execute_reply":"2022-10-31T09:22:55.677200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_impute(df, used_cols:dict):\n    print(f\"Impute missing values with fillna\")\n    for col in used_cols:\n        df[col] = df[col].fillna(0).astype('float16')\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.679837Z","iopub.execute_input":"2022-10-31T09:22:55.680451Z","iopub.status.idle":"2022-10-31T09:22:55.687955Z","shell.execute_reply.started":"2022-10-31T09:22:55.680415Z","shell.execute_reply":"2022-10-31T09:22:55.686914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_print_missing_values(df):\n    cols_with_missing_values =  df.columns[df.isnull().any()]\n\n    for col in cols_with_missing_values:\n        print(col, train_df[col].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.689340Z","iopub.execute_input":"2022-10-31T09:22:55.689936Z","iopub.status.idle":"2022-10-31T09:22:55.699077Z","shell.execute_reply.started":"2022-10-31T09:22:55.689893Z","shell.execute_reply":"2022-10-31T09:22:55.698075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if print_missing_values:\n    do_print_missing_values(train_df)\n    do_print_missing_values(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.700585Z","iopub.execute_input":"2022-10-31T09:22:55.701234Z","iopub.status.idle":"2022-10-31T09:22:55.708714Z","shell.execute_reply.started":"2022-10-31T09:22:55.701198Z","shell.execute_reply":"2022-10-31T09:22:55.707711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if impute_missing_values:\n    train_df[:] = do_impute(train_df, used_cols)\n    test_df[:] = do_impute(test_df, used_cols)\nelse:\n    print(\"No missing values are imputed\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:22:55.710380Z","iopub.execute_input":"2022-10-31T09:22:55.711087Z","iopub.status.idle":"2022-10-31T09:23:01.515946Z","shell.execute_reply.started":"2022-10-31T09:22:55.711039Z","shell.execute_reply":"2022-10-31T09:23:01.514858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if scale_position_values:\n    train_df = do_scale_boost_timer(train_df)\n    test_df = do_scale_boost_timer(test_df)\nelse:\n    print(\"No scaling of the position values\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:01.517764Z","iopub.execute_input":"2022-10-31T09:23:01.518455Z","iopub.status.idle":"2022-10-31T09:23:02.170985Z","shell.execute_reply.started":"2022-10-31T09:23:01.518417Z","shell.execute_reply":"2022-10-31T09:23:02.169827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if scale_timer_values:\n    train_df = do_scale_position(train_df)\n    test_df = do_scale_position(test_df)\nelse: \n    print(\"No scaling of boost timer values\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:02.172870Z","iopub.execute_input":"2022-10-31T09:23:02.173662Z","iopub.status.idle":"2022-10-31T09:23:06.734757Z","shell.execute_reply.started":"2022-10-31T09:23:02.173617Z","shell.execute_reply":"2022-10-31T09:23:06.733686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if add_new_features:\n    train_df = do_add_new_features(train_df)\n    test_df = do_add_new_features(test_df)\nelse: \n    print(\"No new features added\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:06.736502Z","iopub.execute_input":"2022-10-31T09:23:06.737194Z","iopub.status.idle":"2022-10-31T09:23:21.206196Z","shell.execute_reply.started":"2022-10-31T09:23:06.737155Z","shell.execute_reply":"2022-10-31T09:23:21.205106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if print_missing_values:\n    do_print_missing_values(train_df)\n    do_print_missing_values(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:21.207958Z","iopub.execute_input":"2022-10-31T09:23:21.208641Z","iopub.status.idle":"2022-10-31T09:23:21.214917Z","shell.execute_reply.started":"2022-10-31T09:23:21.208600Z","shell.execute_reply":"2022-10-31T09:23:21.213800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I need a list of the column names, which are candidates for category variables and which are no candidates, also called continous variables. The Fastai library offers the function 'cont_cat_split' to do this for us. You can use the optional parameter 'max_card' to specify the maximum number of unique values a column can have for a category variable. I will use the default value 25, which is sufficient for this data set. Both lists are used later, to create a corresponding  model. The category variables are mapped into embeddings, the continous variables are mapped to simple linear model. The value for max_cards specifies the ratio between the category and continous variables. Lower max_card values reduces the number of categories and and increases the number of continous variables. The value max_card=1 produces an empty continous variable list. All columns of the data frame are handled as continous variables.\nThe parameter dep_vars specifies our depended variables, these column will be skiped when the category and contious variables are determined.","metadata":{}},{"cell_type":"code","source":"cont_vars, cat_vars = cont_cat_split(train_df, dep_var=dep_vars)\ncat_vars.remove('game_num')\nlen(cont_vars), len(cat_vars),cont_vars,cat_vars","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:21.217426Z","iopub.execute_input":"2022-10-31T09:23:21.219362Z","iopub.status.idle":"2022-10-31T09:23:21.240363Z","shell.execute_reply.started":"2022-10-31T09:23:21.219324Z","shell.execute_reply":"2022-10-31T09:23:21.239228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"he next step is to create a data loader. The Fastai library offers a powerful helper called 'TabularPandas'. It needs the data frame, list of the category and continous variables, the depened variable and a splitter. The splitter divides the data set into two parts: one for the training and one for the validation and for internal optimization step in each epoch. Let's use a rate of 4 to 1. I need a dataloader also, which is created from this TabularPandas instance. The helper function getData does this job and allows you, to get a small dataloader if you want to do a quick prototyping of your model. ","metadata":{}},{"cell_type":"code","source":"def getData(df, batchSize=8196):\n    \n    max_training_game_num = 0.8*df['game_num'].max()\n    cond = (df.game_num <max_training_game_num)\n    train_idx = np.where(cond)[0]\n    valid_idx = np.where(~cond)[0]\n\n    game_splits = (list(train_idx), list(valid_idx))\n    \n    to_train = TabularPandas(df, \n                             [Normalize],\n                             cat_names=cat_vars,\n                             cont_names=cont_vars, \n                             splits=RandomSplitter(valid_pct=0.20)(df),  \n                             # splits=game_splits,\n                             device = device,\n                             y_block=RegressionBlock(),\n                             y_names=dep_vars) \n\n    return to_train.dataloaders(bs=batchSize)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:21.244342Z","iopub.execute_input":"2022-10-31T09:23:21.246981Z","iopub.status.idle":"2022-10-31T09:23:21.256570Z","shell.execute_reply.started":"2022-10-31T09:23:21.246925Z","shell.execute_reply":"2022-10-31T09:23:21.255615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = getData(train_df)\nlen(dls.train), len(dls.valid), dls.train.device","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:21.258071Z","iopub.execute_input":"2022-10-31T09:23:21.258832Z","iopub.status.idle":"2022-10-31T09:23:59.962260Z","shell.execute_reply.started":"2022-10-31T09:23:21.258794Z","shell.execute_reply":"2022-10-31T09:23:59.961083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:23:59.966786Z","iopub.execute_input":"2022-10-31T09:23:59.969561Z","iopub.status.idle":"2022-10-31T09:24:06.398576Z","shell.execute_reply.started":"2022-10-31T09:23:59.969510Z","shell.execute_reply":"2022-10-31T09:24:06.397304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At least i create a learner pasing the dataloader into it. I use the default values for the internal layers as you can see in the reported summary. The model has two hidden layers with 200 and 100 elements as the default. You can change the structure of the hidden layer, using the paramter layers liks this 'layers=[1024,512,16]'. The hidden layers uses a batch normalization and the ReLU activation function.","metadata":{}},{"cell_type":"code","source":"my_config = tabular_config(ps=0.1, \n                           act_cls = Mish(),\n                           use_bn=True, \n                           bn_cont=True, \n                           y_range=(0,1))\n\nlearn = tabular_learner(dls,   n_out=dls.c,\n                        config = my_config,\n                         metrics=[mae, rmse],\n                     #   loss_func=BCEWithLogitsLossFlat(),\n                     #    layers=[1024,256,16],\n                         layers = [2048, 2048, 1024, 1024],\n#                         layers=[512, 256, 128,64],\n                        )\n\nlearn.summary()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:24:06.400267Z","iopub.execute_input":"2022-10-31T09:24:06.400797Z","iopub.status.idle":"2022-10-31T09:24:12.976277Z","shell.execute_reply.started":"2022-10-31T09:24:06.400739Z","shell.execute_reply":"2022-10-31T09:24:12.975133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_min,lr_steep = learn.lr_find(suggest_funcs=(minimum, steep))\nprint(f\"Minimum: {lr_min:.2e}, steepest point: {lr_steep:.2e}\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:24:12.980800Z","iopub.execute_input":"2022-10-31T09:24:12.983326Z","iopub.status.idle":"2022-10-31T09:24:30.840043Z","shell.execute_reply.started":"2022-10-31T09:24:12.983280Z","shell.execute_reply":"2022-10-31T09:24:30.838967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will use a maximum learning rate of 1e-2. \nStarting the learning process is quite easy, i will run for 100 epochs and i will save the model with the best, with the lowest validation lost value. The Fastai library offers the SaveModelCallback callback. You must specify the file name only. The option with_opt=True stores the values of the optimizer also.\nYou will find the new file under models/kaggle_tps2022_oct.pth","metadata":{}},{"cell_type":"code","source":"learn.fit_one_cycle(num_epochs, 2e-2, wd=0.1, cbs=SaveModelCallback(fname='kaggle_tps2022_oct', with_opt=True))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T09:24:30.844800Z","iopub.execute_input":"2022-10-31T09:24:30.847336Z","iopub.status.idle":"2022-10-31T10:07:27.515317Z","shell.execute_reply.started":"2022-10-31T09:24:30.847294Z","shell.execute_reply":"2022-10-31T10:07:27.514157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To calculate the predictions for this competition, i will load the best model from the training process. Best model means the model where the validation loss has the lowest value.","metadata":{}},{"cell_type":"code","source":"learn.load('kaggle_tps2022_oct')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:27.520846Z","iopub.execute_input":"2022-10-31T10:07:27.523507Z","iopub.status.idle":"2022-10-31T10:07:27.608766Z","shell.execute_reply.started":"2022-10-31T10:07:27.523464Z","shell.execute_reply":"2022-10-31T10:07:27.607648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dlt = learn.dls.test_dl(test_df) \npreds = learn.get_preds(dl=dlt) \n\npreds[0]","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:27.613557Z","iopub.execute_input":"2022-10-31T10:07:27.616216Z","iopub.status.idle":"2022-10-31T10:07:34.059825Z","shell.execute_reply.started":"2022-10-31T10:07:27.616170Z","shell.execute_reply":"2022-10-31T10:07:34.058594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(os.path.join(path, 'sample_submission.csv')).set_index('id')\npx = pd.DataFrame(preds[0])\nsample_submission[dep_var_A] = px[0]\nsample_submission[dep_var_B] = px[1]\n\nsample_submission.to_csv(\"submission.csv\")\nsample_submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:34.065158Z","iopub.execute_input":"2022-10-31T10:07:34.067901Z","iopub.status.idle":"2022-10-31T10:07:37.983035Z","shell.execute_reply.started":"2022-10-31T10:07:34.067858Z","shell.execute_reply":"2022-10-31T10:07:37.981919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission[dep_var_A].hist()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:37.988041Z","iopub.execute_input":"2022-10-31T10:07:37.990622Z","iopub.status.idle":"2022-10-31T10:07:38.366723Z","shell.execute_reply.started":"2022-10-31T10:07:37.990577Z","shell.execute_reply":"2022-10-31T10:07:38.365629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission[dep_var_B].hist() ","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:38.371516Z","iopub.execute_input":"2022-10-31T10:07:38.373929Z","iopub.status.idle":"2022-10-31T10:07:38.724642Z","shell.execute_reply.started":"2022-10-31T10:07:38.373888Z","shell.execute_reply":"2022-10-31T10:07:38.723524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"prediction {dep_var_A} mean: {sample_submission[dep_var_A].mean():.6}\")\nprint(f\"prediction {dep_var_B} mean: {sample_submission[dep_var_B].mean():.6}\")","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:38.729884Z","iopub.execute_input":"2022-10-31T10:07:38.732536Z","iopub.status.idle":"2022-10-31T10:07:38.747728Z","shell.execute_reply.started":"2022-10-31T10:07:38.732491Z","shell.execute_reply":"2022-10-31T10:07:38.746660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -la ","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:38.752227Z","iopub.execute_input":"2022-10-31T10:07:38.754731Z","iopub.status.idle":"2022-10-31T10:07:40.091873Z","shell.execute_reply.started":"2022-10-31T10:07:38.754689Z","shell.execute_reply":"2022-10-31T10:07:40.089876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next class is a direct copy from Zachary website: https://walkwithfastai.com/Regression_and_Permutation_Importance#Permutation-Importance\n\nThe implementation shuffles each column/feature in my training data and calculates how the prediction is changed. The more the prediction is changed, the more important a column/feature is. ","metadata":{}},{"cell_type":"code","source":"class PermutationImportance():\n      \"Calculate and plot the permutation importance\"\n      def __init__(self, learn:Learner, df=None, bs=None):\n        \"Initialize with a test dataframe, a learner, and a metric\"\n        self.learn = learn\n        self.df = df\n        bs = bs if bs is not None else learn.dls.bs\n        if self.df is not None:\n          self.dl = learn.dls.test_dl(self.df, bs=bs)\n        else:\n          self.dl = learn.dls[1]\n        self.x_names = learn.dls.x_names.filter(lambda x: '_na' not in x)\n        self.na = learn.dls.x_names.filter(lambda x: '_na' in x)\n        self.y = dls.y_names\n        self.results = self.calc_feat_importance()\n        self.plot_importance(self.ord_dic_to_df(self.results))\n    \n        \n      def get_results(self): return self.results\n        \n      def measure_col(self, name:str):\n          \"Measures change after column shuffle\"\n          col = [name]\n          if f'{name}_na' in self.na: col.append(name)\n          orig = self.dl.items[col].values\n          perm = np.random.permutation(len(orig))\n          self.dl.items[col] = self.dl.items[col].values[perm]\n          metric = learn.validate(dl=self.dl)[1]\n          self.dl.items[col] = orig\n          clear_output()\n          return metric\n\n      def calc_feat_importance(self):\n          \"Calculates permutation importance by shuffling a column on a percentage scale\"\n          print('Getting base error')\n          base_error = self.learn.validate(dl=self.dl)[1]\n          self.importance = {}\n          pbar = progress_bar(self.x_names)\n          print('Calculating Permutation Importance')\n          for col in pbar:\n            self.importance[col] = self.measure_col(col)\n          for key, value in self.importance.items():\n            self.importance[key] = np.abs(base_error-value)/base_error #this can be adjusted\n          return OrderedDict(sorted(self.importance.items(), key=lambda kv: kv[1], reverse=True))\n\n      def ord_dic_to_df(self, dict:OrderedDict):\n          return pd.DataFrame([[k, v] for k, v in dict.items()], columns=['feature', 'importance'])\n\n      def plot_importance(self, df:pd.DataFrame, limit=20, asc=False, **kwargs):\n          \"Plot importance with an optional limit to how many variables shown\"\n          df_copy = df.copy()\n          df_copy['feature'] = df_copy['feature'].str.slice(0,40)\n          df_copy = df_copy.sort_values(by='importance', ascending=asc)[:limit].sort_values(by='importance', ascending=not(asc))\n          ax = df_copy.plot.barh(x='feature', y='importance', sort_columns=True, **kwargs)\n          for p in ax.patches:\n            ax.annotate(f'{p.get_width():.4f}', ((p.get_width() * 1.005), p.get_y()  * 1.005))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:40.098026Z","iopub.execute_input":"2022-10-31T10:07:40.098430Z","iopub.status.idle":"2022-10-31T10:07:40.126556Z","shell.execute_reply.started":"2022-10-31T10:07:40.098389Z","shell.execute_reply":"2022-10-31T10:07:40.125464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = PermutationImportance(learn, train_df.iloc[:150000], bs=2048)\nres.get_results()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T10:07:40.139182Z","iopub.execute_input":"2022-10-31T10:07:40.141752Z","iopub.status.idle":"2022-10-31T10:09:19.490519Z","shell.execute_reply.started":"2022-10-31T10:07:40.141712Z","shell.execute_reply":"2022-10-31T10:09:19.489338Z"},"trusted":true},"execution_count":null,"outputs":[]}]}