{"cells":[{"metadata":{},"cell_type":"markdown","source":"TabNet (https://github.com/dreamquark-ai/tabnet/) starter kernel for RiiiD challenge\n\nFeature engineering is based on:\n- https://www.kaggle.com/code1110/riiid-gbdt-pipeline-baseline\n- https://www.kaggle.com/lgreig/simple-lgbm-baseline\n- https://www.kaggle.com/jsylas/riiid-lgbm-starter"},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install ../input/python-datatable/datatable-0.11.0-cp37-cp37m-manylinux2010_x86_64.whl >/dev/null\n!pip install \"../input/pytorch-tabnet/pytorch_tabnet-1.2.0-py3-none-any.whl\" >> quit","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preprocess"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Used most of coding from this kernel \nimport random\nimport os\nimport operator\nimport riiideducation\n\nimport datatable as dt\nimport dask.dataframe as dd\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.style as style\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score\nfrom matplotlib import pyplot\nfrom matplotlib.ticker import ScalarFormatter\n\nimport torch\nfrom pytorch_tabnet.tab_model import TabNetClassifier\n\nsns.set_context(\"talk\")\nstyle.use('fivethirtyeight')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Config"},{"metadata":{"trusted":true},"cell_type":"code","source":"class CFG:\n    START_IDX = 90000000\n    SEED = 42\n    TEST_SIZE = 0.2\n    N_EPOCHS = 5\n    BATCH_SZ = 1024\n    PATIENCE = 3\n    VIRTUAL_BS = 128\n    LR = 0.01\n    ND = 8  # Width of the decision prediction layer. Bigger values gives more capacity to the model with the risk of overfitting. \n    NA = 8  # Width of the attention embedding for each mask. According to the paper n_d=n_a is usually a good choice. \n    N_STEPS = 3 # Number of steps in the architecture (usually between 3 and 10)\n    GAMMA = 1.3 # This is the coefficient for feature reusage in the masks. A value close to 1 will make mask selection least correlated between layers. \n    #Values range from 1.0 to 2.0.\n    N_INDEPENDENT = 1 # Number of independent Gated Linear Units layers at each step. Usual values range from 1 to 5.\n    LAMBDA = 0\n    N_SHARED = 3 # Number of shared Gated Linear Units at each step Usual values range from 1 to 5\n    MOMENTUM = 0.1\n    CLIP = 1.0\n    MASK_TYPE = 'sparsemax' #(default='sparsemax') Either \"sparsemax\" or \"entmax\" : this is the masking function to use for selecting features","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# UTILS"},{"metadata":{"trusted":true},"cell_type":"code","source":"def seed_everything(seed_value):\n    random.seed(seed_value)\n    np.random.seed(seed_value)\n    torch.manual_seed(seed_value)\n    os.environ['PYTHONHASHSEED'] = str(seed_value)\n    \n    if torch.cuda.is_available(): \n        torch.cuda.manual_seed(seed_value)\n        torch.cuda.manual_seed_all(seed_value)\n        torch.backends.cudnn.deterministic = False\n        torch.backends.cudnn.benchmark = False\n        \nseed_everything(CFG.SEED)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# PREPARE TRAIN SET"},{"metadata":{"trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()\ntrain = dt.fread(\"../input/riiid-test-answer-prediction/train.csv\").to_pandas()\n\n# Remove lectures\ntrain = train[train.content_type_id == False]\n\n# Arrange by timestamp\ntrain = train.sort_values(['timestamp'], ascending=True)\n\n# Drop useless columns\ntrain.drop(['timestamp','content_type_id'], axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Average of correct answers for each content_id\nresults_c = train[['content_id', 'answered_correctly']].groupby(['content_id']).agg(['mean'])\nresults_c.columns = [\"answered_correctly_content\"]\n\n# Number of correct answers for each suser\nresults_u = train[['user_id','answered_correctly']].groupby(['user_id']).agg(['mean', 'sum'])\nresults_u.columns = [\"answered_correctly_user\", 'sum']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Select 10% of the training set (last interactions)\nX = train.iloc[CFG.START_IDX:,:]\n\n#del train\n\n# Merge with features previously computer\nX = pd.merge(X, results_u, on=['user_id'], how=\"left\")\nX = pd.merge(X, results_c, on=['content_id'], how=\"left\")\n\n# Remove all lectures\nX = X[X.answered_correctly!= -1 ]\nX = X.sort_values(['user_id'])\n\n# Get ground truth\nY = X[[\"answered_correctly\"]]\nX = X.drop([\"answered_correctly\"], axis=1)\n\n# Categorical encoding\nlb_make = LabelEncoder()\nX[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(X[\"prior_question_had_explanation\"])\n\n# Keep relevant features\nX = X[['answered_correctly_user', 'answered_correctly_content', 'sum', 'prior_question_elapsed_time', 'prior_question_had_explanation_enc']] \nX.fillna(0.5, inplace=True)\n\nX.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# TRAIN SPLIT"},{"metadata":{"trusted":true},"cell_type":"code","source":"#X = X[:2000]\n#Y = Y[:2000]\nXt, Xv, Yt, Yv = train_test_split(X, Y, test_size = CFG.TEST_SIZE, shuffle = False, random_state=CFG.SEED)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MODEL"},{"metadata":{"trusted":true},"cell_type":"code","source":"cat_idxs = Xt.columns.get_loc('prior_question_had_explanation_enc')\ncat_dims = Xt['prior_question_had_explanation_enc'].nunique()\n\nprint(cat_idxs, cat_dims)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = TabNetClassifier(\n    n_d = CFG.ND,\n    n_a = CFG.NA,\n    n_steps = CFG.N_STEPS,\n    gamma = CFG.GAMMA, \n    n_independent = CFG.N_INDEPENDENT,\n    n_shared = CFG.N_SHARED,\n    cat_dims=[cat_dims],\n    cat_emb_dim=1,\n    optimizer_params=dict(lr=CFG.LR),\n    momentum=CFG.MOMENTUM,\n    cat_idxs=[cat_idxs],\n    verbose=1,\n    #scheduler_params=dict(milestones=[20, 50, 80], gamma=0.5), \n    #scheduler_fn=torch.optim.lr_scheduler.MultiStepLR,\n    mask_type = CFG.MASK_TYPE,\n    lambda_sparse = CFG.LAMBDA,\n    clip_value = CFG.CLIP\n)\n\nmodel.fit(\n    X_train = Xt.values, \n    y_train = Yt['answered_correctly'].values,\n    X_valid = Xv.values, \n    y_valid = Yv['answered_correctly'].values,\n    max_epochs = CFG.N_EPOCHS, \n    patience = CFG.PATIENCE,\n    batch_size = CFG.BATCH_SZ, \n    virtual_batch_size = CFG.VIRTUAL_BS,\n    num_workers = 0,\n    weights = 1,\n    drop_last = False\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot losses\n#plt.plot(model.history['train']['loss'])\n#plt.plot(model.history['valid']['loss'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot learning rate\n#plt.plot(model.history['train']['lr'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot metric\n#plt.plot(np.array(model.history['train']['metric']) * -1)\n#plt.plot(np.array(model.history['valid']['metric']) * -1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# SUBMIT"},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df = pd.merge(test_df, results_u, on=['user_id'], how=\"left\")\n    test_df = pd.merge(test_df, results_c, on=['content_id'], how=\"left\")\n    test_df['answered_correctly_user'].fillna(0.5, inplace=True)\n    test_df['answered_correctly_content'].fillna(0.5, inplace=True)\n    test_df['sum'].fillna(0, inplace=True)\n    test_df['prior_question_had_explanation'].fillna(False, inplace=True)\n    test_df[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(test_df[\"prior_question_had_explanation\"])\n    \n    test_ = test_df[['answered_correctly_user', 'answered_correctly_content', 'sum','prior_question_elapsed_time','prior_question_had_explanation_enc']]\n    test_.fillna(0.5, inplace=True)   # should be modified !\n    test_df['answered_correctly'] = model.predict(test_.values)\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}